Skip to content

Rate Limit Planner — Swift source

Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.

This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Rate Limit Planner — turn provider rate limits plus a workload into a
// concrete schedule: batch size, spacing, what caps it, and finish time.
// Deterministic — no clock reads.
//
// Language: Swift (5.9+, zero dependencies)
// Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner

/// Errors mirroring the TS RangeError contract.
enum PlanError: Error, CustomStringConvertible {
    case negativeInputs
    case safetyFactor

    var description: String {
        switch self {
        case .negativeInputs: return "requests and avgTokensPerRequest must be >= 0"
        case .safetyFactor: return "safetyFactor must be in (0, 1]"
        }
    }
}

/// Requests/tokens per minute; nil = not limited.
struct RateLimits {
    var rpm: Double?
    var tpm: Double?
}

struct Workload {
    var requests: Int
    var avgTokensPerRequest: Int
}

struct PlanOptions {
    /// Fraction of the limits to target, leaving headroom for retries.
    var safetyFactor: Double?
}

struct BatchSlice {
    var batch: Int
    var atMs: Int
    var requests: Int
    var tokens: Int
}

struct RateLimitPlan {
    /// Requests to send per 60s window (0 when the workload cannot run).
    var batchSize: Int
    /// Steady-state spacing between individual requests, in ms.
    var intervalMs: Int
    /// Sustainable concurrent in-flight requests under even spacing.
    var maxConcurrent: Int
    /// Which limit binds first: "rpm", "tpm", "both", or "none".
    var boundedBy: String
    /// First batches of the schedule (max 10).
    var timeline: [BatchSlice]
    /// Estimated total wall time, in ms (.infinity when impossible).
    var totalMs: Double
    var warnings: [String]
}

private let windowMs = 60_000.0
private let defaultSafety = 0.8

/// Group an integer with thousands separators (toLocaleString stand-in).
private func fmt(_ v: Int) -> String {
    let s = String(v)
    var out = ""
    for (offset, ch) in s.enumerated().reversed() {
        out.append(ch)
        let remaining = s.count - offset - 1
        if remaining > 0 && remaining % 3 == 0 { out.append(",") }
    }
    return String(out.reversed())
}

/// Plan a schedule. Throws the PlanError mirror of the TS RangeError.
func planRateLimit(_ limits: RateLimits, _ workload: Workload,
                   _ opts: PlanOptions = PlanOptions()) throws -> RateLimitPlan
{
    let sf = opts.safetyFactor ?? defaultSafety
    var warnings: [String] = []
    if workload.requests < 0 || workload.avgTokensPerRequest < 0 {
        throw PlanError.negativeInputs
    }
    if sf <= 0 || sf > 1 { throw PlanError.safetyFactor }

    let rpmEff = limits.rpm.map { $0 * sf }
    let tpmEff = limits.tpm.map { $0 * sf }

    // Impossible: one request alone exceeds the token budget.
    if let tpmEff = tpmEff, Double(workload.avgTokensPerRequest) > tpmEff,
       workload.requests > 0
    {
        return RateLimitPlan(
            batchSize: 0, intervalMs: 0, maxConcurrent: 0, boundedBy: "tpm",
            timeline: [], totalMs: .infinity,
            warnings: [
                "A single request averages \(fmt(workload.avgTokensPerRequest)) tokens but " +
                    "the effective token limit is \(fmt(Int(tpmEff.rounded(.down))))/min — " +
                    "no schedule can run this. Shrink requests or raise the tier.",
            ])
    }

    let byRpm = rpmEff ?? Double.infinity
    let byTokens = tpmEff == nil || workload.avgTokensPerRequest == 0
        ? Double.infinity
        : tpmEff! / Double(workload.avgTokensPerRequest)

    let noRpm = byRpm.isInfinite
    let noTokens = byTokens.isInfinite
    if noRpm && noTokens {
        warnings.append(
            "No limits set — the plan assumes an unbounded endpoint. " +
                "Add RPM or TPM for a real schedule.")
    }

    let steady = max(1, Int(min(byRpm, byTokens).rounded(.down)))
    let boundedBy: String
    if noRpm && noTokens {
        boundedBy = "none"
    } else if Int(byRpm.rounded(.down)) == Int(byTokens.rounded(.down)) {
        boundedBy = "both"
    } else if byRpm < byTokens {
        boundedBy = "rpm"
    } else {
        boundedBy = "tpm"
    }

    // Even pacing inside the window: batchSize requests spread over 60s.
    let intervalMs = Int((windowMs / Double(steady)).rounded())
    // Concurrency >1 only helps sub-interval latencies; the safe published
    // floor is 1 — batch bursts raise it to batchSize/4.
    let maxConcurrent = steady == 1 ? 1 : min(steady, Int((Double(steady) / 4.0).rounded(.up)))

    var timeline: [BatchSlice] = []
    var remaining = workload.requests
    var batch = 0
    while remaining > 0 && batch < 10 {
        let take = min(steady, remaining)
        timeline.append(BatchSlice(
            batch: batch + 1,
            atMs: batch * Int(windowMs),
            requests: take,
            tokens: take * workload.avgTokensPerRequest))
        remaining -= take
        batch += 1
    }

    let windowsNeeded = workload.requests > 0
        ? Int((Double(workload.requests) / Double(steady)).rounded(.up)) : 0
    let lastWindowRequests = windowsNeeded > 0
        ? workload.requests - (windowsNeeded - 1) * steady : 0
    let totalMs: Double = windowsNeeded > 0
        ? Double((windowsNeeded - 1) * Int(windowMs))
            + Double(intervalMs * lastWindowRequests)
        : 0

    if rpmEff != nil && workload.requests > 0 && Double(steady) > byRpm {
        warnings.append(
            "Rounded up to at least one request per window — even a single request per " +
                "minute keeps the schedule honest.")
    }

    return RateLimitPlan(batchSize: steady, intervalMs: intervalMs,
                         maxConcurrent: maxConcurrent, boundedBy: boundedBy,
                         timeline: timeline, totalMs: totalMs, warnings: warnings)
}

/// Human summary line for the plan.
func describePlan(_ plan: RateLimitPlan) -> String {
    if plan.batchSize == 0 { return "No viable schedule." }
    if plan.boundedBy == "none" {
        return "\(plan.batchSize)+ requests per window — endpoint treated as unbounded."
    }
    let limiter = plan.boundedBy == "both"
        ? "both limits bind together"
        : "the \(plan.boundedBy.uppercased()) limit binds first"
    return "\(plan.batchSize) requests per 60s window (one every \(plan.intervalMs)ms) — \(limiter)."
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →