Skip to content

Rate Limit Planner — Zig source

Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Rate Limit Planner — turn provider rate limits plus a workload into a
// concrete schedule: batch size, spacing, what caps it, and finish time.
// Deterministic — no clock reads.
//
// Language: Zig (0.13+, stdlib only)
// Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
// implementation). The timeline (max 10 slices) is allocated from the
// caller's allocator; free with freePlan. Infinity maps to
// std.math.inf(f64). RangeError maps to error unions.
//
// Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner

const std = @import("std");

pub const BoundedBy = enum { rpm, tpm, both, none };

/// Requests/tokens per minute; null = not limited.
pub const Limits = struct {
    rpm: ?f64 = null,
    tpm: ?f64 = null,
};

pub const Workload = struct {
    /// Total requests to run.
    requests: i64,
    /// Average tokens per request (prompt + completion).
    avg_tokens_per_request: i64,
};

pub const Options = struct {
    /// Fraction of the limits to target, leaving headroom for retries.
    safety_factor: ?f64 = null,
};

pub const BatchSlice = struct {
    batch: i64,
    at_ms: i64,
    requests: i64,
    tokens: i64,
};

pub const Plan = struct {
    /// Requests to send per 60s window (0 when the workload cannot run).
    batch_size: i64,
    /// Steady-state spacing between individual requests, in ms.
    interval_ms: i64,
    /// Sustainable concurrent in-flight requests under even spacing.
    max_concurrent: i64,
    bounded_by: BoundedBy,
    /// First batches of the schedule (max 10). Allocated.
    timeline: []BatchSlice,
    /// Estimated total wall time, in ms (inf when impossible).
    total_ms: f64,
    /// Allocated strings; free with freePlan.
    warnings: [][]const u8,
};

pub const PlanError = error{
    NegativeInputs,
    SafetyFactor,
    OutOfMemory,
};

const WINDOW_MS: i64 = 60_000;
const DEFAULT_SAFETY: f64 = 0.8;

const INF = std.math.inf(f64);

/// Group an integer with thousands separators into `buf` (>= 24 bytes).
fn commaFormat(buf: []u8, v: i64) []const u8 {
    var tmp: [24]u8 = undefined;
    const digits = std.fmt.bufPrint(&tmp, "{d}", .{v}) catch unreachable;
    var out_len: usize = 0;
    const n = digits.len;
    for (digits, 0..) |ch, i| {
        buf[out_len] = ch;
        out_len += 1;
        const remaining = n - i - 1;
        if (remaining > 0 and remaining % 3 == 0) {
            buf[out_len] = ',';
            out_len += 1;
        }
    }
    return buf[0..out_len];
}

/// Plan a schedule. Deterministic; all allocations come from `alloc`.
pub fn planRateLimit(
    alloc: std.mem.Allocator,
    limits: Limits,
    workload: Workload,
    opts: Options,
) PlanError!Plan {
    const sf = opts.safety_factor orelse DEFAULT_SAFETY;
    var warnings: std.ArrayList([]const u8) = .init(alloc);
    errdefer {
        for (warnings.items) |w| alloc.free(w);
        warnings.deinit();
    }
    if (workload.requests < 0 or workload.avg_tokens_per_request < 0) {
        return error.NegativeInputs;
    }
    if (sf <= 0 or sf > 1) return error.SafetyFactor;

    const rpm_eff: ?f64 = if (limits.rpm) |r| r * sf else null;
    const tpm_eff: ?f64 = if (limits.tpm) |t| t * sf else null;

    // Impossible: one request alone exceeds the token budget.
    if (tpm_eff) |te| {
        if (@as(f64, @floatFromInt(workload.avg_tokens_per_request)) > te
            and workload.requests > 0)
        {
            var b1: [24]u8 = undefined;
            var b2: [24]u8 = undefined;
            const w = std.fmt.allocPrint(alloc,
                "A single request averages {s} tokens but the effective token limit is {s}/min — no schedule can run this. Shrink requests or raise the tier.", .{
                    commaFormat(&b1, workload.avg_tokens_per_request),
                    commaFormat(&b2, @intFromFloat(@floor(te))),
                }) catch return error.OutOfMemory;
            try warnings.append(w);
            return .{
                .batch_size = 0,
                .interval_ms = 0,
                .max_concurrent = 0,
                .bounded_by = .tpm,
                .timeline = &.{},
                .total_ms = INF,
                .warnings = try warnings.toOwnedSlice(),
            };
        }
    }

    const by_rpm: f64 = rpm_eff orelse INF;
    const by_tokens: f64 = if (tpm_eff == null or workload.avg_tokens_per_request == 0)
        INF
    else
        tpm_eff.? / @as(f64, @floatFromInt(workload.avg_tokens_per_request));

    const no_rpm = std.math.isInf(by_rpm);
    const no_tokens = std.math.isInf(by_tokens);
    if (no_rpm and no_tokens) {
        try warnings.append(try alloc.dupe(u8,
            "No limits set — the plan assumes an unbounded endpoint. Add RPM or TPM for a real schedule."));
    }

    const steady: i64 = @max(1, @as(i64, @intFromFloat(@floor(@min(by_rpm, by_tokens)))));
    const bounded_by: BoundedBy = blk: {
        if (no_rpm and no_tokens) break :blk .none;
        const rpm_floor: i64 = @intFromFloat(@floor(by_rpm));
        const tpm_floor: i64 = @intFromFloat(@floor(by_tokens));
        if (rpm_floor == tpm_floor) break :blk .both;
        break :blk if (by_rpm < by_tokens) .rpm else .tpm;
    };

    // Even pacing inside the window: batch_size requests spread over 60s.
    const interval_ms: i64 = @intFromFloat(@round(@as(f64, WINDOW_MS) / @as(f64, @floatFromInt(steady))));
    // Concurrency >1 only helps sub-interval latencies; the safe published
    // floor is 1 — batch bursts raise it to batch_size/4.
    const max_concurrent: i64 = if (steady == 1) 1 else @min(steady, @divTrunc(steady + 3, 4));

    var timeline: std.ArrayList(BatchSlice) = .init(alloc);
    errdefer timeline.deinit();
    var remaining = workload.requests;
    var batch: i64 = 0;
    while (remaining > 0 and batch < 10) {
        const take = @min(steady, remaining);
        try timeline.append(.{
            .batch = batch + 1,
            .at_ms = batch * WINDOW_MS,
            .requests = take,
            .tokens = take * workload.avg_tokens_per_request,
        });
        remaining -= take;
        batch += 1;
    }

    const windows_needed: i64 = if (workload.requests > 0)
        @intFromFloat(@ceil(@as(f64, @floatFromInt(workload.requests)) / @as(f64, @floatFromInt(steady))))
    else
        0;
    const last_window_requests: i64 = if (windows_needed > 0)
        workload.requests - (windows_needed - 1) * steady
    else
        0;
    const total_ms: f64 = if (windows_needed > 0)
        @as(f64, @floatFromInt((windows_needed - 1) * WINDOW_MS))
            + @as(f64, @floatFromInt(interval_ms * last_window_requests))
    else
        0;

    if (rpm_eff != null and workload.requests > 0 and @as(f64, @floatFromInt(steady)) > by_rpm) {
        try warnings.append(try alloc.dupe(u8,
            "Rounded up to at least one request per window — even a single request per minute keeps the schedule honest."));
    }

    return .{
        .batch_size = steady,
        .interval_ms = interval_ms,
        .max_concurrent = max_concurrent,
        .bounded_by = bounded_by,
        .timeline = try timeline.toOwnedSlice(),
        .total_ms = total_ms,
        .warnings = try warnings.toOwnedSlice(),
    };
}

/// Free a plan's allocations.
pub fn freePlan(alloc: std.mem.Allocator, plan: Plan) void {
    alloc.free(plan.timeline);
    for (plan.warnings) |w| alloc.free(w);
    alloc.free(plan.warnings);
}

/// Human summary line for the plan. Caller owns the returned string.
pub fn describePlan(alloc: std.mem.Allocator, plan: Plan) ![]const u8 {
    if (plan.batch_size == 0) return alloc.dupe(u8, "No viable schedule.");
    if (plan.bounded_by == .none) {
        return std.fmt.allocPrint(alloc, "{d}+ requests per window — endpoint treated as unbounded.", .{plan.batch_size});
    }
    const limiter: []const u8 = switch (plan.bounded_by) {
        .both => "both limits bind together",
        .rpm => "the RPM limit binds first",
        .tpm => "the TPM limit binds first",
        .none => unreachable,
    };
    return std.fmt.allocPrint(alloc, "{d} requests per 60s window (one every {d}ms) — {s}.", .{
        plan.batch_size, plan.interval_ms, limiter,
    });
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →