Rate Limit Planner — Zig source
Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Rate Limit Planner — turn provider rate limits plus a workload into a
// concrete schedule: batch size, spacing, what caps it, and finish time.
// Deterministic — no clock reads.
//
// Language: Zig (0.13+, stdlib only)
// Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
// implementation). The timeline (max 10 slices) is allocated from the
// caller's allocator; free with freePlan. Infinity maps to
// std.math.inf(f64). RangeError maps to error unions.
//
// Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner
const std = @import("std");
pub const BoundedBy = enum { rpm, tpm, both, none };
/// Requests/tokens per minute; null = not limited.
pub const Limits = struct {
rpm: ?f64 = null,
tpm: ?f64 = null,
};
pub const Workload = struct {
/// Total requests to run.
requests: i64,
/// Average tokens per request (prompt + completion).
avg_tokens_per_request: i64,
};
pub const Options = struct {
/// Fraction of the limits to target, leaving headroom for retries.
safety_factor: ?f64 = null,
};
pub const BatchSlice = struct {
batch: i64,
at_ms: i64,
requests: i64,
tokens: i64,
};
pub const Plan = struct {
/// Requests to send per 60s window (0 when the workload cannot run).
batch_size: i64,
/// Steady-state spacing between individual requests, in ms.
interval_ms: i64,
/// Sustainable concurrent in-flight requests under even spacing.
max_concurrent: i64,
bounded_by: BoundedBy,
/// First batches of the schedule (max 10). Allocated.
timeline: []BatchSlice,
/// Estimated total wall time, in ms (inf when impossible).
total_ms: f64,
/// Allocated strings; free with freePlan.
warnings: [][]const u8,
};
pub const PlanError = error{
NegativeInputs,
SafetyFactor,
OutOfMemory,
};
const WINDOW_MS: i64 = 60_000;
const DEFAULT_SAFETY: f64 = 0.8;
const INF = std.math.inf(f64);
/// Group an integer with thousands separators into `buf` (>= 24 bytes).
fn commaFormat(buf: []u8, v: i64) []const u8 {
var tmp: [24]u8 = undefined;
const digits = std.fmt.bufPrint(&tmp, "{d}", .{v}) catch unreachable;
var out_len: usize = 0;
const n = digits.len;
for (digits, 0..) |ch, i| {
buf[out_len] = ch;
out_len += 1;
const remaining = n - i - 1;
if (remaining > 0 and remaining % 3 == 0) {
buf[out_len] = ',';
out_len += 1;
}
}
return buf[0..out_len];
}
/// Plan a schedule. Deterministic; all allocations come from `alloc`.
pub fn planRateLimit(
alloc: std.mem.Allocator,
limits: Limits,
workload: Workload,
opts: Options,
) PlanError!Plan {
const sf = opts.safety_factor orelse DEFAULT_SAFETY;
var warnings: std.ArrayList([]const u8) = .init(alloc);
errdefer {
for (warnings.items) |w| alloc.free(w);
warnings.deinit();
}
if (workload.requests < 0 or workload.avg_tokens_per_request < 0) {
return error.NegativeInputs;
}
if (sf <= 0 or sf > 1) return error.SafetyFactor;
const rpm_eff: ?f64 = if (limits.rpm) |r| r * sf else null;
const tpm_eff: ?f64 = if (limits.tpm) |t| t * sf else null;
// Impossible: one request alone exceeds the token budget.
if (tpm_eff) |te| {
if (@as(f64, @floatFromInt(workload.avg_tokens_per_request)) > te
and workload.requests > 0)
{
var b1: [24]u8 = undefined;
var b2: [24]u8 = undefined;
const w = std.fmt.allocPrint(alloc,
"A single request averages {s} tokens but the effective token limit is {s}/min — no schedule can run this. Shrink requests or raise the tier.", .{
commaFormat(&b1, workload.avg_tokens_per_request),
commaFormat(&b2, @intFromFloat(@floor(te))),
}) catch return error.OutOfMemory;
try warnings.append(w);
return .{
.batch_size = 0,
.interval_ms = 0,
.max_concurrent = 0,
.bounded_by = .tpm,
.timeline = &.{},
.total_ms = INF,
.warnings = try warnings.toOwnedSlice(),
};
}
}
const by_rpm: f64 = rpm_eff orelse INF;
const by_tokens: f64 = if (tpm_eff == null or workload.avg_tokens_per_request == 0)
INF
else
tpm_eff.? / @as(f64, @floatFromInt(workload.avg_tokens_per_request));
const no_rpm = std.math.isInf(by_rpm);
const no_tokens = std.math.isInf(by_tokens);
if (no_rpm and no_tokens) {
try warnings.append(try alloc.dupe(u8,
"No limits set — the plan assumes an unbounded endpoint. Add RPM or TPM for a real schedule."));
}
const steady: i64 = @max(1, @as(i64, @intFromFloat(@floor(@min(by_rpm, by_tokens)))));
const bounded_by: BoundedBy = blk: {
if (no_rpm and no_tokens) break :blk .none;
const rpm_floor: i64 = @intFromFloat(@floor(by_rpm));
const tpm_floor: i64 = @intFromFloat(@floor(by_tokens));
if (rpm_floor == tpm_floor) break :blk .both;
break :blk if (by_rpm < by_tokens) .rpm else .tpm;
};
// Even pacing inside the window: batch_size requests spread over 60s.
const interval_ms: i64 = @intFromFloat(@round(@as(f64, WINDOW_MS) / @as(f64, @floatFromInt(steady))));
// Concurrency >1 only helps sub-interval latencies; the safe published
// floor is 1 — batch bursts raise it to batch_size/4.
const max_concurrent: i64 = if (steady == 1) 1 else @min(steady, @divTrunc(steady + 3, 4));
var timeline: std.ArrayList(BatchSlice) = .init(alloc);
errdefer timeline.deinit();
var remaining = workload.requests;
var batch: i64 = 0;
while (remaining > 0 and batch < 10) {
const take = @min(steady, remaining);
try timeline.append(.{
.batch = batch + 1,
.at_ms = batch * WINDOW_MS,
.requests = take,
.tokens = take * workload.avg_tokens_per_request,
});
remaining -= take;
batch += 1;
}
const windows_needed: i64 = if (workload.requests > 0)
@intFromFloat(@ceil(@as(f64, @floatFromInt(workload.requests)) / @as(f64, @floatFromInt(steady))))
else
0;
const last_window_requests: i64 = if (windows_needed > 0)
workload.requests - (windows_needed - 1) * steady
else
0;
const total_ms: f64 = if (windows_needed > 0)
@as(f64, @floatFromInt((windows_needed - 1) * WINDOW_MS))
+ @as(f64, @floatFromInt(interval_ms * last_window_requests))
else
0;
if (rpm_eff != null and workload.requests > 0 and @as(f64, @floatFromInt(steady)) > by_rpm) {
try warnings.append(try alloc.dupe(u8,
"Rounded up to at least one request per window — even a single request per minute keeps the schedule honest."));
}
return .{
.batch_size = steady,
.interval_ms = interval_ms,
.max_concurrent = max_concurrent,
.bounded_by = bounded_by,
.timeline = try timeline.toOwnedSlice(),
.total_ms = total_ms,
.warnings = try warnings.toOwnedSlice(),
};
}
/// Free a plan's allocations.
pub fn freePlan(alloc: std.mem.Allocator, plan: Plan) void {
alloc.free(plan.timeline);
for (plan.warnings) |w| alloc.free(w);
alloc.free(plan.warnings);
}
/// Human summary line for the plan. Caller owns the returned string.
pub fn describePlan(alloc: std.mem.Allocator, plan: Plan) ![]const u8 {
if (plan.batch_size == 0) return alloc.dupe(u8, "No viable schedule.");
if (plan.bounded_by == .none) {
return std.fmt.allocPrint(alloc, "{d}+ requests per window — endpoint treated as unbounded.", .{plan.batch_size});
}
const limiter: []const u8 = switch (plan.bounded_by) {
.both => "both limits bind together",
.rpm => "the RPM limit binds first",
.tpm => "the TPM limit binds first",
.none => unreachable,
};
return std.fmt.allocPrint(alloc, "{d} requests per 60s window (one every {d}ms) — {s}.", .{
plan.batch_size, plan.interval_ms, limiter,
});
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →