Rate Limit Planner — C source
Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/*
* Rate Limit Planner — turn provider rate limits plus a workload into a
* concrete schedule: batch size, spacing, timeline and wall time.
*
* Language: C (C11, standard library only)
* Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
* implementation).
* Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner
*
* Deterministic — no time reads. The TS lib throws RangeError; C returns
* a zeroed plan (bounded_by "none") and sets *err to a static message
* (NULL on success) — the caller checks err != NULL. Numeric plan fields
* are doubles because the original returns Infinity for the unbounded
* (no limits) case.
*/
#include <math.h>
#include <stdio.h>
#include <string.h>
#define RATE_LIMIT_WINDOW_MS 60000.0
#define RATE_LIMIT_DEFAULT_SAFETY 0.8
#define RATE_LIMIT_MAX_TIMELINE 10
#define RATE_LIMIT_MAX_WARNINGS 2
#define RATE_LIMIT_WARNING_LEN 256
typedef struct {
double rpm, tpm; /* limits per minute (ignored when unset) */
int has_rpm, has_tpm; /* 0 = not limited */
} RateLimits;
typedef struct {
int requests; /* total requests to run (>= 0) */
double avg_tokens_per_request; /* prompt + completion */
} Workload;
typedef struct {
double safety_factor; /* fraction of the limits to target */
int has_safety; /* 0 = use the default */
} PlanOptions;
typedef struct {
int batch;
double at_ms;
int requests;
double tokens;
} BatchSlice;
typedef struct {
double batch_size; /* requests per 60s window (0 = cannot run) */
double interval_ms; /* spacing between requests, ms */
double max_concurrent; /* sustainable in-flight requests */
const char *bounded_by; /* "rpm" | "tpm" | "both" | "none" */
BatchSlice timeline[RATE_LIMIT_MAX_TIMELINE];
int timeline_len;
double total_ms; /* estimated wall time, ms */
char warnings[RATE_LIMIT_MAX_WARNINGS][RATE_LIMIT_WARNING_LEN];
int warnings_len;
} RateLimitPlan;
/* Format like TS `toLocaleString('en-US')`: thousands separators. */
static void fmt_num(double v, char *buf, size_t n) {
if (n == 0) return;
if (isinf(v)) { snprintf(buf, n, v > 0 ? "Infinity" : "-Infinity"); return; }
if (isnan(v)) { snprintf(buf, n, "NaN"); return; }
size_t pos = 0;
if (v < 0.0 && pos + 1 < n) buf[pos++] = '-';
double a = fabs(v);
long long ip = (long long)floor(a);
char digits[48];
snprintf(digits, sizeof digits, "%lld", ip);
size_t len = strlen(digits);
for (size_t i = 0; i < len; i++) {
if (i > 0 && (len - i) % 3 == 0 && pos + 1 < n) buf[pos++] = ',';
if (pos + 1 < n) buf[pos++] = digits[i];
}
double frac = a - (double)ip;
if (frac > 0.0) {
char f[40];
snprintf(f, sizeof f, "%g", frac);
for (size_t i = 0; f[i] != '\0' && pos + 1 < n; i++) buf[pos++] = f[i];
}
buf[pos] = '\0';
}
RateLimitPlan plan_rate_limit(RateLimits limits, Workload workload, PlanOptions opts, const char **err) {
RateLimitPlan plan;
memset(&plan, 0, sizeof plan);
plan.bounded_by = "none";
if (err != NULL) *err = NULL;
double sf = opts.has_safety ? opts.safety_factor : RATE_LIMIT_DEFAULT_SAFETY;
if (workload.requests < 0 || workload.avg_tokens_per_request < 0.0) {
if (err != NULL) *err = "requests and avgTokensPerRequest must be >= 0";
return plan;
}
if (sf <= 0.0 || sf > 1.0) {
if (err != NULL) *err = "safetyFactor must be in (0, 1]";
return plan;
}
int has_rpm_eff = limits.has_rpm;
int has_tpm_eff = limits.has_tpm;
double rpm_eff = has_rpm_eff ? limits.rpm * sf : 0.0;
double tpm_eff = has_tpm_eff ? limits.tpm * sf : 0.0;
/* Impossible: one request alone exceeds the token budget. */
if (has_tpm_eff && workload.avg_tokens_per_request > tpm_eff && workload.requests > 0) {
plan.bounded_by = "tpm";
plan.total_ms = INFINITY;
char avg_s[48], lim_s[48];
fmt_num(workload.avg_tokens_per_request, avg_s, sizeof avg_s);
fmt_num(floor(tpm_eff), lim_s, sizeof lim_s);
snprintf(plan.warnings[0], RATE_LIMIT_WARNING_LEN,
"A single request averages %s tokens but the effective token limit is %s/min — "
"no schedule can run this. Shrink requests or raise the tier.",
avg_s, lim_s);
plan.warnings_len = 1;
return plan;
}
double by_rpm = has_rpm_eff ? rpm_eff : INFINITY;
double by_tokens = (!has_tpm_eff || workload.avg_tokens_per_request == 0.0)
? INFINITY
: tpm_eff / workload.avg_tokens_per_request;
if (isinf(by_rpm) && isinf(by_tokens)) {
snprintf(plan.warnings[plan.warnings_len], RATE_LIMIT_WARNING_LEN,
"No limits set — the plan assumes an unbounded endpoint. "
"Add RPM or TPM for a real schedule.");
plan.warnings_len++;
}
double steady = fmax(1.0, floor(fmin(by_rpm, by_tokens)));
if (isinf(by_rpm) && isinf(by_tokens)) {
plan.bounded_by = "none";
} else if (floor(by_rpm) == floor(by_tokens)) {
plan.bounded_by = "both";
} else {
plan.bounded_by = by_rpm < by_tokens ? "rpm" : "tpm";
}
/* Even pacing inside the window: batch_size requests spread over 60s. */
plan.batch_size = steady;
plan.interval_ms = round(RATE_LIMIT_WINDOW_MS / steady);
/* With even spacing and a per-request latency near interval_ms, one
* request is in flight at a time; concurrency >1 only helps
* sub-interval latencies, so the safe published floor is 1 — batch
* bursts raise it to batch_size. */
plan.max_concurrent = steady == 1.0 ? 1.0 : fmin(steady, ceil(steady / 4.0));
double remaining = (double)workload.requests;
int batch = 0;
while (remaining > 0.0 && batch < RATE_LIMIT_MAX_TIMELINE) {
double take = fmin(steady, remaining);
plan.timeline[batch].batch = batch + 1;
plan.timeline[batch].at_ms = (double)batch * RATE_LIMIT_WINDOW_MS;
plan.timeline[batch].requests = (int)take;
plan.timeline[batch].tokens = take * workload.avg_tokens_per_request;
plan.timeline_len = batch + 1;
remaining -= take;
batch++;
}
int windows_needed =
workload.requests > 0 ? (int)ceil((double)workload.requests / steady) : 0;
double last_window_requests =
windows_needed > 0 ? (double)workload.requests - (double)(windows_needed - 1) * steady : 0.0;
plan.total_ms = windows_needed > 0
? (double)(windows_needed - 1) * RATE_LIMIT_WINDOW_MS +
plan.interval_ms * last_window_requests
: 0.0;
if (has_rpm_eff && workload.requests > 0 && steady > by_rpm) {
snprintf(plan.warnings[plan.warnings_len], RATE_LIMIT_WARNING_LEN,
"Rounded up to at least one request per window — even a single "
"request per minute keeps the schedule honest.");
plan.warnings_len++;
}
return plan;
}
/* Human summary line for the plan (used by the island + docs).
* Writes into `out` (at most out_size bytes) and returns `out`. */
/* Plain interpolation like TS `${v}` (no separators, no trailing ".0"). */
static void plain_num(double v, char *buf, size_t n) {
if (isinf(v)) {
snprintf(buf, n, v > 0 ? "Infinity" : "-Infinity");
} else if (v == floor(v)) {
snprintf(buf, n, "%.0f", v);
} else {
snprintf(buf, n, "%g", v);
}
}
const char *describe_plan(const RateLimitPlan *plan, char *out, size_t out_size) {
if (plan->batch_size == 0.0) {
snprintf(out, out_size, "No viable schedule.");
} else if (strcmp(plan->bounded_by, "none") == 0) {
char b[48];
plain_num(plan->batch_size, b, sizeof b);
snprintf(out, out_size, "%s+ requests per window — endpoint treated as unbounded.", b);
} else {
char limiter[64];
if (strcmp(plan->bounded_by, "both") == 0) {
snprintf(limiter, sizeof limiter, "both limits bind together");
} else {
const char *up = strcmp(plan->bounded_by, "rpm") == 0 ? "RPM" : "TPM";
snprintf(limiter, sizeof limiter, "the %s limit binds first", up);
}
char bs[48], iv[48];
plain_num(plan->batch_size, bs, sizeof bs);
plain_num(plan->interval_ms, iv, sizeof iv);
snprintf(out, out_size, "%s requests per 60s window (one every %sms) — %s.", bs, iv, limiter);
}
return out;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →