Skip to content

Rate Limit Planner — C source

Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * Rate Limit Planner — turn provider rate limits plus a workload into a
 * concrete schedule: batch size, spacing, timeline and wall time.
 *
 * Language: C (C11, standard library only)
 * Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
 * implementation).
 * Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner
 *
 * Deterministic — no time reads. The TS lib throws RangeError; C returns
 * a zeroed plan (bounded_by "none") and sets *err to a static message
 * (NULL on success) — the caller checks err != NULL. Numeric plan fields
 * are doubles because the original returns Infinity for the unbounded
 * (no limits) case.
 */

#include <math.h>
#include <stdio.h>
#include <string.h>

#define RATE_LIMIT_WINDOW_MS 60000.0
#define RATE_LIMIT_DEFAULT_SAFETY 0.8
#define RATE_LIMIT_MAX_TIMELINE 10
#define RATE_LIMIT_MAX_WARNINGS 2
#define RATE_LIMIT_WARNING_LEN 256

typedef struct {
    double rpm, tpm;      /* limits per minute (ignored when unset) */
    int has_rpm, has_tpm; /* 0 = not limited */
} RateLimits;

typedef struct {
    int requests;                    /* total requests to run (>= 0) */
    double avg_tokens_per_request;   /* prompt + completion */
} Workload;

typedef struct {
    double safety_factor; /* fraction of the limits to target */
    int has_safety;       /* 0 = use the default */
} PlanOptions;

typedef struct {
    int batch;
    double at_ms;
    int requests;
    double tokens;
} BatchSlice;

typedef struct {
    double batch_size;      /* requests per 60s window (0 = cannot run) */
    double interval_ms;     /* spacing between requests, ms */
    double max_concurrent;  /* sustainable in-flight requests */
    const char *bounded_by; /* "rpm" | "tpm" | "both" | "none" */
    BatchSlice timeline[RATE_LIMIT_MAX_TIMELINE];
    int timeline_len;
    double total_ms; /* estimated wall time, ms */
    char warnings[RATE_LIMIT_MAX_WARNINGS][RATE_LIMIT_WARNING_LEN];
    int warnings_len;
} RateLimitPlan;

/* Format like TS `toLocaleString('en-US')`: thousands separators. */
static void fmt_num(double v, char *buf, size_t n) {
    if (n == 0) return;
    if (isinf(v)) { snprintf(buf, n, v > 0 ? "Infinity" : "-Infinity"); return; }
    if (isnan(v)) { snprintf(buf, n, "NaN"); return; }
    size_t pos = 0;
    if (v < 0.0 && pos + 1 < n) buf[pos++] = '-';
    double a = fabs(v);
    long long ip = (long long)floor(a);
    char digits[48];
    snprintf(digits, sizeof digits, "%lld", ip);
    size_t len = strlen(digits);
    for (size_t i = 0; i < len; i++) {
        if (i > 0 && (len - i) % 3 == 0 && pos + 1 < n) buf[pos++] = ',';
        if (pos + 1 < n) buf[pos++] = digits[i];
    }
    double frac = a - (double)ip;
    if (frac > 0.0) {
        char f[40];
        snprintf(f, sizeof f, "%g", frac);
        for (size_t i = 0; f[i] != '\0' && pos + 1 < n; i++) buf[pos++] = f[i];
    }
    buf[pos] = '\0';
}

RateLimitPlan plan_rate_limit(RateLimits limits, Workload workload, PlanOptions opts, const char **err) {
    RateLimitPlan plan;
    memset(&plan, 0, sizeof plan);
    plan.bounded_by = "none";
    if (err != NULL) *err = NULL;

    double sf = opts.has_safety ? opts.safety_factor : RATE_LIMIT_DEFAULT_SAFETY;
    if (workload.requests < 0 || workload.avg_tokens_per_request < 0.0) {
        if (err != NULL) *err = "requests and avgTokensPerRequest must be >= 0";
        return plan;
    }
    if (sf <= 0.0 || sf > 1.0) {
        if (err != NULL) *err = "safetyFactor must be in (0, 1]";
        return plan;
    }

    int has_rpm_eff = limits.has_rpm;
    int has_tpm_eff = limits.has_tpm;
    double rpm_eff = has_rpm_eff ? limits.rpm * sf : 0.0;
    double tpm_eff = has_tpm_eff ? limits.tpm * sf : 0.0;

    /* Impossible: one request alone exceeds the token budget. */
    if (has_tpm_eff && workload.avg_tokens_per_request > tpm_eff && workload.requests > 0) {
        plan.bounded_by = "tpm";
        plan.total_ms = INFINITY;
        char avg_s[48], lim_s[48];
        fmt_num(workload.avg_tokens_per_request, avg_s, sizeof avg_s);
        fmt_num(floor(tpm_eff), lim_s, sizeof lim_s);
        snprintf(plan.warnings[0], RATE_LIMIT_WARNING_LEN,
                 "A single request averages %s tokens but the effective token limit is %s/min — "
                 "no schedule can run this. Shrink requests or raise the tier.",
                 avg_s, lim_s);
        plan.warnings_len = 1;
        return plan;
    }

    double by_rpm = has_rpm_eff ? rpm_eff : INFINITY;
    double by_tokens = (!has_tpm_eff || workload.avg_tokens_per_request == 0.0)
                           ? INFINITY
                           : tpm_eff / workload.avg_tokens_per_request;

    if (isinf(by_rpm) && isinf(by_tokens)) {
        snprintf(plan.warnings[plan.warnings_len], RATE_LIMIT_WARNING_LEN,
                 "No limits set — the plan assumes an unbounded endpoint. "
                 "Add RPM or TPM for a real schedule.");
        plan.warnings_len++;
    }

    double steady = fmax(1.0, floor(fmin(by_rpm, by_tokens)));
    if (isinf(by_rpm) && isinf(by_tokens)) {
        plan.bounded_by = "none";
    } else if (floor(by_rpm) == floor(by_tokens)) {
        plan.bounded_by = "both";
    } else {
        plan.bounded_by = by_rpm < by_tokens ? "rpm" : "tpm";
    }

    /* Even pacing inside the window: batch_size requests spread over 60s. */
    plan.batch_size = steady;
    plan.interval_ms = round(RATE_LIMIT_WINDOW_MS / steady);
    /* With even spacing and a per-request latency near interval_ms, one
     * request is in flight at a time; concurrency >1 only helps
     * sub-interval latencies, so the safe published floor is 1 — batch
     * bursts raise it to batch_size. */
    plan.max_concurrent = steady == 1.0 ? 1.0 : fmin(steady, ceil(steady / 4.0));

    double remaining = (double)workload.requests;
    int batch = 0;
    while (remaining > 0.0 && batch < RATE_LIMIT_MAX_TIMELINE) {
        double take = fmin(steady, remaining);
        plan.timeline[batch].batch = batch + 1;
        plan.timeline[batch].at_ms = (double)batch * RATE_LIMIT_WINDOW_MS;
        plan.timeline[batch].requests = (int)take;
        plan.timeline[batch].tokens = take * workload.avg_tokens_per_request;
        plan.timeline_len = batch + 1;
        remaining -= take;
        batch++;
    }

    int windows_needed =
        workload.requests > 0 ? (int)ceil((double)workload.requests / steady) : 0;
    double last_window_requests =
        windows_needed > 0 ? (double)workload.requests - (double)(windows_needed - 1) * steady : 0.0;
    plan.total_ms = windows_needed > 0
                        ? (double)(windows_needed - 1) * RATE_LIMIT_WINDOW_MS +
                              plan.interval_ms * last_window_requests
                        : 0.0;

    if (has_rpm_eff && workload.requests > 0 && steady > by_rpm) {
        snprintf(plan.warnings[plan.warnings_len], RATE_LIMIT_WARNING_LEN,
                 "Rounded up to at least one request per window — even a single "
                 "request per minute keeps the schedule honest.");
        plan.warnings_len++;
    }
    return plan;
}

/* Human summary line for the plan (used by the island + docs).
 * Writes into `out` (at most out_size bytes) and returns `out`. */
/* Plain interpolation like TS `${v}` (no separators, no trailing ".0"). */
static void plain_num(double v, char *buf, size_t n) {
    if (isinf(v)) {
        snprintf(buf, n, v > 0 ? "Infinity" : "-Infinity");
    } else if (v == floor(v)) {
        snprintf(buf, n, "%.0f", v);
    } else {
        snprintf(buf, n, "%g", v);
    }
}

const char *describe_plan(const RateLimitPlan *plan, char *out, size_t out_size) {
    if (plan->batch_size == 0.0) {
        snprintf(out, out_size, "No viable schedule.");
    } else if (strcmp(plan->bounded_by, "none") == 0) {
        char b[48];
        plain_num(plan->batch_size, b, sizeof b);
        snprintf(out, out_size, "%s+ requests per window — endpoint treated as unbounded.", b);
    } else {
        char limiter[64];
        if (strcmp(plan->bounded_by, "both") == 0) {
            snprintf(limiter, sizeof limiter, "both limits bind together");
        } else {
            const char *up = strcmp(plan->bounded_by, "rpm") == 0 ? "RPM" : "TPM";
            snprintf(limiter, sizeof limiter, "the %s limit binds first", up);
        }
        char bs[48], iv[48];
        plain_num(plan->batch_size, bs, sizeof bs);
        plain_num(plan->interval_ms, iv, sizeof iv);
        snprintf(out, out_size, "%s requests per 60s window (one every %sms) — %s.", bs, iv, limiter);
    }
    return out;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →