Skip to content

Rate Limit Planner — C++ source

Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Rate Limit Planner — turn provider rate limits plus a workload into a
// concrete schedule: batch size, spacing, timeline and wall time.
//
// Language: C++ (C++17, standard library only)
// Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
// implementation).
// Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner
//
// Deterministic — no time reads. The TS lib throws RangeError; this port
// throws std::range_error with the same messages. Numeric plan fields are
// double because the original returns Infinity for the unbounded (no
// limits) case.

#include <cmath>
#include <cstdio>
#include <optional>
#include <stdexcept>
#include <string>
#include <vector>

namespace rate_limit_planner {

inline constexpr double kWindowMs = 60000.0;
inline constexpr double kDefaultSafety = 0.8;
inline constexpr int kMaxTimeline = 10;

/// Provider rate limits; std::nullopt = not limited.
struct RateLimits {
    std::optional<double> rpm;  // requests per minute
    std::optional<double> tpm;  // tokens per minute
};

/// The workload to schedule.
struct Workload {
    long long requests = 0;             // total requests to run (>= 0)
    double avg_tokens_per_request = 0;  // prompt + completion
};

/// Optional tweaks; std::nullopt selects the default.
struct PlanOptions {
    std::optional<double> safety_factor;  // fraction of the limits to target
};

/// One batch of the schedule: how many requests, when, and their tokens.
struct BatchSlice {
    long long batch;
    double at_ms;
    long long requests;
    double tokens;
};

/// A concrete schedule for the workload.
struct RateLimitPlan {
    double batch_size;      // requests per 60s window (0 = cannot run)
    double interval_ms;     // spacing between requests, ms
    double max_concurrent;  // sustainable in-flight requests
    std::string bounded_by;  // "rpm" | "tpm" | "both" | "none"
    std::vector<BatchSlice> timeline;  // first batches (max 10)
    double total_ms;                   // estimated wall time, ms
    std::vector<std::string> warnings;
};

/// Format like TS `toLocaleString('en-US')`: thousands separators.
inline std::string fmtNum(double v) {
    if (std::isinf(v)) return v > 0 ? "Infinity" : "-Infinity";
    bool neg = v < 0;
    double a = std::fabs(v);
    long long ip = static_cast<long long>(std::floor(a));
    std::string digits = std::to_string(ip);
    std::string grouped;
    for (std::size_t i = 0; i < digits.size(); ++i) {
        if (i > 0 && (digits.size() - i) % 3 == 0) grouped += ',';
        grouped += digits[i];
    }
    double frac = a - static_cast<double>(ip);
    if (frac > 0.0) {
        char f[40];
        std::snprintf(f, sizeof f, "%g", frac);
        grouped += f;
    }
    return neg ? "-" + grouped : grouped;
}

/// Plan a schedule under the given limits. `safety_factor` must be in
/// (0, 1] (default 0.8). Throws std::range_error on negative workload
/// numbers or an out-of-range safety factor.
inline RateLimitPlan planRateLimit(const RateLimits& limits, const Workload& workload,
                                   const PlanOptions& opts = {}) {
    const double sf = opts.safety_factor.value_or(kDefaultSafety);
    std::vector<std::string> warnings;
    if (workload.requests < 0 || workload.avg_tokens_per_request < 0.0) {
        throw std::range_error("requests and avgTokensPerRequest must be >= 0");
    }
    if (!(sf > 0.0 && sf <= 1.0)) {
        throw std::range_error("safetyFactor must be in (0, 1]");
    }

    const std::optional<double> rpmEff =
        limits.rpm ? std::optional<double>(*limits.rpm * sf) : std::nullopt;
    const std::optional<double> tpmEff =
        limits.tpm ? std::optional<double>(*limits.tpm * sf) : std::nullopt;

    // Impossible: one request alone exceeds the token budget.
    if (tpmEff && workload.avg_tokens_per_request > *tpmEff && workload.requests > 0) {
        return RateLimitPlan{
            0.0,
            0.0,
            0.0,
            "tpm",
            {},
            INFINITY,
            {"A single request averages " + fmtNum(workload.avg_tokens_per_request) +
             " tokens but the effective token limit is " + fmtNum(std::floor(*tpmEff)) +
             "/min — no schedule can run this. Shrink requests or raise the tier."},
        };
    }

    const double byRpm = rpmEff.value_or(INFINITY);
    const double byTokens = (!tpmEff || workload.avg_tokens_per_request == 0.0)
                                ? INFINITY
                                : *tpmEff / workload.avg_tokens_per_request;

    if (std::isinf(byRpm) && std::isinf(byTokens)) {
        warnings.push_back(
            "No limits set — the plan assumes an unbounded endpoint. "
            "Add RPM or TPM for a real schedule.");
    }

    const double steady = std::max(1.0, std::floor(std::min(byRpm, byTokens)));
    std::string boundedBy;
    if (std::isinf(byRpm) && std::isinf(byTokens)) {
        boundedBy = "none";
    } else if (std::floor(byRpm) == std::floor(byTokens)) {
        boundedBy = "both";
    } else {
        boundedBy = byRpm < byTokens ? "rpm" : "tpm";
    }

    // Even pacing inside the window: batch_size requests spread over 60s.
    const double intervalMs = std::round(kWindowMs / steady);
    // With even spacing and a per-request latency near intervalMs, one
    // request is in flight at a time; concurrency >1 only helps
    // sub-interval latencies, so the safe published floor is 1 — batch
    // bursts raise it to batch_size.
    const double maxConcurrent =
        steady == 1.0 ? 1.0 : std::min(steady, std::ceil(steady / 4.0));

    std::vector<BatchSlice> timeline;
    double remaining = static_cast<double>(workload.requests);
    long long batch = 0;
    while (remaining > 0.0 && batch < kMaxTimeline) {
        const double take = std::min(steady, remaining);
        timeline.push_back(BatchSlice{
            batch + 1,
            static_cast<double>(batch) * kWindowMs,
            static_cast<long long>(take),
            take * workload.avg_tokens_per_request,
        });
        remaining -= take;
        batch++;
    }

    const int windowsNeeded =
        workload.requests > 0 ? static_cast<int>(std::ceil(static_cast<double>(workload.requests) / steady)) : 0;
    const double lastWindowRequests =
        windowsNeeded > 0 ? static_cast<double>(workload.requests) -
                                static_cast<double>(windowsNeeded - 1) * steady
                          : 0.0;
    const double totalMs = windowsNeeded > 0
                               ? static_cast<double>(windowsNeeded - 1) * kWindowMs +
                                     intervalMs * lastWindowRequests
                               : 0.0;

    if (rpmEff && workload.requests > 0 && steady > byRpm) {
        warnings.push_back(
            "Rounded up to at least one request per window — even a single "
            "request per minute keeps the schedule honest.");
    }

    return RateLimitPlan{steady, intervalMs, maxConcurrent, boundedBy,
                         std::move(timeline), totalMs, std::move(warnings)};
}

/// Human summary line for the plan (used by the island + docs).
/// Plain interpolation like TS `${v}` (no separators, no trailing ".0").
inline std::string plainNum(double v) {
    if (std::isinf(v)) return v > 0 ? "Infinity" : "-Infinity";
    char b[48];
    if (v == std::floor(v)) {
        std::snprintf(b, sizeof b, "%.0f", v);
    } else {
        std::snprintf(b, sizeof b, "%g", v);
    }
    return b;
}

inline std::string describePlan(const RateLimitPlan& plan) {
    if (plan.batch_size == 0.0) return "No viable schedule.";
    if (plan.bounded_by == "none") {
        return plainNum(plan.batch_size) +
               "+ requests per window — endpoint treated as unbounded.";
    }
    const std::string limiter =
        plan.bounded_by == "both"
            ? "both limits bind together"
            : "the " + [&] {
                  std::string s = plan.bounded_by;
                  for (char& c : s) c = static_cast<char>(c >= 'a' && c <= 'z' ? c - 32 : c);
                  return s;
              }() + " limit binds first";
    return plainNum(plan.batch_size) + " requests per 60s window (one every " +
           plainNum(plan.interval_ms) + "ms) — " + limiter + ".";
}

}  // namespace rate_limit_planner

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →