Rate Limit Planner — C++ source
Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Rate Limit Planner — turn provider rate limits plus a workload into a
// concrete schedule: batch size, spacing, timeline and wall time.
//
// Language: C++ (C++17, standard library only)
// Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
// implementation).
// Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner
//
// Deterministic — no time reads. The TS lib throws RangeError; this port
// throws std::range_error with the same messages. Numeric plan fields are
// double because the original returns Infinity for the unbounded (no
// limits) case.
#include <cmath>
#include <cstdio>
#include <optional>
#include <stdexcept>
#include <string>
#include <vector>
namespace rate_limit_planner {
inline constexpr double kWindowMs = 60000.0;
inline constexpr double kDefaultSafety = 0.8;
inline constexpr int kMaxTimeline = 10;
/// Provider rate limits; std::nullopt = not limited.
struct RateLimits {
std::optional<double> rpm; // requests per minute
std::optional<double> tpm; // tokens per minute
};
/// The workload to schedule.
struct Workload {
long long requests = 0; // total requests to run (>= 0)
double avg_tokens_per_request = 0; // prompt + completion
};
/// Optional tweaks; std::nullopt selects the default.
struct PlanOptions {
std::optional<double> safety_factor; // fraction of the limits to target
};
/// One batch of the schedule: how many requests, when, and their tokens.
struct BatchSlice {
long long batch;
double at_ms;
long long requests;
double tokens;
};
/// A concrete schedule for the workload.
struct RateLimitPlan {
double batch_size; // requests per 60s window (0 = cannot run)
double interval_ms; // spacing between requests, ms
double max_concurrent; // sustainable in-flight requests
std::string bounded_by; // "rpm" | "tpm" | "both" | "none"
std::vector<BatchSlice> timeline; // first batches (max 10)
double total_ms; // estimated wall time, ms
std::vector<std::string> warnings;
};
/// Format like TS `toLocaleString('en-US')`: thousands separators.
inline std::string fmtNum(double v) {
if (std::isinf(v)) return v > 0 ? "Infinity" : "-Infinity";
bool neg = v < 0;
double a = std::fabs(v);
long long ip = static_cast<long long>(std::floor(a));
std::string digits = std::to_string(ip);
std::string grouped;
for (std::size_t i = 0; i < digits.size(); ++i) {
if (i > 0 && (digits.size() - i) % 3 == 0) grouped += ',';
grouped += digits[i];
}
double frac = a - static_cast<double>(ip);
if (frac > 0.0) {
char f[40];
std::snprintf(f, sizeof f, "%g", frac);
grouped += f;
}
return neg ? "-" + grouped : grouped;
}
/// Plan a schedule under the given limits. `safety_factor` must be in
/// (0, 1] (default 0.8). Throws std::range_error on negative workload
/// numbers or an out-of-range safety factor.
inline RateLimitPlan planRateLimit(const RateLimits& limits, const Workload& workload,
const PlanOptions& opts = {}) {
const double sf = opts.safety_factor.value_or(kDefaultSafety);
std::vector<std::string> warnings;
if (workload.requests < 0 || workload.avg_tokens_per_request < 0.0) {
throw std::range_error("requests and avgTokensPerRequest must be >= 0");
}
if (!(sf > 0.0 && sf <= 1.0)) {
throw std::range_error("safetyFactor must be in (0, 1]");
}
const std::optional<double> rpmEff =
limits.rpm ? std::optional<double>(*limits.rpm * sf) : std::nullopt;
const std::optional<double> tpmEff =
limits.tpm ? std::optional<double>(*limits.tpm * sf) : std::nullopt;
// Impossible: one request alone exceeds the token budget.
if (tpmEff && workload.avg_tokens_per_request > *tpmEff && workload.requests > 0) {
return RateLimitPlan{
0.0,
0.0,
0.0,
"tpm",
{},
INFINITY,
{"A single request averages " + fmtNum(workload.avg_tokens_per_request) +
" tokens but the effective token limit is " + fmtNum(std::floor(*tpmEff)) +
"/min — no schedule can run this. Shrink requests or raise the tier."},
};
}
const double byRpm = rpmEff.value_or(INFINITY);
const double byTokens = (!tpmEff || workload.avg_tokens_per_request == 0.0)
? INFINITY
: *tpmEff / workload.avg_tokens_per_request;
if (std::isinf(byRpm) && std::isinf(byTokens)) {
warnings.push_back(
"No limits set — the plan assumes an unbounded endpoint. "
"Add RPM or TPM for a real schedule.");
}
const double steady = std::max(1.0, std::floor(std::min(byRpm, byTokens)));
std::string boundedBy;
if (std::isinf(byRpm) && std::isinf(byTokens)) {
boundedBy = "none";
} else if (std::floor(byRpm) == std::floor(byTokens)) {
boundedBy = "both";
} else {
boundedBy = byRpm < byTokens ? "rpm" : "tpm";
}
// Even pacing inside the window: batch_size requests spread over 60s.
const double intervalMs = std::round(kWindowMs / steady);
// With even spacing and a per-request latency near intervalMs, one
// request is in flight at a time; concurrency >1 only helps
// sub-interval latencies, so the safe published floor is 1 — batch
// bursts raise it to batch_size.
const double maxConcurrent =
steady == 1.0 ? 1.0 : std::min(steady, std::ceil(steady / 4.0));
std::vector<BatchSlice> timeline;
double remaining = static_cast<double>(workload.requests);
long long batch = 0;
while (remaining > 0.0 && batch < kMaxTimeline) {
const double take = std::min(steady, remaining);
timeline.push_back(BatchSlice{
batch + 1,
static_cast<double>(batch) * kWindowMs,
static_cast<long long>(take),
take * workload.avg_tokens_per_request,
});
remaining -= take;
batch++;
}
const int windowsNeeded =
workload.requests > 0 ? static_cast<int>(std::ceil(static_cast<double>(workload.requests) / steady)) : 0;
const double lastWindowRequests =
windowsNeeded > 0 ? static_cast<double>(workload.requests) -
static_cast<double>(windowsNeeded - 1) * steady
: 0.0;
const double totalMs = windowsNeeded > 0
? static_cast<double>(windowsNeeded - 1) * kWindowMs +
intervalMs * lastWindowRequests
: 0.0;
if (rpmEff && workload.requests > 0 && steady > byRpm) {
warnings.push_back(
"Rounded up to at least one request per window — even a single "
"request per minute keeps the schedule honest.");
}
return RateLimitPlan{steady, intervalMs, maxConcurrent, boundedBy,
std::move(timeline), totalMs, std::move(warnings)};
}
/// Human summary line for the plan (used by the island + docs).
/// Plain interpolation like TS `${v}` (no separators, no trailing ".0").
inline std::string plainNum(double v) {
if (std::isinf(v)) return v > 0 ? "Infinity" : "-Infinity";
char b[48];
if (v == std::floor(v)) {
std::snprintf(b, sizeof b, "%.0f", v);
} else {
std::snprintf(b, sizeof b, "%g", v);
}
return b;
}
inline std::string describePlan(const RateLimitPlan& plan) {
if (plan.batch_size == 0.0) return "No viable schedule.";
if (plan.bounded_by == "none") {
return plainNum(plan.batch_size) +
"+ requests per window — endpoint treated as unbounded.";
}
const std::string limiter =
plan.bounded_by == "both"
? "both limits bind together"
: "the " + [&] {
std::string s = plan.bounded_by;
for (char& c : s) c = static_cast<char>(c >= 'a' && c <= 'z' ? c - 32 : c);
return s;
}() + " limit binds first";
return plainNum(plan.batch_size) + " requests per 60s window (one every " +
plainNum(plan.interval_ms) + "ms) — " + limiter + ".";
}
} // namespace rate_limit_planner
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →