Rate Limit Planner — Python source
Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""
Rate Limit Planner — turn provider rate limits plus a workload into a
concrete schedule: batch size, spacing, timeline and wall time.
Language: Python 3.10+ (standard library only)
Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
implementation).
Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner
Deterministic — no time reads. The TS lib throws RangeError; this port
raises ValueError with the same messages. With no limits at all the plan
is unbounded: batch_size/max_concurrent become math.inf, matching the
TypeScript original's Infinity.
"""
from __future__ import annotations
import math
from typing import Any
WINDOW_MS = 60_000
DEFAULT_SAFETY = 0.8
def _fmt(v: float) -> str:
"""Format like TS `toLocaleString('en-US')`: thousands separators."""
if v == math.inf:
return "Infinity"
if v == -math.inf:
return "-Infinity"
return f"{v:,}"
def _plain(v: float) -> str:
"""Plain interpolation like TS `${v}` (no separators, no trailing .0)."""
if v == math.inf:
return "Infinity"
if v == -math.inf:
return "-Infinity"
if v == math.floor(v):
return str(int(v))
return str(v)
def _fl(x: float) -> float:
"""floor() that tolerates infinity (Python's math.floor(inf) raises)."""
return math.inf if math.isinf(x) else math.floor(x)
def plan_rate_limit(
limits: dict[str, Any] | None,
workload: dict[str, Any],
opts: dict[str, Any] | None = None,
) -> dict[str, Any]:
"""Plan a schedule under the given limits.
``limits`` keys (both optional): ``rpm`` / ``tpm`` — requests / tokens
per minute; absent or None = not limited. ``workload``:
``{"requests": int, "avgTokensPerRequest": float}``.
``opts["safetyFactor"]`` must be in (0, 1] (default 0.8). Raises
ValueError on negative workload numbers or an out-of-range safety
factor.
"""
limits = limits or {}
opts = opts or {}
sf = opts.get("safetyFactor", DEFAULT_SAFETY)
warnings: list[str] = []
requests = workload["requests"]
avg_tokens = workload["avgTokensPerRequest"]
if requests < 0 or avg_tokens < 0:
raise ValueError("requests and avgTokensPerRequest must be >= 0")
if sf <= 0 or sf > 1:
raise ValueError("safetyFactor must be in (0, 1]")
rpm_eff = limits["rpm"] * sf if limits.get("rpm") is not None else None
tpm_eff = limits["tpm"] * sf if limits.get("tpm") is not None else None
# Impossible: one request alone exceeds the token budget.
if tpm_eff is not None and avg_tokens > tpm_eff and requests > 0:
return {
"batch_size": 0,
"interval_ms": 0,
"max_concurrent": 0,
"bounded_by": "tpm",
"timeline": [],
"total_ms": math.inf,
"warnings": [
f"A single request averages {_fmt(avg_tokens)} tokens but the "
f"effective token limit is {_fmt(math.floor(tpm_eff))}/min — no "
f"schedule can run this. Shrink requests or raise the tier.",
],
}
by_rpm = math.inf if rpm_eff is None else rpm_eff
if tpm_eff is None or avg_tokens == 0:
by_tokens = math.inf
else:
by_tokens = tpm_eff / avg_tokens
if math.isinf(by_rpm) and math.isinf(by_tokens):
warnings.append(
"No limits set — the plan assumes an unbounded endpoint. "
"Add RPM or TPM for a real schedule."
)
cap = min(by_rpm, by_tokens)
# floor(max(1, cap)); keep inf as inf (Python's math.floor(inf) raises).
steady: float = cap if math.isinf(cap) else max(1, math.floor(cap))
if math.isinf(by_rpm) and math.isinf(by_tokens):
bounded_by = "none"
elif _fl(by_rpm) == _fl(by_tokens):
bounded_by = "both"
else:
bounded_by = "rpm" if by_rpm < by_tokens else "tpm"
# Even pacing inside the window: batch_size requests spread over 60s.
interval_ms = round(WINDOW_MS / steady)
# With even spacing and a per-request latency near interval_ms, one
# request is in flight at a time; concurrency >1 only helps sub-interval
# latencies, so the safe published floor is 1 — batch bursts raise it.
if steady == 1:
max_concurrent: float = 1
elif math.isinf(steady):
max_concurrent = math.inf
else:
max_concurrent = min(steady, math.ceil(steady / 4))
timeline: list[dict[str, Any]] = []
remaining = requests
batch = 0
while remaining > 0 and batch < 10:
take = min(steady, remaining)
timeline.append(
{
"batch": batch + 1,
"at_ms": batch * WINDOW_MS,
"requests": take,
"tokens": take * avg_tokens,
}
)
remaining -= take
batch += 1
windows_needed = math.ceil(requests / steady) if requests > 0 else 0
last_window_requests = (
requests - (windows_needed - 1) * steady if windows_needed > 0 else 0
)
total_ms = (
(windows_needed - 1) * WINDOW_MS + interval_ms * last_window_requests
if windows_needed > 0
else 0
)
if rpm_eff is not None and requests > 0 and steady > by_rpm:
warnings.append(
"Rounded up to at least one request per window — even a single "
"request per minute keeps the schedule honest."
)
return {
"batch_size": steady,
"interval_ms": interval_ms,
"max_concurrent": max_concurrent,
"bounded_by": bounded_by,
"timeline": timeline,
"total_ms": total_ms,
"warnings": warnings,
}
def describe_plan(plan: dict[str, Any]) -> str:
"""Human summary line for the plan (used by the island + docs)."""
if plan["batch_size"] == 0:
return "No viable schedule."
if plan["bounded_by"] == "none":
return (
f"{_plain(plan['batch_size'])}+ requests per window — "
"endpoint treated as unbounded."
)
bounded_by = plan["bounded_by"]
limiter = (
"both limits bind together"
if bounded_by == "both"
else f"the {bounded_by.upper()} limit binds first"
)
return (
f"{_plain(plan['batch_size'])} requests per 60s window "
f"(one every {_plain(plan['interval_ms'])}ms) — {limiter}."
)
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →