Skip to content

Rate Limit Planner — Python source

Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""
Rate Limit Planner — turn provider rate limits plus a workload into a
concrete schedule: batch size, spacing, timeline and wall time.

Language: Python 3.10+ (standard library only)
Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
implementation).
Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner

Deterministic — no time reads. The TS lib throws RangeError; this port
raises ValueError with the same messages. With no limits at all the plan
is unbounded: batch_size/max_concurrent become math.inf, matching the
TypeScript original's Infinity.
"""

from __future__ import annotations

import math
from typing import Any

WINDOW_MS = 60_000
DEFAULT_SAFETY = 0.8


def _fmt(v: float) -> str:
    """Format like TS `toLocaleString('en-US')`: thousands separators."""
    if v == math.inf:
        return "Infinity"
    if v == -math.inf:
        return "-Infinity"
    return f"{v:,}"


def _plain(v: float) -> str:
    """Plain interpolation like TS `${v}` (no separators, no trailing .0)."""
    if v == math.inf:
        return "Infinity"
    if v == -math.inf:
        return "-Infinity"
    if v == math.floor(v):
        return str(int(v))
    return str(v)


def _fl(x: float) -> float:
    """floor() that tolerates infinity (Python's math.floor(inf) raises)."""
    return math.inf if math.isinf(x) else math.floor(x)


def plan_rate_limit(
    limits: dict[str, Any] | None,
    workload: dict[str, Any],
    opts: dict[str, Any] | None = None,
) -> dict[str, Any]:
    """Plan a schedule under the given limits.

    ``limits`` keys (both optional): ``rpm`` / ``tpm`` — requests / tokens
    per minute; absent or None = not limited. ``workload``:
    ``{"requests": int, "avgTokensPerRequest": float}``.
    ``opts["safetyFactor"]`` must be in (0, 1] (default 0.8). Raises
    ValueError on negative workload numbers or an out-of-range safety
    factor.
    """
    limits = limits or {}
    opts = opts or {}
    sf = opts.get("safetyFactor", DEFAULT_SAFETY)
    warnings: list[str] = []
    requests = workload["requests"]
    avg_tokens = workload["avgTokensPerRequest"]
    if requests < 0 or avg_tokens < 0:
        raise ValueError("requests and avgTokensPerRequest must be >= 0")
    if sf <= 0 or sf > 1:
        raise ValueError("safetyFactor must be in (0, 1]")

    rpm_eff = limits["rpm"] * sf if limits.get("rpm") is not None else None
    tpm_eff = limits["tpm"] * sf if limits.get("tpm") is not None else None

    # Impossible: one request alone exceeds the token budget.
    if tpm_eff is not None and avg_tokens > tpm_eff and requests > 0:
        return {
            "batch_size": 0,
            "interval_ms": 0,
            "max_concurrent": 0,
            "bounded_by": "tpm",
            "timeline": [],
            "total_ms": math.inf,
            "warnings": [
                f"A single request averages {_fmt(avg_tokens)} tokens but the "
                f"effective token limit is {_fmt(math.floor(tpm_eff))}/min — no "
                f"schedule can run this. Shrink requests or raise the tier.",
            ],
        }

    by_rpm = math.inf if rpm_eff is None else rpm_eff
    if tpm_eff is None or avg_tokens == 0:
        by_tokens = math.inf
    else:
        by_tokens = tpm_eff / avg_tokens

    if math.isinf(by_rpm) and math.isinf(by_tokens):
        warnings.append(
            "No limits set — the plan assumes an unbounded endpoint. "
            "Add RPM or TPM for a real schedule."
        )

    cap = min(by_rpm, by_tokens)
    # floor(max(1, cap)); keep inf as inf (Python's math.floor(inf) raises).
    steady: float = cap if math.isinf(cap) else max(1, math.floor(cap))
    if math.isinf(by_rpm) and math.isinf(by_tokens):
        bounded_by = "none"
    elif _fl(by_rpm) == _fl(by_tokens):
        bounded_by = "both"
    else:
        bounded_by = "rpm" if by_rpm < by_tokens else "tpm"

    # Even pacing inside the window: batch_size requests spread over 60s.
    interval_ms = round(WINDOW_MS / steady)
    # With even spacing and a per-request latency near interval_ms, one
    # request is in flight at a time; concurrency >1 only helps sub-interval
    # latencies, so the safe published floor is 1 — batch bursts raise it.
    if steady == 1:
        max_concurrent: float = 1
    elif math.isinf(steady):
        max_concurrent = math.inf
    else:
        max_concurrent = min(steady, math.ceil(steady / 4))

    timeline: list[dict[str, Any]] = []
    remaining = requests
    batch = 0
    while remaining > 0 and batch < 10:
        take = min(steady, remaining)
        timeline.append(
            {
                "batch": batch + 1,
                "at_ms": batch * WINDOW_MS,
                "requests": take,
                "tokens": take * avg_tokens,
            }
        )
        remaining -= take
        batch += 1

    windows_needed = math.ceil(requests / steady) if requests > 0 else 0
    last_window_requests = (
        requests - (windows_needed - 1) * steady if windows_needed > 0 else 0
    )
    total_ms = (
        (windows_needed - 1) * WINDOW_MS + interval_ms * last_window_requests
        if windows_needed > 0
        else 0
    )

    if rpm_eff is not None and requests > 0 and steady > by_rpm:
        warnings.append(
            "Rounded up to at least one request per window — even a single "
            "request per minute keeps the schedule honest."
        )

    return {
        "batch_size": steady,
        "interval_ms": interval_ms,
        "max_concurrent": max_concurrent,
        "bounded_by": bounded_by,
        "timeline": timeline,
        "total_ms": total_ms,
        "warnings": warnings,
    }


def describe_plan(plan: dict[str, Any]) -> str:
    """Human summary line for the plan (used by the island + docs)."""
    if plan["batch_size"] == 0:
        return "No viable schedule."
    if plan["bounded_by"] == "none":
        return (
            f"{_plain(plan['batch_size'])}+ requests per window — "
            "endpoint treated as unbounded."
        )
    bounded_by = plan["bounded_by"]
    limiter = (
        "both limits bind together"
        if bounded_by == "both"
        else f"the {bounded_by.upper()} limit binds first"
    )
    return (
        f"{_plain(plan['batch_size'])} requests per 60s window "
        f"(one every {_plain(plan['interval_ms'])}ms) — {limiter}."
    )

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →