Skip to content

Rate Limit Planner — TypeScript source

Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Pure logic for the Rate Limit Planner tool (slug: rate-limit-planner).
// Turns provider rate limits (requests/tokens per minute) plus a workload
// into a concrete schedule: how many requests per batch, how far apart, what
// caps it, and when the work finishes. Deterministic — no time reads.
export interface RateLimits {
  /** Requests per minute; undefined = not limited. */
  rpm?: number;
  /** Tokens per minute; undefined = not limited. */
  tpm?: number;
}

export interface Workload {
  /** Total requests to run. */
  requests: number;
  /** Average tokens per request (prompt + completion). */
  avgTokensPerRequest: number;
}

export interface PlanOptions {
  /** Fraction of the limits to target, leaving headroom for retries. */
  safetyFactor?: number;
}

export interface BatchSlice {
  batch: number;
  atMs: number;
  requests: number;
  tokens: number;
}

export interface RateLimitPlan {
  /** Requests to send per 60s window (0 when the workload cannot run). */
  batchSize: number;
  /** Steady-state spacing between individual requests, in ms. */
  intervalMs: number;
  /** Sustainable concurrent in-flight requests under even spacing. */
  maxConcurrent: number;
  /** Which limit binds first. */
  boundedBy: 'rpm' | 'tpm' | 'both' | 'none';
  /** First batches of the schedule (max 10) — enough to act on. */
  timeline: BatchSlice[];
  /** Estimated total wall time, in ms. */
  totalMs: number;
  warnings: string[];
}

const WINDOW_MS = 60_000;
const DEFAULT_SAFETY = 0.8;

export function planRateLimit(
  limits: RateLimits,
  workload: Workload,
  opts: PlanOptions = {},
): RateLimitPlan {
  const sf = opts.safetyFactor ?? DEFAULT_SAFETY;
  const warnings: string[] = [];
  if (workload.requests < 0 || workload.avgTokensPerRequest < 0) {
    throw new RangeError('requests and avgTokensPerRequest must be >= 0');
  }
  if (sf <= 0 || sf > 1) {
    throw new RangeError('safetyFactor must be in (0, 1]');
  }

  const rpmEff = limits.rpm !== undefined ? limits.rpm * sf : undefined;
  const tpmEff = limits.tpm !== undefined ? limits.tpm * sf : undefined;

  // Impossible: one request alone exceeds the token budget.
  if (tpmEff !== undefined && workload.avgTokensPerRequest > tpmEff && workload.requests > 0) {
    return {
      batchSize: 0,
      intervalMs: 0,
      maxConcurrent: 0,
      boundedBy: 'tpm',
      timeline: [],
      totalMs: Infinity,
      warnings: [
        `A single request averages ${workload.avgTokensPerRequest.toLocaleString('en-US')} tokens but the effective token limit is ${Math.floor(tpmEff).toLocaleString('en-US')}/min — no schedule can run this. Shrink requests or raise the tier.`,
      ],
    };
  }

  const byRpm = rpmEff ?? Infinity;
  const byTokens =
    tpmEff === undefined || workload.avgTokensPerRequest === 0
      ? Infinity
      : tpmEff / workload.avgTokensPerRequest;

  if (!Number.isFinite(byRpm) && !Number.isFinite(byTokens)) {
    warnings.push('No limits set — the plan assumes an unbounded endpoint. Add RPM or TPM for a real schedule.');
  }

  const steady = Math.max(1, Math.floor(Math.min(byRpm, byTokens)));
  const boundedBy: RateLimitPlan['boundedBy'] =
    !Number.isFinite(byRpm) && !Number.isFinite(byTokens)
      ? 'none'
      : Math.floor(byRpm) === Math.floor(byTokens)
        ? 'both'
        : byRpm < byTokens
          ? 'rpm'
          : 'tpm';

  // Even pacing inside the window: batchSize requests spread over 60s.
  const intervalMs = Math.round(WINDOW_MS / steady);
  // With even spacing and a per-request latency near intervalMs, one request
  // is in flight at a time; concurrency >1 only helps sub-interval latencies,
  // so the safe published floor is 1 — batch bursts raise it to batchSize.
  const maxConcurrent = steady === 1 ? 1 : Math.min(steady, Math.ceil(steady / 4));

  const timeline: BatchSlice[] = [];
  let remaining = workload.requests;
  let batch = 0;
  while (remaining > 0 && batch < 10) {
    const take = Math.min(steady, remaining);
    timeline.push({
      batch: batch + 1,
      atMs: batch * WINDOW_MS,
      requests: take,
      tokens: take * workload.avgTokensPerRequest,
    });
    remaining -= take;
    batch += 1;
  }

  const windowsNeeded = workload.requests > 0 ? Math.ceil(workload.requests / steady) : 0;
  const lastWindowRequests = windowsNeeded > 0 ? workload.requests - (windowsNeeded - 1) * steady : 0;
  const totalMs =
    windowsNeeded > 0 ? (windowsNeeded - 1) * WINDOW_MS + intervalMs * lastWindowRequests : 0;

  if (rpmEff !== undefined && workload.requests > 0 && steady > byRpm) {
    warnings.push('Rounded up to at least one request per window — even a single request per minute keeps the schedule honest.');
  }

  return { batchSize: steady, intervalMs, maxConcurrent, boundedBy, timeline, totalMs, warnings };
}

/** Human summary line for the plan (used by the island + docs). */
export function describePlan(plan: RateLimitPlan): string {
  if (plan.batchSize === 0) return 'No viable schedule.';
  if (plan.boundedBy === 'none') return `${plan.batchSize}+ requests per window — endpoint treated as unbounded.`;
  const limiter =
    plan.boundedBy === 'both' ? 'both limits bind together' : `the ${plan.boundedBy.toUpperCase()} limit binds first`;
  return `${plan.batchSize} requests per 60s window (one every ${plan.intervalMs}ms) — ${limiter}.`;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →