Skip to content

Rate Limit Planner — Kotlin source

Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.

This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Rate Limit Planner — turn provider rate limits plus a workload into a
// concrete schedule: batch size, spacing, what caps it, and finish time.
// Deterministic — no clock reads.
//
// Language: Kotlin (JVM 17+, zero dependencies)
// Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner

import kotlin.math.ceil
import kotlin.math.floor
import kotlin.math.max
import kotlin.math.min
import kotlin.math.roundToInt

/** Requests/tokens per minute; null = not limited. */
data class RateLimits(val rpm: Double? = null, val tpm: Double? = null)

data class Workload(val requests: Long, val avgTokensPerRequest: Long)

data class PlanOptions(val safetyFactor: Double? = null)

data class BatchSlice(val batch: Long, val atMs: Long, val requests: Long, val tokens: Long)

data class RateLimitPlan(
    /** Requests to send per 60s window (0 when the workload cannot run). */
    val batchSize: Long,
    /** Steady-state spacing between individual requests, in ms. */
    val intervalMs: Long,
    /** Sustainable concurrent in-flight requests under even spacing. */
    val maxConcurrent: Long,
    /** Which limit binds first: "rpm", "tpm", "both", or "none". */
    val boundedBy: String,
    /** First batches of the schedule (max 10). */
    val timeline: List<BatchSlice>,
    /** Estimated total wall time, in ms (Infinity when impossible). */
    val totalMs: Double,
    val warnings: List<String>,
)

private const val WINDOW_MS = 60_000.0
private const val DEFAULT_SAFETY = 0.8

private fun fmt(v: Long): String = String.format("%,d", v)

/**
 * Plan a schedule.
 * @throws IllegalArgumentException on impossible inputs (the TS RangeError contract).
 */
fun planRateLimit(limits: RateLimits, workload: Workload, opts: PlanOptions = PlanOptions()): RateLimitPlan {
    val sf = opts.safetyFactor ?: DEFAULT_SAFETY
    val warnings = mutableListOf<String>()
    require(!(workload.requests < 0 || workload.avgTokensPerRequest < 0)) {
        "requests and avgTokensPerRequest must be >= 0"
    }
    require(sf in 0.0..1.0 && sf != 0.0) { "safetyFactor must be in (0, 1]" }

    val rpmEff = limits.rpm?.times(sf)
    val tpmEff = limits.tpm?.times(sf)

    // Impossible: one request alone exceeds the token budget.
    if (tpmEff != null && workload.avgTokensPerRequest > tpmEff && workload.requests > 0) {
        return RateLimitPlan(
            batchSize = 0, intervalMs = 0, maxConcurrent = 0, boundedBy = "tpm",
            timeline = emptyList(), totalMs = Double.POSITIVE_INFINITY,
            warnings = listOf(
                "A single request averages ${fmt(workload.avgTokensPerRequest)} tokens but " +
                    "the effective token limit is ${fmt(floor(tpmEff).toLong())}/min — no " +
                    "schedule can run this. Shrink requests or raise the tier.",
            ),
        )
    }

    val byRpm = rpmEff ?: Double.POSITIVE_INFINITY
    val byTokens = if (tpmEff == null || workload.avgTokensPerRequest == 0.0) {
        Double.POSITIVE_INFINITY
    } else {
        tpmEff / workload.avgTokensPerRequest
    }

    val noRpm = byRpm.isInfinite()
    val noTokens = byTokens.isInfinite()
    if (noRpm && noTokens) {
        warnings.add(
            "No limits set — the plan assumes an unbounded endpoint. " +
                "Add RPM or TPM for a real schedule."
        )
    }

    val steady = max(1L, floor(min(byRpm, byTokens)).toLong())
    val boundedBy = when {
        noRpm && noTokens -> "none"
        floor(byRpm).toLong() == floor(byTokens).toLong() -> "both"
        byRpm < byTokens -> "rpm"
        else -> "tpm"
    }

    // Even pacing inside the window: batchSize requests spread over 60s.
    val intervalMs = (WINDOW_MS / steady).roundToInt().toLong()
    // Concurrency >1 only helps sub-interval latencies; the safe published
    // floor is 1 — batch bursts raise it to batchSize/4.
    val maxConcurrent = if (steady == 1L) 1L else min(steady, ceil(steady / 4.0).toLong())

    val timeline = mutableListOf<BatchSlice>()
    var remaining = workload.requests
    var batch = 0L
    while (remaining > 0 && batch < 10) {
        val take = min(steady, remaining)
        timeline.add(
            BatchSlice(
                batch = batch + 1,
                atMs = (batch * WINDOW_MS).toLong(),
                requests = take,
                tokens = take * workload.avgTokensPerRequest,
            )
        )
        remaining -= take
        batch += 1
    }

    val windowsNeeded = if (workload.requests > 0) ceil(workload.requests / steady.toDouble()).toLong() else 0
    val lastWindowRequests = if (windowsNeeded > 0) workload.requests - (windowsNeeded - 1) * steady else 0
    val totalMs = if (windowsNeeded > 0) {
        (windowsNeeded - 1) * WINDOW_MS + intervalMs * lastWindowRequests
    } else 0.0

    if (rpmEff != null && workload.requests > 0 && steady > byRpm) {
        warnings.add(
            "Rounded up to at least one request per window — even a single request per " +
                "minute keeps the schedule honest."
        )
    }

    return RateLimitPlan(steady, intervalMs, maxConcurrent, boundedBy, timeline, totalMs, warnings)
}

/** Human summary line for the plan. */
fun describePlan(plan: RateLimitPlan): String {
    if (plan.batchSize == 0L) return "No viable schedule."
    if (plan.boundedBy == "none") {
        return "${plan.batchSize}+ requests per window — endpoint treated as unbounded."
    }
    val limiter = if (plan.boundedBy == "both") "both limits bind together"
    else "the ${plan.boundedBy.uppercase()} limit binds first"
    return "${plan.batchSize} requests per 60s window (one every ${plan.intervalMs}ms) — $limiter."
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →