Rate Limit Planner — Kotlin source
Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.
This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Rate Limit Planner — turn provider rate limits plus a workload into a
// concrete schedule: batch size, spacing, what caps it, and finish time.
// Deterministic — no clock reads.
//
// Language: Kotlin (JVM 17+, zero dependencies)
// Port of src/lib/rateLimitPlanner.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/rate-limit-planner
import kotlin.math.ceil
import kotlin.math.floor
import kotlin.math.max
import kotlin.math.min
import kotlin.math.roundToInt
/** Requests/tokens per minute; null = not limited. */
data class RateLimits(val rpm: Double? = null, val tpm: Double? = null)
data class Workload(val requests: Long, val avgTokensPerRequest: Long)
data class PlanOptions(val safetyFactor: Double? = null)
data class BatchSlice(val batch: Long, val atMs: Long, val requests: Long, val tokens: Long)
data class RateLimitPlan(
/** Requests to send per 60s window (0 when the workload cannot run). */
val batchSize: Long,
/** Steady-state spacing between individual requests, in ms. */
val intervalMs: Long,
/** Sustainable concurrent in-flight requests under even spacing. */
val maxConcurrent: Long,
/** Which limit binds first: "rpm", "tpm", "both", or "none". */
val boundedBy: String,
/** First batches of the schedule (max 10). */
val timeline: List<BatchSlice>,
/** Estimated total wall time, in ms (Infinity when impossible). */
val totalMs: Double,
val warnings: List<String>,
)
private const val WINDOW_MS = 60_000.0
private const val DEFAULT_SAFETY = 0.8
private fun fmt(v: Long): String = String.format("%,d", v)
/**
* Plan a schedule.
* @throws IllegalArgumentException on impossible inputs (the TS RangeError contract).
*/
fun planRateLimit(limits: RateLimits, workload: Workload, opts: PlanOptions = PlanOptions()): RateLimitPlan {
val sf = opts.safetyFactor ?: DEFAULT_SAFETY
val warnings = mutableListOf<String>()
require(!(workload.requests < 0 || workload.avgTokensPerRequest < 0)) {
"requests and avgTokensPerRequest must be >= 0"
}
require(sf in 0.0..1.0 && sf != 0.0) { "safetyFactor must be in (0, 1]" }
val rpmEff = limits.rpm?.times(sf)
val tpmEff = limits.tpm?.times(sf)
// Impossible: one request alone exceeds the token budget.
if (tpmEff != null && workload.avgTokensPerRequest > tpmEff && workload.requests > 0) {
return RateLimitPlan(
batchSize = 0, intervalMs = 0, maxConcurrent = 0, boundedBy = "tpm",
timeline = emptyList(), totalMs = Double.POSITIVE_INFINITY,
warnings = listOf(
"A single request averages ${fmt(workload.avgTokensPerRequest)} tokens but " +
"the effective token limit is ${fmt(floor(tpmEff).toLong())}/min — no " +
"schedule can run this. Shrink requests or raise the tier.",
),
)
}
val byRpm = rpmEff ?: Double.POSITIVE_INFINITY
val byTokens = if (tpmEff == null || workload.avgTokensPerRequest == 0.0) {
Double.POSITIVE_INFINITY
} else {
tpmEff / workload.avgTokensPerRequest
}
val noRpm = byRpm.isInfinite()
val noTokens = byTokens.isInfinite()
if (noRpm && noTokens) {
warnings.add(
"No limits set — the plan assumes an unbounded endpoint. " +
"Add RPM or TPM for a real schedule."
)
}
val steady = max(1L, floor(min(byRpm, byTokens)).toLong())
val boundedBy = when {
noRpm && noTokens -> "none"
floor(byRpm).toLong() == floor(byTokens).toLong() -> "both"
byRpm < byTokens -> "rpm"
else -> "tpm"
}
// Even pacing inside the window: batchSize requests spread over 60s.
val intervalMs = (WINDOW_MS / steady).roundToInt().toLong()
// Concurrency >1 only helps sub-interval latencies; the safe published
// floor is 1 — batch bursts raise it to batchSize/4.
val maxConcurrent = if (steady == 1L) 1L else min(steady, ceil(steady / 4.0).toLong())
val timeline = mutableListOf<BatchSlice>()
var remaining = workload.requests
var batch = 0L
while (remaining > 0 && batch < 10) {
val take = min(steady, remaining)
timeline.add(
BatchSlice(
batch = batch + 1,
atMs = (batch * WINDOW_MS).toLong(),
requests = take,
tokens = take * workload.avgTokensPerRequest,
)
)
remaining -= take
batch += 1
}
val windowsNeeded = if (workload.requests > 0) ceil(workload.requests / steady.toDouble()).toLong() else 0
val lastWindowRequests = if (windowsNeeded > 0) workload.requests - (windowsNeeded - 1) * steady else 0
val totalMs = if (windowsNeeded > 0) {
(windowsNeeded - 1) * WINDOW_MS + intervalMs * lastWindowRequests
} else 0.0
if (rpmEff != null && workload.requests > 0 && steady > byRpm) {
warnings.add(
"Rounded up to at least one request per window — even a single request per " +
"minute keeps the schedule honest."
)
}
return RateLimitPlan(steady, intervalMs, maxConcurrent, boundedBy, timeline, totalMs, warnings)
}
/** Human summary line for the plan. */
fun describePlan(plan: RateLimitPlan): String {
if (plan.batchSize == 0L) return "No viable schedule."
if (plan.boundedBy == "none") {
return "${plan.batchSize}+ requests per window — endpoint treated as unbounded."
}
val limiter = if (plan.boundedBy == "both") "both limits bind together"
else "the ${plan.boundedBy.uppercase()} limit binds first"
return "${plan.batchSize} requests per 60s window (one every ${plan.intervalMs}ms) — $limiter."
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →