Skip to content

Rate Limit Planner — Go source

Turn RPM/TPM limits into a concrete request schedule — batch size, spacing, binding limit, and total run time, with a safety factor for retries. 100% client-side.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package ratelimitplanner is the Go twin of CosmoDev's
// src/lib/rateLimitPlanner.ts (dual source: the web lib is TypeScript, the
// CLI lib is Go — kept in lock-step). Pure + deterministic, never panics.
// The table-driven tests in ratelimitplanner_test.go share vectors with
// src/lib/rateLimitPlanner.test.ts so the two implementations are held to
// the same contract.
//
// The planner turns provider rate limits (requests/tokens per minute) plus a
// workload into a concrete schedule: how many requests per batch, how far
// apart, what caps it, and when the work finishes. No time reads.
//
// Mapping notes (TS → Go): optional numbers become *float64 with nil meaning
// "unset" (TS undefined); the RangeError throws become errors wrapping
// ErrRange; TS's Infinity batch size (no limits bind) saturates to
// unlimitedBatch since Go has no integer infinity.
package ratelimitplanner

import (
	"errors"
	"fmt"
	"math"
	"strconv"
	"strings"
)

const (
	// windowMS is the rate-limit window every plan paces against (TS: WINDOW_MS).
	windowMS = 60_000
	// defaultSafety is the fraction of the limits targeted, leaving headroom
	// for retries (TS: DEFAULT_SAFETY).
	defaultSafety = 0.8
	// timelineCap bounds the returned schedule preview (TS: max 10 batches).
	timelineCap = 10
	// unlimitedBatch stands in for TS's Infinity when no limit binds any
	// axis: Go has no integer infinity, so the unbounded steady rate
	// saturates here.
	unlimitedBatch = math.MaxInt
)

// ErrRange marks out-of-domain inputs. The TS lib throws RangeError for the
// same conditions, and the shared vectors assert on it via errors.Is.
var ErrRange = errors.New("range error")

var (
	errWorkloadRange = fmt.Errorf("%w: requests and avgTokensPerRequest must be >= 0", ErrRange)
	errSafetyRange   = fmt.Errorf("%w: safetyFactor must be in (0, 1]", ErrRange)
)

// RateLimits mirrors the TS RateLimits interface. A nil pointer means the
// axis is not limited (TS: undefined).
type RateLimits struct {
	RPM *float64 // requests per minute; nil = not limited
	TPM *float64 // tokens per minute; nil = not limited
}

// Workload mirrors the TS Workload interface.
type Workload struct {
	Requests            int     // total requests to run
	AvgTokensPerRequest float64 // average tokens per request (prompt + completion)
}

// PlanOptions mirrors the TS PlanOptions interface. The zero value is the
// TS default ({}): a nil SafetyFactor targets defaultSafety (0.8), while an
// explicit 0 is rejected exactly like the TS lib rejects safetyFactor: 0.
type PlanOptions struct {
	SafetyFactor *float64 // nil → 0.8; explicit values must be in (0, 1]
}

// BatchSlice mirrors the TS BatchSlice interface: one window of the schedule.
type BatchSlice struct {
	Batch    int
	AtMs     int
	Requests int
	Tokens   float64
}

// BoundedBy names which limit binds first (TS: 'rpm' | 'tpm' | 'both' | 'none').
type BoundedBy string

const (
	BoundedByRPM  BoundedBy = "rpm"
	BoundedByTPM  BoundedBy = "tpm"
	BoundedByBoth BoundedBy = "both"
	BoundedByNone BoundedBy = "none"
)

// RateLimitPlan mirrors the TS RateLimitPlan interface.
type RateLimitPlan struct {
	BatchSize     int          // requests per 60s window (0 when the workload cannot run)
	IntervalMs    int          // steady-state spacing between individual requests, in ms
	MaxConcurrent int          // sustainable concurrent in-flight requests under even spacing
	BoundedBy     BoundedBy    // which limit binds first
	Timeline      []BatchSlice // first batches of the schedule (max timelineCap)
	TotalMs       float64      // estimated total wall time, in ms (+Inf = never)
	Warnings      []string
}

// PlanRateLimit is the Go twin of planRateLimit() in src/lib/rateLimitPlanner.ts
// and must agree with it on every shared vector. Out-of-domain inputs return a
// zero plan plus an error wrapping ErrRange (TS: throws RangeError); a plan
// that cannot run at all is a valid result with BatchSize 0 and TotalMs +Inf.
func PlanRateLimit(limits RateLimits, workload Workload, opts PlanOptions) (RateLimitPlan, error) {
	sf := defaultSafety
	if opts.SafetyFactor != nil {
		sf = *opts.SafetyFactor
	}

	if workload.Requests < 0 || workload.AvgTokensPerRequest < 0 {
		return RateLimitPlan{}, errWorkloadRange
	}
	if sf <= 0 || sf > 1 {
		return RateLimitPlan{}, errSafetyRange
	}

	rpmSet, tpmSet := limits.RPM != nil, limits.TPM != nil
	rpmEff, tpmEff := math.Inf(1), math.Inf(1)
	if rpmSet {
		rpmEff = *limits.RPM * sf
	}
	if tpmSet {
		tpmEff = *limits.TPM * sf
	}

	// Impossible: one request alone exceeds the token budget.
	if tpmSet && workload.AvgTokensPerRequest > tpmEff && workload.Requests > 0 {
		return RateLimitPlan{
			BatchSize:     0,
			IntervalMs:    0,
			MaxConcurrent: 0,
			BoundedBy:     BoundedByTPM,
			Timeline:      []BatchSlice{},
			TotalMs:       math.Inf(1),
			Warnings: []string{
				"A single request averages " + formatNumber(workload.AvgTokensPerRequest) +
					" tokens but the effective token limit is " + formatNumber(math.Floor(tpmEff)) +
					"/min — no schedule can run this. Shrink requests or raise the tier.",
			},
		}, nil
	}

	byRpm := rpmEff
	byTokens := math.Inf(1)
	if tpmSet && workload.AvgTokensPerRequest != 0 {
		byTokens = tpmEff / workload.AvgTokensPerRequest
	}

	warnings := []string{}
	if math.IsInf(byRpm, 1) && math.IsInf(byTokens, 1) {
		warnings = append(warnings,
			"No limits set — the plan assumes an unbounded endpoint. Add RPM or TPM for a real schedule.")
	}

	// steady = max(1, floor(min(byRpm, byTokens))), saturated to unlimitedBatch
	// when unbounded (TS keeps Infinity) so every downstream int stays finite.
	steadyF := math.Max(1, math.Floor(math.Min(byRpm, byTokens)))
	batchSize := unlimitedBatch
	if steadyF < float64(unlimitedBatch) {
		batchSize = int(steadyF)
	}

	var boundedBy BoundedBy = BoundedByTPM
	switch {
	case math.IsInf(byRpm, 1) && math.IsInf(byTokens, 1):
		boundedBy = BoundedByNone
	case math.Floor(byRpm) == math.Floor(byTokens):
		boundedBy = BoundedByBoth
	case byRpm < byTokens:
		boundedBy = BoundedByRPM
	}

	// Even pacing inside the window: batchSize requests spread over 60s.
	intervalMs := int(math.Round(float64(windowMS) / float64(batchSize)))
	// With even spacing and a per-request latency near intervalMs, one request
	// is in flight at a time; concurrency >1 only helps sub-interval latencies,
	// so the safe published floor is 1 — batch bursts raise it to batchSize/4.
	maxConcurrent := 1
	if batchSize != 1 {
		maxConcurrent = int(math.Ceil(float64(batchSize) / 4))
		if maxConcurrent > batchSize {
			maxConcurrent = batchSize
		}
	}

	timeline := []BatchSlice{}
	remaining := workload.Requests
	batch := 0
	for remaining > 0 && batch < timelineCap {
		take := batchSize
		if remaining < batchSize {
			take = remaining
		}
		timeline = append(timeline, BatchSlice{
			Batch:    batch + 1,
			AtMs:     batch * windowMS,
			Requests: take,
			Tokens:   float64(take) * workload.AvgTokensPerRequest,
		})
		remaining -= take
		batch++
	}

	windowsNeeded := 0
	if workload.Requests > 0 {
		if batchSize >= workload.Requests {
			windowsNeeded = 1
		} else {
			windowsNeeded = 1 + (workload.Requests-1)/batchSize
		}
	}
	lastWindowRequests := 0
	if windowsNeeded > 0 {
		lastWindowRequests = workload.Requests - (windowsNeeded-1)*batchSize
	}
	totalMs := 0.0
	if windowsNeeded > 0 {
		totalMs = float64(windowsNeeded-1)*float64(windowMS) +
			float64(intervalMs)*float64(lastWindowRequests)
	}

	if rpmSet && workload.Requests > 0 && float64(batchSize) > byRpm {
		warnings = append(warnings,
			"Rounded up to at least one request per window — even a single request per minute keeps the schedule honest.")
	}

	return RateLimitPlan{
		BatchSize:     batchSize,
		IntervalMs:    intervalMs,
		MaxConcurrent: maxConcurrent,
		BoundedBy:     boundedBy,
		Timeline:      timeline,
		TotalMs:       totalMs,
		Warnings:      warnings,
	}, nil
}

// DescribePlan is the Go twin of describePlan() in src/lib/rateLimitPlanner.ts:
// a human summary line for the plan (used by the island + docs). When no limit
// binds, TS prints Infinity where the saturated batch size renders "unlimited".
func DescribePlan(plan RateLimitPlan) string {
	if plan.BatchSize == 0 {
		return "No viable schedule."
	}
	if plan.BoundedBy == BoundedByNone {
		size := strconv.Itoa(plan.BatchSize)
		if plan.BatchSize == unlimitedBatch {
			size = "unlimited"
		}
		return size + "+ requests per window — endpoint treated as unbounded."
	}
	limiter := "both limits bind together"
	if plan.BoundedBy != BoundedByBoth {
		limiter = fmt.Sprintf("the %s limit binds first", strings.ToUpper(string(plan.BoundedBy)))
	}
	return fmt.Sprintf("%d requests per 60s window (one every %dms) — %s.",
		plan.BatchSize, plan.IntervalMs, limiter)
}

// formatNumber renders a non-negative float like
// Number.prototype.toLocaleString('en-US'): integer part grouped with commas,
// fraction kept as-is. Used only inside the impossible-plan warning.
func formatNumber(n float64) string {
	s := strconv.FormatFloat(n, 'f', -1, 64)
	intPart, frac := s, ""
	if i := strings.IndexByte(s, '.'); i >= 0 {
		intPart, frac = s[:i], s[i:]
	}
	var b strings.Builder
	for i, r := range intPart {
		if i > 0 && (len(intPart)-i)%3 == 0 {
			b.WriteByte(',')
		}
		b.WriteRune(r)
	}
	return b.String() + frac
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →