LLM Cost Calculator — TypeScript source
Estimate LLM costs per request or per month at billion-token scale — with realistic prompt-cache hit rates, four-lane pricing, and side-by-side model comparison from a dated pricing snapshot.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
/**
* Pure cost math for the LLM Cost Calculator. All rates flow from the model
* snapshot accessor (src/lib/ai/models.ts) — never hardcoded here. Custom
* rates are the one exception: the caller supplies them as a CostRates value.
*/
import { allModels, getModel, tokensPerDollar, type AiModel } from './ai/models';
/** Multiplier applied to Batch API pricing (the standard 50% discount). */
export const BATCH_DISCOUNT = 0.5;
export interface CostRates {
inputPerM: number | null;
outputPerM: number | null;
}
export interface CostInput {
/** Input tokens per request. */
inputTokens: number;
/** Output tokens per request. */
outputTokens: number;
/** Request count; defaults to 1. */
requests?: number;
/** Apply the BATCH_DISCOUNT multiplier. */
batch?: boolean;
}
/**
* Cost in USD for a workload, or null when either rate is unpriced:
* ((inputTokens/1e6)·inputPerM + (outputTokens/1e6)·outputPerM) × requests
* × BATCH_DISCOUNT when batch.
*/
export function costFor(rates: CostRates, input: CostInput): number | null {
if (rates.inputPerM === null || rates.outputPerM === null) return null;
const base =
(input.inputTokens / 1_000_000) * rates.inputPerM +
(input.outputTokens / 1_000_000) * rates.outputPerM;
return base * (input.requests ?? 1) * (input.batch ? BATCH_DISCOUNT : 1);
}
export interface ModelCost {
id: string;
inputPerM: number | null;
outputPerM: number | null;
/** costFor with the model's rates; null when unpriced. */
cost: number | null;
/** Output tokens per USD: 1e6 / outputPerM (null-safe). */
tokensPerDollar: number | null;
}
/** Project a model onto its CostRates pair. */
export function ratesFor(m: AiModel): CostRates {
return { inputPerM: m.inputPerM, outputPerM: m.outputPerM };
}
// -- Cache-aware monthly estimator (FEAT-070) --
/** All four pricing lanes, cache fields null when a model has no caching. */
export interface CacheRates {
inputPerM: number | null;
outputPerM: number | null;
cacheReadPerM: number | null;
cacheWritePerM: number | null;
}
/** A monthly workload at scale: total tokens/month, cache profile, batch flag. */
export interface MonthlyInput {
/** Total input tokens per month. */
inputTokensPerMonth: number;
/** Total output tokens per month. */
outputTokensPerMonth: number;
/** Share of input served from cache (0..1). */
cacheHit: number;
/** Share of input eligible for caching; default 0.9. */
cacheableFraction?: number;
/**
* Explicit cache-write tokens/month. Default derives from this month's
* misses: inputTokensPerMonth × (1 − cacheHit) × cacheableFraction — each
* cacheable miss is written once, then served by reads.
*/
cacheWriteTokensPerMonth?: number;
/** Apply the BATCH_DISCOUNT multiplier. */
batch?: boolean;
}
export interface MonthlyCost {
total: number | null;
uncachedInput: number | null;
cacheReads: number | null;
cacheWrites: number | null;
output: number | null;
/** True when cacheHit > 0 was requested but the model has no cache pricing. */
cacheUnavailable: boolean;
/** total / (input+output tokens in millions); null when total is null. */
blendedPerM: number | null;
}
/** Project a model onto all four lanes. */
export function cacheRatesFor(m: AiModel): CacheRates {
return {
inputPerM: m.inputPerM,
outputPerM: m.outputPerM,
cacheReadPerM: m.cacheReadPerM,
cacheWritePerM: m.cacheWritePerM,
};
}
/**
* Monthly cost with cache economics. Caching is active iff cacheHit > 0 AND
* both cache rates are priced; otherwise the estimate degrades to plain
* no-cache math with cacheUnavailable set (total stays a number — the
* estimate is still useful, just less precise). Input/output unpriced →
* total null, all lanes null (the costFor convention).
*/
export function monthlyCost(rates: CacheRates, w: MonthlyInput): MonthlyCost {
const unpriced = rates.inputPerM === null || rates.outputPerM === null;
const cacheable = w.cacheableFraction ?? 0.9;
const wantedCache = w.cacheHit > 0;
const hasCacheRates = rates.cacheReadPerM !== null && rates.cacheWritePerM !== null;
const active = wantedCache && hasCacheRates;
const hit = active ? w.cacheHit : 0;
if (unpriced) {
return {
total: null,
uncachedInput: null,
cacheReads: null,
cacheWrites: null,
output: null,
cacheUnavailable: wantedCache && !hasCacheRates,
blendedPerM: null,
};
}
const inT = w.inputTokensPerMonth;
const missTokens = inT * (1 - hit);
const writeTokens =
w.cacheWriteTokensPerMonth ?? (active ? missTokens * cacheable : 0);
const uncachedInput = (missTokens / 1e6) * rates.inputPerM!;
// Lane semantics: real 0 when caching is simply off; null only on the
// degrade path (cache wanted, rates absent — the price is unknowable).
const degrade = wantedCache && !hasCacheRates;
const cacheReads = active ? (inT * hit / 1e6) * rates.cacheReadPerM! : degrade ? null : 0;
const cacheWrites = active
? (writeTokens / 1e6) * rates.cacheWritePerM!
: degrade
? null
: 0;
const output = (w.outputTokensPerMonth / 1e6) * rates.outputPerM!;
// Degrade lanes are null (unknowable, not free) — they contribute nothing.
const lanes = uncachedInput + output + (cacheReads ?? 0) + (cacheWrites ?? 0);
const total = lanes * (w.batch ? BATCH_DISCOUNT : 1);
const totalMTok = (inT + w.outputTokensPerMonth) / 1e6;
return {
total,
uncachedInput,
cacheReads,
cacheWrites,
output,
cacheUnavailable: wantedCache && !hasCacheRates,
blendedPerM: totalMTok > 0 ? total / totalMTok : null,
};
}
/**
* Parse a tokens/month string: "2B"/"1.5b"/"500M"/"10m"/"2K"/"2000"/"1_000_000".
* Case-insensitive suffix; underscores allowed; throws on anything else.
*/
export function parseTokensPerMonth(s: string): number {
const m = /^(\d[\d_]*(?:\.\d+)?)([bmk])?$/i.exec(s.trim());
if (!m) throw new Error(`unparseable tokens/month: ${JSON.stringify(s)}`);
const base = Number(m[1].replace(/_/g, ''));
const mult = { b: 1e9, m: 1e6, k: 1e3 }[((m[2] ?? '') as string).toLowerCase()] ?? 1;
return base * mult;
}
export interface WorkloadPreset {
id: string;
label: string;
cacheHit: number;
cacheableFraction: number;
rationale: string;
}
/**
* Archetype defaults for realistic cache modeling. Hit rates are planning
* numbers from published prompt-caching writeups and observed agent logs —
* refine against live data as it accumulates.
*/
export const WORKLOAD_PRESETS: readonly WorkloadPreset[] = [
{
id: 'agent-coding',
label: 'Agent / coding',
cacheHit: 0.75,
cacheableFraction: 0.95,
rationale: 'Large stable system prompt plus repo context, re-sent every turn — the highest-hit profile.',
},
{
id: 'chat',
label: 'Chat / assistant',
cacheHit: 0.3,
cacheableFraction: 0.6,
rationale: 'Short varied conversations; only the system prompt and early turns stay hot.',
},
{
id: 'rag',
label: 'RAG / retrieval',
cacheHit: 0.5,
cacheableFraction: 0.8,
rationale: 'Preamble caches well but each query pulls fresh retrieved chunks.',
},
{
id: 'batch-summarize',
label: 'Batch summarize',
cacheHit: 0.6,
cacheableFraction: 0.5,
rationale: 'Shared instruction prefix across many unique documents.',
},
] as const;
/**
* Cost every requested model for one workload. Unknown ids are dropped.
* Sort: cost asc, nulls last, ties by id asc.
*/
export function compareModels(
modelIds: string[],
input: CostInput,
models: readonly AiModel[] = allModels(),
): ModelCost[] {
const rows: ModelCost[] = [];
for (const id of modelIds) {
const m = getModel(id, models);
if (m === undefined) continue;
rows.push({
id,
inputPerM: m.inputPerM,
outputPerM: m.outputPerM,
cost: costFor(ratesFor(m), input),
tokensPerDollar: tokensPerDollar(m),
});
}
return rows.sort((a, b) => {
if (a.cost === null || b.cost === null) {
if (a.cost === null && b.cost === null) return a.id.localeCompare(b.id);
return a.cost === null ? 1 : -1;
}
if (a.cost === b.cost) return a.id.localeCompare(b.id);
return a.cost - b.cost;
});
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →