Skip to content

LLM Cost Calculator — TypeScript source

Estimate LLM costs per request or per month at billion-token scale — with realistic prompt-cache hit rates, four-lane pricing, and side-by-side model comparison from a dated pricing snapshot.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

/**
 * Pure cost math for the LLM Cost Calculator. All rates flow from the model
 * snapshot accessor (src/lib/ai/models.ts) — never hardcoded here. Custom
 * rates are the one exception: the caller supplies them as a CostRates value.
 */
import { allModels, getModel, tokensPerDollar, type AiModel } from './ai/models';

/** Multiplier applied to Batch API pricing (the standard 50% discount). */
export const BATCH_DISCOUNT = 0.5;

export interface CostRates {
  inputPerM: number | null;
  outputPerM: number | null;
}

export interface CostInput {
  /** Input tokens per request. */
  inputTokens: number;
  /** Output tokens per request. */
  outputTokens: number;
  /** Request count; defaults to 1. */
  requests?: number;
  /** Apply the BATCH_DISCOUNT multiplier. */
  batch?: boolean;
}

/**
 * Cost in USD for a workload, or null when either rate is unpriced:
 * ((inputTokens/1e6)·inputPerM + (outputTokens/1e6)·outputPerM) × requests
 * × BATCH_DISCOUNT when batch.
 */
export function costFor(rates: CostRates, input: CostInput): number | null {
  if (rates.inputPerM === null || rates.outputPerM === null) return null;
  const base =
    (input.inputTokens / 1_000_000) * rates.inputPerM +
    (input.outputTokens / 1_000_000) * rates.outputPerM;
  return base * (input.requests ?? 1) * (input.batch ? BATCH_DISCOUNT : 1);
}

export interface ModelCost {
  id: string;
  inputPerM: number | null;
  outputPerM: number | null;
  /** costFor with the model's rates; null when unpriced. */
  cost: number | null;
  /** Output tokens per USD: 1e6 / outputPerM (null-safe). */
  tokensPerDollar: number | null;
}

/** Project a model onto its CostRates pair. */
export function ratesFor(m: AiModel): CostRates {
  return { inputPerM: m.inputPerM, outputPerM: m.outputPerM };
}

// -- Cache-aware monthly estimator (FEAT-070) --

/** All four pricing lanes, cache fields null when a model has no caching. */
export interface CacheRates {
  inputPerM: number | null;
  outputPerM: number | null;
  cacheReadPerM: number | null;
  cacheWritePerM: number | null;
}

/** A monthly workload at scale: total tokens/month, cache profile, batch flag. */
export interface MonthlyInput {
  /** Total input tokens per month. */
  inputTokensPerMonth: number;
  /** Total output tokens per month. */
  outputTokensPerMonth: number;
  /** Share of input served from cache (0..1). */
  cacheHit: number;
  /** Share of input eligible for caching; default 0.9. */
  cacheableFraction?: number;
  /**
   * Explicit cache-write tokens/month. Default derives from this month's
   * misses: inputTokensPerMonth × (1 − cacheHit) × cacheableFraction — each
   * cacheable miss is written once, then served by reads.
   */
  cacheWriteTokensPerMonth?: number;
  /** Apply the BATCH_DISCOUNT multiplier. */
  batch?: boolean;
}

export interface MonthlyCost {
  total: number | null;
  uncachedInput: number | null;
  cacheReads: number | null;
  cacheWrites: number | null;
  output: number | null;
  /** True when cacheHit > 0 was requested but the model has no cache pricing. */
  cacheUnavailable: boolean;
  /** total / (input+output tokens in millions); null when total is null. */
  blendedPerM: number | null;
}

/** Project a model onto all four lanes. */
export function cacheRatesFor(m: AiModel): CacheRates {
  return {
    inputPerM: m.inputPerM,
    outputPerM: m.outputPerM,
    cacheReadPerM: m.cacheReadPerM,
    cacheWritePerM: m.cacheWritePerM,
  };
}

/**
 * Monthly cost with cache economics. Caching is active iff cacheHit > 0 AND
 * both cache rates are priced; otherwise the estimate degrades to plain
 * no-cache math with cacheUnavailable set (total stays a number — the
 * estimate is still useful, just less precise). Input/output unpriced →
 * total null, all lanes null (the costFor convention).
 */
export function monthlyCost(rates: CacheRates, w: MonthlyInput): MonthlyCost {
  const unpriced = rates.inputPerM === null || rates.outputPerM === null;
  const cacheable = w.cacheableFraction ?? 0.9;
  const wantedCache = w.cacheHit > 0;
  const hasCacheRates = rates.cacheReadPerM !== null && rates.cacheWritePerM !== null;
  const active = wantedCache && hasCacheRates;
  const hit = active ? w.cacheHit : 0;

  if (unpriced) {
    return {
      total: null,
      uncachedInput: null,
      cacheReads: null,
      cacheWrites: null,
      output: null,
      cacheUnavailable: wantedCache && !hasCacheRates,
      blendedPerM: null,
    };
  }

  const inT = w.inputTokensPerMonth;
  const missTokens = inT * (1 - hit);
  const writeTokens =
    w.cacheWriteTokensPerMonth ?? (active ? missTokens * cacheable : 0);

  const uncachedInput = (missTokens / 1e6) * rates.inputPerM!;
  // Lane semantics: real 0 when caching is simply off; null only on the
  // degrade path (cache wanted, rates absent — the price is unknowable).
  const degrade = wantedCache && !hasCacheRates;
  const cacheReads = active ? (inT * hit / 1e6) * rates.cacheReadPerM! : degrade ? null : 0;
  const cacheWrites = active
    ? (writeTokens / 1e6) * rates.cacheWritePerM!
    : degrade
      ? null
      : 0;
  const output = (w.outputTokensPerMonth / 1e6) * rates.outputPerM!;

  // Degrade lanes are null (unknowable, not free) — they contribute nothing.
  const lanes = uncachedInput + output + (cacheReads ?? 0) + (cacheWrites ?? 0);
  const total = lanes * (w.batch ? BATCH_DISCOUNT : 1);
  const totalMTok = (inT + w.outputTokensPerMonth) / 1e6;

  return {
    total,
    uncachedInput,
    cacheReads,
    cacheWrites,
    output,
    cacheUnavailable: wantedCache && !hasCacheRates,
    blendedPerM: totalMTok > 0 ? total / totalMTok : null,
  };
}

/**
 * Parse a tokens/month string: "2B"/"1.5b"/"500M"/"10m"/"2K"/"2000"/"1_000_000".
 * Case-insensitive suffix; underscores allowed; throws on anything else.
 */
export function parseTokensPerMonth(s: string): number {
  const m = /^(\d[\d_]*(?:\.\d+)?)([bmk])?$/i.exec(s.trim());
  if (!m) throw new Error(`unparseable tokens/month: ${JSON.stringify(s)}`);
  const base = Number(m[1].replace(/_/g, ''));
  const mult = { b: 1e9, m: 1e6, k: 1e3 }[((m[2] ?? '') as string).toLowerCase()] ?? 1;
  return base * mult;
}

export interface WorkloadPreset {
  id: string;
  label: string;
  cacheHit: number;
  cacheableFraction: number;
  rationale: string;
}

/**
 * Archetype defaults for realistic cache modeling. Hit rates are planning
 * numbers from published prompt-caching writeups and observed agent logs —
 * refine against live data as it accumulates.
 */
export const WORKLOAD_PRESETS: readonly WorkloadPreset[] = [
  {
    id: 'agent-coding',
    label: 'Agent / coding',
    cacheHit: 0.75,
    cacheableFraction: 0.95,
    rationale: 'Large stable system prompt plus repo context, re-sent every turn — the highest-hit profile.',
  },
  {
    id: 'chat',
    label: 'Chat / assistant',
    cacheHit: 0.3,
    cacheableFraction: 0.6,
    rationale: 'Short varied conversations; only the system prompt and early turns stay hot.',
  },
  {
    id: 'rag',
    label: 'RAG / retrieval',
    cacheHit: 0.5,
    cacheableFraction: 0.8,
    rationale: 'Preamble caches well but each query pulls fresh retrieved chunks.',
  },
  {
    id: 'batch-summarize',
    label: 'Batch summarize',
    cacheHit: 0.6,
    cacheableFraction: 0.5,
    rationale: 'Shared instruction prefix across many unique documents.',
  },
] as const;

/**
 * Cost every requested model for one workload. Unknown ids are dropped.
 * Sort: cost asc, nulls last, ties by id asc.
 */
export function compareModels(
  modelIds: string[],
  input: CostInput,
  models: readonly AiModel[] = allModels(),
): ModelCost[] {
  const rows: ModelCost[] = [];
  for (const id of modelIds) {
    const m = getModel(id, models);
    if (m === undefined) continue;
    rows.push({
      id,
      inputPerM: m.inputPerM,
      outputPerM: m.outputPerM,
      cost: costFor(ratesFor(m), input),
      tokensPerDollar: tokensPerDollar(m),
    });
  }
  return rows.sort((a, b) => {
    if (a.cost === null || b.cost === null) {
      if (a.cost === null && b.cost === null) return a.id.localeCompare(b.id);
      return a.cost === null ? 1 : -1;
    }
    if (a.cost === b.cost) return a.id.localeCompare(b.id);
    return a.cost - b.cost;
  });
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →