Skip to content

LLM Cost Calculator — Python source

Estimate LLM costs per request or per month at billion-token scale — with realistic prompt-cache hit rates, four-lane pricing, and side-by-side model comparison from a dated pricing snapshot.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""LLM Cost Calculator — per-million-token cost math for LLM workloads.

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the LLM Cost Calculator tool,
          ported from src/lib/llmCost.ts (the canonical TypeScript
          implementation), with the tokens_per_dollar helper inlined from
          src/lib/ai/models.ts so this module stays dependency-free.
Tool page: https://dev.cosmolabs.org/tools/llm-cost-calculator
License:  display source — part of CosmoDev's polyglot tool pages.

Design goals:
  - Pure + deterministic; never raises.
  - Functionally equivalent to the TS reference: same inputs -> same outputs
    (a None rate means "unpriced" and propagates to a None cost).
  - Self-contained: stdlib only — and no model-snapshot import.

Formula: ((input_tokens / 1e6) * input_per_m + (output_tokens / 1e6) * output_per_m)
         * requests * (0.5 if batch else 1). Model rates are NEVER hardcoded
         here — the caller supplies them (in the TS lib they flow from the
         model snapshot accessor; custom rates are the one exception).
"""

from __future__ import annotations

from dataclasses import dataclass
from typing import List, Optional

__all__ = [
    "BATCH_DISCOUNT",
    "CostRates",
    "CostInput",
    "Model",
    "ModelCost",
    "cost_for",
    "rates_for",
    "tokens_per_dollar",
    "compare_models",
]

#: Multiplier applied to Batch API pricing (the standard 50% discount).
BATCH_DISCOUNT: float = 0.5


@dataclass(frozen=True)
class CostRates:
    """Price pair for a model or a custom rate card. ``None`` = unpriced."""

    input_per_m: Optional[float]
    """USD per 1M input tokens."""

    output_per_m: Optional[float]
    """USD per 1M output tokens."""


@dataclass(frozen=True)
class CostInput:
    """One workload to price."""

    input_tokens: int
    """Input tokens per request."""

    output_tokens: int
    """Output tokens per request."""

    requests: int = 1
    """Request count."""

    batch: bool = False
    """Apply the BATCH_DISCOUNT multiplier."""


@dataclass(frozen=True)
class Model:
    """Pricing projection of a model. The full AiModel record
    (src/lib/ai/models.ts) carries a dozen non-pricing fields; only these
    three matter to cost math."""

    id: str
    input_per_m: Optional[float]
    output_per_m: Optional[float]


@dataclass(frozen=True)
class ModelCost:
    """One row of a :func:`compare_models` result."""

    id: str
    input_per_m: Optional[float]
    output_per_m: Optional[float]
    cost: Optional[float]
    """cost_for with the model's rates; None when unpriced."""

    tokens_per_dollar: Optional[float]
    """Output tokens per USD: 1e6 / output_per_m (None-safe)."""


def cost_for(rates: CostRates, workload: CostInput) -> Optional[float]:
    """Cost in USD for a workload, or None when either rate is unpriced:

    ((input_tokens/1e6)·input_per_m + (output_tokens/1e6)·output_per_m)
    × requests × BATCH_DISCOUNT when batch.
    """
    if rates.input_per_m is None or rates.output_per_m is None:
        return None
    base = (
        workload.input_tokens / 1_000_000 * rates.input_per_m
        + workload.output_tokens / 1_000_000 * rates.output_per_m
    )
    multiplier = workload.requests * (BATCH_DISCOUNT if workload.batch else 1.0)
    return base * multiplier


def rates_for(m: Model) -> CostRates:
    """Project a model onto its CostRates pair."""
    return CostRates(input_per_m=m.input_per_m, output_per_m=m.output_per_m)


def tokens_per_dollar(m: Model) -> Optional[float]:
    """Output tokens per USD: 1e6 / output_per_m. None when unpriced."""
    return None if m.output_per_m is None else 1_000_000 / m.output_per_m


def _sort_key(row: ModelCost) -> tuple:
    """Ordering key for compare_models rows: cost asc, nulls last, id asc.

    TS compares nulls explicitly; here ``row.cost is None`` sorts after every
    priced row (False < True), priced rows compare by cost then id, and two
    unpriced rows fall through to id — the comparator's exact semantics.
    Ids are ASCII slugs, so plain str comparison matches the TS localeCompare.
    """
    return (row.cost is None, row.cost if row.cost is not None else 0.0, row.id)


def compare_models(
    model_ids: List[str],
    workload: CostInput,
    models: List[Model],
) -> List[ModelCost]:
    """Cost every requested model for one workload. Unknown ids are dropped.

    The TS original defaults ``models`` to the live snapshot (allModels());
    this port has no snapshot dependency, so the model list is always explicit.
    """
    by_id = {m.id: m for m in models}
    rows = []
    for mid in model_ids:
        m = by_id.get(mid)
        if m is None:
            continue
        rows.append(
            ModelCost(
                id=mid,
                input_per_m=m.input_per_m,
                output_per_m=m.output_per_m,
                cost=cost_for(rates_for(m), workload),
                tokens_per_dollar=tokens_per_dollar(m),
            )
        )
    rows.sort(key=_sort_key)
    return rows

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →