Skip to content

LLM Cost Calculator — Rust source

Estimate LLM costs per request or per month at billion-token scale — with realistic prompt-cache hit rates, four-lane pricing, and side-by-side model comparison from a dated pricing snapshot.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! LLM Cost Calculator — per-million-token cost math for LLM workloads.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source:   CosmoDev polyglot showcase port of the LLM Cost Calculator tool,
//!           ported from src/lib/llmCost.ts (the canonical TypeScript
//!           implementation), with the `tokens_per_dollar` helper inlined
//!           from src/lib/ai/models.ts so this module stays dependency-free.
//! Tool page: https://dev.cosmolabs.org/tools/llm-cost-calculator
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; never panics (public API returns plain values,
//!     no Result — unpriced is a value, not an error).
//!   - Functionally equivalent to the TS reference: same inputs -> same outputs
//!     (a `None` rate means "unpriced" and propagates to a `None` cost).
//!   - Self-contained: std only (no crates.io dependencies, no model snapshot).
//!
//! Formula: ((input_tokens / 1e6) * input_per_m + (output_tokens / 1e6) * output_per_m)
//!          * requests * (0.5 if batch). Model rates are NEVER hardcoded here —
//!          the caller supplies them (in the TS lib they flow from the model
//!          snapshot accessor; custom rates are the one exception).

use std::cmp::Ordering;

/// Multiplier applied to Batch API pricing (the standard 50% discount).
pub const BATCH_DISCOUNT: f64 = 0.5;

/// Price pair for a model or a custom rate card. `None` = unpriced.
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct CostRates {
    /// USD per 1M input tokens.
    pub input_per_m: Option<f64>,
    /// USD per 1M output tokens.
    pub output_per_m: Option<f64>,
}

/// One workload to price.
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct CostInput {
    /// Input tokens per request.
    pub input_tokens: u64,
    /// Output tokens per request.
    pub output_tokens: u64,
    /// Request count; `None` defaults to 1 (mirrors the TS `requests?`).
    pub requests: Option<u64>,
    /// Apply the BATCH_DISCOUNT multiplier.
    pub batch: bool,
}

impl CostInput {
    /// Convenience constructor for the common single-request, non-batch case.
    pub fn new(input_tokens: u64, output_tokens: u64) -> Self {
        Self { input_tokens, output_tokens, requests: None, batch: false }
    }
}

/// Pricing projection of a model. The full AiModel record
/// (src/lib/ai/models.ts) carries a dozen non-pricing fields; only these
/// three matter to cost math.
#[derive(Debug, Clone, PartialEq)]
pub struct Model {
    pub id: String,
    pub input_per_m: Option<f64>,
    pub output_per_m: Option<f64>,
}

/// One row of a `compare_models` result.
#[derive(Debug, Clone, PartialEq)]
pub struct ModelCost {
    pub id: String,
    pub input_per_m: Option<f64>,
    pub output_per_m: Option<f64>,
    /// `cost_for` with the model's rates; `None` when unpriced.
    pub cost: Option<f64>,
    /// Output tokens per USD: 1e6 / output_per_m (`None`-safe).
    pub tokens_per_dollar: Option<f64>,
}

/// Cost in USD for a workload, or `None` when either rate is unpriced:
/// ((input_tokens/1e6)·input_per_m + (output_tokens/1e6)·output_per_m)
/// × requests × BATCH_DISCOUNT when batch.
pub fn cost_for(rates: &CostRates, input: &CostInput) -> Option<f64> {
    let (in_rate, out_rate) = (rates.input_per_m?, rates.output_per_m?);
    let base = (input.input_tokens as f64 / 1_000_000.0) * in_rate
        + (input.output_tokens as f64 / 1_000_000.0) * out_rate;
    let multiplier = input.requests.unwrap_or(1) as f64
        * if input.batch { BATCH_DISCOUNT } else { 1.0 };
    Some(base * multiplier)
}

/// Project a model onto its CostRates pair.
pub fn rates_for(m: &Model) -> CostRates {
    CostRates { input_per_m: m.input_per_m, output_per_m: m.output_per_m }
}

/// Output tokens per USD: 1e6 / output_per_m. `None` when unpriced.
pub fn tokens_per_dollar(m: &Model) -> Option<f64> {
    Some(1_000_000.0 / m.output_per_m?)
}

/// Cost every requested model for one workload. Unknown ids are dropped.
/// Sort: cost asc, nulls last, ties by id asc.
///
/// The TS original defaults `models` to the live snapshot (allModels()); this
/// port has no snapshot dependency, so the model list is always explicit.
/// `sort_by` is stable and ids are ASCII slugs compared byte-wise, matching
/// the TS Array.prototype.sort + localeCompare ordering for this domain.
/// Costs are finite whenever `Some` (finite rates × finite tokens), so the
/// `partial_cmp` fallback to `Equal` on NaN is unreachable in practice.
pub fn compare_models(
    model_ids: &[&str],
    input: &CostInput,
    models: &[Model],
) -> Vec<ModelCost> {
    let mut rows: Vec<ModelCost> = model_ids
        .iter()
        .filter_map(|id| {
            let m = models.iter().find(|m| m.id == *id)?;
            Some(ModelCost {
                id: (*id).to_string(),
                input_per_m: m.input_per_m,
                output_per_m: m.output_per_m,
                cost: cost_for(&rates_for(m), input),
                tokens_per_dollar: tokens_per_dollar(m),
            })
        })
        .collect();

    rows.sort_by(|a, b| match (&a.cost, &b.cost) {
        (Some(x), Some(y)) => x
            .partial_cmp(y)
            .unwrap_or(Ordering::Equal)
            .then_with(|| a.id.cmp(&b.id)),
        (None, None) => a.id.cmp(&b.id),
        (None, Some(_)) => Ordering::Greater,
        (Some(_), None) => Ordering::Less,
    });

    rows
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →