Skip to content

Cache Savings Calculator — Zig source

See what prompt caching saves — uncached vs cached cost over N requests, with the write-premium break-even point.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! cache_savings — uncached vs prompt-cached LLM cost comparison.
//!
//! Language: Zig (0.13, standard library only)
//! Source:   CosmoDev polyglot showcase port of the Cache Savings Calculator
//!           tool, ported from src/lib/cacheSavings.ts (the canonical
//!           TypeScript implementation).
//! Tool:     https://dev.cosmolabs.org/tools/cache-savings-calculator
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; never panics (plain f64 math, no unchecked
//!     division — the only division guards `uncached == 0`).
//!   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//!   - Self-contained: std only (`?f64` mirrors the TS `null`).
//!
//! The TS original takes a full AiModel record but reads only its four pricing
//! rates, so this port narrows the parameter to exactly those fields. Any
//! null rate makes every output null — the caller renders an explanatory
//! empty state instead of partial math. All rates are per-1M-token USD,
//! mirroring the cost conventions of llmCost.ts.
//!
//! Numeric mapping: TS `number` is f64, so tokens and hits stay `f64`
//! (fractional hits clamp up to 1.0 exactly like `Math.max(1, hits)` —
//! `@max` here); breakEvenHits is a whole hit count, so it lands in `?u64`
//! via `@ceil` + `@intFromFloat`.

const std = @import("std");

/// The four per-1M-token USD pricing rates `cacheMath` reads from the TS
/// AiModel record.
pub const ModelRates = struct {
    input_per_m: ?f64, // Uncached prompt (input) rate, USD per 1M tokens.
    output_per_m: ?f64, // Completion (output) rate, USD per 1M tokens.
    cache_read_per_m: ?f64, // Cached prompt read rate, USD per 1M tokens.
    cache_write_per_m: ?f64, // Cache write premium rate, USD per 1M tokens.
};

/// Request shape (TS `CacheInput`).
pub const CacheInput = struct {
    prompt_tokens: f64, // Prompt (input) tokens per request.
    output_tokens: f64, // Completion (output) tokens per request.
    hits: f64, // Requests reusing the cached prompt; < 1 counts as 1.
};

/// Result shape. Any missing rate nulls every field.
pub const CacheMath = struct {
    uncached: ?f64, // `hits × (prompt·in$/M + output·out$/M) / 1e6`.
    cached: ?f64, // `(prompt·write$/M + hits × (prompt·read$/M + output·out$/M)) / 1e6` —
                  // one cache write, `hits` cache reads, output billed every request.
    savings: ?f64, // uncached − cached (negative when caching costs more).
    savings_pct: ?f64, // savings / uncached × 100; 0 when uncached is 0.
    break_even_hits: ?u64, // `ceil(write$/M / read$/M)` when read$/M > 0 — hits needed for
                           // cumulative READ spend to equal ONE write premium; null otherwise.
};

/// The all-null result used when any pricing rate is missing.
fn nulled() CacheMath {
    return .{
        .uncached = null,
        .cached = null,
        .savings = null,
        .savings_pct = null,
        .break_even_hits = null,
    };
}

/// Compare uncached vs prompt-cached cost for one model. Any missing rate
/// (input, output, cache read, cache write) nulls every field — the caller
/// renders an explanatory empty state instead of partial math.
pub fn cacheMath(model: ModelRates, input: CacheInput) CacheMath {
    const ipm = model.input_per_m orelse return nulled();
    const opm = model.output_per_m orelse return nulled();
    const cr = model.cache_read_per_m orelse return nulled();
    const cw = model.cache_write_per_m orelse return nulled();

    const hits = @max(1.0, input.hits);
    const in_t = input.prompt_tokens;
    const out_t = input.output_tokens;

    // One cache write, `hits` cache reads; output tokens are billed on every request.
    const uncached = hits * (in_t * ipm + out_t * opm) / 1_000_000.0;
    const cached = (in_t * cw + hits * (in_t * cr + out_t * opm)) / 1_000_000.0;
    const savings = uncached - cached;
    const savings_pct = if (uncached == 0.0) 0.0 else savings / uncached * 100.0;
    const break_even_hits: ?u64 = if (cr > 0.0)
        @intFromFloat(@ceil(cw / cr))
    else
        null;

    return .{
        .uncached = uncached,
        .cached = cached,
        .savings = savings,
        .savings_pct = savings_pct,
        .break_even_hits = break_even_hits,
    };
}

// ----------------------------------------------------------------------
// Self-test — the reference vectors shared with cacheSavings.test.ts (the
// lock-step contract every port mirrors). `zig run zig.zig` prints "ok".
// ----------------------------------------------------------------------

fn check(ok: bool, what: []const u8) void {
    if (!ok) {
        std.debug.print("cache-savings self-test failed: {s}\n", .{what});
        std.process.exit(1);
    }
}

fn close(a: f64, b: f64) bool {
    return @abs(a - b) < 1e-9;
}

/// Fixture model F: input_per_m 10, output_per_m 50, cache_read_per_m 1,
/// cache_write_per_m 12.5. Null one rate to test the unpriced path.
fn base() ModelRates {
    return .{
        .input_per_m = 10.0,
        .output_per_m = 50.0,
        .cache_read_per_m = 1.0,
        .cache_write_per_m = 12.5,
    };
}

fn request(p: f64, o: f64, h: f64) CacheInput {
    return .{ .prompt_tokens = p, .output_tokens = o, .hits = h };
}

pub fn main() void {
    // Spec vector: 10k in / 1k out / 5 hits -> uncached 0.75, cached 0.425,
    // savings 0.325, 43.333...% saved, break-even 13 hits.
    var r = cacheMath(base(), request(10_000.0, 1_000.0, 5.0));
    check(close(r.uncached.?, 0.75), "uncached");
    check(close(r.cached.?, 0.425), "cached");
    check(close(r.savings.?, 0.325), "savings");
    check(close(r.savings_pct.?, 43.3333333333), "savingsPct");
    check(r.break_even_hits.? == 13, "breakEven 13"); // ceil(12.5 / 1)

    // At 1 hit caching LOSES 0.035 — an honest negative saving.
    r = cacheMath(base(), request(10_000.0, 1_000.0, 1.0));
    check(close(r.uncached.?, 0.15), "one-hit uncached");
    check(close(r.cached.?, 0.185), "one-hit cached");
    check(close(r.savings.?, -0.035), "one-hit negative saving");
    check(close(r.savings_pct.?, -23.3333333333), "one-hit negative pct");
    check(r.break_even_hits.? == 13, "one-hit breakEven");

    // Each missing rate in turn nulls every field.
    const variants = [_]ModelRates{
        .{ .input_per_m = null, .output_per_m = 50.0, .cache_read_per_m = 1.0, .cache_write_per_m = 12.5 },
        .{ .input_per_m = 10.0, .output_per_m = null, .cache_read_per_m = 1.0, .cache_write_per_m = 12.5 },
        .{ .input_per_m = 10.0, .output_per_m = 50.0, .cache_read_per_m = null, .cache_write_per_m = 12.5 },
        .{ .input_per_m = 10.0, .output_per_m = 50.0, .cache_read_per_m = 1.0, .cache_write_per_m = null },
    };
    for (variants) |v| {
        const x = cacheMath(v, request(10_000.0, 1_000.0, 5.0));
        check(x.uncached == null and x.cached == null and x.savings == null and
            x.savings_pct == null and x.break_even_hits == null,
            "missing rate nulls everything");
    }

    // hits < 1 counts as 1.
    const r0 = cacheMath(base(), request(10_000.0, 1_000.0, 0.0));
    const r1 = cacheMath(base(), request(10_000.0, 1_000.0, 1.0));
    check(std.meta.eql(r0, r1), "hits < 1 clamps to 1");

    // Zero tokens -> zero costs with 0%, no division error.
    r = cacheMath(base(), request(0.0, 0.0, 5.0));
    check(r.uncached.? == 0.0 and r.cached.? == 0.0 and
        r.savings.? == 0.0 and r.savings_pct.? == 0.0,
        "zero tokens");
    check(r.break_even_hits.? == 13, "zero-tokens breakEven");

    // cache_read 0 -> break-even null but costs kept (10k×$12.5 + 5×(0 + 1k×$50)).
    r = cacheMath(.{
        .input_per_m = 10.0,
        .output_per_m = 50.0,
        .cache_read_per_m = 0.0,
        .cache_write_per_m = 12.5,
    }, request(10_000.0, 1_000.0, 5.0));
    check(r.break_even_hits == null, "zero read rate nulls breakEven"); // premium never repaid
    check(close(r.uncached.?, 0.75), "zero-read uncached");
    check(close(r.cached.?, 0.375), "zero-read cached");

    // ceil(4/2) stays 2 — no rounding up at the exact integer boundary.
    r = cacheMath(.{
        .input_per_m = 10.0,
        .output_per_m = 50.0,
        .cache_read_per_m = 2.0,
        .cache_write_per_m = 4.0,
    }, request(1_000.0, 0.0, 3.0));
    check(r.break_even_hits.? == 2, "integer boundary");

    std.debug.print("ok\n", .{});
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →