Cache Savings Calculator — Zig source
See what prompt caching saves — uncached vs cached cost over N requests, with the write-premium break-even point.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! cache_savings — uncached vs prompt-cached LLM cost comparison.
//!
//! Language: Zig (0.13, standard library only)
//! Source: CosmoDev polyglot showcase port of the Cache Savings Calculator
//! tool, ported from src/lib/cacheSavings.ts (the canonical
//! TypeScript implementation).
//! Tool: https://dev.cosmolabs.org/tools/cache-savings-calculator
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; never panics (plain f64 math, no unchecked
//! division — the only division guards `uncached == 0`).
//! - Functionally equivalent to the TS reference: same inputs -> same outputs.
//! - Self-contained: std only (`?f64` mirrors the TS `null`).
//!
//! The TS original takes a full AiModel record but reads only its four pricing
//! rates, so this port narrows the parameter to exactly those fields. Any
//! null rate makes every output null — the caller renders an explanatory
//! empty state instead of partial math. All rates are per-1M-token USD,
//! mirroring the cost conventions of llmCost.ts.
//!
//! Numeric mapping: TS `number` is f64, so tokens and hits stay `f64`
//! (fractional hits clamp up to 1.0 exactly like `Math.max(1, hits)` —
//! `@max` here); breakEvenHits is a whole hit count, so it lands in `?u64`
//! via `@ceil` + `@intFromFloat`.
const std = @import("std");
/// The four per-1M-token USD pricing rates `cacheMath` reads from the TS
/// AiModel record.
pub const ModelRates = struct {
input_per_m: ?f64, // Uncached prompt (input) rate, USD per 1M tokens.
output_per_m: ?f64, // Completion (output) rate, USD per 1M tokens.
cache_read_per_m: ?f64, // Cached prompt read rate, USD per 1M tokens.
cache_write_per_m: ?f64, // Cache write premium rate, USD per 1M tokens.
};
/// Request shape (TS `CacheInput`).
pub const CacheInput = struct {
prompt_tokens: f64, // Prompt (input) tokens per request.
output_tokens: f64, // Completion (output) tokens per request.
hits: f64, // Requests reusing the cached prompt; < 1 counts as 1.
};
/// Result shape. Any missing rate nulls every field.
pub const CacheMath = struct {
uncached: ?f64, // `hits × (prompt·in$/M + output·out$/M) / 1e6`.
cached: ?f64, // `(prompt·write$/M + hits × (prompt·read$/M + output·out$/M)) / 1e6` —
// one cache write, `hits` cache reads, output billed every request.
savings: ?f64, // uncached − cached (negative when caching costs more).
savings_pct: ?f64, // savings / uncached × 100; 0 when uncached is 0.
break_even_hits: ?u64, // `ceil(write$/M / read$/M)` when read$/M > 0 — hits needed for
// cumulative READ spend to equal ONE write premium; null otherwise.
};
/// The all-null result used when any pricing rate is missing.
fn nulled() CacheMath {
return .{
.uncached = null,
.cached = null,
.savings = null,
.savings_pct = null,
.break_even_hits = null,
};
}
/// Compare uncached vs prompt-cached cost for one model. Any missing rate
/// (input, output, cache read, cache write) nulls every field — the caller
/// renders an explanatory empty state instead of partial math.
pub fn cacheMath(model: ModelRates, input: CacheInput) CacheMath {
const ipm = model.input_per_m orelse return nulled();
const opm = model.output_per_m orelse return nulled();
const cr = model.cache_read_per_m orelse return nulled();
const cw = model.cache_write_per_m orelse return nulled();
const hits = @max(1.0, input.hits);
const in_t = input.prompt_tokens;
const out_t = input.output_tokens;
// One cache write, `hits` cache reads; output tokens are billed on every request.
const uncached = hits * (in_t * ipm + out_t * opm) / 1_000_000.0;
const cached = (in_t * cw + hits * (in_t * cr + out_t * opm)) / 1_000_000.0;
const savings = uncached - cached;
const savings_pct = if (uncached == 0.0) 0.0 else savings / uncached * 100.0;
const break_even_hits: ?u64 = if (cr > 0.0)
@intFromFloat(@ceil(cw / cr))
else
null;
return .{
.uncached = uncached,
.cached = cached,
.savings = savings,
.savings_pct = savings_pct,
.break_even_hits = break_even_hits,
};
}
// ----------------------------------------------------------------------
// Self-test — the reference vectors shared with cacheSavings.test.ts (the
// lock-step contract every port mirrors). `zig run zig.zig` prints "ok".
// ----------------------------------------------------------------------
fn check(ok: bool, what: []const u8) void {
if (!ok) {
std.debug.print("cache-savings self-test failed: {s}\n", .{what});
std.process.exit(1);
}
}
fn close(a: f64, b: f64) bool {
return @abs(a - b) < 1e-9;
}
/// Fixture model F: input_per_m 10, output_per_m 50, cache_read_per_m 1,
/// cache_write_per_m 12.5. Null one rate to test the unpriced path.
fn base() ModelRates {
return .{
.input_per_m = 10.0,
.output_per_m = 50.0,
.cache_read_per_m = 1.0,
.cache_write_per_m = 12.5,
};
}
fn request(p: f64, o: f64, h: f64) CacheInput {
return .{ .prompt_tokens = p, .output_tokens = o, .hits = h };
}
pub fn main() void {
// Spec vector: 10k in / 1k out / 5 hits -> uncached 0.75, cached 0.425,
// savings 0.325, 43.333...% saved, break-even 13 hits.
var r = cacheMath(base(), request(10_000.0, 1_000.0, 5.0));
check(close(r.uncached.?, 0.75), "uncached");
check(close(r.cached.?, 0.425), "cached");
check(close(r.savings.?, 0.325), "savings");
check(close(r.savings_pct.?, 43.3333333333), "savingsPct");
check(r.break_even_hits.? == 13, "breakEven 13"); // ceil(12.5 / 1)
// At 1 hit caching LOSES 0.035 — an honest negative saving.
r = cacheMath(base(), request(10_000.0, 1_000.0, 1.0));
check(close(r.uncached.?, 0.15), "one-hit uncached");
check(close(r.cached.?, 0.185), "one-hit cached");
check(close(r.savings.?, -0.035), "one-hit negative saving");
check(close(r.savings_pct.?, -23.3333333333), "one-hit negative pct");
check(r.break_even_hits.? == 13, "one-hit breakEven");
// Each missing rate in turn nulls every field.
const variants = [_]ModelRates{
.{ .input_per_m = null, .output_per_m = 50.0, .cache_read_per_m = 1.0, .cache_write_per_m = 12.5 },
.{ .input_per_m = 10.0, .output_per_m = null, .cache_read_per_m = 1.0, .cache_write_per_m = 12.5 },
.{ .input_per_m = 10.0, .output_per_m = 50.0, .cache_read_per_m = null, .cache_write_per_m = 12.5 },
.{ .input_per_m = 10.0, .output_per_m = 50.0, .cache_read_per_m = 1.0, .cache_write_per_m = null },
};
for (variants) |v| {
const x = cacheMath(v, request(10_000.0, 1_000.0, 5.0));
check(x.uncached == null and x.cached == null and x.savings == null and
x.savings_pct == null and x.break_even_hits == null,
"missing rate nulls everything");
}
// hits < 1 counts as 1.
const r0 = cacheMath(base(), request(10_000.0, 1_000.0, 0.0));
const r1 = cacheMath(base(), request(10_000.0, 1_000.0, 1.0));
check(std.meta.eql(r0, r1), "hits < 1 clamps to 1");
// Zero tokens -> zero costs with 0%, no division error.
r = cacheMath(base(), request(0.0, 0.0, 5.0));
check(r.uncached.? == 0.0 and r.cached.? == 0.0 and
r.savings.? == 0.0 and r.savings_pct.? == 0.0,
"zero tokens");
check(r.break_even_hits.? == 13, "zero-tokens breakEven");
// cache_read 0 -> break-even null but costs kept (10k×$12.5 + 5×(0 + 1k×$50)).
r = cacheMath(.{
.input_per_m = 10.0,
.output_per_m = 50.0,
.cache_read_per_m = 0.0,
.cache_write_per_m = 12.5,
}, request(10_000.0, 1_000.0, 5.0));
check(r.break_even_hits == null, "zero read rate nulls breakEven"); // premium never repaid
check(close(r.uncached.?, 0.75), "zero-read uncached");
check(close(r.cached.?, 0.375), "zero-read cached");
// ceil(4/2) stays 2 — no rounding up at the exact integer boundary.
r = cacheMath(.{
.input_per_m = 10.0,
.output_per_m = 50.0,
.cache_read_per_m = 2.0,
.cache_write_per_m = 4.0,
}, request(1_000.0, 0.0, 3.0));
check(r.break_even_hits.? == 2, "integer boundary");
std.debug.print("ok\n", .{});
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →