Cache Savings Calculator — C++ source
See what prompt caching saves — uncached vs cached cost over N requests, with the write-premium break-even point.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// cache_savings — uncached vs prompt-cached LLM cost comparison.
//
// Language: C++ (C++17, standard library only)
// Source: CosmoDev polyglot showcase port of the Cache Savings Calculator
// tool, ported from src/lib/cacheSavings.ts (the canonical
// TypeScript implementation).
// Tool: https://dev.cosmolabs.org/tools/cache-savings-calculator
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; never throws (plain double math, no allocation,
// no unchecked division — the only division guards uncached == 0).
// - Functionally equivalent to the TS reference: same inputs -> same outputs.
// - Self-contained: std only (std::optional mirrors the TS `null`).
//
// The TS original takes a full AiModel record but reads only its four pricing
// rates, so this port narrows the parameter to exactly those fields. Any
// nullopt rate makes every output nullopt — the caller renders an explanatory
// empty state instead of partial math. All rates are per-1M-token USD,
// mirroring the cost conventions of llmCost.ts.
//
// Numeric mapping: TS `number` is a double, so tokens and hits stay `double`
// (fractional hits clamp up to 1.0 exactly like `Math.max(1, hits)`);
// breakEvenHits is a whole hit count, so it lands in std::int64_t.
#include <algorithm>
#include <cassert>
#include <cmath>
#include <cstdint>
#include <cstdio>
#include <optional>
// The four per-1M-token USD pricing rates cache_math reads from the TS AiModel
// record.
struct ModelRates {
std::optional<double> input_per_m; // Uncached prompt (input) rate.
std::optional<double> output_per_m; // Completion (output) rate.
std::optional<double> cache_read_per_m; // Cached prompt read rate.
std::optional<double> cache_write_per_m; // Cache write premium rate.
};
// Request shape (TS CacheInput).
struct CacheInput {
double prompt_tokens; // Prompt (input) tokens per request.
double output_tokens; // Completion (output) tokens per request.
double hits; // Requests reusing the cached prompt; < 1 counts as 1.
};
// Result shape. Any missing rate nulls every field.
struct CacheMath {
std::optional<double> uncached; // hits × (prompt·in$/M + output·out$/M) / 1e6.
std::optional<double> cached; // (prompt·write$/M + hits × (prompt·read$/M + output·out$/M)) / 1e6 —
// one cache write, `hits` cache reads, output billed every request.
std::optional<double> savings; // uncached − cached (negative when caching costs more).
std::optional<double> savings_pct; // savings / uncached × 100; 0 when uncached is 0.
std::optional<std::int64_t> break_even_hits; // ceil(write$/M / read$/M) when read$/M > 0 — hits
// needed for cumulative READ spend to equal ONE write
// premium; nullopt otherwise.
// The all-nullopt result used when any pricing rate is missing.
static CacheMath nulled() { return CacheMath{}; }
};
// Compare uncached vs prompt-cached cost for one model. Any missing rate
// (input, output, cache read, cache write) nulls every field — the caller
// renders an explanatory empty state instead of partial math.
CacheMath cache_math(const ModelRates &model, const CacheInput &input)
{
const auto ipm = model.input_per_m;
const auto opm = model.output_per_m;
const auto cr = model.cache_read_per_m;
const auto cw = model.cache_write_per_m;
if (!ipm || !opm || !cr || !cw) {
return CacheMath::nulled();
}
const double hits = std::max(1.0, input.hits);
const double in_t = input.prompt_tokens;
const double out_t = input.output_tokens;
// One cache write, `hits` cache reads; output tokens are billed on every request.
const double uncached = hits * (in_t * *ipm + out_t * *opm) / 1e6;
const double cached = (in_t * *cw + hits * (in_t * *cr + out_t * *opm)) / 1e6;
const double savings = uncached - cached;
const double savings_pct = uncached == 0.0 ? 0.0 : savings / uncached * 100.0;
const std::optional<std::int64_t> break_even =
*cr > 0.0
? std::optional<std::int64_t>(static_cast<std::int64_t>(std::ceil(*cw / *cr)))
: std::nullopt;
return CacheMath{uncached, cached, savings, savings_pct, break_even};
}
// ----------------------------------------------------------------------
// Self-test — the reference vectors shared with cacheSavings.test.ts (the
// lock-step contract every port mirrors).
// ----------------------------------------------------------------------
namespace {
// Fixture model F: inputPerM 10, outputPerM 50, cacheReadPerM 1,
// cacheWritePerM 12.5. Null one rate to test the unpriced path.
ModelRates base_rates()
{
return ModelRates{10.0, 50.0, 1.0, 12.5};
}
CacheInput mk_input(double p, double o, double h)
{
return CacheInput{p, o, h};
}
bool close(double a, double b)
{
return std::fabs(a - b) < 1e-9;
}
} // namespace
int main()
{
// Spec vector: 10k in / 1k out / 5 hits -> uncached 0.75, cached 0.425,
// savings 0.325, 43.333...% saved, break-even 13 hits.
CacheMath r = cache_math(base_rates(), mk_input(10'000.0, 1'000.0, 5.0));
assert(r.uncached && close(*r.uncached, 0.75));
assert(r.cached && close(*r.cached, 0.425));
assert(r.savings && close(*r.savings, 0.325));
assert(r.savings_pct && close(*r.savings_pct, 43.3333333333));
assert(r.break_even_hits && *r.break_even_hits == 13); // ceil(12.5 / 1)
// At 1 hit caching LOSES 0.035 — an honest negative saving.
r = cache_math(base_rates(), mk_input(10'000.0, 1'000.0, 1.0));
assert(close(*r.uncached, 0.15));
assert(close(*r.cached, 0.185));
assert(close(*r.savings, -0.035));
assert(close(*r.savings_pct, -23.3333333333));
assert(*r.break_even_hits == 13);
// Each missing rate in turn nulls every field.
const ModelRates variants[] = {
ModelRates{std::nullopt, 50.0, 1.0, 12.5},
ModelRates{10.0, std::nullopt, 1.0, 12.5},
ModelRates{10.0, 50.0, std::nullopt, 12.5},
ModelRates{10.0, 50.0, 1.0, std::nullopt},
};
for (const ModelRates &m : variants) {
const CacheMath x = cache_math(m, mk_input(10'000.0, 1'000.0, 5.0));
assert(!x.uncached && !x.cached && !x.savings && !x.savings_pct && !x.break_even_hits);
}
// hits < 1 counts as 1.
const CacheMath r0 = cache_math(base_rates(), mk_input(10'000.0, 1'000.0, 0.0));
const CacheMath r1 = cache_math(base_rates(), mk_input(10'000.0, 1'000.0, 1.0));
assert(r0.uncached == r1.uncached && r0.cached == r1.cached &&
r0.savings == r1.savings && r0.savings_pct == r1.savings_pct &&
r0.break_even_hits == r1.break_even_hits);
// Zero tokens -> zero costs with 0%, no division error.
r = cache_math(base_rates(), mk_input(0.0, 0.0, 5.0));
assert(*r.uncached == 0.0 && *r.cached == 0.0 && *r.savings == 0.0 && *r.savings_pct == 0.0);
assert(*r.break_even_hits == 13);
// cache_read 0 -> break-even absent but costs kept (10k×$12.5 + 5×(0 + 1k×$50)).
r = cache_math(ModelRates{10.0, 50.0, 0.0, 12.5}, mk_input(10'000.0, 1'000.0, 5.0));
assert(!r.break_even_hits); // write premium never repaid
assert(close(*r.uncached, 0.75));
assert(close(*r.cached, 0.375));
// ceil(4/2) stays 2 — no rounding up at the exact integer boundary.
r = cache_math(ModelRates{10.0, 50.0, 2.0, 4.0}, mk_input(1'000.0, 0.0, 3.0));
assert(r.break_even_hits && *r.break_even_hits == 2);
std::printf("ok\n");
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →