Skip to content

Cache Savings Calculator — C++ source

See what prompt caching saves — uncached vs cached cost over N requests, with the write-premium break-even point.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// cache_savings — uncached vs prompt-cached LLM cost comparison.
//
// Language: C++ (C++17, standard library only)
// Source:   CosmoDev polyglot showcase port of the Cache Savings Calculator
//           tool, ported from src/lib/cacheSavings.ts (the canonical
//           TypeScript implementation).
// Tool:     https://dev.cosmolabs.org/tools/cache-savings-calculator
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; never throws (plain double math, no allocation,
//     no unchecked division — the only division guards uncached == 0).
//   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//   - Self-contained: std only (std::optional mirrors the TS `null`).
//
// The TS original takes a full AiModel record but reads only its four pricing
// rates, so this port narrows the parameter to exactly those fields. Any
// nullopt rate makes every output nullopt — the caller renders an explanatory
// empty state instead of partial math. All rates are per-1M-token USD,
// mirroring the cost conventions of llmCost.ts.
//
// Numeric mapping: TS `number` is a double, so tokens and hits stay `double`
// (fractional hits clamp up to 1.0 exactly like `Math.max(1, hits)`);
// breakEvenHits is a whole hit count, so it lands in std::int64_t.

#include <algorithm>
#include <cassert>
#include <cmath>
#include <cstdint>
#include <cstdio>
#include <optional>

// The four per-1M-token USD pricing rates cache_math reads from the TS AiModel
// record.
struct ModelRates {
    std::optional<double> input_per_m;       // Uncached prompt (input) rate.
    std::optional<double> output_per_m;      // Completion (output) rate.
    std::optional<double> cache_read_per_m;  // Cached prompt read rate.
    std::optional<double> cache_write_per_m; // Cache write premium rate.
};

// Request shape (TS CacheInput).
struct CacheInput {
    double prompt_tokens; // Prompt (input) tokens per request.
    double output_tokens; // Completion (output) tokens per request.
    double hits;          // Requests reusing the cached prompt; < 1 counts as 1.
};

// Result shape. Any missing rate nulls every field.
struct CacheMath {
    std::optional<double> uncached;    // hits × (prompt·in$/M + output·out$/M) / 1e6.
    std::optional<double> cached;      // (prompt·write$/M + hits × (prompt·read$/M + output·out$/M)) / 1e6 —
                                       // one cache write, `hits` cache reads, output billed every request.
    std::optional<double> savings;     // uncached − cached (negative when caching costs more).
    std::optional<double> savings_pct; // savings / uncached × 100; 0 when uncached is 0.
    std::optional<std::int64_t> break_even_hits; // ceil(write$/M / read$/M) when read$/M > 0 — hits
                                                 // needed for cumulative READ spend to equal ONE write
                                                 // premium; nullopt otherwise.

    // The all-nullopt result used when any pricing rate is missing.
    static CacheMath nulled() { return CacheMath{}; }
};

// Compare uncached vs prompt-cached cost for one model. Any missing rate
// (input, output, cache read, cache write) nulls every field — the caller
// renders an explanatory empty state instead of partial math.
CacheMath cache_math(const ModelRates &model, const CacheInput &input)
{
    const auto ipm = model.input_per_m;
    const auto opm = model.output_per_m;
    const auto cr = model.cache_read_per_m;
    const auto cw = model.cache_write_per_m;
    if (!ipm || !opm || !cr || !cw) {
        return CacheMath::nulled();
    }

    const double hits = std::max(1.0, input.hits);
    const double in_t = input.prompt_tokens;
    const double out_t = input.output_tokens;

    // One cache write, `hits` cache reads; output tokens are billed on every request.
    const double uncached = hits * (in_t * *ipm + out_t * *opm) / 1e6;
    const double cached = (in_t * *cw + hits * (in_t * *cr + out_t * *opm)) / 1e6;
    const double savings = uncached - cached;
    const double savings_pct = uncached == 0.0 ? 0.0 : savings / uncached * 100.0;
    const std::optional<std::int64_t> break_even =
        *cr > 0.0
            ? std::optional<std::int64_t>(static_cast<std::int64_t>(std::ceil(*cw / *cr)))
            : std::nullopt;

    return CacheMath{uncached, cached, savings, savings_pct, break_even};
}

// ----------------------------------------------------------------------
// Self-test — the reference vectors shared with cacheSavings.test.ts (the
// lock-step contract every port mirrors).
// ----------------------------------------------------------------------

namespace {

// Fixture model F: inputPerM 10, outputPerM 50, cacheReadPerM 1,
// cacheWritePerM 12.5. Null one rate to test the unpriced path.
ModelRates base_rates()
{
    return ModelRates{10.0, 50.0, 1.0, 12.5};
}

CacheInput mk_input(double p, double o, double h)
{
    return CacheInput{p, o, h};
}

bool close(double a, double b)
{
    return std::fabs(a - b) < 1e-9;
}

} // namespace

int main()
{
    // Spec vector: 10k in / 1k out / 5 hits -> uncached 0.75, cached 0.425,
    // savings 0.325, 43.333...% saved, break-even 13 hits.
    CacheMath r = cache_math(base_rates(), mk_input(10'000.0, 1'000.0, 5.0));
    assert(r.uncached && close(*r.uncached, 0.75));
    assert(r.cached && close(*r.cached, 0.425));
    assert(r.savings && close(*r.savings, 0.325));
    assert(r.savings_pct && close(*r.savings_pct, 43.3333333333));
    assert(r.break_even_hits && *r.break_even_hits == 13); // ceil(12.5 / 1)

    // At 1 hit caching LOSES 0.035 — an honest negative saving.
    r = cache_math(base_rates(), mk_input(10'000.0, 1'000.0, 1.0));
    assert(close(*r.uncached, 0.15));
    assert(close(*r.cached, 0.185));
    assert(close(*r.savings, -0.035));
    assert(close(*r.savings_pct, -23.3333333333));
    assert(*r.break_even_hits == 13);

    // Each missing rate in turn nulls every field.
    const ModelRates variants[] = {
        ModelRates{std::nullopt, 50.0, 1.0, 12.5},
        ModelRates{10.0, std::nullopt, 1.0, 12.5},
        ModelRates{10.0, 50.0, std::nullopt, 12.5},
        ModelRates{10.0, 50.0, 1.0, std::nullopt},
    };
    for (const ModelRates &m : variants) {
        const CacheMath x = cache_math(m, mk_input(10'000.0, 1'000.0, 5.0));
        assert(!x.uncached && !x.cached && !x.savings && !x.savings_pct && !x.break_even_hits);
    }

    // hits < 1 counts as 1.
    const CacheMath r0 = cache_math(base_rates(), mk_input(10'000.0, 1'000.0, 0.0));
    const CacheMath r1 = cache_math(base_rates(), mk_input(10'000.0, 1'000.0, 1.0));
    assert(r0.uncached == r1.uncached && r0.cached == r1.cached &&
           r0.savings == r1.savings && r0.savings_pct == r1.savings_pct &&
           r0.break_even_hits == r1.break_even_hits);

    // Zero tokens -> zero costs with 0%, no division error.
    r = cache_math(base_rates(), mk_input(0.0, 0.0, 5.0));
    assert(*r.uncached == 0.0 && *r.cached == 0.0 && *r.savings == 0.0 && *r.savings_pct == 0.0);
    assert(*r.break_even_hits == 13);

    // cache_read 0 -> break-even absent but costs kept (10k×$12.5 + 5×(0 + 1k×$50)).
    r = cache_math(ModelRates{10.0, 50.0, 0.0, 12.5}, mk_input(10'000.0, 1'000.0, 5.0));
    assert(!r.break_even_hits); // write premium never repaid
    assert(close(*r.uncached, 0.75));
    assert(close(*r.cached, 0.375));

    // ceil(4/2) stays 2 — no rounding up at the exact integer boundary.
    r = cache_math(ModelRates{10.0, 50.0, 2.0, 4.0}, mk_input(1'000.0, 0.0, 3.0));
    assert(r.break_even_hits && *r.break_even_hits == 2);

    std::printf("ok\n");
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →