Skip to content

Cache Savings Calculator — Swift source

See what prompt caching saves — uncached vs cached cost over N requests, with the write-premium break-even point.

This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.

// cache_savings — uncached vs prompt-cached LLM cost comparison.
//
// Language: Swift (5.9, standard library only)
// Source:   CosmoDev polyglot showcase port of the Cache Savings Calculator
//           tool, ported from src/lib/cacheSavings.ts (the canonical
//           TypeScript implementation).
// Tool:     https://dev.cosmolabs.org/tools/cache-savings-calculator
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; never traps (plain Double math, no force unwraps,
//     no unchecked division — the only division guards uncached == 0).
//   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//   - Self-contained: the standard library only (Double? mirrors the TS `null`;
//     .rounded(.up) replaces Math.ceil without importing Foundation).
//
// The TS original takes a full AiModel record but reads only its four pricing
// rates, so this port narrows the parameter to exactly those fields. Any nil
// rate makes every output nil — the caller renders an explanatory empty state
// instead of partial math. All rates are per-1M-token USD, mirroring the cost
// conventions of llmCost.ts.
//
// Numeric mapping: TS `number` is a double, so tokens and hits stay `Double`
// (fractional hits clamp up to 1.0 exactly like `Math.max(1, hits)`);
// breakEvenHits is a whole hit count, so it lands in `Int?`.

/// The four per-1M-token USD pricing rates `cacheMath(model:input:)` reads
/// from the TS AiModel record.
public struct ModelRates: Equatable {
    /// Uncached prompt (input) rate, USD per 1M tokens.
    public var inputPerM: Double?
    /// Completion (output) rate, USD per 1M tokens.
    public var outputPerM: Double?
    /// Cached prompt read rate, USD per 1M tokens.
    public var cacheReadPerM: Double?
    /// Cache write premium rate, USD per 1M tokens.
    public var cacheWritePerM: Double?

    public init(inputPerM: Double?, outputPerM: Double?, cacheReadPerM: Double?, cacheWritePerM: Double?) {
        self.inputPerM = inputPerM
        self.outputPerM = outputPerM
        self.cacheReadPerM = cacheReadPerM
        self.cacheWritePerM = cacheWritePerM
    }
}

/// Request shape (TS `CacheInput`).
public struct CacheInput: Equatable {
    /// Prompt (input) tokens per request.
    public var promptTokens: Double
    /// Completion (output) tokens per request.
    public var outputTokens: Double
    /// Requests reusing the cached prompt; values < 1 count as 1.
    public var hits: Double

    public init(promptTokens: Double, outputTokens: Double, hits: Double) {
        self.promptTokens = promptTokens
        self.outputTokens = outputTokens
        self.hits = hits
    }
}

/// Result shape. Any missing rate nils every field.
public struct CacheMath: Equatable {
    /// `hits × (prompt·in$/M + output·out$/M) / 1e6`.
    public var uncached: Double?
    /// `(prompt·write$/M + hits × (prompt·read$/M + output·out$/M)) / 1e6` —
    /// one cache write, `hits` cache reads, output billed every request.
    public var cached: Double?
    /// uncached − cached (negative when caching costs more).
    public var savings: Double?
    /// savings / uncached × 100; 0 when uncached is 0.
    public var savingsPct: Double?
    /// `ceil(write$/M / read$/M)` when read$/M > 0 — hits needed for
    /// cumulative READ spend to equal ONE write premium; nil otherwise.
    public var breakEvenHits: Int?

    public init(uncached: Double?, cached: Double?, savings: Double?, savingsPct: Double?, breakEvenHits: Int?) {
        self.uncached = uncached
        self.cached = cached
        self.savings = savings
        self.savingsPct = savingsPct
        self.breakEvenHits = breakEvenHits
    }
}

/// The all-nil result used when any pricing rate is missing.
func nulled() -> CacheMath {
    CacheMath(uncached: nil, cached: nil, savings: nil, savingsPct: nil, breakEvenHits: nil)
}

/// Compare uncached vs prompt-cached cost for one model. Any missing rate
/// (input, output, cache read, cache write) nils every field — the caller
/// renders an explanatory empty state instead of partial math.
public func cacheMath(model: ModelRates, input: CacheInput) -> CacheMath {
    guard let ipm = model.inputPerM,
          let opm = model.outputPerM,
          let cr = model.cacheReadPerM,
          let cw = model.cacheWritePerM else {
        return nulled()
    }

    let hits = max(1.0, input.hits)
    let inT = input.promptTokens
    let outT = input.outputTokens

    // One cache write, `hits` cache reads; output tokens are billed on every request.
    let uncached = hits * (inT * ipm + outT * opm) / 1_000_000
    let cached = (inT * cw + hits * (inT * cr + outT * opm)) / 1_000_000
    let savings = uncached - cached
    let savingsPct = uncached == 0 ? 0 : savings / uncached * 100
    let breakEvenHits: Int? = cr > 0 ? Int((cw / cr).rounded(.up)) : nil

    return CacheMath(uncached: uncached, cached: cached, savings: savings,
                     savingsPct: savingsPct, breakEvenHits: breakEvenHits)
}

// ----------------------------------------------------------------------
// Self-test — the reference vectors shared with cacheSavings.test.ts (the
// lock-step contract every port mirrors). Like the sibling ports, this file
// is a main-style script: `swift swift.swift` compiles and prints "ok".
// ----------------------------------------------------------------------

func check(_ ok: Bool, _ what: String) {
    if !ok { fatalError("cache-savings self-test failed: \(what)") }
}

func close(_ a: Double, _ b: Double) -> Bool { abs(a - b) < 1e-9 }

// Fixture model F: inputPerM 10, outputPerM 50, cacheReadPerM 1,
// cacheWritePerM 12.5. Nil one rate to test the unpriced path.
let base = ModelRates(inputPerM: 10, outputPerM: 50, cacheReadPerM: 1, cacheWritePerM: 12.5)
func request(_ p: Double, _ o: Double, _ h: Double) -> CacheInput {
    CacheInput(promptTokens: p, outputTokens: o, hits: h)
}

// Spec vector: 10k in / 1k out / 5 hits -> uncached 0.75, cached 0.425,
// savings 0.325, 43.333...% saved, break-even 13 hits.
var r = cacheMath(model: base, input: request(10_000, 1_000, 5))
check(close(r.uncached!, 0.75), "uncached \(r.uncached!)")
check(close(r.cached!, 0.425), "cached \(r.cached!)")
check(close(r.savings!, 0.325), "savings \(r.savings!)")
check(close(r.savingsPct!, 43.3333333333), "savingsPct \(r.savingsPct!)")
check(r.breakEvenHits == 13, "breakEven \(String(describing: r.breakEvenHits))") // ceil(12.5 / 1)

// At 1 hit caching LOSES 0.035 — an honest negative saving.
r = cacheMath(model: base, input: request(10_000, 1_000, 1))
check(close(r.uncached!, 0.15), "one-hit uncached")
check(close(r.cached!, 0.185), "one-hit cached")
check(close(r.savings!, -0.035), "one-hit negative saving")
check(close(r.savingsPct!, -23.3333333333), "one-hit negative pct")
check(r.breakEvenHits == 13, "one-hit breakEven")

// Each missing rate in turn nils every field.
let variants = [
    ModelRates(inputPerM: nil, outputPerM: 50, cacheReadPerM: 1, cacheWritePerM: 12.5),
    ModelRates(inputPerM: 10, outputPerM: nil, cacheReadPerM: 1, cacheWritePerM: 12.5),
    ModelRates(inputPerM: 10, outputPerM: 50, cacheReadPerM: nil, cacheWritePerM: 12.5),
    ModelRates(inputPerM: 10, outputPerM: 50, cacheReadPerM: 1, cacheWritePerM: nil),
]
for v in variants {
    let x = cacheMath(model: v, input: request(10_000, 1_000, 5))
    check(x == nulled(), "missing rate nils everything")
}

// hits < 1 counts as 1.
check(cacheMath(model: base, input: request(10_000, 1_000, 0))
    == cacheMath(model: base, input: request(10_000, 1_000, 1)),
    "hits < 1 clamps to 1")

// Zero tokens -> zero costs with 0%, no division error.
r = cacheMath(model: base, input: request(0, 0, 5))
check(r.uncached == 0 && r.cached == 0 && r.savings == 0 && r.savingsPct == 0, "zero tokens")
check(r.breakEvenHits == 13, "zero-tokens breakEven")

// cacheRead 0 -> break-even nil but costs kept (10k×$12.5 + 5×(0 + 1k×$50)).
r = cacheMath(model: ModelRates(inputPerM: 10, outputPerM: 50, cacheReadPerM: 0, cacheWritePerM: 12.5),
              input: request(10_000, 1_000, 5))
check(r.breakEvenHits == nil, "zero read rate nils breakEven") // premium never repaid
check(close(r.uncached!, 0.75), "zero-read uncached")
check(close(r.cached!, 0.375), "zero-read cached")

// ceil(4/2) stays 2 — no rounding up at the exact integer boundary.
r = cacheMath(model: ModelRates(inputPerM: 10, outputPerM: 50, cacheReadPerM: 2, cacheWritePerM: 4),
              input: request(1_000, 0, 3))
check(r.breakEvenHits == 2, "integer boundary")

print("ok")

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →