LLM Cost Calculator — Zig source
Estimate LLM costs per request or per month at billion-token scale — with realistic prompt-cache hit rates, four-lane pricing, and side-by-side model comparison from a dated pricing snapshot.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
// llm-cost-calculator — Zig port: per-million-token cost math for LLM workloads.
const std = @import("std");
/// Multiplier applied to Batch API pricing (the standard 50% discount).
pub const batch_discount: f64 = 0.5;
/// Price pair for a model or a custom rate card. null = unpriced.
pub const CostRates = struct {
input_per_m: ?f64, // USD per 1M input tokens.
output_per_m: ?f64, // USD per 1M output tokens.
};
/// One workload to price.
pub const CostInput = struct {
input_tokens: f64, // Input tokens per request.
output_tokens: f64, // Output tokens per request.
requests: f64 = 1,
batch: bool = false,
};
/// Pricing projection of a model — the only three fields cost math needs.
pub const Model = struct {
id: []const u8,
input_per_m: ?f64,
output_per_m: ?f64,
};
/// One row of a compareModels result.
pub const ModelCost = struct {
id: []const u8,
cost: ?f64, // costFor with the model's rates.
tokens_per_dollar: ?f64, // 1e6 / output_per_m.
};
/// Cost in USD, or null when either rate is unpriced:
/// ((in/1e6)·input + (out/1e6)·output) × requests × batch discount.
pub fn costFor(rates: CostRates, work: CostInput) ?f64 {
const input = rates.input_per_m orelse return null;
const output = rates.output_per_m orelse return null;
const base = work.input_tokens / 1_000_000 * input + work.output_tokens / 1_000_000 * output;
return base * work.requests * (if (work.batch) batch_discount else 1.0);
}
/// Output tokens per USD: 1e6 / output_per_m, or null when unpriced.
pub fn tokensPerDollar(m: Model) ?f64 {
const output = m.output_per_m orelse return null;
return 1_000_000 / output;
}
/// Cost ordering: cost asc, unpriced last, ties by id (ASCII slug ids).
fn lessThan(_: void, a: ModelCost, b: ModelCost) bool {
if (a.cost == null or b.cost == null) {
if (a.cost == null and b.cost == null) return std.mem.lessThan(u8, a.id, b.id);
return b.cost == null; // priced rows sort before unpriced ones
}
if (a.cost.? == b.cost.?) return std.mem.lessThan(u8, a.id, b.id);
return a.cost.? < b.cost.?;
}
/// Cost every model for one workload, sorted. `out` must fit `models`.
pub fn compareModels(models: []const Model, work: CostInput, out: []ModelCost) []ModelCost {
var n: usize = 0;
for (models) |m| {
const rates = CostRates{ .input_per_m = m.input_per_m, .output_per_m = m.output_per_m };
out[n] = .{ .id = m.id, .cost = costFor(rates, work), .tokens_per_dollar = tokensPerDollar(m) };
n += 1;
}
std.mem.sort(ModelCost, out[0..n], {}, lessThan);
return out[0..n];
}
pub fn main() !void {
// Rate cards flow from the model snapshot in the TS lib; one is unpriced.
const models = [_]Model{
.{ .id = "sonnet-class", .input_per_m = 3.0, .output_per_m = 15.0 },
.{ .id = "budget-class", .input_per_m = 0.5, .output_per_m = 2.0 },
.{ .id = "preview", .input_per_m = null, .output_per_m = null },
};
const work = CostInput{ .input_tokens = 12_000, .output_tokens = 3_000, .requests = 250, .batch = true };
const stdout = std.io.getStdOut().writer();
var rows: [models.len]ModelCost = undefined;
for (compareModels(&models, work, &rows)) |row| {
if (row.cost) |cost|
try stdout.print("{s:<14}${d:.4} ({d:.0} output tokens/USD)\n", .{ row.id, cost, row.tokens_per_dollar.? })
else
try stdout.print("{s:<14}unpriced\n", .{row.id});
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →