Skip to content

LLM Cost Calculator — Zig source

Estimate LLM costs per request or per month at billion-token scale — with realistic prompt-cache hit rates, four-lane pricing, and side-by-side model comparison from a dated pricing snapshot.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// llm-cost-calculator — Zig port: per-million-token cost math for LLM workloads.
const std = @import("std");
/// Multiplier applied to Batch API pricing (the standard 50% discount).
pub const batch_discount: f64 = 0.5;
/// Price pair for a model or a custom rate card. null = unpriced.
pub const CostRates = struct {
    input_per_m: ?f64,  // USD per 1M input tokens.
    output_per_m: ?f64, // USD per 1M output tokens.
};
/// One workload to price.
pub const CostInput = struct {
    input_tokens: f64,  // Input tokens per request.
    output_tokens: f64, // Output tokens per request.
    requests: f64 = 1,
    batch: bool = false,
};
/// Pricing projection of a model — the only three fields cost math needs.
pub const Model = struct {
    id: []const u8,
    input_per_m: ?f64,
    output_per_m: ?f64,
};
/// One row of a compareModels result.
pub const ModelCost = struct {
    id: []const u8,
    cost: ?f64,              // costFor with the model's rates.
    tokens_per_dollar: ?f64, // 1e6 / output_per_m.
};

/// Cost in USD, or null when either rate is unpriced:
/// ((in/1e6)·input + (out/1e6)·output) × requests × batch discount.
pub fn costFor(rates: CostRates, work: CostInput) ?f64 {
    const input = rates.input_per_m orelse return null;
    const output = rates.output_per_m orelse return null;
    const base = work.input_tokens / 1_000_000 * input + work.output_tokens / 1_000_000 * output;
    return base * work.requests * (if (work.batch) batch_discount else 1.0);
}
/// Output tokens per USD: 1e6 / output_per_m, or null when unpriced.
pub fn tokensPerDollar(m: Model) ?f64 {
    const output = m.output_per_m orelse return null;
    return 1_000_000 / output;
}
/// Cost ordering: cost asc, unpriced last, ties by id (ASCII slug ids).
fn lessThan(_: void, a: ModelCost, b: ModelCost) bool {
    if (a.cost == null or b.cost == null) {
        if (a.cost == null and b.cost == null) return std.mem.lessThan(u8, a.id, b.id);
        return b.cost == null; // priced rows sort before unpriced ones
    }
    if (a.cost.? == b.cost.?) return std.mem.lessThan(u8, a.id, b.id);
    return a.cost.? < b.cost.?;
}
/// Cost every model for one workload, sorted. `out` must fit `models`.
pub fn compareModels(models: []const Model, work: CostInput, out: []ModelCost) []ModelCost {
    var n: usize = 0;
    for (models) |m| {
        const rates = CostRates{ .input_per_m = m.input_per_m, .output_per_m = m.output_per_m };
        out[n] = .{ .id = m.id, .cost = costFor(rates, work), .tokens_per_dollar = tokensPerDollar(m) };
        n += 1;
    }
    std.mem.sort(ModelCost, out[0..n], {}, lessThan);
    return out[0..n];
}

pub fn main() !void {
    // Rate cards flow from the model snapshot in the TS lib; one is unpriced.
    const models = [_]Model{
        .{ .id = "sonnet-class", .input_per_m = 3.0, .output_per_m = 15.0 },
        .{ .id = "budget-class", .input_per_m = 0.5, .output_per_m = 2.0 },
        .{ .id = "preview", .input_per_m = null, .output_per_m = null },
    };
    const work = CostInput{ .input_tokens = 12_000, .output_tokens = 3_000, .requests = 250, .batch = true };
    const stdout = std.io.getStdOut().writer();
    var rows: [models.len]ModelCost = undefined;
    for (compareModels(&models, work, &rows)) |row| {
        if (row.cost) |cost|
            try stdout.print("{s:<14}${d:.4}  ({d:.0} output tokens/USD)\n", .{ row.id, cost, row.tokens_per_dollar.? })
        else
            try stdout.print("{s:<14}unpriced\n", .{row.id});
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →