Skip to content

Eval Metrics — Zig source

The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: Zig (0.13+, stdlib only)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive). Errors use error unions —
// Zig's stand-in for the TS RangeError contract; callers get the zero value
// alongside the error, mirroring TS.

const std = @import("std");

/// Precision / recall / F1 triple.
pub const Prf1 = struct {
    precision: f64 = 0,
    recall: f64 = 0,
    f1: f64 = 0,
};

/// Confusion counts; tn is accepted but unused by P/R/F1.
pub const ConfusionCounts = struct {
    tp: i64,
    fp: i64,
    fn_: i64,
    tn: i64 = 0,
};

pub const EvalError = error{
    NMustBePositive,
    COutOfRange,
    KOutOfRange,
    CountsNegative,
    LengthMismatch,
};

/// Unbiased pass@k: probability that at least one of k samples drawn without
/// replacement from n (of which c are correct) passes.
///
///   1                                       when n - c < k  (a wrong draw is impossible)
///   1 - Π_{i=0..k-1} (n - c - i) / (n - i)  otherwise
pub fn passAtK(n: i64, c: i64, k: i64) EvalError!f64 {
    if (n <= 0) return error.NMustBePositive;
    if (c < 0 or c > n) return error.COutOfRange;
    if (k <= 0 or k > n) return error.KOutOfRange;
    if (n - c < k) return 1.0;
    var product: f64 = 1.0;
    var i: i64 = 0;
    while (i < k) : (i += 1) {
        product *= @as(f64, @floatFromInt(n - c - i)) / @as(f64, @floatFromInt(n - i));
    }
    return 1.0 - product;
}

/// Precision/recall/F1 over confusion counts. Zero denominators score 0.
pub fn precisionRecall(counts: ConfusionCounts) EvalError!Prf1 {
    if (counts.tp < 0 or counts.fp < 0 or counts.fn_ < 0) {
        return error.CountsNegative;
    }
    const precision: f64 = if (counts.tp + counts.fp > 0)
        @as(f64, @floatFromInt(counts.tp)) / @as(f64, @floatFromInt(counts.tp + counts.fp))
    else
        0.0;
    const recall: f64 = if (counts.tp + counts.fn_ > 0)
        @as(f64, @floatFromInt(counts.tp)) / @as(f64, @floatFromInt(counts.tp + counts.fn_))
    else
        0.0;
    const f1: f64 = if (precision + recall > 0.0)
        (2.0 * precision * recall) / (precision + recall)
    else
        0.0;
    return .{ .precision = precision, .recall = recall, .f1 = f1 };
}

/// Micro-averaged P/R/F1 across per-class confusion counts.
pub fn microAverage(per_class: []const ConfusionCounts) EvalError!Prf1 {
    var sums = ConfusionCounts{ .tp = 0, .fp = 0, .fn_ = 0 };
    for (per_class) |c| {
        sums.tp += c.tp;
        sums.fp += c.fp;
        sums.fn_ += c.fn_;
    }
    return precisionRecall(sums);
}

/// Exact-match rate over paired predictions/references (case-sensitive).
/// Empty input scores 0; length mismatch errors.
pub fn exactMatchRate(predictions: []const []const u8, references: []const []const u8) EvalError!f64 {
    if (predictions.len != references.len) return error.LengthMismatch;
    if (predictions.len == 0) return 0.0;
    var hits: usize = 0;
    for (predictions, references) |p, r| {
        if (std.mem.eql(u8, p, r)) hits += 1;
    }
    return @as(f64, @floatFromInt(hits)) / @as(f64, @floatFromInt(predictions.len));
}

/// P/R/F1 over label SETS — the standard multi-label / extraction metric.
/// Linear scans stand in for hash sets: O(|p| * |r|), fine at label scale.
pub fn setMatch(prediction: []const []const u8, reference: []const []const u8) EvalError!Prf1 {
    var tp: i64 = 0;
    var fp: i64 = 0;
    var fn_: i64 = 0;
    for (reference) |label| {
        var in_p = false;
        for (prediction) |cand| {
            if (std.mem.eql(u8, label, cand)) {
                in_p = true;
                break;
            }
        }
        if (in_p) tp += 1 else fn_ += 1;
    }
    for (prediction) |label| {
        var in_r = false;
        for (reference) |cand| {
            if (std.mem.eql(u8, label, cand)) {
                in_r = true;
                break;
            }
        }
        if (!in_r) fp += 1;
    }
    return precisionRecall(.{ .tp = tp, .fp = fp, .fn_ = fn_ });
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →