Eval Metrics — Zig source
The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: Zig (0.13+, stdlib only)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive). Errors use error unions —
// Zig's stand-in for the TS RangeError contract; callers get the zero value
// alongside the error, mirroring TS.
const std = @import("std");
/// Precision / recall / F1 triple.
pub const Prf1 = struct {
precision: f64 = 0,
recall: f64 = 0,
f1: f64 = 0,
};
/// Confusion counts; tn is accepted but unused by P/R/F1.
pub const ConfusionCounts = struct {
tp: i64,
fp: i64,
fn_: i64,
tn: i64 = 0,
};
pub const EvalError = error{
NMustBePositive,
COutOfRange,
KOutOfRange,
CountsNegative,
LengthMismatch,
};
/// Unbiased pass@k: probability that at least one of k samples drawn without
/// replacement from n (of which c are correct) passes.
///
/// 1 when n - c < k (a wrong draw is impossible)
/// 1 - Π_{i=0..k-1} (n - c - i) / (n - i) otherwise
pub fn passAtK(n: i64, c: i64, k: i64) EvalError!f64 {
if (n <= 0) return error.NMustBePositive;
if (c < 0 or c > n) return error.COutOfRange;
if (k <= 0 or k > n) return error.KOutOfRange;
if (n - c < k) return 1.0;
var product: f64 = 1.0;
var i: i64 = 0;
while (i < k) : (i += 1) {
product *= @as(f64, @floatFromInt(n - c - i)) / @as(f64, @floatFromInt(n - i));
}
return 1.0 - product;
}
/// Precision/recall/F1 over confusion counts. Zero denominators score 0.
pub fn precisionRecall(counts: ConfusionCounts) EvalError!Prf1 {
if (counts.tp < 0 or counts.fp < 0 or counts.fn_ < 0) {
return error.CountsNegative;
}
const precision: f64 = if (counts.tp + counts.fp > 0)
@as(f64, @floatFromInt(counts.tp)) / @as(f64, @floatFromInt(counts.tp + counts.fp))
else
0.0;
const recall: f64 = if (counts.tp + counts.fn_ > 0)
@as(f64, @floatFromInt(counts.tp)) / @as(f64, @floatFromInt(counts.tp + counts.fn_))
else
0.0;
const f1: f64 = if (precision + recall > 0.0)
(2.0 * precision * recall) / (precision + recall)
else
0.0;
return .{ .precision = precision, .recall = recall, .f1 = f1 };
}
/// Micro-averaged P/R/F1 across per-class confusion counts.
pub fn microAverage(per_class: []const ConfusionCounts) EvalError!Prf1 {
var sums = ConfusionCounts{ .tp = 0, .fp = 0, .fn_ = 0 };
for (per_class) |c| {
sums.tp += c.tp;
sums.fp += c.fp;
sums.fn_ += c.fn_;
}
return precisionRecall(sums);
}
/// Exact-match rate over paired predictions/references (case-sensitive).
/// Empty input scores 0; length mismatch errors.
pub fn exactMatchRate(predictions: []const []const u8, references: []const []const u8) EvalError!f64 {
if (predictions.len != references.len) return error.LengthMismatch;
if (predictions.len == 0) return 0.0;
var hits: usize = 0;
for (predictions, references) |p, r| {
if (std.mem.eql(u8, p, r)) hits += 1;
}
return @as(f64, @floatFromInt(hits)) / @as(f64, @floatFromInt(predictions.len));
}
/// P/R/F1 over label SETS — the standard multi-label / extraction metric.
/// Linear scans stand in for hash sets: O(|p| * |r|), fine at label scale.
pub fn setMatch(prediction: []const []const u8, reference: []const []const u8) EvalError!Prf1 {
var tp: i64 = 0;
var fp: i64 = 0;
var fn_: i64 = 0;
for (reference) |label| {
var in_p = false;
for (prediction) |cand| {
if (std.mem.eql(u8, label, cand)) {
in_p = true;
break;
}
}
if (in_p) tp += 1 else fn_ += 1;
}
for (prediction) |label| {
var in_r = false;
for (reference) |cand| {
if (std.mem.eql(u8, label, cand)) {
in_r = true;
break;
}
}
if (!in_r) fp += 1;
}
return precisionRecall(.{ .tp = tp, .fp = fp, .fn_ = fn_ });
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →