Eval Metrics — TypeScript source
The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Pure logic for the Eval Metrics tool (slug: eval-metrics). The standard
// LLM-eval metrics with exact, testable formulas:
// - pass@k: the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form.
// - precision / recall / F1 over confusion counts, and over label sets.
// - exact-match rate over paired strings.
export interface PRF1 {
precision: number;
recall: number;
f1: number;
}
/**
* Unbiased pass@k: probability that at least one of k samples drawn without
* replacement from n (of which c are correct) passes.
*
* 1 when n - c < k (a wrong draw is impossible)
* 1 - Π_{i=0..k-1} (n - c - i) / (n - i) otherwise
*/
export function passAtK(n: number, c: number, k: number): number {
if (n <= 0) throw new RangeError('n must be > 0');
if (c < 0 || c > n) throw new RangeError('c must be in [0, n]');
if (k <= 0 || k > n) throw new RangeError('k must be in [1, n]');
if (n - c < k) return 1;
let product = 1;
for (let i = 0; i < k; i++) {
product *= (n - c - i) / (n - i);
}
return 1 - product;
}
export interface ConfusionCounts {
tp: number;
fp: number;
fn: number;
tn?: number;
}
/** Precision/recall/F1 over confusion counts. Zero denominators score 0. */
export function precisionRecall(counts: ConfusionCounts): PRF1 {
const { tp, fp, fn } = counts;
if ([tp, fp, fn].some((v) => v < 0)) throw new RangeError('counts must be >= 0');
const precision = tp + fp > 0 ? tp / (tp + fp) : 0;
const recall = tp + fn > 0 ? tp / (tp + fn) : 0;
const f1 = precision + recall > 0 ? (2 * precision * recall) / (precision + recall) : 0;
return { precision, recall, f1 };
}
/** Micro-averaged P/R/F1 across per-class confusion counts. */
export function microAverage(perClass: readonly ConfusionCounts[]): PRF1 {
const sums = perClass.reduce(
(acc, c) => ({ tp: acc.tp + c.tp, fp: acc.fp + c.fp, fn: acc.fn + c.fn }),
{ tp: 0, fp: 0, fn: 0 },
);
return precisionRecall(sums);
}
/** Exact-match rate over paired predictions/references (case-sensitive). */
export function exactMatchRate(predictions: readonly string[], references: readonly string[]): number {
if (predictions.length !== references.length) {
throw new RangeError('predictions and references must have the same length');
}
if (predictions.length === 0) return 0;
let hits = 0;
for (let i = 0; i < predictions.length; i++) {
if (predictions[i] === references[i]) hits++;
}
return hits / predictions.length;
}
/** P/R/F1 over label SETS — the standard multi-label / extraction metric. */
export function setMatch(prediction: readonly string[], reference: readonly string[]): PRF1 {
const p = new Set(prediction);
const r = new Set(reference);
let tp = 0;
for (const label of r) if (p.has(label)) tp++;
const fp = [...p].filter((l) => !r.has(l)).length;
const fn = [...r].filter((l) => !p.has(l)).length;
return precisionRecall({ tp, fp, fn });
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →