Eval Metrics — C# source
The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.
This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: C# (.NET 8, zero dependencies)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive).
using System;
using System.Collections.Generic;
using System.Linq;
namespace CosmoDev.EvalMetrics;
/// <summary>Precision / recall / F1 triple.</summary>
public readonly record struct Prf1(double Precision, double Recall, double F1);
/// <summary>Confusion counts; Tn is accepted but unused by P/R/F1.</summary>
public readonly record struct ConfusionCounts(long Tp, long Fp, long Fn, long Tn = 0);
public static class EvalMetrics
{
/// <summary>
/// Unbiased pass@k: probability that at least one of k samples drawn
/// without replacement from n (of which c are correct) passes.
///
/// 1 when n - c < k (a wrong draw is impossible)
/// 1 - Π_{i=0..k-1} (n - c - i) / (n - i) otherwise
///
/// Throws <see cref="ArgumentOutOfRangeException"/> on impossible inputs.
/// </summary>
public static double PassAtK(long n, long c, long k)
{
if (n <= 0) throw new ArgumentOutOfRangeException(nameof(n), "n must be > 0");
if (c < 0 || c > n) throw new ArgumentOutOfRangeException(nameof(c), "c must be in [0, n]");
if (k <= 0 || k > n) throw new ArgumentOutOfRangeException(nameof(k), "k must be in [1, n]");
if (n - c < k) return 1.0;
double product = 1.0;
for (long i = 0; i < k; i++)
{
product *= (double)(n - c - i) / (n - i);
}
return 1.0 - product;
}
/// <summary>Precision/recall/F1 over confusion counts. Zero denominators score 0.</summary>
public static Prf1 PrecisionRecall(ConfusionCounts counts)
{
if (counts.Tp < 0 || counts.Fp < 0 || counts.Fn < 0)
throw new ArgumentException("counts must be >= 0");
double precision = counts.Tp + counts.Fp > 0
? (double)counts.Tp / (counts.Tp + counts.Fp) : 0.0;
double recall = counts.Tp + counts.Fn > 0
? (double)counts.Tp / (counts.Tp + counts.Fn) : 0.0;
double f1 = precision + recall > 0.0
? (2.0 * precision * recall) / (precision + recall) : 0.0;
return new Prf1(precision, recall, f1);
}
/// <summary>Micro-averaged P/R/F1 across per-class confusion counts.</summary>
public static Prf1 MicroAverage(IEnumerable<ConfusionCounts> perClass)
{
long tp = 0, fp = 0, fn = 0;
foreach (var c in perClass)
{
tp += c.Tp; fp += c.Fp; fn += c.Fn;
}
return PrecisionRecall(new ConfusionCounts(tp, fp, fn));
}
/// <summary>
/// Exact-match rate over paired predictions/references (case-sensitive).
/// Empty input scores 0; length mismatch throws.
/// </summary>
public static double ExactMatchRate(IReadOnlyList<string> predictions, IReadOnlyList<string> references)
{
if (predictions.Count != references.Count)
throw new ArgumentException("predictions and references must have the same length");
if (predictions.Count == 0) return 0.0;
long hits = 0;
for (int i = 0; i < predictions.Count; i++)
{
if (predictions[i] == references[i]) hits++;
}
return (double)hits / predictions.Count;
}
/// <summary>P/R/F1 over label SETS — the standard multi-label / extraction metric.</summary>
public static Prf1 SetMatch(IEnumerable<string> prediction, IEnumerable<string> reference)
{
var p = prediction.ToHashSet();
var r = reference.ToHashSet();
long tp = r.Count(label => p.Contains(label));
long fp = p.Count(label => !r.Contains(label));
long fn = r.Count(label => !p.Contains(label));
return PrecisionRecall(new ConfusionCounts(tp, fp, fn));
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →