Skip to content

Eval Metrics — C# source

The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.

This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: C# (.NET 8, zero dependencies)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive).

using System;
using System.Collections.Generic;
using System.Linq;

namespace CosmoDev.EvalMetrics;

/// <summary>Precision / recall / F1 triple.</summary>
public readonly record struct Prf1(double Precision, double Recall, double F1);

/// <summary>Confusion counts; Tn is accepted but unused by P/R/F1.</summary>
public readonly record struct ConfusionCounts(long Tp, long Fp, long Fn, long Tn = 0);

public static class EvalMetrics
{
    /// <summary>
    /// Unbiased pass@k: probability that at least one of k samples drawn
    /// without replacement from n (of which c are correct) passes.
    ///
    ///   1                                       when n - c &lt; k  (a wrong draw is impossible)
    ///   1 - Π_{i=0..k-1} (n - c - i) / (n - i)  otherwise
    ///
    /// Throws <see cref="ArgumentOutOfRangeException"/> on impossible inputs.
    /// </summary>
    public static double PassAtK(long n, long c, long k)
    {
        if (n <= 0) throw new ArgumentOutOfRangeException(nameof(n), "n must be > 0");
        if (c < 0 || c > n) throw new ArgumentOutOfRangeException(nameof(c), "c must be in [0, n]");
        if (k <= 0 || k > n) throw new ArgumentOutOfRangeException(nameof(k), "k must be in [1, n]");
        if (n - c < k) return 1.0;
        double product = 1.0;
        for (long i = 0; i < k; i++)
        {
            product *= (double)(n - c - i) / (n - i);
        }
        return 1.0 - product;
    }

    /// <summary>Precision/recall/F1 over confusion counts. Zero denominators score 0.</summary>
    public static Prf1 PrecisionRecall(ConfusionCounts counts)
    {
        if (counts.Tp < 0 || counts.Fp < 0 || counts.Fn < 0)
            throw new ArgumentException("counts must be >= 0");
        double precision = counts.Tp + counts.Fp > 0
            ? (double)counts.Tp / (counts.Tp + counts.Fp) : 0.0;
        double recall = counts.Tp + counts.Fn > 0
            ? (double)counts.Tp / (counts.Tp + counts.Fn) : 0.0;
        double f1 = precision + recall > 0.0
            ? (2.0 * precision * recall) / (precision + recall) : 0.0;
        return new Prf1(precision, recall, f1);
    }

    /// <summary>Micro-averaged P/R/F1 across per-class confusion counts.</summary>
    public static Prf1 MicroAverage(IEnumerable<ConfusionCounts> perClass)
    {
        long tp = 0, fp = 0, fn = 0;
        foreach (var c in perClass)
        {
            tp += c.Tp; fp += c.Fp; fn += c.Fn;
        }
        return PrecisionRecall(new ConfusionCounts(tp, fp, fn));
    }

    /// <summary>
    /// Exact-match rate over paired predictions/references (case-sensitive).
    /// Empty input scores 0; length mismatch throws.
    /// </summary>
    public static double ExactMatchRate(IReadOnlyList<string> predictions, IReadOnlyList<string> references)
    {
        if (predictions.Count != references.Count)
            throw new ArgumentException("predictions and references must have the same length");
        if (predictions.Count == 0) return 0.0;
        long hits = 0;
        for (int i = 0; i < predictions.Count; i++)
        {
            if (predictions[i] == references[i]) hits++;
        }
        return (double)hits / predictions.Count;
    }

    /// <summary>P/R/F1 over label SETS — the standard multi-label / extraction metric.</summary>
    public static Prf1 SetMatch(IEnumerable<string> prediction, IEnumerable<string> reference)
    {
        var p = prediction.ToHashSet();
        var r = reference.ToHashSet();
        long tp = r.Count(label => p.Contains(label));
        long fp = p.Count(label => !r.Contains(label));
        long fn = r.Count(label => !p.Contains(label));
        return PrecisionRecall(new ConfusionCounts(tp, fp, fn));
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →