Skip to content

Eval Metrics — Java source

The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.

This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: Java (17+, zero dependencies)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive).

import java.util.HashSet;
import java.util.List;
import java.util.Set;

public final class EvalMetrics {

    /** Precision / recall / F1 triple. */
    public record Prf1(double precision, double recall, double f1) {}

    /** Confusion counts; tn is accepted but unused by P/R/F1. */
    public record ConfusionCounts(long tp, long fp, long fn, long tn) {
        public ConfusionCounts(long tp, long fp, long fn) { this(tp, fp, fn, 0L); }
    }

    /**
     * Unbiased pass@k: probability that at least one of k samples drawn
     * without replacement from n (of which c are correct) passes.
     *
     *   1                                       when n - c < k  (a wrong draw is impossible)
     *   1 - Π_{i=0..k-1} (n - c - i) / (n - i)  otherwise
     *
     * @throws IllegalArgumentException on impossible inputs (the TS RangeError contract).
     */
    public static double passAtK(long n, long c, long k) {
        if (n <= 0) throw new IllegalArgumentException("n must be > 0");
        if (c < 0 || c > n) throw new IllegalArgumentException("c must be in [0, n]");
        if (k <= 0 || k > n) throw new IllegalArgumentException("k must be in [1, n]");
        if (n - c < k) return 1.0;
        double product = 1.0;
        for (long i = 0; i < k; i++) {
            product *= (double) (n - c - i) / (double) (n - i);
        }
        return 1.0 - product;
    }

    /** Precision/recall/F1 over confusion counts. Zero denominators score 0. */
    public static Prf1 precisionRecall(ConfusionCounts counts) {
        if (counts.tp() < 0 || counts.fp() < 0 || counts.fn() < 0)
            throw new IllegalArgumentException("counts must be >= 0");
        double precision = counts.tp() + counts.fp() > 0
                ? (double) counts.tp() / (counts.tp() + counts.fp()) : 0.0;
        double recall = counts.tp() + counts.fn() > 0
                ? (double) counts.tp() / (counts.tp() + counts.fn()) : 0.0;
        double f1 = precision + recall > 0.0
                ? (2.0 * precision * recall) / (precision + recall) : 0.0;
        return new Prf1(precision, recall, f1);
    }

    /** Micro-averaged P/R/F1 across per-class confusion counts. */
    public static Prf1 microAverage(List<ConfusionCounts> perClass) {
        long tp = 0, fp = 0, fn = 0;
        for (ConfusionCounts c : perClass) {
            tp += c.tp(); fp += c.fp(); fn += c.fn();
        }
        return precisionRecall(new ConfusionCounts(tp, fp, fn));
    }

    /**
     * Exact-match rate over paired predictions/references (case-sensitive).
     * Empty input scores 0; length mismatch throws.
     */
    public static double exactMatchRate(List<String> predictions, List<String> references) {
        if (predictions.size() != references.size())
            throw new IllegalArgumentException(
                    "predictions and references must have the same length");
        if (predictions.isEmpty()) return 0.0;
        long hits = 0;
        for (int i = 0; i < predictions.size(); i++) {
            if (predictions.get(i).equals(references.get(i))) hits++;
        }
        return (double) hits / predictions.size();
    }

    /** P/R/F1 over label SETS — the standard multi-label / extraction metric. */
    public static Prf1 setMatch(List<String> prediction, List<String> reference) {
        Set<String> p = new HashSet<>(prediction);
        Set<String> r = new HashSet<>(reference);
        long tp = 0, fp = 0, fn = 0;
        for (String label : r) if (p.contains(label)) tp++;
        for (String label : p) if (!r.contains(label)) fp++;
        for (String label : r) if (!p.contains(label)) fn++;
        return precisionRecall(new ConfusionCounts(tp, fp, fn));
    }

    private EvalMetrics() {}
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →