Eval Metrics — Java source
The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.
This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: Java (17+, zero dependencies)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive).
import java.util.HashSet;
import java.util.List;
import java.util.Set;
public final class EvalMetrics {
/** Precision / recall / F1 triple. */
public record Prf1(double precision, double recall, double f1) {}
/** Confusion counts; tn is accepted but unused by P/R/F1. */
public record ConfusionCounts(long tp, long fp, long fn, long tn) {
public ConfusionCounts(long tp, long fp, long fn) { this(tp, fp, fn, 0L); }
}
/**
* Unbiased pass@k: probability that at least one of k samples drawn
* without replacement from n (of which c are correct) passes.
*
* 1 when n - c < k (a wrong draw is impossible)
* 1 - Π_{i=0..k-1} (n - c - i) / (n - i) otherwise
*
* @throws IllegalArgumentException on impossible inputs (the TS RangeError contract).
*/
public static double passAtK(long n, long c, long k) {
if (n <= 0) throw new IllegalArgumentException("n must be > 0");
if (c < 0 || c > n) throw new IllegalArgumentException("c must be in [0, n]");
if (k <= 0 || k > n) throw new IllegalArgumentException("k must be in [1, n]");
if (n - c < k) return 1.0;
double product = 1.0;
for (long i = 0; i < k; i++) {
product *= (double) (n - c - i) / (double) (n - i);
}
return 1.0 - product;
}
/** Precision/recall/F1 over confusion counts. Zero denominators score 0. */
public static Prf1 precisionRecall(ConfusionCounts counts) {
if (counts.tp() < 0 || counts.fp() < 0 || counts.fn() < 0)
throw new IllegalArgumentException("counts must be >= 0");
double precision = counts.tp() + counts.fp() > 0
? (double) counts.tp() / (counts.tp() + counts.fp()) : 0.0;
double recall = counts.tp() + counts.fn() > 0
? (double) counts.tp() / (counts.tp() + counts.fn()) : 0.0;
double f1 = precision + recall > 0.0
? (2.0 * precision * recall) / (precision + recall) : 0.0;
return new Prf1(precision, recall, f1);
}
/** Micro-averaged P/R/F1 across per-class confusion counts. */
public static Prf1 microAverage(List<ConfusionCounts> perClass) {
long tp = 0, fp = 0, fn = 0;
for (ConfusionCounts c : perClass) {
tp += c.tp(); fp += c.fp(); fn += c.fn();
}
return precisionRecall(new ConfusionCounts(tp, fp, fn));
}
/**
* Exact-match rate over paired predictions/references (case-sensitive).
* Empty input scores 0; length mismatch throws.
*/
public static double exactMatchRate(List<String> predictions, List<String> references) {
if (predictions.size() != references.size())
throw new IllegalArgumentException(
"predictions and references must have the same length");
if (predictions.isEmpty()) return 0.0;
long hits = 0;
for (int i = 0; i < predictions.size(); i++) {
if (predictions.get(i).equals(references.get(i))) hits++;
}
return (double) hits / predictions.size();
}
/** P/R/F1 over label SETS — the standard multi-label / extraction metric. */
public static Prf1 setMatch(List<String> prediction, List<String> reference) {
Set<String> p = new HashSet<>(prediction);
Set<String> r = new HashSet<>(reference);
long tp = 0, fp = 0, fn = 0;
for (String label : r) if (p.contains(label)) tp++;
for (String label : p) if (!r.contains(label)) fp++;
for (String label : r) if (!p.contains(label)) fn++;
return precisionRecall(new ConfusionCounts(tp, fp, fn));
}
private EvalMetrics() {}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →