Skip to content

Eval Metrics — Kotlin source

The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.

This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: Kotlin (JVM 17+, zero dependencies)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive).

/** Precision / recall / F1 triple. */
data class Prf1(val precision: Double, val recall: Double, val f1: Double)

/** Confusion counts; tn is accepted but unused by P/R/F1. */
data class ConfusionCounts(val tp: Long, val fp: Long, val fn: Long, val tn: Long = 0)

/**
 * Unbiased pass@k: probability that at least one of k samples drawn without
 * replacement from n (of which c are correct) passes.
 *
 *   1                                       when n - c < k  (a wrong draw is impossible)
 *   1 - Π_{i=0..k-1} (n - c - i) / (n - i)  otherwise
 *
 * @throws IllegalArgumentException on impossible inputs (the TS RangeError contract).
 */
fun passAtK(n: Long, c: Long, k: Long): Double {
    require(n > 0) { "n must be > 0" }
    require(c in 0..n) { "c must be in [0, n]" }
    require(k in 1..n) { "k must be in [1, n]" }
    if (n - c < k) return 1.0
    var product = 1.0
    for (i in 0 until k) {
        product *= (n - c - i).toDouble() / (n - i).toDouble()
    }
    return 1.0 - product
}

/** Precision/recall/F1 over confusion counts. Zero denominators score 0. */
fun precisionRecall(counts: ConfusionCounts): Prf1 {
    require(!(counts.tp < 0 || counts.fp < 0 || counts.fn < 0)) { "counts must be >= 0" }
    val precision = if (counts.tp + counts.fp > 0) counts.tp.toDouble() / (counts.tp + counts.fp) else 0.0
    val recall = if (counts.tp + counts.fn > 0) counts.tp.toDouble() / (counts.tp + counts.fn) else 0.0
    val f1 = if (precision + recall > 0.0) (2.0 * precision * recall) / (precision + recall) else 0.0
    return Prf1(precision, recall, f1)
}

/** Micro-averaged P/R/F1 across per-class confusion counts. */
fun microAverage(perClass: List<ConfusionCounts>): Prf1 {
    var tp = 0L; var fp = 0L; var fn = 0L
    for (c in perClass) { tp += c.tp; fp += c.fp; fn += c.fn }
    return precisionRecall(ConfusionCounts(tp, fp, fn))
}

/**
 * Exact-match rate over paired predictions/references (case-sensitive).
 * Empty input scores 0; length mismatch throws.
 */
fun exactMatchRate(predictions: List<String>, references: List<String>): Double {
    require(predictions.size == references.size) {
        "predictions and references must have the same length"
    }
    if (predictions.isEmpty()) return 0.0
    val hits = predictions.indices.count { predictions[it] == references[it] }
    return hits.toDouble() / predictions.size
}

/** P/R/F1 over label SETS — the standard multi-label / extraction metric. */
fun setMatch(prediction: List<String>, reference: List<String>): Prf1 {
    val p = prediction.toSet()
    val r = reference.toSet()
    val tp = r.count { it in p }
    val fp = p.count { it !in r }
    val fn = r.count { it !in p }
    return precisionRecall(ConfusionCounts(tp.toLong(), fp.toLong(), fn.toLong()))
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →