Skip to content

Eval Metrics — Swift source

The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.

This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: Swift (5.9+, zero dependencies)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive).

/// Precision / recall / F1 triple.
struct Prf1: Equatable {
    var precision: Double
    var recall: Double
    var f1: Double
}

/// Confusion counts; tn is accepted but unused by P/R/F1.
struct ConfusionCounts {
    var tp: Int
    var fp: Int
    var fn: Int
    var tn: Int = 0
}

/// Errors mirroring the TS suite's RangeError contract.
enum EvalError: Error, CustomStringConvertible {
    case nMustBePositive
    case cOutOfRange
    case kOutOfRange
    case countsNegative
    case lengthMismatch

    var description: String {
        switch self {
        case .nMustBePositive: return "n must be > 0"
        case .cOutOfRange: return "c must be in [0, n]"
        case .kOutOfRange: return "k must be in [1, n]"
        case .countsNegative: return "counts must be >= 0"
        case .lengthMismatch: return "predictions and references must have the same length"
        }
    }
}

/// Unbiased pass@k: probability that at least one of k samples drawn without
/// replacement from n (of which c are correct) passes.
///
///   1                                       when n - c < k  (a wrong draw is impossible)
///   1 - Π_{i=0..k-1} (n - c - i) / (n - i)  otherwise
func passAtK(_ n: Int, _ c: Int, _ k: Int) throws -> Double {
    if n <= 0 { throw EvalError.nMustBePositive }
    if c < 0 || c > n { throw EvalError.cOutOfRange }
    if k <= 0 || k > n { throw EvalError.kOutOfRange }
    if n - c < k { return 1.0 }
    var product = 1.0
    for i in 0..<k {
        product *= Double(n - c - i) / Double(n - i)
    }
    return 1.0 - product
}

/// Precision/recall/F1 over confusion counts. Zero denominators score 0.
func precisionRecall(_ counts: ConfusionCounts) throws -> Prf1 {
    if counts.tp < 0 || counts.fp < 0 || counts.fn < 0 {
        throw EvalError.countsNegative
    }
    let precision = counts.tp + counts.fp > 0
        ? Double(counts.tp) / Double(counts.tp + counts.fp) : 0.0
    let recall = counts.tp + counts.fn > 0
        ? Double(counts.tp) / Double(counts.tp + counts.fn) : 0.0
    let f1 = precision + recall > 0.0
        ? (2.0 * precision * recall) / (precision + recall) : 0.0
    return Prf1(precision: precision, recall: recall, f1: f1)
}

/// Micro-averaged P/R/F1 across per-class confusion counts.
func microAverage(_ perClass: [ConfusionCounts]) throws -> Prf1 {
    var sums = ConfusionCounts(tp: 0, fp: 0, fn: 0)
    for c in perClass {
        sums.tp += c.tp
        sums.fp += c.fp
        sums.fn += c.fn
    }
    return try precisionRecall(sums)
}

/// Exact-match rate over paired predictions/references (case-sensitive).
/// Empty input scores 0; length mismatch throws.
func exactMatchRate(_ predictions: [String], _ references: [String]) throws -> Double {
    if predictions.count != references.count {
        throw EvalError.lengthMismatch
    }
    if predictions.isEmpty { return 0.0 }
    var hits = 0
    for (p, r) in zip(predictions, references) where p == r {
        hits += 1
    }
    return Double(hits) / Double(predictions.count)
}

/// P/R/F1 over label SETS — the standard multi-label / extraction metric.
func setMatch(_ prediction: [String], _ reference: [String]) throws -> Prf1 {
    let p = Set(prediction)
    let r = Set(reference)
    let tp = r.filter { p.contains($0) }.count
    let fp = p.filter { !r.contains($0) }.count
    let fn = r.filter { !p.contains($0) }.count
    return try precisionRecall(ConfusionCounts(tp: tp, fp: fp, fn: fn))
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →