Eval Metrics — Swift source
The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.
This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: Swift (5.9+, zero dependencies)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive).
/// Precision / recall / F1 triple.
struct Prf1: Equatable {
var precision: Double
var recall: Double
var f1: Double
}
/// Confusion counts; tn is accepted but unused by P/R/F1.
struct ConfusionCounts {
var tp: Int
var fp: Int
var fn: Int
var tn: Int = 0
}
/// Errors mirroring the TS suite's RangeError contract.
enum EvalError: Error, CustomStringConvertible {
case nMustBePositive
case cOutOfRange
case kOutOfRange
case countsNegative
case lengthMismatch
var description: String {
switch self {
case .nMustBePositive: return "n must be > 0"
case .cOutOfRange: return "c must be in [0, n]"
case .kOutOfRange: return "k must be in [1, n]"
case .countsNegative: return "counts must be >= 0"
case .lengthMismatch: return "predictions and references must have the same length"
}
}
}
/// Unbiased pass@k: probability that at least one of k samples drawn without
/// replacement from n (of which c are correct) passes.
///
/// 1 when n - c < k (a wrong draw is impossible)
/// 1 - Π_{i=0..k-1} (n - c - i) / (n - i) otherwise
func passAtK(_ n: Int, _ c: Int, _ k: Int) throws -> Double {
if n <= 0 { throw EvalError.nMustBePositive }
if c < 0 || c > n { throw EvalError.cOutOfRange }
if k <= 0 || k > n { throw EvalError.kOutOfRange }
if n - c < k { return 1.0 }
var product = 1.0
for i in 0..<k {
product *= Double(n - c - i) / Double(n - i)
}
return 1.0 - product
}
/// Precision/recall/F1 over confusion counts. Zero denominators score 0.
func precisionRecall(_ counts: ConfusionCounts) throws -> Prf1 {
if counts.tp < 0 || counts.fp < 0 || counts.fn < 0 {
throw EvalError.countsNegative
}
let precision = counts.tp + counts.fp > 0
? Double(counts.tp) / Double(counts.tp + counts.fp) : 0.0
let recall = counts.tp + counts.fn > 0
? Double(counts.tp) / Double(counts.tp + counts.fn) : 0.0
let f1 = precision + recall > 0.0
? (2.0 * precision * recall) / (precision + recall) : 0.0
return Prf1(precision: precision, recall: recall, f1: f1)
}
/// Micro-averaged P/R/F1 across per-class confusion counts.
func microAverage(_ perClass: [ConfusionCounts]) throws -> Prf1 {
var sums = ConfusionCounts(tp: 0, fp: 0, fn: 0)
for c in perClass {
sums.tp += c.tp
sums.fp += c.fp
sums.fn += c.fn
}
return try precisionRecall(sums)
}
/// Exact-match rate over paired predictions/references (case-sensitive).
/// Empty input scores 0; length mismatch throws.
func exactMatchRate(_ predictions: [String], _ references: [String]) throws -> Double {
if predictions.count != references.count {
throw EvalError.lengthMismatch
}
if predictions.isEmpty { return 0.0 }
var hits = 0
for (p, r) in zip(predictions, references) where p == r {
hits += 1
}
return Double(hits) / Double(predictions.count)
}
/// P/R/F1 over label SETS — the standard multi-label / extraction metric.
func setMatch(_ prediction: [String], _ reference: [String]) throws -> Prf1 {
let p = Set(prediction)
let r = Set(reference)
let tp = r.filter { p.contains($0) }.count
let fp = p.filter { !r.contains($0) }.count
let fn = r.filter { !p.contains($0) }.count
return try precisionRecall(ConfusionCounts(tp: tp, fp: fp, fn: fn))
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →