Eval Metrics — PHP source
The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.
This is the PHP implementation — the same logic the interactive tool runs, in a shareable, citable form.
<?php
/**
* Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
*
* Language: PHP (8.1+, standard library only)
* Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
* Tool page: https://dev.cosmolabs.org/tools/eval-metrics
*
* pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
* identical to the HumanEval implementation's combinatorial form. Precision,
* recall, and F1 run over confusion counts and over label sets; exact-match
* rate runs over paired strings (case-sensitive).
*/
declare(strict_types=1);
/**
* Unbiased pass@k: probability that at least one of k samples drawn without
* replacement from n (of which c are correct) passes.
*
* 1 when n - c < k (a wrong draw is impossible)
* 1 - Π_{i=0..k-1} (n - c - i) / (n - i) otherwise
*
* @throws RangeException on impossible inputs (the TS RangeError contract).
*/
function pass_at_k(int $n, int $c, int $k): float
{
if ($n <= 0) {
throw new RangeException('n must be > 0');
}
if ($c < 0 || $c > $n) {
throw new RangeException('c must be in [0, n]');
}
if ($k <= 0 || $k > $n) {
throw new RangeException('k must be in [1, n]');
}
if ($n - $c < $k) {
return 1.0;
}
$product = 1.0;
for ($i = 0; $i < $k; $i++) {
$product *= ($n - $c - $i) / ($n - $i);
}
return 1.0 - $product;
}
/**
* Precision/recall/F1 over confusion counts. Zero denominators score 0.
* Returns an associative array {precision, recall, f1}.
*
* @param array{tp: int, fp: int, fn: int, tn?: int} $counts
* @return array{precision: float, recall: float, f1: float}
* @throws RangeException on negative counts.
*/
function precision_recall(array $counts): array
{
$tp = $counts['tp'];
$fp = $counts['fp'];
$fn = $counts['fn'];
if ($tp < 0 || $fp < 0 || $fn < 0) {
throw new RangeException('counts must be >= 0');
}
$precision = ($tp + $fp > 0) ? $tp / ($tp + $fp) : 0.0;
$recall = ($tp + $fn > 0) ? $tp / ($tp + $fn) : 0.0;
$f1 = ($precision + $recall > 0.0)
? (2.0 * $precision * $recall) / ($precision + $recall)
: 0.0;
return ['precision' => $precision, 'recall' => $recall, 'f1' => $f1];
}
/**
* Micro-averaged P/R/F1 across per-class confusion counts.
*
* @param array<array{tp: int, fp: int, fn: int, tn?: int}> $perClass
* @return array{precision: float, recall: float, f1: float}
*/
function micro_average(array $perClass): array
{
$sums = ['tp' => 0, 'fp' => 0, 'fn' => 0];
foreach ($perClass as $c) {
$sums['tp'] += $c['tp'];
$sums['fp'] += $c['fp'];
$sums['fn'] += $c['fn'];
}
return precision_recall($sums);
}
/**
* Exact-match rate over paired predictions/references (case-sensitive).
* Empty input scores 0; length mismatch throws.
*
* @param list<string> $predictions
* @param list<string> $references
* @throws RangeException when the lengths differ.
*/
function exact_match_rate(array $predictions, array $references): float
{
if (count($predictions) !== count($references)) {
throw new RangeException('predictions and references must have the same length');
}
if ($predictions === []) {
return 0.0;
}
$hits = 0;
foreach ($predictions as $i => $p) {
if ($p === $references[$i]) {
$hits++;
}
}
return $hits / count($predictions);
}
/**
* P/R/F1 over label SETS — the standard multi-label / extraction metric.
*
* @param list<string> $prediction
* @param list<string> $reference
* @return array{precision: float, recall: float, f1: float}
*/
function set_match(array $prediction, array $reference): array
{
$p = array_unique($prediction);
$r = array_unique($reference);
$tp = 0;
foreach ($r as $label) {
if (in_array($label, $p, true)) {
$tp++;
}
}
$fp = count(array_diff($p, $r));
$fn = count(array_diff($r, $p));
return precision_recall(['tp' => $tp, 'fp' => $fp, 'fn' => $fn]);
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →