Eval Metrics — C++ source
The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: C++ (C++20, standard library only)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive). Errors use std::invalid_
// argument, mirroring the TS RangeError contract.
#include <algorithm>
#include <cmath>
#include <stdexcept>
#include <string>
#include <string_view>
#include <unordered_set>
#include <vector>
namespace eval_metrics {
/// Precision / recall / F1 triple.
struct Prf1 {
double precision;
double recall;
double f1;
};
/// Confusion counts; tn is accepted but unused by P/R/F1.
struct ConfusionCounts {
long tp;
long fp;
long fn;
long tn = 0; // unused, kept for schema parity with the TS lib
};
/// Unbiased pass@k: probability that at least one of k samples drawn without
/// replacement from n (of which c are correct) passes.
///
/// 1 when n - c < k (a wrong draw is impossible)
/// 1 - Π_{i=0..k-1} (n - c - i) / (n - i) otherwise
inline double pass_at_k(long n, long c, long k) {
if (n <= 0) throw std::invalid_argument("n must be > 0");
if (c < 0 || c > n) throw std::invalid_argument("c must be in [0, n]");
if (k <= 0 || k > n) throw std::invalid_argument("k must be in [1, n]");
if (n - c < k) return 1.0;
double product = 1.0;
for (long i = 0; i < k; i++) {
product *= static_cast<double>(n - c - i) / static_cast<double>(n - i);
}
return 1.0 - product;
}
/// Precision/recall/F1 over confusion counts. Zero denominators score 0.
inline Prf1 precision_recall(const ConfusionCounts &counts) {
if (counts.tp < 0 || counts.fp < 0 || counts.fn < 0) {
throw std::invalid_argument("counts must be >= 0");
}
double precision = counts.tp + counts.fp > 0
? static_cast<double>(counts.tp) / static_cast<double>(counts.tp + counts.fp)
: 0.0;
double recall = counts.tp + counts.fn > 0
? static_cast<double>(counts.tp) / static_cast<double>(counts.tp + counts.fn)
: 0.0;
double f1 = precision + recall > 0.0
? (2.0 * precision * recall) / (precision + recall)
: 0.0;
return {precision, recall, f1};
}
/// Micro-averaged P/R/F1 across per-class confusion counts.
inline Prf1 micro_average(const std::vector<ConfusionCounts> &per_class) {
ConfusionCounts sums{0, 0, 0};
for (const auto &c : per_class) {
sums.tp += c.tp;
sums.fp += c.fp;
sums.fn += c.fn;
}
return precision_recall(sums);
}
/// Exact-match rate over paired predictions/references (case-sensitive).
/// Empty input scores 0; length mismatch throws.
inline double exact_match_rate(const std::vector<std::string> &predictions,
const std::vector<std::string> &references) {
if (predictions.size() != references.size()) {
throw std::invalid_argument("predictions and references must have the same length");
}
if (predictions.empty()) return 0.0;
size_t hits = 0;
for (size_t i = 0; i < predictions.size(); i++) {
if (predictions[i] == references[i]) hits++;
}
return static_cast<double>(hits) / static_cast<double>(predictions.size());
}
/// P/R/F1 over label SETS — the standard multi-label / extraction metric.
inline Prf1 set_match(const std::vector<std::string_view> &prediction,
const std::vector<std::string_view> &reference) {
std::unordered_set<std::string_view> p(prediction.begin(), prediction.end());
std::unordered_set<std::string_view> r(reference.begin(), reference.end());
long tp = 0;
for (const auto &label : r) {
if (p.count(label) > 0) tp++;
}
long fp = 0, fn = 0;
for (const auto &label : p) {
if (r.count(label) == 0) fp++;
}
for (const auto &label : r) {
if (p.count(label) == 0) fn++;
}
return precision_recall(ConfusionCounts{tp, fp, fn});
}
} // namespace eval_metrics
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →