Skip to content

Eval Metrics — C++ source

The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
//
// Language: C++ (C++20, standard library only)
// Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/eval-metrics
//
// pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
// identical to the HumanEval implementation's combinatorial form. Precision,
// recall, and F1 run over confusion counts and over label sets; exact-match
// rate runs over paired strings (case-sensitive). Errors use std::invalid_
// argument, mirroring the TS RangeError contract.

#include <algorithm>
#include <cmath>
#include <stdexcept>
#include <string>
#include <string_view>
#include <unordered_set>
#include <vector>

namespace eval_metrics {

/// Precision / recall / F1 triple.
struct Prf1 {
    double precision;
    double recall;
    double f1;
};

/// Confusion counts; tn is accepted but unused by P/R/F1.
struct ConfusionCounts {
    long tp;
    long fp;
    long fn;
    long tn = 0; // unused, kept for schema parity with the TS lib
};

/// Unbiased pass@k: probability that at least one of k samples drawn without
/// replacement from n (of which c are correct) passes.
///
///   1                                       when n - c < k  (a wrong draw is impossible)
///   1 - Π_{i=0..k-1} (n - c - i) / (n - i)  otherwise
inline double pass_at_k(long n, long c, long k) {
    if (n <= 0) throw std::invalid_argument("n must be > 0");
    if (c < 0 || c > n) throw std::invalid_argument("c must be in [0, n]");
    if (k <= 0 || k > n) throw std::invalid_argument("k must be in [1, n]");
    if (n - c < k) return 1.0;
    double product = 1.0;
    for (long i = 0; i < k; i++) {
        product *= static_cast<double>(n - c - i) / static_cast<double>(n - i);
    }
    return 1.0 - product;
}

/// Precision/recall/F1 over confusion counts. Zero denominators score 0.
inline Prf1 precision_recall(const ConfusionCounts &counts) {
    if (counts.tp < 0 || counts.fp < 0 || counts.fn < 0) {
        throw std::invalid_argument("counts must be >= 0");
    }
    double precision = counts.tp + counts.fp > 0
        ? static_cast<double>(counts.tp) / static_cast<double>(counts.tp + counts.fp)
        : 0.0;
    double recall = counts.tp + counts.fn > 0
        ? static_cast<double>(counts.tp) / static_cast<double>(counts.tp + counts.fn)
        : 0.0;
    double f1 = precision + recall > 0.0
        ? (2.0 * precision * recall) / (precision + recall)
        : 0.0;
    return {precision, recall, f1};
}

/// Micro-averaged P/R/F1 across per-class confusion counts.
inline Prf1 micro_average(const std::vector<ConfusionCounts> &per_class) {
    ConfusionCounts sums{0, 0, 0};
    for (const auto &c : per_class) {
        sums.tp += c.tp;
        sums.fp += c.fp;
        sums.fn += c.fn;
    }
    return precision_recall(sums);
}

/// Exact-match rate over paired predictions/references (case-sensitive).
/// Empty input scores 0; length mismatch throws.
inline double exact_match_rate(const std::vector<std::string> &predictions,
                               const std::vector<std::string> &references) {
    if (predictions.size() != references.size()) {
        throw std::invalid_argument("predictions and references must have the same length");
    }
    if (predictions.empty()) return 0.0;
    size_t hits = 0;
    for (size_t i = 0; i < predictions.size(); i++) {
        if (predictions[i] == references[i]) hits++;
    }
    return static_cast<double>(hits) / static_cast<double>(predictions.size());
}

/// P/R/F1 over label SETS — the standard multi-label / extraction metric.
inline Prf1 set_match(const std::vector<std::string_view> &prediction,
                      const std::vector<std::string_view> &reference) {
    std::unordered_set<std::string_view> p(prediction.begin(), prediction.end());
    std::unordered_set<std::string_view> r(reference.begin(), reference.end());
    long tp = 0;
    for (const auto &label : r) {
        if (p.count(label) > 0) tp++;
    }
    long fp = 0, fn = 0;
    for (const auto &label : p) {
        if (r.count(label) == 0) fp++;
    }
    for (const auto &label : r) {
        if (p.count(label) == 0) fn++;
    }
    return precision_recall(ConfusionCounts{tp, fp, fn});
}

} // namespace eval_metrics

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →