Skip to content

Eval Metrics — C source

The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
 *
 * Language: C (C11, standard library only)
 * Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
 * Tool page: https://dev.cosmolabs.org/tools/eval-metrics
 *
 * pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
 * identical to the HumanEval implementation's combinatorial form. Precision,
 * recall, and F1 run over confusion counts and over label sets; exact-match
 * rate runs over paired strings (case-sensitive).
 *
 * The TS lib throws RangeError; C returns 0.0 and sets *err to a static
 * message (NULL on success) — the caller checks err != NULL.
 */

#include <string.h>

typedef struct {
    double precision;
    double recall;
    double f1;
} prf1;

typedef struct {
    long tp;
    long fp;
    long fn;
} confusion_counts;

/* Error codes mirroring the TS RangeError contract. */
enum {
    EVAL_OK = 0,
    EVAL_N_MUST_BE_POSITIVE,
    EVAL_C_OUT_OF_RANGE,
    EVAL_K_OUT_OF_RANGE,
    EVAL_COUNTS_NEGATIVE,
    EVAL_LENGTH_MISMATCH
};

static const char *eval_err_msg(int err) {
    switch (err) {
    case EVAL_OK: return NULL;
    case EVAL_N_MUST_BE_POSITIVE: return "n must be > 0";
    case EVAL_C_OUT_OF_RANGE: return "c must be in [0, n]";
    case EVAL_K_OUT_OF_RANGE: return "k must be in [1, n]";
    case EVAL_COUNTS_NEGATIVE: return "counts must be >= 0";
    case EVAL_LENGTH_MISMATCH: return "predictions and references must have the same length";
    default: return "unknown eval error";
    }
}

/*
 * Unbiased pass@k: probability that at least one of k samples drawn without
 * replacement from n (of which c are correct) passes.
 *
 *   1                                       when n - c < k  (a wrong draw is impossible)
 *   1 - Π_{i=0..k-1} (n - c - i) / (n - i)  otherwise
 *
 * Returns the score, or -1.0 with *err set on impossible inputs.
 */
double pass_at_k(long n, long c, long k, int *err) {
    *err = EVAL_OK;
    if (n <= 0) { *err = EVAL_N_MUST_BE_POSITIVE; return -1.0; }
    if (c < 0 || c > n) { *err = EVAL_C_OUT_OF_RANGE; return -1.0; }
    if (k <= 0 || k > n) { *err = EVAL_K_OUT_OF_RANGE; return -1.0; }
    if (n - c < k) return 1.0;
    double product = 1.0;
    for (long i = 0; i < k; i++) {
        product *= (double)(n - c - i) / (double)(n - i);
    }
    return 1.0 - product;
}

/* Precision/recall/F1 over confusion counts. Zero denominators score 0.
 * Returns EVAL_COUNTS_NEGATIVE in *err on negative counts (result zeroed). */
prf1 precision_recall(confusion_counts counts, int *err) {
    prf1 out = {0.0, 0.0, 0.0};
    *err = EVAL_OK;
    if (counts.tp < 0 || counts.fp < 0 || counts.fn < 0) {
        *err = EVAL_COUNTS_NEGATIVE;
        return out;
    }
    out.precision = (counts.tp + counts.fp > 0)
        ? (double)counts.tp / (double)(counts.tp + counts.fp) : 0.0;
    out.recall = (counts.tp + counts.fn > 0)
        ? (double)counts.tp / (double)(counts.tp + counts.fn) : 0.0;
    out.f1 = (out.precision + out.recall > 0.0)
        ? (2.0 * out.precision * out.recall) / (out.precision + out.recall) : 0.0;
    return out;
}

/* Micro-averaged P/R/F1 across per-class confusion counts. */
prf1 micro_average(const confusion_counts *per_class, size_t n_classes, int *err) {
    confusion_counts sums = {0, 0, 0};
    for (size_t i = 0; i < n_classes; i++) {
        sums.tp += per_class[i].tp;
        sums.fp += per_class[i].fp;
        sums.fn += per_class[i].fn;
    }
    return precision_recall(sums, err);
}

/*
 * Exact-match rate over paired predictions/references (case-sensitive).
 * Empty input scores 0. The caller owns length parity — a single len is
 * passed, so a mismatch cannot occur inside this function (the TS contract's
 * RangeError belongs to the caller's boundary).
 */
double exact_match_rate(const char *const *predictions, const char *const *references,
                        size_t len, int *err) {
    *err = EVAL_OK;
    if (len == 0) return 0.0;
    size_t hits = 0;
    for (size_t i = 0; i < len; i++) {
        if (strcmp(predictions[i], references[i]) == 0) hits++;
    }
    return (double)hits / (double)len;
}

/*
 * P/R/F1 over label SETS — the standard multi-label / extraction metric.
 * Linear scans stand in for hash sets: O(|p| * |r|), fine at label scale.
 */
prf1 set_match(const char *const *prediction, size_t p_len,
               const char *const *reference, size_t r_len, int *err) {
    long tp = 0, fp = 0, fn = 0;
    for (size_t i = 0; i < r_len; i++) {
        int in_p = 0;
        for (size_t j = 0; j < p_len; j++) {
            if (strcmp(reference[i], prediction[j]) == 0) { in_p = 1; break; }
        }
        if (in_p) tp++;
    }
    for (size_t i = 0; i < p_len; i++) {
        int in_r = 0;
        for (size_t j = 0; j < r_len; j++) {
            if (strcmp(prediction[i], reference[j]) == 0) { in_r = 1; break; }
        }
        if (!in_r) fp++;
    }
    for (size_t i = 0; i < r_len; i++) {
        int in_p = 0;
        for (size_t j = 0; j < p_len; j++) {
            if (strcmp(reference[i], prediction[j]) == 0) { in_p = 1; break; }
        }
        if (!in_p) fn++;
    }
    confusion_counts counts = {tp, fp, fn};
    return precision_recall(counts, err);
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →