Eval Metrics — C source
The standard LLM-eval numbers with exact math — unbiased pass@k, precision/recall/F1 from confusion counts, exact-match and label-set micro-F1. 100% client-side.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/*
* Eval Metrics — the standard LLM-eval metrics with exact, testable formulas.
*
* Language: C (C11, standard library only)
* Port of src/lib/evalMetrics.ts (the canonical TypeScript implementation).
* Tool page: https://dev.cosmolabs.org/tools/eval-metrics
*
* pass@k is the unbiased estimator from the Codex paper (Chen et al. 2021),
* identical to the HumanEval implementation's combinatorial form. Precision,
* recall, and F1 run over confusion counts and over label sets; exact-match
* rate runs over paired strings (case-sensitive).
*
* The TS lib throws RangeError; C returns 0.0 and sets *err to a static
* message (NULL on success) — the caller checks err != NULL.
*/
#include <string.h>
typedef struct {
double precision;
double recall;
double f1;
} prf1;
typedef struct {
long tp;
long fp;
long fn;
} confusion_counts;
/* Error codes mirroring the TS RangeError contract. */
enum {
EVAL_OK = 0,
EVAL_N_MUST_BE_POSITIVE,
EVAL_C_OUT_OF_RANGE,
EVAL_K_OUT_OF_RANGE,
EVAL_COUNTS_NEGATIVE,
EVAL_LENGTH_MISMATCH
};
static const char *eval_err_msg(int err) {
switch (err) {
case EVAL_OK: return NULL;
case EVAL_N_MUST_BE_POSITIVE: return "n must be > 0";
case EVAL_C_OUT_OF_RANGE: return "c must be in [0, n]";
case EVAL_K_OUT_OF_RANGE: return "k must be in [1, n]";
case EVAL_COUNTS_NEGATIVE: return "counts must be >= 0";
case EVAL_LENGTH_MISMATCH: return "predictions and references must have the same length";
default: return "unknown eval error";
}
}
/*
* Unbiased pass@k: probability that at least one of k samples drawn without
* replacement from n (of which c are correct) passes.
*
* 1 when n - c < k (a wrong draw is impossible)
* 1 - Π_{i=0..k-1} (n - c - i) / (n - i) otherwise
*
* Returns the score, or -1.0 with *err set on impossible inputs.
*/
double pass_at_k(long n, long c, long k, int *err) {
*err = EVAL_OK;
if (n <= 0) { *err = EVAL_N_MUST_BE_POSITIVE; return -1.0; }
if (c < 0 || c > n) { *err = EVAL_C_OUT_OF_RANGE; return -1.0; }
if (k <= 0 || k > n) { *err = EVAL_K_OUT_OF_RANGE; return -1.0; }
if (n - c < k) return 1.0;
double product = 1.0;
for (long i = 0; i < k; i++) {
product *= (double)(n - c - i) / (double)(n - i);
}
return 1.0 - product;
}
/* Precision/recall/F1 over confusion counts. Zero denominators score 0.
* Returns EVAL_COUNTS_NEGATIVE in *err on negative counts (result zeroed). */
prf1 precision_recall(confusion_counts counts, int *err) {
prf1 out = {0.0, 0.0, 0.0};
*err = EVAL_OK;
if (counts.tp < 0 || counts.fp < 0 || counts.fn < 0) {
*err = EVAL_COUNTS_NEGATIVE;
return out;
}
out.precision = (counts.tp + counts.fp > 0)
? (double)counts.tp / (double)(counts.tp + counts.fp) : 0.0;
out.recall = (counts.tp + counts.fn > 0)
? (double)counts.tp / (double)(counts.tp + counts.fn) : 0.0;
out.f1 = (out.precision + out.recall > 0.0)
? (2.0 * out.precision * out.recall) / (out.precision + out.recall) : 0.0;
return out;
}
/* Micro-averaged P/R/F1 across per-class confusion counts. */
prf1 micro_average(const confusion_counts *per_class, size_t n_classes, int *err) {
confusion_counts sums = {0, 0, 0};
for (size_t i = 0; i < n_classes; i++) {
sums.tp += per_class[i].tp;
sums.fp += per_class[i].fp;
sums.fn += per_class[i].fn;
}
return precision_recall(sums, err);
}
/*
* Exact-match rate over paired predictions/references (case-sensitive).
* Empty input scores 0. The caller owns length parity — a single len is
* passed, so a mismatch cannot occur inside this function (the TS contract's
* RangeError belongs to the caller's boundary).
*/
double exact_match_rate(const char *const *predictions, const char *const *references,
size_t len, int *err) {
*err = EVAL_OK;
if (len == 0) return 0.0;
size_t hits = 0;
for (size_t i = 0; i < len; i++) {
if (strcmp(predictions[i], references[i]) == 0) hits++;
}
return (double)hits / (double)len;
}
/*
* P/R/F1 over label SETS — the standard multi-label / extraction metric.
* Linear scans stand in for hash sets: O(|p| * |r|), fine at label scale.
*/
prf1 set_match(const char *const *prediction, size_t p_len,
const char *const *reference, size_t r_len, int *err) {
long tp = 0, fp = 0, fn = 0;
for (size_t i = 0; i < r_len; i++) {
int in_p = 0;
for (size_t j = 0; j < p_len; j++) {
if (strcmp(reference[i], prediction[j]) == 0) { in_p = 1; break; }
}
if (in_p) tp++;
}
for (size_t i = 0; i < p_len; i++) {
int in_r = 0;
for (size_t j = 0; j < r_len; j++) {
if (strcmp(prediction[i], reference[j]) == 0) { in_r = 1; break; }
}
if (!in_r) fp++;
}
for (size_t i = 0; i < r_len; i++) {
int in_p = 0;
for (size_t j = 0; j < p_len; j++) {
if (strcmp(reference[i], prediction[j]) == 0) { in_p = 1; break; }
}
if (!in_p) fn++;
}
confusion_counts counts = {tp, fp, fn};
return precision_recall(counts, err);
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →