Statistics Calculator — Rust source
Compute descriptive statistics - count, sum, mean, median, mode, min/max, range, variance, standard deviation, and quartiles (Q1/Q3/IQR) - from any list of numbers. Tolerates mixed separators and flags unparseable tokens. Choose sample (n−1) or population (n) variance. Everything runs 100% client-side.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
// statistics — polyglot showcase port (Rust).
//
// Pure descriptive-statistics logic for the Statistics Calculator tool on
// CosmoDev (dev.cosmolabs.org). Ported from the canonical TypeScript source at
// src/lib/statistics.rs so the tool page can display the same logic across six
// languages.
//
// Self-contained: standard library only, no external crates (no rand, no
// num-traits, nothing from crates.io).
//
// License/usage: display source — part of CosmoDev's polyglot tool pages.
//! Descriptive statistics over a sample of numbers.
//!
//! Two public functions mirror the TypeScript reference:
//! - [`parse_numbers`] splits a free-form list into finite values and junk.
//! - [`summarize`] computes count, mean, median, mode, quartiles, and spread.
use std::collections::HashMap;
/// Outcome of splitting a free-form number list into valid + invalid tokens.
#[derive(Clone, Debug)]
pub struct ParseResult {
/// Finite numbers, in the order they appeared.
pub values: Vec<f64>,
/// Tokens that could not be parsed as finite numbers, in order.
pub invalid: Vec<String>,
}
/// Descriptive statistics over a sample of numbers. Numeric fields are `NaN`
/// when `count == 0`; `mode` is empty when there is no mode.
#[derive(Clone, Debug)]
pub struct Stats {
pub count: usize,
pub sum: f64,
pub mean: f64,
pub median: f64,
/// Most frequent value(s), ascending. Empty when uniform / no mode.
pub mode: Vec<f64>,
pub min: f64,
pub max: f64,
pub range: f64,
pub variance: f64,
pub stddev: f64,
pub q1: f64,
pub q3: f64,
pub iqr: f64,
}
/// Split a free-form number list into finite values and unparseable tokens.
///
/// Separators are any run of whitespace and/or commas. Tokens like `inf` and
/// `nan` are not finite, so they land in `invalid`. Empty input yields no
/// values and no invalid tokens.
pub fn parse_numbers(input: &str) -> ParseResult {
if input.trim().is_empty() {
return ParseResult { values: Vec::new(), invalid: Vec::new() };
}
let mut result = ParseResult { values: Vec::new(), invalid: Vec::new() };
// split() on a char predicate yields empties between consecutive
// separators and at the ends, so skip them — matching the TS
// `.filter(|t| t.len() > 0)`.
for tok in input.split(|c: char| c.is_whitespace() || c == ',') {
if tok.is_empty() {
continue;
}
// f64::parse accepts decimals, exponents, signs, and also "inf"/"nan".
// Keep only finite results so overflow (1e400 → inf) and the literal
// "inf"/"nan" are reported as invalid — mirroring JS Number().
match tok.parse::<f64>() {
Ok(v) if v.is_finite() => result.values.push(v),
_ => result.invalid.push(tok.to_string()),
}
}
result
}
/// Linear-interpolation quantile — the R-7 convention used by NumPy, Excel's
/// PERCENTILE, and most stats textbooks. `sorted` must be ascending and
/// non-empty; `p` is in [0, 1].
fn quantile(sorted: &[f64], p: f64) -> f64 {
let n = sorted.len();
// Position along the n-1 gaps between the n sorted samples.
let h = (n - 1) as f64 * p;
let lower = h.floor() as usize;
let upper = h.ceil() as usize;
if lower == upper {
return sorted[lower];
}
// Interpolate between the two bracketing samples.
sorted[lower] + (h - lower as f64) * (sorted[upper] - sorted[lower])
}
/// Most frequent value(s), ascending. Returns an empty `Vec` when there is no
/// mode — i.e. when every value is distinct, or when all distinct values share
/// the same frequency (a flat / uniform distribution with ≥2 distinct values).
/// A single repeated value (e.g. `[5,5,5]`) does have a mode: `[5]`.
fn compute_mode(values: &[f64]) -> Vec<f64> {
// Rust's f64 has no Hash/Eq (NaN != NaN), so we key on the bit pattern.
// JS keys its Map by SameValueZero: -0 and +0 collapse, NaNs collapse.
// parse_numbers guarantees finite inputs, so the only finite edge case is
// -0.0 vs +0.0 — normalize zeros to share one bit key to match JS.
let mut freq: HashMap<u64, (f64, usize)> = HashMap::new();
for &v in values {
let key = if v == 0.0 { 0u64 } else { v.to_bits() };
match freq.get_mut(&key) {
Some((_, count)) => *count += 1,
None => {
freq.insert(key, (v, 1));
}
}
}
// A single distinct value is always the mode (covers [7] and [5,5,5]).
if freq.len() == 1 {
return vec![values[0]];
}
let distinct = freq.len();
let max = freq.values().map(|(_, count)| *count).max().unwrap();
let mut modes: Vec<f64> = freq
.into_iter()
.filter(|(_, (_, count))| *count == max)
.map(|(_, (value, _))| value)
.collect();
// All distinct values share the max frequency → uniform → no mode.
if modes.len() == distinct {
return Vec::new();
}
// total_cmp gives a total, NaN-aware ordering — safe for f64 sorting.
modes.sort_by(|a, b| a.total_cmp(b));
modes
}
/// Compute descriptive statistics over `values`.
///
/// With `sample = true` variance/stddev use the sample estimator (divide by
/// n−1); with `sample = false` they use the population estimator (divide by n).
/// Empty input returns count 0 with every numeric field NaN and `mode: []` —
/// never panics.
pub fn summarize(values: &[f64], sample: bool) -> Stats {
let n = values.len();
if n == 0 {
return empty_stats();
}
// Sort a copy so min/max/quantiles are cheap and the caller's slice is
// untouched.
let mut sorted = values.to_vec();
sorted.sort_by(|a, b| a.total_cmp(&b));
// Sum over the ORIGINAL order to match the TS reference and keep
// floating-point summation deterministic across languages.
let sum: f64 = values.iter().copied().sum();
let mean = sum / n as f64;
let min = sorted[0];
let max = sorted[n - 1];
let median = quantile(&sorted, 0.5);
let q1 = quantile(&sorted, 0.25);
let q3 = quantile(&sorted, 0.75);
// Sum of squared deviations from the mean (original order).
let ss: f64 = values
.iter()
.map(|&x| {
let d = x - mean;
d * d
})
.sum();
// Sample variance is undefined for n < 2; population variance is always ss/n.
let variance = if sample {
if n >= 2 {
ss / (n - 1) as f64
} else {
f64::NAN
}
} else {
ss / n as f64
};
let stddev = if variance.is_finite() {
variance.sqrt()
} else {
f64::NAN
};
Stats {
count: n,
sum,
mean,
median,
mode: compute_mode(values),
min,
max,
range: max - min,
variance,
stddev,
q1,
q3,
iqr: q3 - q1,
}
}
/// All-NaN Stats block for the empty-input case (count 0).
fn empty_stats() -> Stats {
Stats {
count: 0,
sum: f64::NAN,
mean: f64::NAN,
median: f64::NAN,
mode: Vec::new(),
min: f64::NAN,
max: f64::NAN,
range: f64::NAN,
variance: f64::NAN,
stddev: f64::NAN,
q1: f64::NAN,
q3: f64::NAN,
iqr: f64::NAN,
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →