Skip to content

Statistics Calculator — Rust source

Compute descriptive statistics - count, sum, mean, median, mode, min/max, range, variance, standard deviation, and quartiles (Q1/Q3/IQR) - from any list of numbers. Tolerates mixed separators and flags unparseable tokens. Choose sample (n−1) or population (n) variance. Everything runs 100% client-side.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

// statistics — polyglot showcase port (Rust).
//
// Pure descriptive-statistics logic for the Statistics Calculator tool on
// CosmoDev (dev.cosmolabs.org). Ported from the canonical TypeScript source at
// src/lib/statistics.rs so the tool page can display the same logic across six
// languages.
//
// Self-contained: standard library only, no external crates (no rand, no
// num-traits, nothing from crates.io).
//
// License/usage: display source — part of CosmoDev's polyglot tool pages.

//! Descriptive statistics over a sample of numbers.
//!
//! Two public functions mirror the TypeScript reference:
//! - [`parse_numbers`] splits a free-form list into finite values and junk.
//! - [`summarize`] computes count, mean, median, mode, quartiles, and spread.

use std::collections::HashMap;

/// Outcome of splitting a free-form number list into valid + invalid tokens.
#[derive(Clone, Debug)]
pub struct ParseResult {
    /// Finite numbers, in the order they appeared.
    pub values: Vec<f64>,
    /// Tokens that could not be parsed as finite numbers, in order.
    pub invalid: Vec<String>,
}

/// Descriptive statistics over a sample of numbers. Numeric fields are `NaN`
/// when `count == 0`; `mode` is empty when there is no mode.
#[derive(Clone, Debug)]
pub struct Stats {
    pub count: usize,
    pub sum: f64,
    pub mean: f64,
    pub median: f64,
    /// Most frequent value(s), ascending. Empty when uniform / no mode.
    pub mode: Vec<f64>,
    pub min: f64,
    pub max: f64,
    pub range: f64,
    pub variance: f64,
    pub stddev: f64,
    pub q1: f64,
    pub q3: f64,
    pub iqr: f64,
}

/// Split a free-form number list into finite values and unparseable tokens.
///
/// Separators are any run of whitespace and/or commas. Tokens like `inf` and
/// `nan` are not finite, so they land in `invalid`. Empty input yields no
/// values and no invalid tokens.
pub fn parse_numbers(input: &str) -> ParseResult {
    if input.trim().is_empty() {
        return ParseResult { values: Vec::new(), invalid: Vec::new() };
    }

    let mut result = ParseResult { values: Vec::new(), invalid: Vec::new() };

    // split() on a char predicate yields empties between consecutive
    // separators and at the ends, so skip them — matching the TS
    // `.filter(|t| t.len() > 0)`.
    for tok in input.split(|c: char| c.is_whitespace() || c == ',') {
        if tok.is_empty() {
            continue;
        }
        // f64::parse accepts decimals, exponents, signs, and also "inf"/"nan".
        // Keep only finite results so overflow (1e400 → inf) and the literal
        // "inf"/"nan" are reported as invalid — mirroring JS Number().
        match tok.parse::<f64>() {
            Ok(v) if v.is_finite() => result.values.push(v),
            _ => result.invalid.push(tok.to_string()),
        }
    }
    result
}

/// Linear-interpolation quantile — the R-7 convention used by NumPy, Excel's
/// PERCENTILE, and most stats textbooks. `sorted` must be ascending and
/// non-empty; `p` is in [0, 1].
fn quantile(sorted: &[f64], p: f64) -> f64 {
    let n = sorted.len();
    // Position along the n-1 gaps between the n sorted samples.
    let h = (n - 1) as f64 * p;
    let lower = h.floor() as usize;
    let upper = h.ceil() as usize;
    if lower == upper {
        return sorted[lower];
    }
    // Interpolate between the two bracketing samples.
    sorted[lower] + (h - lower as f64) * (sorted[upper] - sorted[lower])
}

/// Most frequent value(s), ascending. Returns an empty `Vec` when there is no
/// mode — i.e. when every value is distinct, or when all distinct values share
/// the same frequency (a flat / uniform distribution with ≥2 distinct values).
/// A single repeated value (e.g. `[5,5,5]`) does have a mode: `[5]`.
fn compute_mode(values: &[f64]) -> Vec<f64> {
    // Rust's f64 has no Hash/Eq (NaN != NaN), so we key on the bit pattern.
    // JS keys its Map by SameValueZero: -0 and +0 collapse, NaNs collapse.
    // parse_numbers guarantees finite inputs, so the only finite edge case is
    // -0.0 vs +0.0 — normalize zeros to share one bit key to match JS.
    let mut freq: HashMap<u64, (f64, usize)> = HashMap::new();
    for &v in values {
        let key = if v == 0.0 { 0u64 } else { v.to_bits() };
        match freq.get_mut(&key) {
            Some((_, count)) => *count += 1,
            None => {
                freq.insert(key, (v, 1));
            }
        }
    }

    // A single distinct value is always the mode (covers [7] and [5,5,5]).
    if freq.len() == 1 {
        return vec![values[0]];
    }

    let distinct = freq.len();
    let max = freq.values().map(|(_, count)| *count).max().unwrap();
    let mut modes: Vec<f64> = freq
        .into_iter()
        .filter(|(_, (_, count))| *count == max)
        .map(|(_, (value, _))| value)
        .collect();

    // All distinct values share the max frequency → uniform → no mode.
    if modes.len() == distinct {
        return Vec::new();
    }

    // total_cmp gives a total, NaN-aware ordering — safe for f64 sorting.
    modes.sort_by(|a, b| a.total_cmp(b));
    modes
}

/// Compute descriptive statistics over `values`.
///
/// With `sample = true` variance/stddev use the sample estimator (divide by
/// n−1); with `sample = false` they use the population estimator (divide by n).
/// Empty input returns count 0 with every numeric field NaN and `mode: []` —
/// never panics.
pub fn summarize(values: &[f64], sample: bool) -> Stats {
    let n = values.len();
    if n == 0 {
        return empty_stats();
    }

    // Sort a copy so min/max/quantiles are cheap and the caller's slice is
    // untouched.
    let mut sorted = values.to_vec();
    sorted.sort_by(|a, b| a.total_cmp(&b));

    // Sum over the ORIGINAL order to match the TS reference and keep
    // floating-point summation deterministic across languages.
    let sum: f64 = values.iter().copied().sum();
    let mean = sum / n as f64;

    let min = sorted[0];
    let max = sorted[n - 1];
    let median = quantile(&sorted, 0.5);
    let q1 = quantile(&sorted, 0.25);
    let q3 = quantile(&sorted, 0.75);

    // Sum of squared deviations from the mean (original order).
    let ss: f64 = values
        .iter()
        .map(|&x| {
            let d = x - mean;
            d * d
        })
        .sum();
    // Sample variance is undefined for n < 2; population variance is always ss/n.
    let variance = if sample {
        if n >= 2 {
            ss / (n - 1) as f64
        } else {
            f64::NAN
        }
    } else {
        ss / n as f64
    };

    let stddev = if variance.is_finite() {
        variance.sqrt()
    } else {
        f64::NAN
    };

    Stats {
        count: n,
        sum,
        mean,
        median,
        mode: compute_mode(values),
        min,
        max,
        range: max - min,
        variance,
        stddev,
        q1,
        q3,
        iqr: q3 - q1,
    }
}

/// All-NaN Stats block for the empty-input case (count 0).
fn empty_stats() -> Stats {
    Stats {
        count: 0,
        sum: f64::NAN,
        mean: f64::NAN,
        median: f64::NAN,
        mode: Vec::new(),
        min: f64::NAN,
        max: f64::NAN,
        range: f64::NAN,
        variance: f64::NAN,
        stddev: f64::NAN,
        q1: f64::NAN,
        q3: f64::NAN,
        iqr: f64::NAN,
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →