Skip to content

Statistics Calculator — Python source

Compute descriptive statistics - count, sum, mean, median, mode, min/max, range, variance, standard deviation, and quartiles (Q1/Q3/IQR) - from any list of numbers. Tolerates mixed separators and flags unparseable tokens. Choose sample (n−1) or population (n) variance. Everything runs 100% client-side.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""statistics — polyglot showcase port (Python).

Pure descriptive-statistics logic for the Statistics Calculator tool on
CosmoDev (dev.cosmolabs.org). Ported from the canonical TypeScript source at
src/lib/statistics.ts so the tool page can display the same logic across six
languages.

Self-contained: standard library only, no external dependencies (no numpy, no
pip packages).

License/usage: display source — part of CosmoDev's polyglot tool pages.
"""

from __future__ import annotations

import math
from dataclasses import dataclass, field
from typing import List


@dataclass
class ParseResult:
    """Outcome of splitting a free-form number list into valid + invalid tokens."""

    # Finite numbers, in the order they appeared.
    values: List[float] = field(default_factory=list)
    # Tokens that could not be parsed as finite numbers, in order.
    invalid: List[str] = field(default_factory=list)


@dataclass
class Stats:
    """Descriptive statistics over a sample of numbers.

    Numeric fields are ``math.nan`` when ``count == 0``; ``mode`` is ``[]``
    when there is no mode.
    """

    count: int = 0
    sum: float = math.nan
    mean: float = math.nan
    median: float = math.nan
    # Most frequent value(s), ascending. Empty when uniform / no mode.
    mode: List[float] = field(default_factory=list)
    min: float = math.nan
    max: float = math.nan
    range: float = math.nan
    variance: float = math.nan
    stddev: float = math.nan
    q1: float = math.nan
    q3: float = math.nan
    iqr: float = math.nan


def parse_numbers(input: str) -> ParseResult:
    """Split a free-form number list into finite values and unparseable tokens.

    Separators are any run of whitespace and/or commas. Tokens like ``inf`` and
    ``nan`` are not finite, so they land in ``invalid``. Empty input yields no
    values and no invalid tokens.
    """
    if not input.strip():
        return ParseResult()

    result = ParseResult()
    # Replace commas with spaces, then split() on whitespace runs. split() with
    # no args collapses runs and strips the ends — matching the TS
    # `.filter(t => len(t) > 0)` without an extra comprehension.
    for tok in input.replace(",", " ").split():
        try:
            v = float(tok)
        except ValueError:
            result.invalid.append(tok)
            continue
        # float() accepts "inf"/"nan"/overflow→inf; keep only finite values so
        # those land in invalid — mirroring JS Number()'s isFinite gate.
        if math.isfinite(v):
            result.values.append(v)
        else:
            result.invalid.append(tok)
    return result


def _quantile(sorted_vals: List[float], p: float) -> float:
    """Linear-interpolation quantile (R-7 / NumPy / Excel PERCENTILE).

    ``sorted_vals`` must be ascending and non-empty; ``p`` is in [0, 1].
    """
    n = len(sorted_vals)
    # Position along the n-1 gaps between the n sorted samples.
    h = (n - 1) * p
    lower = math.floor(h)
    upper = math.ceil(h)
    if lower == upper:
        return sorted_vals[lower]
    # Interpolate between the two bracketing samples.
    return sorted_vals[lower] + (h - lower) * (sorted_vals[upper] - sorted_vals[lower])


def _compute_mode(values: List[float]) -> List[float]:
    """Most frequent value(s), ascending.

    Returns ``[]`` when there is no mode: when every value is distinct, or when
    all distinct values share the same frequency (a flat / uniform distribution
    with ≥2 distinct values). A single repeated value (e.g. ``[5, 5, 5]``) does
    have a mode: ``[5]``.
    """
    freq: dict[float, int] = {}
    for v in values:
        # Python dict keys use float equality: -0.0 == 0.0 and hash equally, so
        # they collapse to one key — matching JS Map's SameValueZero keying.
        freq[v] = freq.get(v, 0) + 1

    # A single distinct value is always the mode (covers [7] and [5,5,5]).
    if len(freq) == 1:
        return [values[0]]

    max_count = max(freq.values())
    modes = [v for v, c in freq.items() if c == max_count]

    # All distinct values share the max frequency → uniform → no mode.
    if len(modes) == len(freq):
        return []

    return sorted(modes)


def summarize(values: List[float], sample: bool = True) -> Stats:
    """Compute descriptive statistics over ``values``.

    With ``sample=True`` (default) variance/stddev use the sample estimator
    (divide by n−1); with ``sample=False`` they use the population estimator
    (divide by n). Empty input returns count 0 with every numeric field
    ``math.nan`` and ``mode: []`` — never raises.
    """
    n = len(values)
    if n == 0:
        return Stats()

    # Sort a copy (not the caller's list) so min/max/quantiles are cheap.
    sorted_vals = sorted(values)

    # Sum over the ORIGINAL order using plain addition — matches the TS reduce
    # and keeps floating-point results consistent across the language ports.
    total = sum(values)
    mean = total / n

    lo = sorted_vals[0]
    hi = sorted_vals[-1]
    median = _quantile(sorted_vals, 0.5)
    q1 = _quantile(sorted_vals, 0.25)
    q3 = _quantile(sorted_vals, 0.75)

    # Sum of squared deviations from the mean (original order).
    ss = sum((x - mean) ** 2 for x in values)
    # Sample variance is undefined for n < 2; population variance is always ss/n.
    if sample:
        variance = ss / (n - 1) if n >= 2 else math.nan
    else:
        variance = ss / n
    stddev = math.sqrt(variance) if math.isfinite(variance) else math.nan

    return Stats(
        count=n,
        sum=total,
        mean=mean,
        median=median,
        mode=_compute_mode(values),
        min=lo,
        max=hi,
        range=hi - lo,
        variance=variance,
        stddev=stddev,
        q1=q1,
        q3=q3,
        iqr=q3 - q1,
    )

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →