Skip to content

Text Statistics & Readability — Rust source

Count words, sentences, paragraphs, characters, lines, and reading time, plus Flesch Reading Ease and Flesch-Kincaid grade-level readability scores.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

// text-stats — Rust port
// Language: Rust (edition 2021, std-only — no external crates)
// CosmoDev polyglot showcase. Ported from src/lib/textStats.ts.
// Display source — part of CosmoDev's polyglot tool pages.
//
// Pure text-statistics & readability logic: counts characters, words, sentences,
// paragraphs, lines and syllables, and derives reading/speaking time plus the
// Flesch readability scores. Deterministic; analyze_text never panics.
//
// Rust's standard library has no regex engine, so every pattern in the
// TypeScript original is implemented here by hand as a small character state
// machine. Each is annotated with the regex it replaces.

/// The full analysis result. The readability fields are `Option` so they can be
/// `None` (mirroring the TypeScript `number | null`) when the input has no
/// words or no sentences to score.
#[derive(Debug, Clone, PartialEq)]
pub struct TextStats {
    pub characters: usize,
    pub characters_no_spaces: usize,
    pub words: usize,
    pub sentences: usize,
    pub paragraphs: usize,
    pub lines: usize,
    pub syllables: usize,
    pub reading_time_ms: i64, // words / 200 wpm
    pub speaking_time_ms: i64, // words / 130 wpm
    pub flesch_reading_ease: Option<f64>,
    pub flesch_kincaid_grade: Option<f64>,
    pub readability_label: Option<String>,
}

/// Replicate JavaScript's `Math.round`, which rounds half-values toward +Inf.
/// Rust's `f64::round` ties half away from zero, so the two disagree on
/// negative half-values — and a negative Flesch-Kincaid grade landing exactly
/// on n.5 is a real possibility. `floor(x + 0.5)` matches `Math.round` for the
/// magnitudes handled here.
fn js_round(x: f64) -> f64 {
    (x + 0.5).floor()
}

/// A character may belong to a "word" if it is an ASCII letter/digit, or one of
/// the three joiners: straight apostrophe ('), typographic apostrophe (’) or
/// hyphen (-). Mirrors the regex class `[A-Za-z0-9''-]`.
fn is_word_char(c: char) -> bool {
    c.is_ascii_alphanumeric() || c == '\'' || c == '\u{2019}' || c == '-'
}

/// Whether a character is in the "vowel" set used for syllable grouping.
fn is_vowel(c: char) -> bool {
    matches!(c, 'a' | 'e' | 'i' | 'o' | 'u' | 'y')
}

/// Whether a character is in the consonant class `[^laeiouy]` — i.e. NOT one of
/// {l, a, e, i, o, u, y}. Used by the silent-suffix rule below.
fn is_non_laeiouy(c: char) -> bool {
    !matches!(c, 'l' | 'a' | 'e' | 'i' | 'o' | 'u' | 'y')
}

/// Extract every maximal run of word-characters as a token. Replaces the
/// regex `[A-Za-z0-9''-]+` applied globally.
fn find_words(text: &str) -> Vec<String> {
    let mut words = Vec::new();
    let mut cur = String::new();
    for c in text.chars() {
        if is_word_char(c) {
            cur.push(c);
        } else if !cur.is_empty() {
            words.push(std::mem::take(&mut cur));
        }
    }
    if !cur.is_empty() {
        words.push(cur);
    }
    words
}

/// Count maximal runs of `[.!?]` that are immediately followed by whitespace or
/// end-of-string. Replaces the regex `[.!?]+(?:\s|$)` (global, non-overlapping).
fn count_sentences(text: &str) -> usize {
    let chars: Vec<char> = text.chars().collect();
    let n = chars.len();
    let mut i = 0;
    let mut count = 0;
    while i < n {
        if matches!(chars[i], '.' | '!' | '?') {
            // Consume the whole terminal-punctuation run.
            while i < n && matches!(chars[i], '.' | '!' | '?') {
                i += 1;
            }
            // The run must be followed by whitespace or EOF to count.
            if i >= n || chars[i].is_whitespace() {
                count += 1;
            }
        } else {
            i += 1;
        }
    }
    count
}

/// Count paragraphs: split on maximal runs of two or more newlines, then drop
/// empty/whitespace-only blocks. Replaces `text.split(/\n{2,}/)` + trim +
/// filter. Whitespace-only input yields zero.
fn count_paragraphs(text: &str) -> usize {
    if text.trim().is_empty() {
        return 0;
    }
    // Newlines are ASCII, so byte-level scanning is safe on UTF-8 input.
    let bytes = text.as_bytes();
    let n = bytes.len();
    let mut count = 0;
    let mut seg_start = 0;
    let mut i = 0;
    while i < n {
        if bytes[i] == b'\n' {
            let run_start = i;
            while i < n && bytes[i] == b'\n' {
                i += 1;
            }
            // A run of >= 2 newlines is a paragraph separator; a single newline
            // is a soft wrap and stays inside the current block.
            if i - run_start >= 2 {
                if !text[seg_start..run_start].trim().is_empty() {
                    count += 1;
                }
                seg_start = i;
            }
        } else {
            i += 1;
        }
    }
    if !text[seg_start..n].trim().is_empty() {
        count += 1;
    }
    count
}

/// Remove the silent suffix matched by `(?:[^laeiouy]es|ed|[^laeiouy]e)$`.
///
/// JS regex semantics: the leftmost match wins, and every match must end at the
/// string boundary (`$`). A 3-character `[^laeiouy]es` match therefore starts
/// earlier than any 2-character match and takes priority; among the 2-character
/// alternatives, `ed` is tried before `[^laeiouy]e`. The branches below mirror
/// that precedence exactly.
fn strip_silent_suffix(w: &str) -> String {
    let chars: Vec<char> = w.chars().collect();
    let n = chars.len();
    if n >= 3 && is_non_laeiouy(chars[n - 3]) && chars[n - 2] == 'e' && chars[n - 1] == 's' {
        return chars[..n - 3].iter().collect();
    }
    if n >= 2 && chars[n - 2] == 'e' && chars[n - 1] == 'd' {
        return chars[..n - 2].iter().collect();
    }
    if n >= 2 && is_non_laeiouy(chars[n - 2]) && chars[n - 1] == 'e' {
        return chars[..n - 2].iter().collect();
    }
    w.to_string()
}

/// Strip a single leading 'y'. Replaces `s.replace(/^y/, '')`.
fn strip_leading_y(w: &str) -> String {
    let mut chars = w.chars();
    match chars.next() {
        Some('y') => chars.collect(),
        _ => w.to_string(),
    }
}

/// Count maximal runs of vowels `[aeiouy]+`. Replaces the regex of the same
/// shape: each transition from a non-vowel into a vowel begins a new group.
fn count_vowel_groups(w: &str) -> usize {
    let mut count = 0;
    let mut prev_vowel = false;
    for c in w.chars() {
        let v = is_vowel(c);
        if v && !prev_vowel {
            count += 1;
        }
        prev_vowel = v;
    }
    count
}

/// Estimate the syllable count of a single word via a vowel-group heuristic.
/// Cheaper than a dictionary and accurate enough for readability scoring.
pub fn count_syllables(word: &str) -> usize {
    // Normalise: lowercase, then keep ASCII letters only (drop digits,
    // apostrophes, hyphens). `flat_map(to_lowercase)` handles Unicode correctly.
    let lower: String = word.chars().flat_map(|c| c.to_lowercase()).collect();
    let w: String = lower.chars().filter(|c| c.is_ascii_lowercase()).collect();

    if w.is_empty() {
        return 0;
    }
    // w is pure ASCII lowercase here, so byte length == char count.
    if w.len() <= 3 {
        return 1;
    }

    // Drop the silent trailing e/es/ed, then a leading consonant-y.
    let w = strip_silent_suffix(&w);
    let w = strip_leading_y(&w);

    let count = count_vowel_groups(&w);
    count.max(1)
}

/// Map a Flesch reading-ease score onto a qualitative label, following the
/// original Flesch interpretation bands.
fn label_for_flesch(f: f64) -> String {
    let label = if f >= 80.0 {
        "Very Easy"
    } else if f >= 70.0 {
        "Easy"
    } else if f >= 60.0 {
        "Standard"
    } else if f >= 50.0 {
        "Fairly Hard"
    } else if f >= 30.0 {
        "Hard"
    } else {
        "Very Hard"
    };
    label.to_string()
}

/// Analyse a string and return its statistics. Accepts the empty string
/// (callers pass "" for null/undefined) and never panics.
pub fn analyze_text(input: &str) -> TextStats {
    let text = input;

    // Code-point count. Note: the TypeScript reference counts UTF-16 code units
    // via string.length; for BMP text the two are identical and only diverge
    // for supplementary-plane characters (emoji, rare CJK extensions).
    let characters = text.chars().count();
    let characters_no_spaces = text.chars().filter(|c| !c.is_whitespace()).count();

    let word_list = find_words(text);
    let words = word_list.len();

    // No words -> zero sentences; otherwise clamp to >= 1 so a word block with
    // no closing punctuation still reads as one sentence.
    let sentences = if words == 0 {
        0
    } else {
        count_sentences(text).max(1)
    };

    let paragraphs = count_paragraphs(text);

    // Lines: count of '\n'-separated rows. str::split includes a trailing empty
    // slice after a final newline, matching JS "a\n".split('\n').length == 2.
    let lines = if text.is_empty() {
        0
    } else {
        text.split('\n').count()
    };

    let syllables: usize = word_list.iter().map(|w| count_syllables(w)).sum();

    let reading_time_ms = js_round(words as f64 / 200.0 * 60000.0) as i64;
    let speaking_time_ms = js_round(words as f64 / 130.0 * 60000.0) as i64;

    // Readability requires at least one word and one sentence.
    let (flesch_reading_ease, flesch_kincaid_grade, readability_label) =
        if words > 0 && sentences > 0 {
            let words_per_sentence = words as f64 / sentences as f64;
            let syllables_per_word = syllables as f64 / words as f64;
            let ease = js_round(
                (206.835 - 1.015 * words_per_sentence - 84.6 * syllables_per_word) * 10.0,
            ) / 10.0;
            let grade = js_round(
                (0.39 * words_per_sentence + 11.8 * syllables_per_word - 15.59) * 10.0,
            ) / 10.0;
            (Some(ease), Some(grade), Some(label_for_flesch(ease)))
        } else {
            (None, None, None)
        };

    TextStats {
        characters,
        characters_no_spaces,
        words,
        sentences,
        paragraphs,
        lines,
        syllables,
        reading_time_ms,
        speaking_time_ms,
        flesch_reading_ease,
        flesch_kincaid_grade,
        readability_label,
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →