Text Statistics & Readability — Rust source
Count words, sentences, paragraphs, characters, lines, and reading time, plus Flesch Reading Ease and Flesch-Kincaid grade-level readability scores.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
// text-stats — Rust port
// Language: Rust (edition 2021, std-only — no external crates)
// CosmoDev polyglot showcase. Ported from src/lib/textStats.ts.
// Display source — part of CosmoDev's polyglot tool pages.
//
// Pure text-statistics & readability logic: counts characters, words, sentences,
// paragraphs, lines and syllables, and derives reading/speaking time plus the
// Flesch readability scores. Deterministic; analyze_text never panics.
//
// Rust's standard library has no regex engine, so every pattern in the
// TypeScript original is implemented here by hand as a small character state
// machine. Each is annotated with the regex it replaces.
/// The full analysis result. The readability fields are `Option` so they can be
/// `None` (mirroring the TypeScript `number | null`) when the input has no
/// words or no sentences to score.
#[derive(Debug, Clone, PartialEq)]
pub struct TextStats {
pub characters: usize,
pub characters_no_spaces: usize,
pub words: usize,
pub sentences: usize,
pub paragraphs: usize,
pub lines: usize,
pub syllables: usize,
pub reading_time_ms: i64, // words / 200 wpm
pub speaking_time_ms: i64, // words / 130 wpm
pub flesch_reading_ease: Option<f64>,
pub flesch_kincaid_grade: Option<f64>,
pub readability_label: Option<String>,
}
/// Replicate JavaScript's `Math.round`, which rounds half-values toward +Inf.
/// Rust's `f64::round` ties half away from zero, so the two disagree on
/// negative half-values — and a negative Flesch-Kincaid grade landing exactly
/// on n.5 is a real possibility. `floor(x + 0.5)` matches `Math.round` for the
/// magnitudes handled here.
fn js_round(x: f64) -> f64 {
(x + 0.5).floor()
}
/// A character may belong to a "word" if it is an ASCII letter/digit, or one of
/// the three joiners: straight apostrophe ('), typographic apostrophe (’) or
/// hyphen (-). Mirrors the regex class `[A-Za-z0-9''-]`.
fn is_word_char(c: char) -> bool {
c.is_ascii_alphanumeric() || c == '\'' || c == '\u{2019}' || c == '-'
}
/// Whether a character is in the "vowel" set used for syllable grouping.
fn is_vowel(c: char) -> bool {
matches!(c, 'a' | 'e' | 'i' | 'o' | 'u' | 'y')
}
/// Whether a character is in the consonant class `[^laeiouy]` — i.e. NOT one of
/// {l, a, e, i, o, u, y}. Used by the silent-suffix rule below.
fn is_non_laeiouy(c: char) -> bool {
!matches!(c, 'l' | 'a' | 'e' | 'i' | 'o' | 'u' | 'y')
}
/// Extract every maximal run of word-characters as a token. Replaces the
/// regex `[A-Za-z0-9''-]+` applied globally.
fn find_words(text: &str) -> Vec<String> {
let mut words = Vec::new();
let mut cur = String::new();
for c in text.chars() {
if is_word_char(c) {
cur.push(c);
} else if !cur.is_empty() {
words.push(std::mem::take(&mut cur));
}
}
if !cur.is_empty() {
words.push(cur);
}
words
}
/// Count maximal runs of `[.!?]` that are immediately followed by whitespace or
/// end-of-string. Replaces the regex `[.!?]+(?:\s|$)` (global, non-overlapping).
fn count_sentences(text: &str) -> usize {
let chars: Vec<char> = text.chars().collect();
let n = chars.len();
let mut i = 0;
let mut count = 0;
while i < n {
if matches!(chars[i], '.' | '!' | '?') {
// Consume the whole terminal-punctuation run.
while i < n && matches!(chars[i], '.' | '!' | '?') {
i += 1;
}
// The run must be followed by whitespace or EOF to count.
if i >= n || chars[i].is_whitespace() {
count += 1;
}
} else {
i += 1;
}
}
count
}
/// Count paragraphs: split on maximal runs of two or more newlines, then drop
/// empty/whitespace-only blocks. Replaces `text.split(/\n{2,}/)` + trim +
/// filter. Whitespace-only input yields zero.
fn count_paragraphs(text: &str) -> usize {
if text.trim().is_empty() {
return 0;
}
// Newlines are ASCII, so byte-level scanning is safe on UTF-8 input.
let bytes = text.as_bytes();
let n = bytes.len();
let mut count = 0;
let mut seg_start = 0;
let mut i = 0;
while i < n {
if bytes[i] == b'\n' {
let run_start = i;
while i < n && bytes[i] == b'\n' {
i += 1;
}
// A run of >= 2 newlines is a paragraph separator; a single newline
// is a soft wrap and stays inside the current block.
if i - run_start >= 2 {
if !text[seg_start..run_start].trim().is_empty() {
count += 1;
}
seg_start = i;
}
} else {
i += 1;
}
}
if !text[seg_start..n].trim().is_empty() {
count += 1;
}
count
}
/// Remove the silent suffix matched by `(?:[^laeiouy]es|ed|[^laeiouy]e)$`.
///
/// JS regex semantics: the leftmost match wins, and every match must end at the
/// string boundary (`$`). A 3-character `[^laeiouy]es` match therefore starts
/// earlier than any 2-character match and takes priority; among the 2-character
/// alternatives, `ed` is tried before `[^laeiouy]e`. The branches below mirror
/// that precedence exactly.
fn strip_silent_suffix(w: &str) -> String {
let chars: Vec<char> = w.chars().collect();
let n = chars.len();
if n >= 3 && is_non_laeiouy(chars[n - 3]) && chars[n - 2] == 'e' && chars[n - 1] == 's' {
return chars[..n - 3].iter().collect();
}
if n >= 2 && chars[n - 2] == 'e' && chars[n - 1] == 'd' {
return chars[..n - 2].iter().collect();
}
if n >= 2 && is_non_laeiouy(chars[n - 2]) && chars[n - 1] == 'e' {
return chars[..n - 2].iter().collect();
}
w.to_string()
}
/// Strip a single leading 'y'. Replaces `s.replace(/^y/, '')`.
fn strip_leading_y(w: &str) -> String {
let mut chars = w.chars();
match chars.next() {
Some('y') => chars.collect(),
_ => w.to_string(),
}
}
/// Count maximal runs of vowels `[aeiouy]+`. Replaces the regex of the same
/// shape: each transition from a non-vowel into a vowel begins a new group.
fn count_vowel_groups(w: &str) -> usize {
let mut count = 0;
let mut prev_vowel = false;
for c in w.chars() {
let v = is_vowel(c);
if v && !prev_vowel {
count += 1;
}
prev_vowel = v;
}
count
}
/// Estimate the syllable count of a single word via a vowel-group heuristic.
/// Cheaper than a dictionary and accurate enough for readability scoring.
pub fn count_syllables(word: &str) -> usize {
// Normalise: lowercase, then keep ASCII letters only (drop digits,
// apostrophes, hyphens). `flat_map(to_lowercase)` handles Unicode correctly.
let lower: String = word.chars().flat_map(|c| c.to_lowercase()).collect();
let w: String = lower.chars().filter(|c| c.is_ascii_lowercase()).collect();
if w.is_empty() {
return 0;
}
// w is pure ASCII lowercase here, so byte length == char count.
if w.len() <= 3 {
return 1;
}
// Drop the silent trailing e/es/ed, then a leading consonant-y.
let w = strip_silent_suffix(&w);
let w = strip_leading_y(&w);
let count = count_vowel_groups(&w);
count.max(1)
}
/// Map a Flesch reading-ease score onto a qualitative label, following the
/// original Flesch interpretation bands.
fn label_for_flesch(f: f64) -> String {
let label = if f >= 80.0 {
"Very Easy"
} else if f >= 70.0 {
"Easy"
} else if f >= 60.0 {
"Standard"
} else if f >= 50.0 {
"Fairly Hard"
} else if f >= 30.0 {
"Hard"
} else {
"Very Hard"
};
label.to_string()
}
/// Analyse a string and return its statistics. Accepts the empty string
/// (callers pass "" for null/undefined) and never panics.
pub fn analyze_text(input: &str) -> TextStats {
let text = input;
// Code-point count. Note: the TypeScript reference counts UTF-16 code units
// via string.length; for BMP text the two are identical and only diverge
// for supplementary-plane characters (emoji, rare CJK extensions).
let characters = text.chars().count();
let characters_no_spaces = text.chars().filter(|c| !c.is_whitespace()).count();
let word_list = find_words(text);
let words = word_list.len();
// No words -> zero sentences; otherwise clamp to >= 1 so a word block with
// no closing punctuation still reads as one sentence.
let sentences = if words == 0 {
0
} else {
count_sentences(text).max(1)
};
let paragraphs = count_paragraphs(text);
// Lines: count of '\n'-separated rows. str::split includes a trailing empty
// slice after a final newline, matching JS "a\n".split('\n').length == 2.
let lines = if text.is_empty() {
0
} else {
text.split('\n').count()
};
let syllables: usize = word_list.iter().map(|w| count_syllables(w)).sum();
let reading_time_ms = js_round(words as f64 / 200.0 * 60000.0) as i64;
let speaking_time_ms = js_round(words as f64 / 130.0 * 60000.0) as i64;
// Readability requires at least one word and one sentence.
let (flesch_reading_ease, flesch_kincaid_grade, readability_label) =
if words > 0 && sentences > 0 {
let words_per_sentence = words as f64 / sentences as f64;
let syllables_per_word = syllables as f64 / words as f64;
let ease = js_round(
(206.835 - 1.015 * words_per_sentence - 84.6 * syllables_per_word) * 10.0,
) / 10.0;
let grade = js_round(
(0.39 * words_per_sentence + 11.8 * syllables_per_word - 15.59) * 10.0,
) / 10.0;
(Some(ease), Some(grade), Some(label_for_flesch(ease)))
} else {
(None, None, None)
};
TextStats {
characters,
characters_no_spaces,
words,
sentences,
paragraphs,
lines,
syllables,
reading_time_ms,
speaking_time_ms,
flesch_reading_ease,
flesch_kincaid_grade,
readability_label,
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →