Text Statistics & Readability — Python source
Count words, sentences, paragraphs, characters, lines, and reading time, plus Flesch Reading Ease and Flesch-Kincaid grade-level readability scores.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""text-stats — Python port.
Language: Python 3.10+ (standard library only — no pip dependencies)
CosmoDev polyglot showcase. Ported from src/lib/textStats.ts.
Display source — part of CosmoDev's polyglot tool pages.
Pure text-statistics & readability logic: counts characters, words, sentences,
paragraphs, lines and syllables, and derives reading/speaking time plus the
Flesch readability scores. Deterministic; analyze_text never raises.
"""
from __future__ import annotations
import math
import re
from dataclasses import dataclass
from typing import Optional
# --- compiled patterns -------------------------------------------------------
# Each mirrors a regex literal from src/lib/textStats.ts. Python's re module is
# the natural stdlib choice here, so unlike the Rust port no hand-rolled state
# machines are needed.
# Words are runs of ASCII letters/digits plus apostrophes (both ' and the
# typographic ') and hyphens, so contractions ("don't") and hyphenated
# compounds ("well-being") stay whole.
WORD_RE = re.compile(r"[A-Za-z0-9'’-]+")
# A sentence ends at a run of terminal punctuation followed by whitespace or EOF.
SENTENCE_END_RE = re.compile(r"[.!?]+(?:\s|$)")
# Paragraphs are separated by two or more newlines.
PARAGRAPH_RE = re.compile(r"\n{2,}")
# Whitespace, used to strip spaces when counting non-space characters.
WHITESPACE_RE = re.compile(r"\s")
# Syllable-heuristic helpers (operate on lowercased ASCII letters only).
NON_ALPHA_RE = re.compile(r"[^a-z]")
SILENT_SUFFIX_RE = re.compile(r"(?:[^laeiouy]es|ed|[^laeiouy]e)$")
LEADING_Y_RE = re.compile(r"^y")
VOWEL_GROUP_RE = re.compile(r"[aeiouy]+")
@dataclass
class TextStats:
"""Full analysis result.
The three readability fields are Optional (mirroring the TypeScript
`number | null`): they are None when the input has no words or no
sentences to score.
"""
characters: int
characters_no_spaces: int
words: int
sentences: int
paragraphs: int
lines: int
syllables: int
reading_time_ms: int # words / 200 wpm
speaking_time_ms: int # words / 130 wpm
flesch_reading_ease: Optional[float]
flesch_kincaid_grade: Optional[float]
readability_label: Optional[str]
def _js_round(x: float) -> int:
"""Replicate JavaScript's Math.round, which rounds half-values toward +Inf.
Python's built-in round() uses banker's rounding (half to even) and Go /
Rust / PHP round half away from zero — all three disagree with Math.round
on half-values. A negative Flesch-Kincaid grade landing exactly on n.5 is a
real case, so we mirror the JS rule via floor(x + 0.5) to keep every port
bit-for-bit aligned with the TypeScript source.
"""
return int(math.floor(x + 0.5))
def _label_for_flesch(f: float) -> str:
"""Map a Flesch reading-ease score onto a qualitative label."""
if f >= 80:
return "Very Easy"
if f >= 70:
return "Easy"
if f >= 60:
return "Standard"
if f >= 50:
return "Fairly Hard"
if f >= 30:
return "Hard"
return "Very Hard"
def count_syllables(word: str) -> int:
"""Estimate the syllable count of a single word via a vowel-group heuristic.
True syllabification needs a dictionary; this heuristic is cheap and
accurate enough to feed the Flesch formulas.
"""
# Normalise to lowercase ASCII letters only, dropping digits, apostrophes
# and hyphens so "don't" / "well-being" are scored on their letter cores.
w = NON_ALPHA_RE.sub("", word.lower())
if not w:
return 0
if len(w) <= 3: # w is pure ASCII here, so len() is the char count
return 1
# Drop a silent trailing 'e'/'es'/'ed'. 'l' is kept out of the consonant
# class so "...le" endings (apple, table) keep their final syllable.
w = SILENT_SUFFIX_RE.sub("", w)
# A leading 'y' acts as a consonant ("yellow"); strip it before grouping.
w = LEADING_Y_RE.sub("", w)
# Each maximal run of vowels is one syllable nucleus.
groups = VOWEL_GROUP_RE.findall(w)
count = len(groups) if groups else 1
return max(1, count)
def analyze_text(input: Optional[str] = None) -> TextStats:
"""Analyse a string and return its statistics.
Accepts None (treated as the empty string) and never raises. The reading
and speaking times use the conventional 200 wpm / 130 wpm rates.
"""
text = "" if input is None else input
# Note on character counting: len() counts Unicode code points, whereas the
# TypeScript reference measures UTF-16 code units via string.length. The two
# agree for BMP text and only diverge for supplementary-plane characters
# (emoji, rare CJK extensions) — an unusual input for a readability tool.
characters = len(text)
characters_no_spaces = len(WHITESPACE_RE.sub("", text))
word_list = WORD_RE.findall(text)
words = len(word_list)
# No words -> zero sentences; otherwise clamp to >= 1 so a word block with
# no closing punctuation still reads as one sentence.
if words == 0:
sentences = 0
else:
sentences = max(1, len(SENTENCE_END_RE.findall(text)))
# Paragraphs: split on blank-line separators, drop empty/whitespace-only
# blocks. Whitespace-only input yields zero paragraphs.
if text.strip() == "":
paragraphs = 0
else:
paragraphs = sum(1 for block in PARAGRAPH_RE.split(text) if block.strip())
# Lines: number of '\n'-separated rows. split mirrors JS exactly, including
# the trailing empty string after a final newline.
lines = 0 if text == "" else len(text.split("\n"))
syllables = sum(count_syllables(w) for w in word_list)
reading_time_ms = _js_round(words / 200 * 60000)
speaking_time_ms = _js_round(words / 130 * 60000)
# Readability requires at least one word and one sentence.
flesch_reading_ease: Optional[float] = None
flesch_kincaid_grade: Optional[float] = None
readability_label: Optional[str] = None
if words > 0 and sentences > 0:
words_per_sentence = words / sentences
syllables_per_word = syllables / words
flesch_reading_ease = _js_round(
(206.835 - 1.015 * words_per_sentence - 84.6 * syllables_per_word) * 10
) / 10
flesch_kincaid_grade = _js_round(
(0.39 * words_per_sentence + 11.8 * syllables_per_word - 15.59) * 10
) / 10
readability_label = _label_for_flesch(flesch_reading_ease)
return TextStats(
characters=characters,
characters_no_spaces=characters_no_spaces,
words=words,
sentences=sentences,
paragraphs=paragraphs,
lines=lines,
syllables=syllables,
reading_time_ms=reading_time_ms,
speaking_time_ms=speaking_time_ms,
flesch_reading_ease=flesch_reading_ease,
flesch_kincaid_grade=flesch_kincaid_grade,
readability_label=readability_label,
)
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →