Skip to content

Text Statistics & Readability — Python source

Count words, sentences, paragraphs, characters, lines, and reading time, plus Flesch Reading Ease and Flesch-Kincaid grade-level readability scores.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""text-stats — Python port.

Language: Python 3.10+ (standard library only — no pip dependencies)
CosmoDev polyglot showcase. Ported from src/lib/textStats.ts.
Display source — part of CosmoDev's polyglot tool pages.

Pure text-statistics & readability logic: counts characters, words, sentences,
paragraphs, lines and syllables, and derives reading/speaking time plus the
Flesch readability scores. Deterministic; analyze_text never raises.
"""

from __future__ import annotations

import math
import re
from dataclasses import dataclass
from typing import Optional

# --- compiled patterns -------------------------------------------------------
# Each mirrors a regex literal from src/lib/textStats.ts. Python's re module is
# the natural stdlib choice here, so unlike the Rust port no hand-rolled state
# machines are needed.

# Words are runs of ASCII letters/digits plus apostrophes (both ' and the
# typographic ') and hyphens, so contractions ("don't") and hyphenated
# compounds ("well-being") stay whole.
WORD_RE = re.compile(r"[A-Za-z0-9'’-]+")
# A sentence ends at a run of terminal punctuation followed by whitespace or EOF.
SENTENCE_END_RE = re.compile(r"[.!?]+(?:\s|$)")
# Paragraphs are separated by two or more newlines.
PARAGRAPH_RE = re.compile(r"\n{2,}")
# Whitespace, used to strip spaces when counting non-space characters.
WHITESPACE_RE = re.compile(r"\s")

# Syllable-heuristic helpers (operate on lowercased ASCII letters only).
NON_ALPHA_RE = re.compile(r"[^a-z]")
SILENT_SUFFIX_RE = re.compile(r"(?:[^laeiouy]es|ed|[^laeiouy]e)$")
LEADING_Y_RE = re.compile(r"^y")
VOWEL_GROUP_RE = re.compile(r"[aeiouy]+")


@dataclass
class TextStats:
    """Full analysis result.

    The three readability fields are Optional (mirroring the TypeScript
    `number | null`): they are None when the input has no words or no
    sentences to score.
    """

    characters: int
    characters_no_spaces: int
    words: int
    sentences: int
    paragraphs: int
    lines: int
    syllables: int
    reading_time_ms: int       # words / 200 wpm
    speaking_time_ms: int      # words / 130 wpm
    flesch_reading_ease: Optional[float]
    flesch_kincaid_grade: Optional[float]
    readability_label: Optional[str]


def _js_round(x: float) -> int:
    """Replicate JavaScript's Math.round, which rounds half-values toward +Inf.

    Python's built-in round() uses banker's rounding (half to even) and Go /
    Rust / PHP round half away from zero — all three disagree with Math.round
    on half-values. A negative Flesch-Kincaid grade landing exactly on n.5 is a
    real case, so we mirror the JS rule via floor(x + 0.5) to keep every port
    bit-for-bit aligned with the TypeScript source.
    """
    return int(math.floor(x + 0.5))


def _label_for_flesch(f: float) -> str:
    """Map a Flesch reading-ease score onto a qualitative label."""
    if f >= 80:
        return "Very Easy"
    if f >= 70:
        return "Easy"
    if f >= 60:
        return "Standard"
    if f >= 50:
        return "Fairly Hard"
    if f >= 30:
        return "Hard"
    return "Very Hard"


def count_syllables(word: str) -> int:
    """Estimate the syllable count of a single word via a vowel-group heuristic.

    True syllabification needs a dictionary; this heuristic is cheap and
    accurate enough to feed the Flesch formulas.
    """
    # Normalise to lowercase ASCII letters only, dropping digits, apostrophes
    # and hyphens so "don't" / "well-being" are scored on their letter cores.
    w = NON_ALPHA_RE.sub("", word.lower())
    if not w:
        return 0
    if len(w) <= 3:  # w is pure ASCII here, so len() is the char count
        return 1

    # Drop a silent trailing 'e'/'es'/'ed'. 'l' is kept out of the consonant
    # class so "...le" endings (apple, table) keep their final syllable.
    w = SILENT_SUFFIX_RE.sub("", w)
    # A leading 'y' acts as a consonant ("yellow"); strip it before grouping.
    w = LEADING_Y_RE.sub("", w)

    # Each maximal run of vowels is one syllable nucleus.
    groups = VOWEL_GROUP_RE.findall(w)
    count = len(groups) if groups else 1
    return max(1, count)


def analyze_text(input: Optional[str] = None) -> TextStats:
    """Analyse a string and return its statistics.

    Accepts None (treated as the empty string) and never raises. The reading
    and speaking times use the conventional 200 wpm / 130 wpm rates.
    """
    text = "" if input is None else input

    # Note on character counting: len() counts Unicode code points, whereas the
    # TypeScript reference measures UTF-16 code units via string.length. The two
    # agree for BMP text and only diverge for supplementary-plane characters
    # (emoji, rare CJK extensions) — an unusual input for a readability tool.
    characters = len(text)
    characters_no_spaces = len(WHITESPACE_RE.sub("", text))

    word_list = WORD_RE.findall(text)
    words = len(word_list)

    # No words -> zero sentences; otherwise clamp to >= 1 so a word block with
    # no closing punctuation still reads as one sentence.
    if words == 0:
        sentences = 0
    else:
        sentences = max(1, len(SENTENCE_END_RE.findall(text)))

    # Paragraphs: split on blank-line separators, drop empty/whitespace-only
    # blocks. Whitespace-only input yields zero paragraphs.
    if text.strip() == "":
        paragraphs = 0
    else:
        paragraphs = sum(1 for block in PARAGRAPH_RE.split(text) if block.strip())

    # Lines: number of '\n'-separated rows. split mirrors JS exactly, including
    # the trailing empty string after a final newline.
    lines = 0 if text == "" else len(text.split("\n"))

    syllables = sum(count_syllables(w) for w in word_list)

    reading_time_ms = _js_round(words / 200 * 60000)
    speaking_time_ms = _js_round(words / 130 * 60000)

    # Readability requires at least one word and one sentence.
    flesch_reading_ease: Optional[float] = None
    flesch_kincaid_grade: Optional[float] = None
    readability_label: Optional[str] = None
    if words > 0 and sentences > 0:
        words_per_sentence = words / sentences
        syllables_per_word = syllables / words
        flesch_reading_ease = _js_round(
            (206.835 - 1.015 * words_per_sentence - 84.6 * syllables_per_word) * 10
        ) / 10
        flesch_kincaid_grade = _js_round(
            (0.39 * words_per_sentence + 11.8 * syllables_per_word - 15.59) * 10
        ) / 10
        readability_label = _label_for_flesch(flesch_reading_ease)

    return TextStats(
        characters=characters,
        characters_no_spaces=characters_no_spaces,
        words=words,
        sentences=sentences,
        paragraphs=paragraphs,
        lines=lines,
        syllables=syllables,
        reading_time_ms=reading_time_ms,
        speaking_time_ms=speaking_time_ms,
        flesch_reading_ease=flesch_reading_ease,
        flesch_kincaid_grade=flesch_kincaid_grade,
        readability_label=readability_label,
    )

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →