Skip to content

Text Statistics & Readability — C++ source

Count words, sentences, paragraphs, characters, lines, and reading time, plus Flesch Reading Ease and Flesch-Kincaid grade-level readability scores.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// text-stats — text statistics & readability: chars, words, sentences, syllables, Flesch. Language: C++ (20). Port of src/lib/textStats.ts — chars count UTF-8 bytes & \s is ASCII-only here, the TS reference counts UTF-16 units & Unicode spaces.
#include <algorithm>
#include <cctype>
#include <cmath>
#include <cstdio>
#include <string>
#include <string_view>

namespace text_stats {

struct TextStats {
    int characters = 0, no_spaces = 0, words = 0, sentences = 0, paragraphs = 0, lines = 0, syllables = 0;
    int reading_ms = 0, speaking_ms = 0;
    double flesch_e = 0, flesch_k = 0;  // 0 when not computable (label == nullptr)
    const char* label = nullptr;
};

inline double js_round(double x) { return std::floor(x + 0.5); }  // JS Math.round: ties toward +infinity
inline bool in(std::string_view set, char c) { return set.find(c) != std::string_view::npos; }

// Word byte: [A-Za-z0-9'-] plus U+2019 (E2 80 99) — keeps contractions whole.
inline bool is_word_byte(std::string_view p) {
    unsigned char c = static_cast<unsigned char>(p[0]);
    if (std::isalnum(c) || c == '\'' || c == '-') return true;
    return p.size() >= 3 && (unsigned char)p[0] == 0xE2 && (unsigned char)p[1] == 0x80 && (unsigned char)p[2] == 0x99;
}

// Syllables in one word via the vowel-group heuristic (only ASCII a-z survive).
inline int count_syllables(std::string_view word) {
    std::string w;
    for (unsigned char c : word) { char l = std::tolower(c); if (l >= 'a' && l <= 'z') w += l; }
    if (w.empty()) return 0;
    if (w.size() <= 3) return 1;
    // drop a silent trailing e/ed/es — 'l' and vowels keep their syllable ("~le")
    if (w.size() >= 3 && w.ends_with("es") && !in("laeiouy", w[w.size() - 3])) w.resize(w.size() - 2);
    else if (w.ends_with("ed")) w.resize(w.size() - 2);
    else if (w.ends_with("e") && !in("laeiouy", w[w.size() - 2])) w.resize(w.size() - 1);
    if (w.starts_with('y')) w.erase(0, 1);  // drop a leading y
    int groups = 0; bool in_vowel = false;
    for (char c : w) { bool v = in("aeiouy", c); if (v && !in_vowel) groups++; in_vowel = v; }
    return std::max(1, groups);
}

inline const char* label_for(double f) {
    return f >= 80 ? "Very Easy" : f >= 70 ? "Easy" : f >= 60 ? "Standard" : f >= 50 ? "Fairly Hard" : f >= 30 ? "Hard" : "Very Hard";
}

inline TextStats analyze(std::string_view text) {
    TextStats s; s.characters = int(text.size());
    for (unsigned char c : text) if (!std::isspace(c)) s.no_spaces++;
    // words = maximal runs of word bytes; syllables sum per word
    for (size_t i = 0; i < text.size();) {
        if (!is_word_byte(text.substr(i))) { i++; continue; }
        size_t start = i;
        while (i < text.size() && is_word_byte(text.substr(i))) i += (unsigned char)text[i] == 0xE2 ? 3 : 1;
        s.words++;
        s.syllables += count_syllables(text.substr(start, i - start));
    }
    // sentences = runs of [.!?] followed by whitespace or end; min 1 when words > 0
    for (size_t i = 0; i < text.size();) {
        if (text[i] != '.' && text[i] != '!' && text[i] != '?') { i++; continue; }
        size_t j = i;
        while (j < text.size() && (text[j] == '.' || text[j] == '!' || text[j] == '?')) j++;
        if (j == text.size() || std::isspace((unsigned char)text[j])) s.sentences++;
        i = j;
    }
    if (s.words == 0) s.sentences = 0; else s.sentences = std::max(1, s.sentences);
    // paragraphs = non-blank chunks split on runs of 2+ newlines; lines = \n count + 1
    for (size_t i = 0; i < text.size();) {
        bool blank = true;
        while (i < text.size() && !(text[i] == '\n' && i + 1 < text.size() && text[i + 1] == '\n')) {
            if (!std::isspace((unsigned char)text[i])) blank = false;
            i++;
        }
        if (!blank) s.paragraphs++;
        while (i < text.size() && text[i] == '\n') i++;
    }
    if (!text.empty()) { s.lines = 1; for (char c : text) if (c == '\n') s.lines++; }
    s.reading_ms = int(js_round(s.words / 200.0 * 60000));   // 200 wpm
    s.speaking_ms = int(js_round(s.words / 130.0 * 60000));  // 130 wpm
    if (s.words > 0 && s.sentences > 0) {
        double wps = double(s.words) / s.sentences, spw = double(s.syllables) / s.words;
        s.flesch_e = js_round((206.835 - 1.015 * wps - 84.6 * spw) * 10) / 10;
        s.flesch_k = js_round((0.39 * wps + 11.8 * spw - 15.59) * 10) / 10;
        s.label = label_for(s.flesch_e);
    }
    return s;
}
}  // namespace text_stats

int main() {
    for (std::string_view text : { std::string_view("The quick brown fox jumps over the lazy dog."),
                                   std::string_view("Hi.\n\nMy name is Inigo Montoya. You killed my father; prepare to die!") }) {
        auto s = text_stats::analyze(text);
        std::printf("chars=%d nospace=%d words=%d sentences=%d paragraphs=%d lines=%d syllables=%d "
                    "reading=%dms speaking=%dms", s.characters, s.no_spaces, s.words, s.sentences,
                    s.paragraphs, s.lines, s.syllables, s.reading_ms, s.speaking_ms);
        if (s.label) std::printf(" flesch=%.1f (%s) grade=%.1f\n", s.flesch_e, s.label, s.flesch_k);
        else std::printf(" flesch=n/a\n");
    }
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →