Skip to content

Text Statistics & Readability — Zig source

Count words, sentences, paragraphs, characters, lines, and reading time, plus Flesch Reading Ease and Flesch-Kincaid grade-level readability scores.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// text-stats — text statistics & readability: chars, words, sentences, syllables, Flesch. Language: Zig (0.12+). Port of src/lib/textStats.ts — chars count UTF-8 bytes & whitespace is ASCII-only here, the TS reference counts UTF-16 units & Unicode spaces.
const std = @import("std");

const TextStats = struct {
    characters: usize = 0, characters_no_spaces: usize = 0, words: usize = 0, sentences: usize = 0,
    paragraphs: usize = 0, lines: usize = 0, syllables: usize = 0, reading_ms: usize = 0, speaking_ms: usize = 0,
    flesch_e: ?f64 = null, flesch_k: ?f64 = null, label: ?[]const u8 = null,
};

fn jsRound(x: f64) f64 { return @floor(x + 0.5); } // JS Math.round: ties toward +infinity

fn inSet(set: []const u8, c: u8) bool { return std.mem.indexOfScalar(u8, set, c) != null; }

fn isWordStart(p: []const u8) bool { // [A-Za-z0-9'-] + U+2019 (E2 80 99) — keeps contractions whole
    const c = p[0];
    if (std.ascii.isAlphanumeric(c) or c == '\'' or c == '-') return true;
    return c == 0xE2 and p.len >= 3 and p[1] == 0x80 and p[2] == 0x99;
}

fn countSyllables(word: []const u8) usize { // vowel-group heuristic; only ASCII a-z survive
    var buf: [64]u8 = undefined;
    var n: usize = 0;
    for (word) |c| {
        const l = std.ascii.toLower(c);
        if (l >= 'a' and l <= 'z' and n < buf.len) { buf[n] = l; n += 1; }
    }
    if (n == 0) return 0;
    if (n <= 3) return 1;
    // drop a silent trailing e/ed/es — 'l' and vowels keep their syllable ("~le")
    if (n >= 3 and buf[n - 1] == 's' and buf[n - 2] == 'e' and !inSet("laeiouy", buf[n - 3])) n -= 2
    else if (n >= 2 and buf[n - 1] == 'd' and buf[n - 2] == 'e') n -= 2
    else if (n >= 2 and buf[n - 1] == 'e' and !inSet("laeiouy", buf[n - 2])) n -= 1;
    var i: usize = if (buf[0] == 'y') 1 else 0; // drop a leading y
    var groups: usize = 0;
    var prev = false;
    while (i < n) : (i += 1) {
        const v = inSet("aeiouy", buf[i]);
        if (v and !prev) groups += 1;
        prev = v;
    }
    return @max(1, groups);
}

fn labelFor(f: f64) []const u8 {
    if (f >= 80) return "Very Easy";
    if (f >= 70) return "Easy";
    if (f >= 60) return "Standard";
    if (f >= 50) return "Fairly Hard";
    if (f >= 30) return "Hard";
    return "Very Hard";
}

fn analyze(text: []const u8) TextStats {
    var st = TextStats{ .characters = text.len };
    for (text) |c| if (!std.ascii.isWhitespace(c)) {
        st.characters_no_spaces += 1;
    };
    var i: usize = 0;
    while (i < text.len) { // words = maximal runs of word bytes; syllables sum per word
        if (!isWordStart(text[i..])) { i += 1; continue; }
        const start = i;
        while (i < text.len and isWordStart(text[i..])) i += if (text[i] == 0xE2) @as(usize, 3) else 1;
        st.words += 1;
        st.syllables += countSyllables(text[start..i]);
    }
    i = 0;
    while (i < text.len) { // sentences = runs of [.!?] followed by whitespace or end; min 1 when words > 0
        const c = text[i];
        if (c != '.' and c != '!' and c != '?') { i += 1; continue; }
        while (i < text.len and (text[i] == '.' or text[i] == '!' or text[i] == '?')) i += 1;
        if (i == text.len or std.ascii.isWhitespace(text[i])) st.sentences += 1;
    }
    if (st.words == 0) st.sentences = 0 else if (st.sentences == 0) st.sentences = 1;
    var blank = true; // paragraphs = non-blank chunks split on runs of 2+ newlines
    var nl_run: usize = 0;
    for (text) |c| {
        if (c == '\n') {
            nl_run += 1;
            if (nl_run >= 2 and !blank) { st.paragraphs += 1; blank = true; }
        } else {
            if (!std.ascii.isWhitespace(c)) blank = false;
            nl_run = 0;
        }
    }
    if (!blank) st.paragraphs += 1;
    if (text.len > 0) { // lines = 0 when empty, else newline count + 1
        st.lines = 1;
        for (text) |c| if (c == '\n') {
            st.lines += 1;
        };
    }
    st.reading_ms = @intFromFloat(jsRound(@as(f64, @floatFromInt(st.words)) / 200.0 * 60000)); // 200 wpm
    st.speaking_ms = @intFromFloat(jsRound(@as(f64, @floatFromInt(st.words)) / 130.0 * 60000)); // 130 wpm
    if (st.words > 0 and st.sentences > 0) {
        const wps = @as(f64, @floatFromInt(st.words)) / @as(f64, @floatFromInt(st.sentences));
        const spw = @as(f64, @floatFromInt(st.syllables)) / @as(f64, @floatFromInt(st.words));
        st.flesch_e = jsRound((206.835 - 1.015 * wps - 84.6 * spw) * 10) / 10;
        st.flesch_k = jsRound((0.39 * wps + 11.8 * spw - 15.59) * 10) / 10;
        st.label = labelFor(st.flesch_e.?);
    }
    return st;
}

pub fn main() void {
    const samples = [_][]const u8{
        "The quick brown fox jumps over the lazy dog.",
        "Hi.\n\nMy name is Inigo Montoya. You killed my father; prepare to die!",
    };
    for (samples) |text| {
        const s = analyze(text);
        std.debug.print("chars={d} nospace={d} words={d} sentences={d} paragraphs={d} lines={d} syllables={d} reading={d}ms speaking={d}ms", .{ s.characters, s.characters_no_spaces, s.words, s.sentences, s.paragraphs, s.lines, s.syllables, s.reading_ms, s.speaking_ms });
        if (s.label) |l| std.debug.print(" flesch={d:.1} ({s}) grade={d:.1}\n", .{ s.flesch_e.?, l, s.flesch_k.? }) else std.debug.print(" flesch=n/a\n", .{});
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →