Text Statistics & Readability — C source
Count words, sentences, paragraphs, characters, lines, and reading time, plus Flesch Reading Ease and Flesch-Kincaid grade-level readability scores.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/* text-stats — text statistics & readability: chars, words, sentences, syllables, Flesch. Language: C (C11). Port of src/lib/textStats.ts — chars count UTF-8 bytes & \s is ASCII-only here, the TS reference counts UTF-16 units & Unicode spaces. */
#include <ctype.h>
#include <math.h>
#include <stdio.h>
#include <string.h>
typedef struct {
int characters, characters_no_spaces, words, sentences, paragraphs, lines, syllables;
int reading_ms, speaking_ms;
double flesch_e, flesch_k; /* 0 when not computable (label NULL) */
const char *label;
} TextStats;
static double js_round(double x) { return floor(x + 0.5); } /* JS Math.round: ties toward +infinity */
/* Word byte: [A-Za-z0-9'-] plus U+2019 (UTF-8 E2 80 99) — keeps contractions whole. */
static int is_word_byte(const unsigned char *p) {
if (isalnum(*p) || *p == '\'' || *p == '-') return 1;
return p[0] == 0xE2 && p[1] == 0x80 && p[2] == 0x99;
}
/* Syllables in one word via the vowel-group heuristic (only ASCII a-z survive). */
static int count_syllables(const char *word) {
char w[64]; size_t n = 0;
for (const unsigned char *p = (const unsigned char *)word; *p && n < sizeof w; p++)
if (islower(*p)) w[n++] = (char)*p;
else if (isupper(*p)) w[n++] = (char)tolower(*p);
if (n == 0) return 0;
if (n <= 3) return 1;
/* drop a silent trailing e/ed/es — 'l' and vowels keep their syllable ("~le") */
if (n >= 3 && w[n-1] == 's' && w[n-2] == 'e' && !strchr("laeiouy", w[n-3])) n -= 2;
else if (n >= 2 && w[n-1] == 'd' && w[n-2] == 'e') n -= 2;
else if (n >= 2 && w[n-1] == 'e' && !strchr("laeiouy", w[n-2])) n -= 1;
size_t i = w[0] == 'y' ? 1 : 0; /* drop a leading y */
int groups = 0, in_vowel = 0;
for (; i < n; i++) { int v = strchr("aeiouy", w[i]) != NULL; if (v && !in_vowel) groups++; in_vowel = v; }
return groups > 1 ? groups : 1;
}
static const char *label_for(double f) {
return f >= 80 ? "Very Easy" : f >= 70 ? "Easy" : f >= 60 ? "Standard"
: f >= 50 ? "Fairly Hard" : f >= 30 ? "Hard" : "Very Hard";
}
static TextStats analyze(const char *text) {
TextStats s = {0};
size_t len = strlen(text);
s.characters = (int)len;
for (size_t i = 0; i < len; i++) if (!isspace((unsigned char)text[i])) s.characters_no_spaces++;
/* words = maximal runs of word bytes; syllables sum per word */
const unsigned char *p = (const unsigned char *)text;
while (*p) {
if (!is_word_byte(p)) { p++; continue; }
char word[64]; size_t n = 0;
while (is_word_byte(p) && n < sizeof word - 1) { word[n++] = (char)*p; p += *p == 0xE2 ? 3 : 1; }
word[n] = '\0';
s.words++;
s.syllables += count_syllables(word);
}
/* sentences = runs of [.!?] followed by whitespace or end; min 1 when words > 0 */
for (size_t i = 0; i < len;) {
if (text[i] != '.' && text[i] != '!' && text[i] != '?') { i++; continue; }
size_t j = i;
while (j < len && (text[j] == '.' || text[j] == '!' || text[j] == '?')) j++;
if (j == len || isspace((unsigned char)text[j])) s.sentences++;
i = j;
}
if (s.words == 0) s.sentences = 0;
else if (s.sentences == 0) s.sentences = 1;
/* paragraphs = non-blank chunks split on runs of 2+ newlines; lines = \n count + 1 */
for (size_t i = 0; i < len;) {
int blank = 1;
while (i < len && !(text[i] == '\n' && i + 1 < len && text[i + 1] == '\n')) {
if (!isspace((unsigned char)text[i])) blank = 0;
i++;
}
if (!blank) s.paragraphs++;
while (i < len && text[i] == '\n') i++;
}
if (len > 0) { s.lines = 1; for (size_t i = 0; i < len; i++) if (text[i] == '\n') s.lines++; }
s.reading_ms = (int)js_round(s.words / 200.0 * 60000.0); /* 200 wpm */
s.speaking_ms = (int)js_round(s.words / 130.0 * 60000.0); /* 130 wpm */
if (s.words > 0 && s.sentences > 0) {
double wps = (double)s.words / s.sentences, spw = (double)s.syllables / s.words;
s.flesch_e = js_round((206.835 - 1.015 * wps - 84.6 * spw) * 10) / 10;
s.flesch_k = js_round((0.39 * wps + 11.8 * spw - 15.59) * 10) / 10;
s.label = label_for(s.flesch_e);
}
return s;
}
int main(void) {
const char *samples[] = {
"The quick brown fox jumps over the lazy dog.",
"Hi.\n\nMy name is Inigo Montoya. You killed my father; prepare to die!",
};
for (size_t i = 0; i < sizeof samples / sizeof *samples; i++) {
TextStats s = analyze(samples[i]);
printf("chars=%d nospace=%d words=%d sentences=%d paragraphs=%d lines=%d syllables=%d "
"reading=%dms speaking=%dms", s.characters, s.characters_no_spaces, s.words,
s.sentences, s.paragraphs, s.lines, s.syllables, s.reading_ms, s.speaking_ms);
if (s.label) printf(" flesch=%.1f (%s) grade=%.1f\n", s.flesch_e, s.label, s.flesch_k);
else printf(" flesch=n/a\n");
}
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →