Skip to content

Text Statistics & Readability — Go source

Count words, sentences, paragraphs, characters, lines, and reading time, plus Flesch Reading Ease and Flesch-Kincaid grade-level readability scores.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package textstats is the Go twin of CosmoDev's src/lib/textStats.ts (dual
// source: the web lib is TypeScript, the CLI lib is Go — kept in lock-step).
// Pure + deterministic, never panics. The table-driven tests in
// text-stats_test.go share vectors with src/lib/textStats.test.ts so the two
// implementations are held to the same contract.
//
// The twin mirrors the TS lib exactly: count characters (UTF-16 code units,
// like JS string .length), words, sentences, paragraphs, lines and syllables,
// then derive reading/speaking time and Flesch readability scores.
package textstats

import (
	"math"
	"regexp"
	"strings"
	"unicode"
)

// TextStats is the Go twin of the TextStats interface in src/lib/textStats.ts.
// The three readability fields are pointers so they can represent the TS
// number|null / string|null union: nil when there are too few words or
// sentences to compute a score.
type TextStats struct {
	Characters         int
	CharactersNoSpaces int
	Words              int
	Sentences          int
	Paragraphs         int
	Lines              int
	Syllables          int
	ReadingTimeMs      int // words / 200 wpm
	SpeakingTimeMs     int // words / 130 wpm
	FleschReadingEase  *float64
	FleschKincaidGrade *float64
	ReadabilityLabel   *string
}

var (
	// wordRe matches a run of word characters, mirroring the TS lib's
	// /[A-Za-z0-9''-]+/ (ASCII letters/digits, apostrophe, right single
	// quote U+2019, hyphen).
	wordRe = regexp.MustCompile("[A-Za-z0-9'’-]+")
	// sentenceRe counts sentence terminators followed by whitespace or end
	// of text, mirroring the TS lib's /[.!?]+(?:\s|$)/.
	sentenceRe = regexp.MustCompile(`[.!?]+(?:\s|$)`)
	// paragraphSepRe splits paragraphs on 2+ newlines, mirroring /\n{2,}/.
	paragraphSepRe = regexp.MustCompile(`\n{2,}`)

	// Syllable heuristics (see CountSyllables), mirroring the regexes in
	// the TS countSyllables(): drop a silent trailing e/ed/es, drop a
	// leading y, then count vowel groups.
	sylTrailingE = regexp.MustCompile(`(?:[^laeiouy]es|ed|[^laeiouy]e)$`)
	sylLeadingY  = regexp.MustCompile(`^y`)
	sylVowels    = regexp.MustCompile(`[aeiouy]+`)
)

// utf16UnitCount returns the number of UTF-16 code units in s, matching
// JavaScript's String .length (each code point below U+10000 is one unit; a
// supplementary-plane code point forms a surrogate pair → two units). The TS
// lib sizes its character counts with string .length, so the twin does too.
func utf16UnitCount(s string) int {
	n := 0
	for _, r := range s {
		if r >= 0x10000 {
			n += 2
		} else {
			n++
		}
	}
	return n
}

// CountSyllables estimates the syllable count of a single word via a
// vowel-group heuristic. It is the Go twin of countSyllables() in
// src/lib/textStats.ts and must agree with it on every shared vector. It
// returns 0 for a word with no letters, otherwise at least 1.
func CountSyllables(word string) int {
	// word.toLowerCase().replace(/[^a-z]/g, '') — keep ASCII a-z only.
	var b strings.Builder
	for _, r := range strings.ToLower(word) {
		if r >= 'a' && r <= 'z' {
			b.WriteRune(r)
		}
	}
	w := b.String()
	if w == "" {
		return 0
	}
	if len(w) <= 3 {
		return 1
	}
	// Drop a silent trailing e/ed/es (but not after l or a vowel), then a
	// leading y, then count vowel groups.
	s := sylTrailingE.ReplaceAllString(w, "")
	s = sylLeadingY.ReplaceAllString(s, "")
	count := 1
	if groups := sylVowels.FindAllString(s, -1); groups != nil {
		count = len(groups)
	}
	return max(count, 1)
}

// labelForFlesch maps a Flesch reading-ease score to a human-readable band. It
// mirrors the unexported labelForFlesch() in src/lib/textStats.ts.
func labelForFlesch(f float64) string {
	switch {
	case f >= 80:
		return "Very Easy"
	case f >= 70:
		return "Easy"
	case f >= 60:
		return "Standard"
	case f >= 50:
		return "Fairly Hard"
	case f >= 30:
		return "Hard"
	default:
		return "Very Hard"
	}
}

// AnalyzeText computes text statistics and readability scores for input. It is
// the Go twin of analyzeText() in src/lib/textStats.ts and must agree with it
// on every shared vector. It never panics; the TS null-input contract (input ??
// '') is modeled by the empty string, since Go has no null string.
func AnalyzeText(input string) TextStats {
	text := input

	stats := TextStats{
		Characters: utf16UnitCount(text),
	}

	// charactersNoSpaces = text.replace(/\s/g, '').length
	noSpace := 0
	for _, r := range text {
		if unicode.IsSpace(r) {
			continue
		}
		if r >= 0x10000 {
			noSpace += 2
		} else {
			noSpace++
		}
	}
	stats.CharactersNoSpaces = noSpace

	wordList := wordRe.FindAllString(text, -1)
	stats.Words = len(wordList)

	// sentences = words === 0 ? 0 : max(1, terminal-punctuation matches)
	if stats.Words == 0 {
		stats.Sentences = 0
	} else {
		stats.Sentences = max(len(sentenceRe.FindAllString(text, -1)), 1)
	}

	// paragraphs = trimmed text empty ? 0 : non-empty chunks split on \n{2,}
	if strings.TrimSpace(text) == "" {
		stats.Paragraphs = 0
	} else {
		for _, p := range paragraphSepRe.Split(text, -1) {
			if strings.TrimSpace(p) != "" {
				stats.Paragraphs++
			}
		}
	}

	// lines = text === '' ? 0 : text.split('\n').length
	if text == "" {
		stats.Lines = 0
	} else {
		stats.Lines = strings.Count(text, "\n") + 1
	}

	// syllables = sum of CountSyllables over each word
	syllables := 0
	for _, w := range wordList {
		syllables += CountSyllables(w)
	}
	stats.Syllables = syllables

	// reading/speaking time, rounded like Math.round.
	stats.ReadingTimeMs = int(math.Round((float64(stats.Words) / 200.0) * 60000.0))
	stats.SpeakingTimeMs = int(math.Round((float64(stats.Words) / 130.0) * 60000.0))

	// Flesch scores only when there are words AND sentences.
	if stats.Words > 0 && stats.Sentences > 0 {
		wordsPerSentence := float64(stats.Words) / float64(stats.Sentences)
		syllablesPerWord := float64(syllables) / float64(stats.Words)
		fre := math.Round((206.835-1.015*wordsPerSentence-84.6*syllablesPerWord)*10) / 10
		fkg := math.Round((0.39*wordsPerSentence+11.8*syllablesPerWord-15.59)*10) / 10
		label := labelForFlesch(fre)
		stats.FleschReadingEase = &fre
		stats.FleschKincaidGrade = &fkg
		stats.ReadabilityLabel = &label
	}

	return stats
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →