RAG Chunk Comparator — Go source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Package ragchunkcomparator is the Go twin of CosmoDev's
// src/lib/ragChunkComparator.ts (dual source: the web lib is TypeScript, the
// CLI lib is Go — kept in lock-step). Pure + deterministic, never panics. The
// table-driven tests in ragchunkcomparator_test.go share vectors with
// src/lib/ragChunkComparator.test.ts so the two implementations are held to
// the same contract.
//
// Chunks one document three ways — fixed-size, sentence-aware, and
// markdown-heading-aware — and reports the stats that matter for retrieval:
// chunk count, size spread, and how often boundaries land on sentence ends
// (mid-sentence cuts are the classic recall killer). Token sizes use the
// tokenEstimator prose heuristic (the Go twin, cosmodev/tokenestimator).
//
// Where the TS lib throws RangeError on bad options, the Go twin returns a
// sentinel error (ErrBadSizeTokens / ErrBadOverlapTokens) with the same
// message. Fixed-chunk stepping indexes UTF-16 code units — the unit TS's
// String.length counts — so astral-plane text chunks identically on both sides.
package ragchunkcomparator
import (
"errors"
"math"
"regexp"
"strings"
"unicode/utf16"
"cosmodev/token-estimator"
)
// ChunkStrategy is the chunking strategy under test. Mirrors the TS union
// 'fixed' | 'sentence' | 'markdown' (string literals, so JSON output matches).
type ChunkStrategy string
const (
StrategyFixed ChunkStrategy = "fixed"
StrategySentence ChunkStrategy = "sentence"
StrategyMarkdown ChunkStrategy = "markdown"
)
// Options configures the chunkers. Mirrors the TS ChunkOptions interface.
// The zero value keeps TS's overlapTokens default of 0; SizeTokens has no TS
// default and must be set to > 0 (zero mirrors TS's missing sizeTokens, which
// throws — here ErrBadSizeTokens).
type Options struct {
// SizeTokens is the target chunk size in tokens. Must be > 0.
SizeTokens int
// OverlapTokens is the overlap between consecutive fixed chunks, in
// tokens (fixed strategy only). Must be in [0, SizeTokens).
OverlapTokens int
}
// Chunk is one piece of the split document. Field-for-field twin of the TS
// Chunk interface (camelCase JSON tags, tokenestimator convention).
type Chunk struct {
Index int `json:"index"`
Text string `json:"text"`
Tokens int `json:"tokens"`
Heading *string `json:"heading,omitempty"` // nearest markdown heading for markdown chunks; nil otherwise
}
// StrategyStats are the comparable per-strategy numbers. Mirrors the TS
// StrategyStats interface.
type StrategyStats struct {
Count int `json:"count"`
MinTokens int `json:"minTokens"`
MaxTokens int `json:"maxTokens"`
AvgTokens int `json:"avgTokens"`
SentenceBoundaryShare float64 `json:"sentenceBoundaryShare"` // share of chunk boundaries that fall on a sentence end (0–1)
}
// StrategyResult pairs a strategy with its chunks and stats. Mirrors the TS
// StrategyResult interface.
type StrategyResult struct {
Strategy ChunkStrategy `json:"strategy"`
Chunks []Chunk `json:"chunks"`
Stats StrategyStats `json:"stats"`
}
// Comparison is the result of CompareStrategies: one entry per strategy.
// Mirrors the TS Record<ChunkStrategy, StrategyResult>.
type Comparison struct {
Fixed StrategyResult `json:"fixed"`
Sentence StrategyResult `json:"sentence"`
Markdown StrategyResult `json:"markdown"`
}
// Sentinel errors carrying the TS RangeError messages verbatim.
var (
ErrBadSizeTokens = errors.New("sizeTokens must be > 0")
ErrBadOverlapTokens = errors.New("overlapTokens must be in [0, sizeTokens)")
)
var (
whitespaceRun = regexp.MustCompile(`\s+`)
headingLine = regexp.MustCompile(`^(#{1,6})\s+(.*)$`)
// Matches text whose last character is a sentence ender, optionally
// wrapped in a closing quote/bracket. Twin of /[.!?]["')\]]?$/.
sentenceEnd = regexp.MustCompile(`[.!?]["')\]]?$`)
)
// tok estimates prose tokens via the tokenEstimator twin of the TS
// estimateTokens(s, { type: 'prose' }).tokens call.
func tok(s string) int {
return tokenestimator.EstimateTokens(s, &tokenestimator.EstimateOptions{
ContentType: tokenestimator.TypeProse,
}).Tokens
}
// SplitSentences splits on sentence enders followed by whitespace or end of
// text. It is the Go twin of splitSentences() in src/lib/ragChunkComparator.ts
// and must agree with it on every shared vector. RE2 has no lookbehind, so the
// TS split(/(?<=[.!?]) +/) is re-expressed as a scan that cuts before each
// run of spaces preceded by an ender.
func SplitSentences(text string) []string {
clean := strings.TrimSpace(whitespaceRun.ReplaceAllString(text, " "))
out := []string{}
start := 0
for i := 0; i < len(clean); i++ {
if clean[i] != ' ' || i <= start || !isEnderByte(clean[i-1]) {
continue
}
j := i
for j < len(clean) && clean[j] == ' ' {
j++
}
out = append(out, clean[start:i])
start = j
i = j - 1
}
if start < len(clean) {
out = append(out, clean[start:])
}
return out
}
// isEnderByte reports whether b is '.', '!', or '?'.
func isEnderByte(b byte) bool { return b == '.' || b == '!' || b == '?' }
// endsSentence reports whether s (trimmed) ends on a sentence ender.
func endsSentence(s string) bool { return sentenceEnd.MatchString(strings.TrimSpace(s)) }
// lastSpaceAt returns the highest index <= end holding a space, or -1 —
// the twin of String.prototype.lastIndexOf(' ', end) over UTF-16 units.
func lastSpaceAt(u []uint16, end int) int {
for i := end; i >= 0; i-- {
if u[i] == ' ' {
return i
}
}
return -1
}
// ChunkFixed greedily accumulates characters to a token target (overlapping
// allowed). It is the Go twin of chunkFixed() in src/lib/ragChunkComparator.ts
// and must agree with it on every shared vector.
func ChunkFixed(text string, opts Options) ([]Chunk, error) {
if opts.SizeTokens <= 0 {
return nil, ErrBadSizeTokens
}
if opts.OverlapTokens < 0 || opts.OverlapTokens >= opts.SizeTokens {
return nil, ErrBadOverlapTokens
}
clean := strings.TrimSpace(text)
if clean == "" {
return []Chunk{}, nil
}
// ~4 chars per prose token: step by tokens, verify with the estimator.
u := utf16.Encode([]rune(clean))
charStep := opts.SizeTokens * 4 // Math.round of a whole multiple is identity
overlapChars := opts.OverlapTokens * 4
chunks := []Chunk{}
start := 0
for start < len(u) {
end := min(start+charStep, len(u))
// Prefer cutting at whitespace near the target — but never trim the
// document's final piece back to a word when it already fits.
if end < len(u) {
if cut := lastSpaceAt(u, end); cut > start {
end = cut
}
}
piece := strings.TrimSpace(string(utf16.Decode(u[start:end])))
if piece != "" {
chunks = append(chunks, Chunk{Index: len(chunks), Text: piece, Tokens: tok(piece)})
}
if end >= len(u) {
break
}
start = max(end-overlapChars, start+1)
}
return chunks, nil
}
// ChunkBySentences groups whole sentences up to the token target; boundaries
// never split a sentence. It is the Go twin of chunkBySentences() in
// src/lib/ragChunkComparator.ts and must agree with it on every shared vector.
func ChunkBySentences(text string, opts Options) ([]Chunk, error) {
if opts.SizeTokens <= 0 {
return nil, ErrBadSizeTokens
}
sentences := SplitSentences(text)
if len(sentences) == 0 {
return []Chunk{}, nil
}
chunks := []Chunk{}
var current []string
currentTokens := 0
flush := func() {
// Guard is defensive: the loop always leaves the last sentence
// pending, so the final flush is never empty by construction.
if len(current) == 0 {
return
}
piece := strings.Join(current, " ")
chunks = append(chunks, Chunk{Index: len(chunks), Text: piece, Tokens: tok(piece)})
current = nil
currentTokens = 0
}
for _, sentence := range sentences {
t := tok(sentence)
if currentTokens > 0 && currentTokens+t > opts.SizeTokens {
flush()
}
current = append(current, sentence)
currentTokens += t
// A single sentence larger than the target becomes its own chunk.
}
flush()
return chunks, nil
}
// ChunkMarkdown splits on markdown headings; oversized sections fall back to
// sentence grouping. It is the Go twin of chunkMarkdown() in
// src/lib/ragChunkComparator.ts and must agree with it on every shared vector.
func ChunkMarkdown(text string, opts Options) ([]Chunk, error) {
if opts.SizeTokens <= 0 {
return nil, ErrBadSizeTokens
}
type section struct {
heading *string
body []string
}
var sections []section
current := section{heading: nil, body: []string{}}
for _, line := range strings.Split(text, "\n") {
if m := headingLine.FindStringSubmatch(line); m != nil {
if len(current.body) > 0 {
sections = append(sections, current)
}
h := strings.TrimSpace(m[2])
current = section{heading: &h, body: []string{}}
} else {
current.body = append(current.body, line)
}
}
if len(current.body) > 0 {
sections = append(sections, current)
}
chunks := []Chunk{}
for _, sec := range sections {
body := strings.TrimSpace(strings.Join(sec.body, "\n"))
if body == "" {
continue
}
whole := body
if sec.heading != nil {
whole = "# " + *sec.heading + "\n" + body
}
if tok(whole) <= opts.SizeTokens {
chunks = append(chunks, Chunk{
Index: len(chunks), Text: whole, Tokens: tok(whole), Heading: sec.heading,
})
continue
}
// Oversized section: sentence-group the body, stamp every chunk with the heading.
sub, err := ChunkBySentences(body, opts)
if err != nil {
return nil, err // unreachable here: opts were validated above
}
for _, c := range sub {
chunks = append(chunks, Chunk{
Index: len(chunks), Text: c.Text, Tokens: c.Tokens, Heading: sec.heading,
})
}
}
return chunks, nil
}
// statsFor summarizes one strategy's chunks. Mirrors the TS statsFor helper.
func statsFor(strategy ChunkStrategy, chunks []Chunk) StrategyResult {
count := len(chunks)
minTokens, maxTokens, sum := 0, 0, 0
if count > 0 {
minTokens, maxTokens = chunks[0].Tokens, chunks[0].Tokens
for _, c := range chunks {
minTokens = min(minTokens, c.Tokens)
maxTokens = max(maxTokens, c.Tokens)
sum += c.Tokens
}
}
avgTokens := 0
if count > 0 {
avgTokens = int(math.Round(float64(sum) / float64(count)))
}
share := 1.0 // a single chunk has no internal boundaries to botch
if boundaries := count - 1; boundaries > 0 {
ends := 0
for _, c := range chunks[:count-1] {
if endsSentence(c.Text) {
ends++
}
}
share = float64(ends) / float64(boundaries)
}
return StrategyResult{
Strategy: strategy,
Chunks: chunks,
Stats: StrategyStats{
Count: count, MinTokens: minTokens, MaxTokens: maxTokens,
AvgTokens: avgTokens, SentenceBoundaryShare: share,
},
}
}
// CompareStrategies runs all three strategies over one document and reports
// comparable stats. It is the Go twin of compareStrategies() in
// src/lib/ragChunkComparator.ts and must agree with it on every shared vector.
func CompareStrategies(text string, opts Options) (Comparison, error) {
fixed, err := ChunkFixed(text, opts)
if err != nil {
return Comparison{}, err
}
sentence, err := ChunkBySentences(text, opts)
if err != nil {
return Comparison{}, err
}
markdown, err := ChunkMarkdown(text, opts)
if err != nil {
return Comparison{}, err
}
return Comparison{
Fixed: statsFor(StrategyFixed, fixed),
Sentence: statsFor(StrategySentence, sentence),
Markdown: statsFor(StrategyMarkdown, markdown),
}, nil
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →