Skip to content

RAG Chunk Comparator — Go source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package ragchunkcomparator is the Go twin of CosmoDev's
// src/lib/ragChunkComparator.ts (dual source: the web lib is TypeScript, the
// CLI lib is Go — kept in lock-step). Pure + deterministic, never panics. The
// table-driven tests in ragchunkcomparator_test.go share vectors with
// src/lib/ragChunkComparator.test.ts so the two implementations are held to
// the same contract.
//
// Chunks one document three ways — fixed-size, sentence-aware, and
// markdown-heading-aware — and reports the stats that matter for retrieval:
// chunk count, size spread, and how often boundaries land on sentence ends
// (mid-sentence cuts are the classic recall killer). Token sizes use the
// tokenEstimator prose heuristic (the Go twin, cosmodev/tokenestimator).
//
// Where the TS lib throws RangeError on bad options, the Go twin returns a
// sentinel error (ErrBadSizeTokens / ErrBadOverlapTokens) with the same
// message. Fixed-chunk stepping indexes UTF-16 code units — the unit TS's
// String.length counts — so astral-plane text chunks identically on both sides.
package ragchunkcomparator

import (
	"errors"
	"math"
	"regexp"
	"strings"
	"unicode/utf16"

	"cosmodev/token-estimator"
)

// ChunkStrategy is the chunking strategy under test. Mirrors the TS union
// 'fixed' | 'sentence' | 'markdown' (string literals, so JSON output matches).
type ChunkStrategy string

const (
	StrategyFixed    ChunkStrategy = "fixed"
	StrategySentence ChunkStrategy = "sentence"
	StrategyMarkdown ChunkStrategy = "markdown"
)

// Options configures the chunkers. Mirrors the TS ChunkOptions interface.
// The zero value keeps TS's overlapTokens default of 0; SizeTokens has no TS
// default and must be set to > 0 (zero mirrors TS's missing sizeTokens, which
// throws — here ErrBadSizeTokens).
type Options struct {
	// SizeTokens is the target chunk size in tokens. Must be > 0.
	SizeTokens int
	// OverlapTokens is the overlap between consecutive fixed chunks, in
	// tokens (fixed strategy only). Must be in [0, SizeTokens).
	OverlapTokens int
}

// Chunk is one piece of the split document. Field-for-field twin of the TS
// Chunk interface (camelCase JSON tags, tokenestimator convention).
type Chunk struct {
	Index   int     `json:"index"`
	Text    string  `json:"text"`
	Tokens  int     `json:"tokens"`
	Heading *string `json:"heading,omitempty"` // nearest markdown heading for markdown chunks; nil otherwise
}

// StrategyStats are the comparable per-strategy numbers. Mirrors the TS
// StrategyStats interface.
type StrategyStats struct {
	Count                 int     `json:"count"`
	MinTokens             int     `json:"minTokens"`
	MaxTokens             int     `json:"maxTokens"`
	AvgTokens             int     `json:"avgTokens"`
	SentenceBoundaryShare float64 `json:"sentenceBoundaryShare"` // share of chunk boundaries that fall on a sentence end (0–1)
}

// StrategyResult pairs a strategy with its chunks and stats. Mirrors the TS
// StrategyResult interface.
type StrategyResult struct {
	Strategy ChunkStrategy `json:"strategy"`
	Chunks   []Chunk       `json:"chunks"`
	Stats    StrategyStats `json:"stats"`
}

// Comparison is the result of CompareStrategies: one entry per strategy.
// Mirrors the TS Record<ChunkStrategy, StrategyResult>.
type Comparison struct {
	Fixed    StrategyResult `json:"fixed"`
	Sentence StrategyResult `json:"sentence"`
	Markdown StrategyResult `json:"markdown"`
}

// Sentinel errors carrying the TS RangeError messages verbatim.
var (
	ErrBadSizeTokens    = errors.New("sizeTokens must be > 0")
	ErrBadOverlapTokens = errors.New("overlapTokens must be in [0, sizeTokens)")
)

var (
	whitespaceRun = regexp.MustCompile(`\s+`)
	headingLine   = regexp.MustCompile(`^(#{1,6})\s+(.*)$`)
	// Matches text whose last character is a sentence ender, optionally
	// wrapped in a closing quote/bracket. Twin of /[.!?]["')\]]?$/.
	sentenceEnd = regexp.MustCompile(`[.!?]["')\]]?$`)
)

// tok estimates prose tokens via the tokenEstimator twin of the TS
// estimateTokens(s, { type: 'prose' }).tokens call.
func tok(s string) int {
	return tokenestimator.EstimateTokens(s, &tokenestimator.EstimateOptions{
		ContentType: tokenestimator.TypeProse,
	}).Tokens
}

// SplitSentences splits on sentence enders followed by whitespace or end of
// text. It is the Go twin of splitSentences() in src/lib/ragChunkComparator.ts
// and must agree with it on every shared vector. RE2 has no lookbehind, so the
// TS split(/(?<=[.!?]) +/) is re-expressed as a scan that cuts before each
// run of spaces preceded by an ender.
func SplitSentences(text string) []string {
	clean := strings.TrimSpace(whitespaceRun.ReplaceAllString(text, " "))
	out := []string{}
	start := 0
	for i := 0; i < len(clean); i++ {
		if clean[i] != ' ' || i <= start || !isEnderByte(clean[i-1]) {
			continue
		}
		j := i
		for j < len(clean) && clean[j] == ' ' {
			j++
		}
		out = append(out, clean[start:i])
		start = j
		i = j - 1
	}
	if start < len(clean) {
		out = append(out, clean[start:])
	}
	return out
}

// isEnderByte reports whether b is '.', '!', or '?'.
func isEnderByte(b byte) bool { return b == '.' || b == '!' || b == '?' }

// endsSentence reports whether s (trimmed) ends on a sentence ender.
func endsSentence(s string) bool { return sentenceEnd.MatchString(strings.TrimSpace(s)) }

// lastSpaceAt returns the highest index <= end holding a space, or -1 —
// the twin of String.prototype.lastIndexOf(' ', end) over UTF-16 units.
func lastSpaceAt(u []uint16, end int) int {
	for i := end; i >= 0; i-- {
		if u[i] == ' ' {
			return i
		}
	}
	return -1
}

// ChunkFixed greedily accumulates characters to a token target (overlapping
// allowed). It is the Go twin of chunkFixed() in src/lib/ragChunkComparator.ts
// and must agree with it on every shared vector.
func ChunkFixed(text string, opts Options) ([]Chunk, error) {
	if opts.SizeTokens <= 0 {
		return nil, ErrBadSizeTokens
	}
	if opts.OverlapTokens < 0 || opts.OverlapTokens >= opts.SizeTokens {
		return nil, ErrBadOverlapTokens
	}
	clean := strings.TrimSpace(text)
	if clean == "" {
		return []Chunk{}, nil
	}
	// ~4 chars per prose token: step by tokens, verify with the estimator.
	u := utf16.Encode([]rune(clean))
	charStep := opts.SizeTokens * 4 // Math.round of a whole multiple is identity
	overlapChars := opts.OverlapTokens * 4
	chunks := []Chunk{}
	start := 0
	for start < len(u) {
		end := min(start+charStep, len(u))
		// Prefer cutting at whitespace near the target — but never trim the
		// document's final piece back to a word when it already fits.
		if end < len(u) {
			if cut := lastSpaceAt(u, end); cut > start {
				end = cut
			}
		}
		piece := strings.TrimSpace(string(utf16.Decode(u[start:end])))
		if piece != "" {
			chunks = append(chunks, Chunk{Index: len(chunks), Text: piece, Tokens: tok(piece)})
		}
		if end >= len(u) {
			break
		}
		start = max(end-overlapChars, start+1)
	}
	return chunks, nil
}

// ChunkBySentences groups whole sentences up to the token target; boundaries
// never split a sentence. It is the Go twin of chunkBySentences() in
// src/lib/ragChunkComparator.ts and must agree with it on every shared vector.
func ChunkBySentences(text string, opts Options) ([]Chunk, error) {
	if opts.SizeTokens <= 0 {
		return nil, ErrBadSizeTokens
	}
	sentences := SplitSentences(text)
	if len(sentences) == 0 {
		return []Chunk{}, nil
	}
	chunks := []Chunk{}
	var current []string
	currentTokens := 0
	flush := func() {
		// Guard is defensive: the loop always leaves the last sentence
		// pending, so the final flush is never empty by construction.
		if len(current) == 0 {
			return
		}
		piece := strings.Join(current, " ")
		chunks = append(chunks, Chunk{Index: len(chunks), Text: piece, Tokens: tok(piece)})
		current = nil
		currentTokens = 0
	}
	for _, sentence := range sentences {
		t := tok(sentence)
		if currentTokens > 0 && currentTokens+t > opts.SizeTokens {
			flush()
		}
		current = append(current, sentence)
		currentTokens += t
		// A single sentence larger than the target becomes its own chunk.
	}
	flush()
	return chunks, nil
}

// ChunkMarkdown splits on markdown headings; oversized sections fall back to
// sentence grouping. It is the Go twin of chunkMarkdown() in
// src/lib/ragChunkComparator.ts and must agree with it on every shared vector.
func ChunkMarkdown(text string, opts Options) ([]Chunk, error) {
	if opts.SizeTokens <= 0 {
		return nil, ErrBadSizeTokens
	}
	type section struct {
		heading *string
		body    []string
	}
	var sections []section
	current := section{heading: nil, body: []string{}}
	for _, line := range strings.Split(text, "\n") {
		if m := headingLine.FindStringSubmatch(line); m != nil {
			if len(current.body) > 0 {
				sections = append(sections, current)
			}
			h := strings.TrimSpace(m[2])
			current = section{heading: &h, body: []string{}}
		} else {
			current.body = append(current.body, line)
		}
	}
	if len(current.body) > 0 {
		sections = append(sections, current)
	}

	chunks := []Chunk{}
	for _, sec := range sections {
		body := strings.TrimSpace(strings.Join(sec.body, "\n"))
		if body == "" {
			continue
		}
		whole := body
		if sec.heading != nil {
			whole = "# " + *sec.heading + "\n" + body
		}
		if tok(whole) <= opts.SizeTokens {
			chunks = append(chunks, Chunk{
				Index: len(chunks), Text: whole, Tokens: tok(whole), Heading: sec.heading,
			})
			continue
		}
		// Oversized section: sentence-group the body, stamp every chunk with the heading.
		sub, err := ChunkBySentences(body, opts)
		if err != nil {
			return nil, err // unreachable here: opts were validated above
		}
		for _, c := range sub {
			chunks = append(chunks, Chunk{
				Index: len(chunks), Text: c.Text, Tokens: c.Tokens, Heading: sec.heading,
			})
		}
	}
	return chunks, nil
}

// statsFor summarizes one strategy's chunks. Mirrors the TS statsFor helper.
func statsFor(strategy ChunkStrategy, chunks []Chunk) StrategyResult {
	count := len(chunks)
	minTokens, maxTokens, sum := 0, 0, 0
	if count > 0 {
		minTokens, maxTokens = chunks[0].Tokens, chunks[0].Tokens
		for _, c := range chunks {
			minTokens = min(minTokens, c.Tokens)
			maxTokens = max(maxTokens, c.Tokens)
			sum += c.Tokens
		}
	}
	avgTokens := 0
	if count > 0 {
		avgTokens = int(math.Round(float64(sum) / float64(count)))
	}
	share := 1.0 // a single chunk has no internal boundaries to botch
	if boundaries := count - 1; boundaries > 0 {
		ends := 0
		for _, c := range chunks[:count-1] {
			if endsSentence(c.Text) {
				ends++
			}
		}
		share = float64(ends) / float64(boundaries)
	}
	return StrategyResult{
		Strategy: strategy,
		Chunks:   chunks,
		Stats: StrategyStats{
			Count: count, MinTokens: minTokens, MaxTokens: maxTokens,
			AvgTokens: avgTokens, SentenceBoundaryShare: share,
		},
	}
}

// CompareStrategies runs all three strategies over one document and reports
// comparable stats. It is the Go twin of compareStrategies() in
// src/lib/ragChunkComparator.ts and must agree with it on every shared vector.
func CompareStrategies(text string, opts Options) (Comparison, error) {
	fixed, err := ChunkFixed(text, opts)
	if err != nil {
		return Comparison{}, err
	}
	sentence, err := ChunkBySentences(text, opts)
	if err != nil {
		return Comparison{}, err
	}
	markdown, err := ChunkMarkdown(text, opts)
	if err != nil {
		return Comparison{}, err
	}
	return Comparison{
		Fixed:    statsFor(StrategyFixed, fixed),
		Sentence: statsFor(StrategySentence, sentence),
		Markdown: statsFor(StrategyMarkdown, markdown),
	}, nil
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →