RAG Chunk Comparator — TypeScript source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Pure logic for the RAG Chunk Comparator tool (slug: rag-chunk-comparator).
// Chunks one document three ways — fixed-size, sentence-aware, and
// markdown-heading-aware — and reports the stats that matter for retrieval:
// chunk count, size spread, and how often boundaries land on sentence ends
// (mid-sentence cuts are the classic recall killer). Token sizes use the
// tokenEstimator prose heuristic.
import { estimateTokens } from './tokenEstimator';
export type ChunkStrategy = 'fixed' | 'sentence' | 'markdown';
export interface ChunkOptions {
/** Target chunk size in tokens. */
sizeTokens: number;
/** Overlap between consecutive fixed chunks, in tokens (fixed strategy only). */
overlapTokens?: number;
}
export interface Chunk {
index: number;
text: string;
tokens: number;
/** Nearest markdown heading for markdown chunks (undefined otherwise). */
heading?: string;
}
export interface StrategyStats {
count: number;
minTokens: number;
maxTokens: number;
avgTokens: number;
/** Share of chunk boundaries that fall on a sentence end (0–1). */
sentenceBoundaryShare: number;
}
export interface StrategyResult {
strategy: ChunkStrategy;
chunks: Chunk[];
stats: StrategyStats;
}
const tok = (s: string): number => estimateTokens(s, { type: 'prose' }).tokens;
/** Split on sentence enders followed by whitespace or end of text. */
export function splitSentences(text: string): string[] {
return text
.replace(/\s+/g, ' ')
.trim()
.split(/(?<=[.!?]) +/)
.filter((s) => s.length > 0);
}
const endsSentence = (s: string): boolean => /[.!?]["')\]]?$/.test(s.trim());
/** Greedy character accumulation to a token target (overlapping allowed). */
export function chunkFixed(text: string, opts: ChunkOptions): Chunk[] {
const { sizeTokens, overlapTokens = 0 } = opts;
if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
if (overlapTokens < 0 || overlapTokens >= sizeTokens) {
throw new RangeError('overlapTokens must be in [0, sizeTokens)');
}
const clean = text.trim();
if (!clean) return [];
// ~4 chars per prose token: step by tokens, verify with the estimator.
const charStep = Math.max(1, Math.round(sizeTokens * 4));
const overlapChars = Math.round(overlapTokens * 4);
const chunks: Chunk[] = [];
let start = 0;
while (start < clean.length) {
let end = Math.min(start + charStep, clean.length);
// Prefer cutting at whitespace near the target — but never trim the
// document's final piece back to a word when it already fits.
if (end < clean.length) {
const cut = clean.lastIndexOf(' ', end);
if (cut > start) end = cut;
}
const piece = clean.slice(start, end).trim();
if (piece) chunks.push({ index: chunks.length, text: piece, tokens: tok(piece) });
if (end >= clean.length) break;
start = Math.max(end - overlapChars, start + 1);
}
return chunks;
}
/** Group whole sentences up to the token target; boundaries never split a sentence. */
export function chunkBySentences(text: string, opts: ChunkOptions): Chunk[] {
const { sizeTokens } = opts;
if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
const sentences = splitSentences(text);
if (sentences.length === 0) return [];
const chunks: Chunk[] = [];
let current: string[] = [];
let currentTokens = 0;
const flush = () => {
// Guard is defensive: the loop always leaves the last sentence pending,
// so the final flush is never empty by construction (enumerated survivor
// for coverage purposes — it must stay for correctness).
if (current.length === 0) return;
const piece = current.join(' ');
chunks.push({ index: chunks.length, text: piece, tokens: tok(piece) });
current = [];
currentTokens = 0;
};
for (const sentence of sentences) {
const t = tok(sentence);
if (currentTokens > 0 && currentTokens + t > sizeTokens) flush();
current.push(sentence);
currentTokens += t;
// A single sentence larger than the target becomes its own chunk.
}
flush();
return chunks;
}
/** Split on markdown headings; oversized sections fall back to sentence grouping. */
export function chunkMarkdown(text: string, opts: ChunkOptions): Chunk[] {
const { sizeTokens } = opts;
if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
const lines = text.split('\n');
const sections: { heading: string | undefined; body: string[] }[] = [];
let current: { heading: string | undefined; body: string[] } = { heading: undefined, body: [] };
for (const line of lines) {
const m = line.match(/^(#{1,6})\s+(.*)$/);
if (m) {
if (current.body.length > 0) sections.push(current);
current = { heading: m[2].trim(), body: [] };
} else {
current.body.push(line);
}
}
if (current.body.length > 0) sections.push(current);
const chunks: Chunk[] = [];
for (const section of sections) {
const body = section.body.join('\n').trim();
if (!body) continue;
const whole = section.heading ? `# ${section.heading}\n${body}` : body;
if (tok(whole) <= sizeTokens) {
chunks.push({ index: chunks.length, text: whole, tokens: tok(whole), heading: section.heading });
continue;
}
// Oversized section: sentence-group the body, stamp every chunk with the heading.
for (const c of chunkBySentences(body, opts)) {
chunks.push({
index: chunks.length,
text: c.text,
tokens: c.tokens,
heading: section.heading,
});
}
}
return chunks;
}
function statsFor(strategy: ChunkStrategy, chunks: Chunk[]): StrategyResult {
const sizes = chunks.map((c) => c.tokens);
const count = chunks.length;
const minTokens = count ? Math.min(...sizes) : 0;
const maxTokens = count ? Math.max(...sizes) : 0;
const avgTokens = count ? Math.round(sizes.reduce((a, b) => a + b, 0) / count) : 0;
const boundaries = chunks.slice(0, -1).map((c) => endsSentence(c.text));
const sentenceBoundaryShare = boundaries.length
? boundaries.filter(Boolean).length / boundaries.length
: 1; // a single chunk has no internal boundaries to botch
return { strategy, chunks, stats: { count, minTokens, maxTokens, avgTokens, sentenceBoundaryShare } };
}
/** Run all three strategies over one document and report comparable stats. */
export function compareStrategies(text: string, opts: ChunkOptions): Record<ChunkStrategy, StrategyResult> {
return {
fixed: statsFor('fixed', chunkFixed(text, opts)),
sentence: statsFor('sentence', chunkBySentences(text, opts)),
markdown: statsFor('markdown', chunkMarkdown(text, opts)),
};
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →