RAG Chunk Comparator — JavaScript source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// RAG Chunk Comparator — chunk one document three ways (fixed-size,
// sentence-aware, markdown-heading-aware) and compare retrieval stats.
//
// Language: JavaScript (ES2022+, ES module; runs unmodified in Node 18+
// and modern browsers)
// Source: CosmoDev polyglot showcase port of the RAG Chunk Comparator
// tool (slug: rag-chunk-comparator).
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
// implementation).
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Token sizes inline the tokenEstimator prose heuristic (~4 chars per
// token, per non-empty line, minimum one token per line) so this file is
// self-contained. The stats that matter for retrieval: chunk count, size
// spread, and how often boundaries land on sentence ends — mid-sentence
// cuts are the classic recall killer.
/** Prose token estimate: chars/4 per non-empty line, min 1 per line. */
const tok = (s) =>
s
.split(/\r?\n/)
.reduce((n, line) => (line.trim() === '' ? n : n + Math.max(1, Math.round(line.length / 4))), 0);
/** Split on sentence enders followed by whitespace or end of text. */
export function splitSentences(text) {
return text
.replace(/\s+/g, ' ')
.trim()
.split(/(?<=[.!?]) +/)
.filter((s) => s.length > 0);
}
const endsSentence = (s) => /[.!?]["')\]]?$/.test(s.trim());
/** Greedy character accumulation to a token target (overlapping allowed). */
export function chunkFixed(text, opts) {
const { sizeTokens, overlapTokens = 0 } = opts;
if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
if (overlapTokens < 0 || overlapTokens >= sizeTokens) {
throw new RangeError('overlapTokens must be in [0, sizeTokens)');
}
const clean = text.trim();
if (!clean) return [];
// ~4 chars per prose token: step by tokens, verify with the estimator.
const charStep = Math.max(1, Math.round(sizeTokens * 4));
const overlapChars = Math.round(overlapTokens * 4);
const chunks = [];
let start = 0;
while (start < clean.length) {
let end = Math.min(start + charStep, clean.length);
// Prefer cutting at whitespace near the target — but never trim the
// document's final piece back to a word when it already fits.
if (end < clean.length) {
const cut = clean.lastIndexOf(' ', end);
if (cut > start) end = cut;
}
const piece = clean.slice(start, end).trim();
if (piece) chunks.push({ index: chunks.length, text: piece, tokens: tok(piece) });
if (end >= clean.length) break;
start = Math.max(end - overlapChars, start + 1);
}
return chunks;
}
/** Group whole sentences up to the token target; boundaries never split a sentence. */
export function chunkBySentences(text, opts) {
const { sizeTokens } = opts;
if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
const sentences = splitSentences(text);
if (sentences.length === 0) return [];
const chunks = [];
let current = [];
let currentTokens = 0;
const flush = () => {
if (current.length === 0) return;
const piece = current.join(' ');
chunks.push({ index: chunks.length, text: piece, tokens: tok(piece) });
current = [];
currentTokens = 0;
};
for (const sentence of sentences) {
const t = tok(sentence);
if (currentTokens > 0 && currentTokens + t > sizeTokens) flush();
current.push(sentence);
currentTokens += t;
// A single sentence larger than the target becomes its own chunk.
}
flush();
return chunks;
}
/** Split on markdown headings; oversized sections fall back to sentence grouping. */
export function chunkMarkdown(text, opts) {
const { sizeTokens } = opts;
if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
const lines = text.split('\n');
const sections = [];
let current = { heading: undefined, body: [] };
for (const line of lines) {
const m = line.match(/^(#{1,6})\s+(.*)$/);
if (m) {
if (current.body.length > 0) sections.push(current);
current = { heading: m[2].trim(), body: [] };
} else {
current.body.push(line);
}
}
if (current.body.length > 0) sections.push(current);
const chunks = [];
for (const section of sections) {
const body = section.body.join('\n').trim();
if (!body) continue;
const whole = section.heading ? `# ${section.heading}\n${body}` : body;
if (tok(whole) <= sizeTokens) {
chunks.push({ index: chunks.length, text: whole, tokens: tok(whole), heading: section.heading });
continue;
}
// Oversized section: sentence-group the body, stamp every chunk with the heading.
for (const c of chunkBySentences(body, opts)) {
chunks.push({
index: chunks.length,
text: c.text,
tokens: c.tokens,
heading: section.heading,
});
}
}
return chunks;
}
function statsFor(strategy, chunks) {
const sizes = chunks.map((c) => c.tokens);
const count = chunks.length;
const minTokens = count ? Math.min(...sizes) : 0;
const maxTokens = count ? Math.max(...sizes) : 0;
const avgTokens = count ? Math.round(sizes.reduce((a, b) => a + b, 0) / count) : 0;
const boundaries = chunks.slice(0, -1).map((c) => endsSentence(c.text));
const sentenceBoundaryShare = boundaries.length
? boundaries.filter(Boolean).length / boundaries.length
: 1; // a single chunk has no internal boundaries to botch
return { strategy, chunks, stats: { count, minTokens, maxTokens, avgTokens, sentenceBoundaryShare } };
}
/** Run all three strategies over one document and report comparable stats. */
export function compareStrategies(text, opts) {
return {
fixed: statsFor('fixed', chunkFixed(text, opts)),
sentence: statsFor('sentence', chunkBySentences(text, opts)),
markdown: statsFor('markdown', chunkMarkdown(text, opts)),
};
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →