Skip to content

RAG Chunk Comparator — JavaScript source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// RAG Chunk Comparator — chunk one document three ways (fixed-size,
// sentence-aware, markdown-heading-aware) and compare retrieval stats.
//
// Language: JavaScript (ES2022+, ES module; runs unmodified in Node 18+
//           and modern browsers)
// Source:   CosmoDev polyglot showcase port of the RAG Chunk Comparator
//           tool (slug: rag-chunk-comparator).
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
//           implementation).
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Token sizes inline the tokenEstimator prose heuristic (~4 chars per
// token, per non-empty line, minimum one token per line) so this file is
// self-contained. The stats that matter for retrieval: chunk count, size
// spread, and how often boundaries land on sentence ends — mid-sentence
// cuts are the classic recall killer.

/** Prose token estimate: chars/4 per non-empty line, min 1 per line. */
const tok = (s) =>
  s
    .split(/\r?\n/)
    .reduce((n, line) => (line.trim() === '' ? n : n + Math.max(1, Math.round(line.length / 4))), 0);

/** Split on sentence enders followed by whitespace or end of text. */
export function splitSentences(text) {
  return text
    .replace(/\s+/g, ' ')
    .trim()
    .split(/(?<=[.!?]) +/)
    .filter((s) => s.length > 0);
}

const endsSentence = (s) => /[.!?]["')\]]?$/.test(s.trim());

/** Greedy character accumulation to a token target (overlapping allowed). */
export function chunkFixed(text, opts) {
  const { sizeTokens, overlapTokens = 0 } = opts;
  if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
  if (overlapTokens < 0 || overlapTokens >= sizeTokens) {
    throw new RangeError('overlapTokens must be in [0, sizeTokens)');
  }
  const clean = text.trim();
  if (!clean) return [];
  // ~4 chars per prose token: step by tokens, verify with the estimator.
  const charStep = Math.max(1, Math.round(sizeTokens * 4));
  const overlapChars = Math.round(overlapTokens * 4);
  const chunks = [];
  let start = 0;
  while (start < clean.length) {
    let end = Math.min(start + charStep, clean.length);
    // Prefer cutting at whitespace near the target — but never trim the
    // document's final piece back to a word when it already fits.
    if (end < clean.length) {
      const cut = clean.lastIndexOf(' ', end);
      if (cut > start) end = cut;
    }
    const piece = clean.slice(start, end).trim();
    if (piece) chunks.push({ index: chunks.length, text: piece, tokens: tok(piece) });
    if (end >= clean.length) break;
    start = Math.max(end - overlapChars, start + 1);
  }
  return chunks;
}

/** Group whole sentences up to the token target; boundaries never split a sentence. */
export function chunkBySentences(text, opts) {
  const { sizeTokens } = opts;
  if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
  const sentences = splitSentences(text);
  if (sentences.length === 0) return [];
  const chunks = [];
  let current = [];
  let currentTokens = 0;
  const flush = () => {
    if (current.length === 0) return;
    const piece = current.join(' ');
    chunks.push({ index: chunks.length, text: piece, tokens: tok(piece) });
    current = [];
    currentTokens = 0;
  };
  for (const sentence of sentences) {
    const t = tok(sentence);
    if (currentTokens > 0 && currentTokens + t > sizeTokens) flush();
    current.push(sentence);
    currentTokens += t;
    // A single sentence larger than the target becomes its own chunk.
  }
  flush();
  return chunks;
}

/** Split on markdown headings; oversized sections fall back to sentence grouping. */
export function chunkMarkdown(text, opts) {
  const { sizeTokens } = opts;
  if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
  const lines = text.split('\n');
  const sections = [];
  let current = { heading: undefined, body: [] };
  for (const line of lines) {
    const m = line.match(/^(#{1,6})\s+(.*)$/);
    if (m) {
      if (current.body.length > 0) sections.push(current);
      current = { heading: m[2].trim(), body: [] };
    } else {
      current.body.push(line);
    }
  }
  if (current.body.length > 0) sections.push(current);

  const chunks = [];
  for (const section of sections) {
    const body = section.body.join('\n').trim();
    if (!body) continue;
    const whole = section.heading ? `# ${section.heading}\n${body}` : body;
    if (tok(whole) <= sizeTokens) {
      chunks.push({ index: chunks.length, text: whole, tokens: tok(whole), heading: section.heading });
      continue;
    }
    // Oversized section: sentence-group the body, stamp every chunk with the heading.
    for (const c of chunkBySentences(body, opts)) {
      chunks.push({
        index: chunks.length,
        text: c.text,
        tokens: c.tokens,
        heading: section.heading,
      });
    }
  }
  return chunks;
}

function statsFor(strategy, chunks) {
  const sizes = chunks.map((c) => c.tokens);
  const count = chunks.length;
  const minTokens = count ? Math.min(...sizes) : 0;
  const maxTokens = count ? Math.max(...sizes) : 0;
  const avgTokens = count ? Math.round(sizes.reduce((a, b) => a + b, 0) / count) : 0;
  const boundaries = chunks.slice(0, -1).map((c) => endsSentence(c.text));
  const sentenceBoundaryShare = boundaries.length
    ? boundaries.filter(Boolean).length / boundaries.length
    : 1; // a single chunk has no internal boundaries to botch
  return { strategy, chunks, stats: { count, minTokens, maxTokens, avgTokens, sentenceBoundaryShare } };
}

/** Run all three strategies over one document and report comparable stats. */
export function compareStrategies(text, opts) {
  return {
    fixed: statsFor('fixed', chunkFixed(text, opts)),
    sentence: statsFor('sentence', chunkBySentences(text, opts)),
    markdown: statsFor('markdown', chunkMarkdown(text, opts)),
  };
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →