Skip to content

RAG Chunk Comparator — TypeScript source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Pure logic for the RAG Chunk Comparator tool (slug: rag-chunk-comparator).
// Chunks one document three ways — fixed-size, sentence-aware, and
// markdown-heading-aware — and reports the stats that matter for retrieval:
// chunk count, size spread, and how often boundaries land on sentence ends
// (mid-sentence cuts are the classic recall killer). Token sizes use the
// tokenEstimator prose heuristic.
import { estimateTokens } from './tokenEstimator';

export type ChunkStrategy = 'fixed' | 'sentence' | 'markdown';

export interface ChunkOptions {
  /** Target chunk size in tokens. */
  sizeTokens: number;
  /** Overlap between consecutive fixed chunks, in tokens (fixed strategy only). */
  overlapTokens?: number;
}

export interface Chunk {
  index: number;
  text: string;
  tokens: number;
  /** Nearest markdown heading for markdown chunks (undefined otherwise). */
  heading?: string;
}

export interface StrategyStats {
  count: number;
  minTokens: number;
  maxTokens: number;
  avgTokens: number;
  /** Share of chunk boundaries that fall on a sentence end (0–1). */
  sentenceBoundaryShare: number;
}

export interface StrategyResult {
  strategy: ChunkStrategy;
  chunks: Chunk[];
  stats: StrategyStats;
}

const tok = (s: string): number => estimateTokens(s, { type: 'prose' }).tokens;

/** Split on sentence enders followed by whitespace or end of text. */
export function splitSentences(text: string): string[] {
  return text
    .replace(/\s+/g, ' ')
    .trim()
    .split(/(?<=[.!?]) +/)
    .filter((s) => s.length > 0);
}

const endsSentence = (s: string): boolean => /[.!?]["')\]]?$/.test(s.trim());

/** Greedy character accumulation to a token target (overlapping allowed). */
export function chunkFixed(text: string, opts: ChunkOptions): Chunk[] {
  const { sizeTokens, overlapTokens = 0 } = opts;
  if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
  if (overlapTokens < 0 || overlapTokens >= sizeTokens) {
    throw new RangeError('overlapTokens must be in [0, sizeTokens)');
  }
  const clean = text.trim();
  if (!clean) return [];
  // ~4 chars per prose token: step by tokens, verify with the estimator.
  const charStep = Math.max(1, Math.round(sizeTokens * 4));
  const overlapChars = Math.round(overlapTokens * 4);
  const chunks: Chunk[] = [];
  let start = 0;
  while (start < clean.length) {
    let end = Math.min(start + charStep, clean.length);
    // Prefer cutting at whitespace near the target — but never trim the
    // document's final piece back to a word when it already fits.
    if (end < clean.length) {
      const cut = clean.lastIndexOf(' ', end);
      if (cut > start) end = cut;
    }
    const piece = clean.slice(start, end).trim();
    if (piece) chunks.push({ index: chunks.length, text: piece, tokens: tok(piece) });
    if (end >= clean.length) break;
    start = Math.max(end - overlapChars, start + 1);
  }
  return chunks;
}

/** Group whole sentences up to the token target; boundaries never split a sentence. */
export function chunkBySentences(text: string, opts: ChunkOptions): Chunk[] {
  const { sizeTokens } = opts;
  if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
  const sentences = splitSentences(text);
  if (sentences.length === 0) return [];
  const chunks: Chunk[] = [];
  let current: string[] = [];
  let currentTokens = 0;
  const flush = () => {
    // Guard is defensive: the loop always leaves the last sentence pending,
    // so the final flush is never empty by construction (enumerated survivor
    // for coverage purposes — it must stay for correctness).
    if (current.length === 0) return;
    const piece = current.join(' ');
    chunks.push({ index: chunks.length, text: piece, tokens: tok(piece) });
    current = [];
    currentTokens = 0;
  };
  for (const sentence of sentences) {
    const t = tok(sentence);
    if (currentTokens > 0 && currentTokens + t > sizeTokens) flush();
    current.push(sentence);
    currentTokens += t;
    // A single sentence larger than the target becomes its own chunk.
  }
  flush();
  return chunks;
}

/** Split on markdown headings; oversized sections fall back to sentence grouping. */
export function chunkMarkdown(text: string, opts: ChunkOptions): Chunk[] {
  const { sizeTokens } = opts;
  if (sizeTokens <= 0) throw new RangeError('sizeTokens must be > 0');
  const lines = text.split('\n');
  const sections: { heading: string | undefined; body: string[] }[] = [];
  let current: { heading: string | undefined; body: string[] } = { heading: undefined, body: [] };
  for (const line of lines) {
    const m = line.match(/^(#{1,6})\s+(.*)$/);
    if (m) {
      if (current.body.length > 0) sections.push(current);
      current = { heading: m[2].trim(), body: [] };
    } else {
      current.body.push(line);
    }
  }
  if (current.body.length > 0) sections.push(current);

  const chunks: Chunk[] = [];
  for (const section of sections) {
    const body = section.body.join('\n').trim();
    if (!body) continue;
    const whole = section.heading ? `# ${section.heading}\n${body}` : body;
    if (tok(whole) <= sizeTokens) {
      chunks.push({ index: chunks.length, text: whole, tokens: tok(whole), heading: section.heading });
      continue;
    }
    // Oversized section: sentence-group the body, stamp every chunk with the heading.
    for (const c of chunkBySentences(body, opts)) {
      chunks.push({
        index: chunks.length,
        text: c.text,
        tokens: c.tokens,
        heading: section.heading,
      });
    }
  }
  return chunks;
}

function statsFor(strategy: ChunkStrategy, chunks: Chunk[]): StrategyResult {
  const sizes = chunks.map((c) => c.tokens);
  const count = chunks.length;
  const minTokens = count ? Math.min(...sizes) : 0;
  const maxTokens = count ? Math.max(...sizes) : 0;
  const avgTokens = count ? Math.round(sizes.reduce((a, b) => a + b, 0) / count) : 0;
  const boundaries = chunks.slice(0, -1).map((c) => endsSentence(c.text));
  const sentenceBoundaryShare = boundaries.length
    ? boundaries.filter(Boolean).length / boundaries.length
    : 1; // a single chunk has no internal boundaries to botch
  return { strategy, chunks, stats: { count, minTokens, maxTokens, avgTokens, sentenceBoundaryShare } };
}

/** Run all three strategies over one document and report comparable stats. */
export function compareStrategies(text: string, opts: ChunkOptions): Record<ChunkStrategy, StrategyResult> {
  return {
    fixed: statsFor('fixed', chunkFixed(text, opts)),
    sentence: statsFor('sentence', chunkBySentences(text, opts)),
    markdown: statsFor('markdown', chunkMarkdown(text, opts)),
  };
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →