Skip to content

RAG Chunk Comparator — Rust source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! RAG Chunk Comparator — chunk one document three ways (fixed-size,
//! sentence-aware, markdown-heading-aware) and compare retrieval stats.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source:   CosmoDev polyglot showcase port of the RAG Chunk Comparator
//!           tool (slug: rag-chunk-comparator).
//! Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
//!           implementation).
//! Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Token sizes inline the tokenEstimator prose heuristic (~4 chars per
//! token, per non-empty line, minimum one token per line) so this file is
//! self-contained. The stats that matter for retrieval: chunk count, size
//! spread, and how often boundaries land on sentence ends — mid-sentence
//! cuts are the classic recall killer. Invalid options panic (the
//! TypeScript throws RangeError; Rust callers that prefer recovery can
//! validate `ChunkOptions` before calling).

/// Which chunking strategy produced a result.
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub enum ChunkStrategy {
    Fixed,
    Sentence,
    Markdown,
}

impl ChunkStrategy {
    pub fn as_str(self) -> &'static str {
        match self {
            ChunkStrategy::Fixed => "fixed",
            ChunkStrategy::Sentence => "sentence",
            ChunkStrategy::Markdown => "markdown",
        }
    }
}

/// Target chunk size in tokens; overlap between consecutive fixed chunks
/// (fixed strategy only).
#[derive(Clone, Copy, Debug)]
pub struct ChunkOptions {
    pub size_tokens: i64,
    pub overlap_tokens: i64,
}

impl ChunkOptions {
    pub fn new(size_tokens: i64) -> Self {
        ChunkOptions { size_tokens, overlap_tokens: 0 }
    }

    pub fn with_overlap(mut self, overlap_tokens: i64) -> Self {
        self.overlap_tokens = overlap_tokens;
        self
    }
}

/// One chunk: text, prose-heuristic token count, optional nearest heading.
#[derive(Clone, Debug)]
pub struct Chunk {
    pub index: usize,
    pub text: String,
    pub tokens: i64,
    /// Nearest markdown heading for markdown chunks (None otherwise).
    pub heading: Option<String>,
}

#[derive(Clone, Copy, Debug)]
pub struct StrategyStats {
    pub count: usize,
    pub min_tokens: i64,
    pub max_tokens: i64,
    pub avg_tokens: i64,
    /// Share of chunk boundaries that fall on a sentence end (0-1).
    pub sentence_boundary_share: f64,
}

#[derive(Clone, Debug)]
pub struct StrategyResult {
    pub strategy: ChunkStrategy,
    pub chunks: Vec<Chunk>,
    pub stats: StrategyStats,
}

/// All three strategies over one document, side by side.
#[derive(Clone, Debug)]
pub struct CompareResults {
    pub fixed: StrategyResult,
    pub sentence: StrategyResult,
    pub markdown: StrategyResult,
}

/// Prose token estimate: chars/4 per non-empty line, min 1 per line.
/// Rounds half up (JavaScript `Math.round`), not to even.
fn tok(s: &str) -> i64 {
    let mut total = 0i64;
    for line in s.split('\n') {
        let line = line.strip_suffix('\r').unwrap_or(line);
        if line.trim().is_empty() {
            continue;
        }
        let n = line.chars().count();
        total += (((n as f64) / 4.0) + 0.5).floor().max(1.0) as i64;
    }
    total
}

/// Split on sentence enders followed by whitespace or end of text.
/// (Hand-rolled: the regex crate has no lookbehind, and we stay stdlib.)
pub fn split_sentences(text: &str) -> Vec<String> {
    // Collapse all whitespace runs to single spaces, then trim.
    let norm = text.split_whitespace().collect::<Vec<_>>().join(" ");
    let mut out: Vec<String> = Vec::new();
    let mut current = String::new();
    let chars: Vec<char> = norm.chars().collect();
    let mut i = 0usize;
    while i < chars.len() {
        let c = chars[i];
        current.push(c);
        if (c == '.' || c == '!' || c == '?') && i + 1 < chars.len() && chars[i + 1] == ' ' {
            out.push(std::mem::take(&mut current));
            i += 1;
            while i < chars.len() && chars[i] == ' ' {
                i += 1; // skip the space run
            }
            continue;
        }
        i += 1;
    }
    if !current.is_empty() {
        out.push(current);
    }
    out.retain(|s| !s.is_empty());
    out
}

fn ends_sentence(s: &str) -> bool {
    let t = s.trim();
    let mut it = t.chars().rev();
    let last = match it.next() {
        Some(c) => c,
        None => return false,
    };
    let is_ender = |c: char| c == '.' || c == '!' || c == '?';
    if is_ender(last) {
        return true;
    }
    if last == '"' || last == '\'' || last == ')' || last == ']' {
        if let Some(prev) = it.next() {
            return is_ender(prev);
        }
    }
    false
}

/// Match ^(#{1,6})\s+(.*)$ and return the trimmed heading text.
fn parse_heading(line: &str) -> Option<String> {
    let hashes = line.chars().take_while(|&c| c == '#').count();
    if hashes < 1 || hashes > 6 {
        return None;
    }
    let rest = &line[hashes..];
    let mut it = rest.char_indices();
    match it.next() {
        Some((_, c)) if c.is_whitespace() => {}
        _ => return None,
    }
    // Skip the whitespace run (the regex's \s+ is greedy; the .trim() after
    // makes the exact run length irrelevant).
    let after = rest.trim_start();
    Some(after.trim().to_string())
}

/// Greedy character accumulation to a token target (overlapping allowed).
pub fn chunk_fixed(text: &str, opts: &ChunkOptions) -> Vec<Chunk> {
    let size_tokens = opts.size_tokens;
    let overlap_tokens = opts.overlap_tokens;
    if size_tokens <= 0 {
        panic!("sizeTokens must be > 0");
    }
    if overlap_tokens < 0 || overlap_tokens >= size_tokens {
        panic!("overlapTokens must be in [0, sizeTokens)");
    }
    let clean: Vec<char> = text.trim().chars().collect();
    if clean.is_empty() {
        return Vec::new();
    }
    // ~4 chars per prose token: step by tokens, verify with the estimator.
    let char_step = (size_tokens * 4).max(1) as usize;
    let overlap_chars = (overlap_tokens * 4) as usize;
    let mut chunks: Vec<Chunk> = Vec::new();
    let mut start: usize = 0;
    while start < clean.len() {
        let mut end = (start + char_step).min(clean.len());
        // Prefer cutting at whitespace near the target — but never trim the
        // document's final piece back to a word when it already fits.
        if end < clean.len() {
            let mut cut: Option<usize> = None;
            let mut j = end as isize;
            while j >= 0 {
                if clean[j as usize] == ' ' {
                    cut = Some(j as usize);
                    break;
                }
                j -= 1;
            }
            if let Some(c) = cut {
                if c > start {
                    end = c;
                }
            }
        }
        let piece: String = clean[start..end].iter().collect::<String>().trim().to_string();
        if !piece.is_empty() {
            let tokens = tok(&piece);
            chunks.push(Chunk { index: chunks.len(), text: piece, tokens, heading: None });
        }
        if end >= clean.len() {
            break;
        }
        start = ((end as isize) - (overlap_chars as isize)).max(start as isize + 1) as usize;
    }
    chunks
}

/// Group whole sentences up to the token target; boundaries never split a
/// sentence.
pub fn chunk_by_sentences(text: &str, opts: &ChunkOptions) -> Vec<Chunk> {
    let size_tokens = opts.size_tokens;
    if size_tokens <= 0 {
        panic!("sizeTokens must be > 0");
    }
    let sentences = split_sentences(text);
    if sentences.is_empty() {
        return Vec::new();
    }
    let mut chunks: Vec<Chunk> = Vec::new();
    let mut current: Vec<String> = Vec::new();
    let mut current_tokens = 0i64;
    for sentence in sentences {
        let t = tok(&sentence);
        if current_tokens > 0 && current_tokens + t > size_tokens {
            let piece = current.join(" ");
            let tokens = tok(&piece);
            chunks.push(Chunk { index: chunks.len(), text: piece, tokens, heading: None });
            current.clear();
            current_tokens = 0;
        }
        current.push(sentence);
        current_tokens += t;
        // A single sentence larger than the target becomes its own chunk.
    }
    if !current.is_empty() {
        let piece = current.join(" ");
        let tokens = tok(&piece);
        chunks.push(Chunk { index: chunks.len(), text: piece, tokens, heading: None });
    }
    chunks
}

/// Split on markdown headings; oversized sections fall back to sentence
/// grouping.
pub fn chunk_markdown(text: &str, opts: &ChunkOptions) -> Vec<Chunk> {
    let size_tokens = opts.size_tokens;
    if size_tokens <= 0 {
        panic!("sizeTokens must be > 0");
    }
    let mut sections: Vec<(Option<String>, Vec<&str>)> = Vec::new();
    let mut current: (Option<String>, Vec<&str>) = (None, Vec::new());
    for line in text.split('\n') {
        if let Some(heading) = parse_heading(line) {
            if !current.1.is_empty() {
                sections.push(current);
            }
            current = (Some(heading), Vec::new());
        } else {
            current.1.push(line);
        }
    }
    if !current.1.is_empty() {
        sections.push(current);
    }

    let mut chunks: Vec<Chunk> = Vec::new();
    for (heading, body_lines) in sections {
        let body = body_lines.join("\n").trim().to_string();
        if body.is_empty() {
            continue;
        }
        let whole = match &heading {
            Some(h) => format!("# {}\n{}", h, body),
            None => body.clone(),
        };
        if tok(&whole) <= size_tokens {
            let tokens = tok(&whole);
            chunks.push(Chunk { index: chunks.len(), text: whole, tokens, heading });
            continue;
        }
        // Oversized section: sentence-group the body, stamp every chunk with
        // the heading.
        for c in chunk_by_sentences(&body, opts) {
            chunks.push(Chunk {
                index: chunks.len(),
                text: c.text,
                tokens: c.tokens,
                heading: heading.clone(),
            });
        }
    }
    chunks
}

fn stats_for(strategy: ChunkStrategy, chunks: Vec<Chunk>) -> StrategyResult {
    let count = chunks.len();
    let min_tokens = chunks.iter().map(|c| c.tokens).min().unwrap_or(0);
    let max_tokens = chunks.iter().map(|c| c.tokens).max().unwrap_or(0);
    let sum: i64 = chunks.iter().map(|c| c.tokens).sum();
    let avg_tokens = if count > 0 {
        ((sum as f64 / count as f64) + 0.5).floor() as i64
    } else {
        0
    };
    let boundaries: Vec<bool> =
        chunks[..count.saturating_sub(1)].iter().map(|c| ends_sentence(&c.text)).collect();
    // A single chunk has no internal boundaries to botch.
    let sentence_boundary_share = if boundaries.is_empty() {
        1.0
    } else {
        boundaries.iter().filter(|b| **b).count() as f64 / boundaries.len() as f64
    };
    StrategyResult {
        strategy,
        chunks,
        stats: StrategyStats { count, min_tokens, max_tokens, avg_tokens, sentence_boundary_share },
    }
}

/// Run all three strategies over one document and report comparable stats.
pub fn compare_strategies(text: &str, opts: &ChunkOptions) -> CompareResults {
    CompareResults {
        fixed: stats_for(ChunkStrategy::Fixed, chunk_fixed(text, opts)),
        sentence: stats_for(ChunkStrategy::Sentence, chunk_by_sentences(text, opts)),
        markdown: stats_for(ChunkStrategy::Markdown, chunk_markdown(text, opts)),
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →