RAG Chunk Comparator — Rust source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! RAG Chunk Comparator — chunk one document three ways (fixed-size,
//! sentence-aware, markdown-heading-aware) and compare retrieval stats.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source: CosmoDev polyglot showcase port of the RAG Chunk Comparator
//! tool (slug: rag-chunk-comparator).
//! Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
//! implementation).
//! Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Token sizes inline the tokenEstimator prose heuristic (~4 chars per
//! token, per non-empty line, minimum one token per line) so this file is
//! self-contained. The stats that matter for retrieval: chunk count, size
//! spread, and how often boundaries land on sentence ends — mid-sentence
//! cuts are the classic recall killer. Invalid options panic (the
//! TypeScript throws RangeError; Rust callers that prefer recovery can
//! validate `ChunkOptions` before calling).
/// Which chunking strategy produced a result.
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub enum ChunkStrategy {
Fixed,
Sentence,
Markdown,
}
impl ChunkStrategy {
pub fn as_str(self) -> &'static str {
match self {
ChunkStrategy::Fixed => "fixed",
ChunkStrategy::Sentence => "sentence",
ChunkStrategy::Markdown => "markdown",
}
}
}
/// Target chunk size in tokens; overlap between consecutive fixed chunks
/// (fixed strategy only).
#[derive(Clone, Copy, Debug)]
pub struct ChunkOptions {
pub size_tokens: i64,
pub overlap_tokens: i64,
}
impl ChunkOptions {
pub fn new(size_tokens: i64) -> Self {
ChunkOptions { size_tokens, overlap_tokens: 0 }
}
pub fn with_overlap(mut self, overlap_tokens: i64) -> Self {
self.overlap_tokens = overlap_tokens;
self
}
}
/// One chunk: text, prose-heuristic token count, optional nearest heading.
#[derive(Clone, Debug)]
pub struct Chunk {
pub index: usize,
pub text: String,
pub tokens: i64,
/// Nearest markdown heading for markdown chunks (None otherwise).
pub heading: Option<String>,
}
#[derive(Clone, Copy, Debug)]
pub struct StrategyStats {
pub count: usize,
pub min_tokens: i64,
pub max_tokens: i64,
pub avg_tokens: i64,
/// Share of chunk boundaries that fall on a sentence end (0-1).
pub sentence_boundary_share: f64,
}
#[derive(Clone, Debug)]
pub struct StrategyResult {
pub strategy: ChunkStrategy,
pub chunks: Vec<Chunk>,
pub stats: StrategyStats,
}
/// All three strategies over one document, side by side.
#[derive(Clone, Debug)]
pub struct CompareResults {
pub fixed: StrategyResult,
pub sentence: StrategyResult,
pub markdown: StrategyResult,
}
/// Prose token estimate: chars/4 per non-empty line, min 1 per line.
/// Rounds half up (JavaScript `Math.round`), not to even.
fn tok(s: &str) -> i64 {
let mut total = 0i64;
for line in s.split('\n') {
let line = line.strip_suffix('\r').unwrap_or(line);
if line.trim().is_empty() {
continue;
}
let n = line.chars().count();
total += (((n as f64) / 4.0) + 0.5).floor().max(1.0) as i64;
}
total
}
/// Split on sentence enders followed by whitespace or end of text.
/// (Hand-rolled: the regex crate has no lookbehind, and we stay stdlib.)
pub fn split_sentences(text: &str) -> Vec<String> {
// Collapse all whitespace runs to single spaces, then trim.
let norm = text.split_whitespace().collect::<Vec<_>>().join(" ");
let mut out: Vec<String> = Vec::new();
let mut current = String::new();
let chars: Vec<char> = norm.chars().collect();
let mut i = 0usize;
while i < chars.len() {
let c = chars[i];
current.push(c);
if (c == '.' || c == '!' || c == '?') && i + 1 < chars.len() && chars[i + 1] == ' ' {
out.push(std::mem::take(&mut current));
i += 1;
while i < chars.len() && chars[i] == ' ' {
i += 1; // skip the space run
}
continue;
}
i += 1;
}
if !current.is_empty() {
out.push(current);
}
out.retain(|s| !s.is_empty());
out
}
fn ends_sentence(s: &str) -> bool {
let t = s.trim();
let mut it = t.chars().rev();
let last = match it.next() {
Some(c) => c,
None => return false,
};
let is_ender = |c: char| c == '.' || c == '!' || c == '?';
if is_ender(last) {
return true;
}
if last == '"' || last == '\'' || last == ')' || last == ']' {
if let Some(prev) = it.next() {
return is_ender(prev);
}
}
false
}
/// Match ^(#{1,6})\s+(.*)$ and return the trimmed heading text.
fn parse_heading(line: &str) -> Option<String> {
let hashes = line.chars().take_while(|&c| c == '#').count();
if hashes < 1 || hashes > 6 {
return None;
}
let rest = &line[hashes..];
let mut it = rest.char_indices();
match it.next() {
Some((_, c)) if c.is_whitespace() => {}
_ => return None,
}
// Skip the whitespace run (the regex's \s+ is greedy; the .trim() after
// makes the exact run length irrelevant).
let after = rest.trim_start();
Some(after.trim().to_string())
}
/// Greedy character accumulation to a token target (overlapping allowed).
pub fn chunk_fixed(text: &str, opts: &ChunkOptions) -> Vec<Chunk> {
let size_tokens = opts.size_tokens;
let overlap_tokens = opts.overlap_tokens;
if size_tokens <= 0 {
panic!("sizeTokens must be > 0");
}
if overlap_tokens < 0 || overlap_tokens >= size_tokens {
panic!("overlapTokens must be in [0, sizeTokens)");
}
let clean: Vec<char> = text.trim().chars().collect();
if clean.is_empty() {
return Vec::new();
}
// ~4 chars per prose token: step by tokens, verify with the estimator.
let char_step = (size_tokens * 4).max(1) as usize;
let overlap_chars = (overlap_tokens * 4) as usize;
let mut chunks: Vec<Chunk> = Vec::new();
let mut start: usize = 0;
while start < clean.len() {
let mut end = (start + char_step).min(clean.len());
// Prefer cutting at whitespace near the target — but never trim the
// document's final piece back to a word when it already fits.
if end < clean.len() {
let mut cut: Option<usize> = None;
let mut j = end as isize;
while j >= 0 {
if clean[j as usize] == ' ' {
cut = Some(j as usize);
break;
}
j -= 1;
}
if let Some(c) = cut {
if c > start {
end = c;
}
}
}
let piece: String = clean[start..end].iter().collect::<String>().trim().to_string();
if !piece.is_empty() {
let tokens = tok(&piece);
chunks.push(Chunk { index: chunks.len(), text: piece, tokens, heading: None });
}
if end >= clean.len() {
break;
}
start = ((end as isize) - (overlap_chars as isize)).max(start as isize + 1) as usize;
}
chunks
}
/// Group whole sentences up to the token target; boundaries never split a
/// sentence.
pub fn chunk_by_sentences(text: &str, opts: &ChunkOptions) -> Vec<Chunk> {
let size_tokens = opts.size_tokens;
if size_tokens <= 0 {
panic!("sizeTokens must be > 0");
}
let sentences = split_sentences(text);
if sentences.is_empty() {
return Vec::new();
}
let mut chunks: Vec<Chunk> = Vec::new();
let mut current: Vec<String> = Vec::new();
let mut current_tokens = 0i64;
for sentence in sentences {
let t = tok(&sentence);
if current_tokens > 0 && current_tokens + t > size_tokens {
let piece = current.join(" ");
let tokens = tok(&piece);
chunks.push(Chunk { index: chunks.len(), text: piece, tokens, heading: None });
current.clear();
current_tokens = 0;
}
current.push(sentence);
current_tokens += t;
// A single sentence larger than the target becomes its own chunk.
}
if !current.is_empty() {
let piece = current.join(" ");
let tokens = tok(&piece);
chunks.push(Chunk { index: chunks.len(), text: piece, tokens, heading: None });
}
chunks
}
/// Split on markdown headings; oversized sections fall back to sentence
/// grouping.
pub fn chunk_markdown(text: &str, opts: &ChunkOptions) -> Vec<Chunk> {
let size_tokens = opts.size_tokens;
if size_tokens <= 0 {
panic!("sizeTokens must be > 0");
}
let mut sections: Vec<(Option<String>, Vec<&str>)> = Vec::new();
let mut current: (Option<String>, Vec<&str>) = (None, Vec::new());
for line in text.split('\n') {
if let Some(heading) = parse_heading(line) {
if !current.1.is_empty() {
sections.push(current);
}
current = (Some(heading), Vec::new());
} else {
current.1.push(line);
}
}
if !current.1.is_empty() {
sections.push(current);
}
let mut chunks: Vec<Chunk> = Vec::new();
for (heading, body_lines) in sections {
let body = body_lines.join("\n").trim().to_string();
if body.is_empty() {
continue;
}
let whole = match &heading {
Some(h) => format!("# {}\n{}", h, body),
None => body.clone(),
};
if tok(&whole) <= size_tokens {
let tokens = tok(&whole);
chunks.push(Chunk { index: chunks.len(), text: whole, tokens, heading });
continue;
}
// Oversized section: sentence-group the body, stamp every chunk with
// the heading.
for c in chunk_by_sentences(&body, opts) {
chunks.push(Chunk {
index: chunks.len(),
text: c.text,
tokens: c.tokens,
heading: heading.clone(),
});
}
}
chunks
}
fn stats_for(strategy: ChunkStrategy, chunks: Vec<Chunk>) -> StrategyResult {
let count = chunks.len();
let min_tokens = chunks.iter().map(|c| c.tokens).min().unwrap_or(0);
let max_tokens = chunks.iter().map(|c| c.tokens).max().unwrap_or(0);
let sum: i64 = chunks.iter().map(|c| c.tokens).sum();
let avg_tokens = if count > 0 {
((sum as f64 / count as f64) + 0.5).floor() as i64
} else {
0
};
let boundaries: Vec<bool> =
chunks[..count.saturating_sub(1)].iter().map(|c| ends_sentence(&c.text)).collect();
// A single chunk has no internal boundaries to botch.
let sentence_boundary_share = if boundaries.is_empty() {
1.0
} else {
boundaries.iter().filter(|b| **b).count() as f64 / boundaries.len() as f64
};
StrategyResult {
strategy,
chunks,
stats: StrategyStats { count, min_tokens, max_tokens, avg_tokens, sentence_boundary_share },
}
}
/// Run all three strategies over one document and report comparable stats.
pub fn compare_strategies(text: &str, opts: &ChunkOptions) -> CompareResults {
CompareResults {
fixed: stats_for(ChunkStrategy::Fixed, chunk_fixed(text, opts)),
sentence: stats_for(ChunkStrategy::Sentence, chunk_by_sentences(text, opts)),
markdown: stats_for(ChunkStrategy::Markdown, chunk_markdown(text, opts)),
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →