RAG Chunk Comparator — Ruby source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the Ruby implementation — the same logic the interactive tool runs, in a shareable, citable form.
# RAG Chunk Comparator — chunk one document three ways and report the stats
# that matter for retrieval.
#
# Language: Ruby (3.x, zero dependencies)
# Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
# implementation). Method names are snake_case per Ruby convention; the TS
# surface maps as chunkFixed -> chunk_fixed, compareStrategies ->
# compare_strategies. Raises ArgumentError where TS raises RangeError.
# Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
module RagChunkComparator
SENTENCE_SPLIT = /(?<=[.!?]) +/.freeze
ENDS_SENTENCE = /[.!?]["'()\]]?$/.freeze
# Target chunk size in tokens + fixed-strategy overlap.
Options = Struct.new(:size_tokens, :overlap_tokens, keyword_init: true) do
def initialize(size_tokens:, overlap_tokens: 0)
super
end
end
# One chunk: index, text, tokens, and (markdown only) the section heading.
Chunk = Struct.new(:index, :text, :tokens, :heading, keyword_init: true)
Stats = Struct.new(
:count, :min_tokens, :max_tokens, :avg_tokens, :sentence_boundary_share,
keyword_init: true
)
Result = Struct.new(:strategy, :chunks, :stats, keyword_init: true)
module_function
# The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
# line costs max(1, round(length / 4)) tokens; empty text is 0.
def tok(s)
return 0 if s.nil? || s.empty?
s.split("\n", -1).sum do |line|
line.empty? ? 0 : [1, (line.length / 4.0).round].max
end
end
# Split on sentence enders followed by whitespace or end of text.
def split_sentences(text)
text.gsub(/\s+/, ' ').strip.split(SENTENCE_SPLIT).reject(&:empty?)
end
def ends_sentence(s)
ENDS_SENTENCE.match?(s.strip)
end
# Greedy character accumulation to a token target (overlapping allowed).
# Raises ArgumentError on impossible options (the TS RangeError contract).
def chunk_fixed(text, opts)
raise ArgumentError, 'sizeTokens must be > 0' if opts.size_tokens <= 0
if opts.overlap_tokens.negative? || opts.overlap_tokens >= opts.size_tokens
raise ArgumentError, 'overlapTokens must be in [0, sizeTokens)'
end
clean = text.strip
return [] if clean.empty?
# ~4 chars per prose token: step by tokens, verify with the estimator.
char_step = [1, (opts.size_tokens * 4.0).round].max
overlap_chars = (opts.overlap_tokens * 4.0).round
chunks = []
start = 0
while start < clean.length
end_ = [start + char_step, clean.length].min
# Prefer cutting at whitespace near the target.
if end_ < clean.length
cut = clean.rindex(' ', end_)
end_ = cut if cut && cut > start
end
piece = clean[start...end_].to_s.strip
chunks << Chunk.new(index: chunks.length, text: piece, tokens: tok(piece)) unless piece.empty?
break if end_ >= clean.length
start = [end_ - overlap_chars, start + 1].max
end
chunks
end
# Group whole sentences up to the token target; boundaries never split a
# sentence. A single sentence larger than the target becomes its own chunk.
def chunk_by_sentences(text, opts)
raise ArgumentError, 'sizeTokens must be > 0' if opts.size_tokens <= 0
sentences = split_sentences(text)
return [] if sentences.empty?
chunks = []
current = []
current_tokens = 0
flush = lambda do
return if current.empty?
piece = current.join(' ')
chunks << Chunk.new(index: chunks.length, text: piece, tokens: tok(piece))
current = []
current_tokens = 0
end
sentences.each do |sentence|
t = tok(sentence)
flush.call if current_tokens.positive? && current_tokens + t > opts.size_tokens
current << sentence
current_tokens += t
end
flush.call
chunks
end
# 1-6 '#' then whitespace then heading text (a backslash keeps '#' out of
# interpolation inside a Ruby regex literal).
HEADING_LINE = /\A(\#{1,6})\s+(.*)\z/.freeze
# Split on markdown headings; oversized sections fall back to sentence
# grouping, and every chunk carries the section heading.
def chunk_markdown(text, opts)
raise ArgumentError, 'sizeTokens must be > 0' if opts.size_tokens <= 0
sections = []
pending_heading = nil
pending_body = []
text.split("\n", -1).each do |line|
m = HEADING_LINE.match(line)
if m
sections << [pending_heading, pending_body] unless pending_body.empty?
pending_heading = m[2].strip
pending_body = []
else
pending_body << line
end
end
sections << [pending_heading, pending_body] unless pending_body.empty?
chunks = []
sections.each do |heading, body|
clean = body.join("\n").strip
next if clean.empty?
whole = heading ? "# #{heading}\n#{clean}" : clean
if tok(whole) <= opts.size_tokens
chunks << Chunk.new(index: chunks.length, text: whole, tokens: tok(whole), heading: heading)
next
end
chunk_by_sentences(clean, opts).each do |c|
chunks << Chunk.new(index: chunks.length, text: c.text, tokens: c.tokens, heading: heading)
end
end
chunks
end
def stats_for(strategy, chunks)
count = chunks.length
sizes = chunks.map(&:tokens)
boundaries = chunks[0...-1].map { |c| ends_sentence(c.text) }
Stats.new(
count: count,
min_tokens: count.positive? ? sizes.min : 0,
max_tokens: count.positive? ? sizes.max : 0,
avg_tokens: count.positive? ? (sizes.sum.to_f / count).round : 0,
sentence_boundary_share: boundaries.empty? ? 1.0 : boundaries.count(true).to_f / boundaries.length
)
end
# Run all three strategies over one document and report comparable stats.
def compare_strategies(text, opts)
%w[fixed sentence markdown].each_with_object({}) do |strategy, out|
chunks =
case strategy
when 'fixed' then chunk_fixed(text, opts)
when 'sentence' then chunk_by_sentences(text, opts)
else chunk_markdown(text, opts)
end
out[strategy] = Result.new(
strategy: strategy, chunks: chunks, stats: stats_for(strategy, chunks)
)
end
end
end
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →