Skip to content

RAG Chunk Comparator — Ruby source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the Ruby implementation — the same logic the interactive tool runs, in a shareable, citable form.

# RAG Chunk Comparator — chunk one document three ways and report the stats
# that matter for retrieval.
#
# Language: Ruby (3.x, zero dependencies)
# Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
# implementation). Method names are snake_case per Ruby convention; the TS
# surface maps as chunkFixed -> chunk_fixed, compareStrategies ->
# compare_strategies. Raises ArgumentError where TS raises RangeError.
# Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator

module RagChunkComparator
  SENTENCE_SPLIT = /(?<=[.!?]) +/.freeze
  ENDS_SENTENCE = /[.!?]["'()\]]?$/.freeze

  # Target chunk size in tokens + fixed-strategy overlap.
  Options = Struct.new(:size_tokens, :overlap_tokens, keyword_init: true) do
    def initialize(size_tokens:, overlap_tokens: 0)
      super
    end
  end

  # One chunk: index, text, tokens, and (markdown only) the section heading.
  Chunk = Struct.new(:index, :text, :tokens, :heading, keyword_init: true)

  Stats = Struct.new(
    :count, :min_tokens, :max_tokens, :avg_tokens, :sentence_boundary_share,
    keyword_init: true
  )

  Result = Struct.new(:strategy, :chunks, :stats, keyword_init: true)

  module_function

  # The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
  # line costs max(1, round(length / 4)) tokens; empty text is 0.
  def tok(s)
    return 0 if s.nil? || s.empty?

    s.split("\n", -1).sum do |line|
      line.empty? ? 0 : [1, (line.length / 4.0).round].max
    end
  end

  # Split on sentence enders followed by whitespace or end of text.
  def split_sentences(text)
    text.gsub(/\s+/, ' ').strip.split(SENTENCE_SPLIT).reject(&:empty?)
  end

  def ends_sentence(s)
    ENDS_SENTENCE.match?(s.strip)
  end

  # Greedy character accumulation to a token target (overlapping allowed).
  # Raises ArgumentError on impossible options (the TS RangeError contract).
  def chunk_fixed(text, opts)
    raise ArgumentError, 'sizeTokens must be > 0' if opts.size_tokens <= 0
    if opts.overlap_tokens.negative? || opts.overlap_tokens >= opts.size_tokens
      raise ArgumentError, 'overlapTokens must be in [0, sizeTokens)'
    end

    clean = text.strip
    return [] if clean.empty?

    # ~4 chars per prose token: step by tokens, verify with the estimator.
    char_step = [1, (opts.size_tokens * 4.0).round].max
    overlap_chars = (opts.overlap_tokens * 4.0).round
    chunks = []
    start = 0
    while start < clean.length
      end_ = [start + char_step, clean.length].min
      # Prefer cutting at whitespace near the target.
      if end_ < clean.length
        cut = clean.rindex(' ', end_)
        end_ = cut if cut && cut > start
      end
      piece = clean[start...end_].to_s.strip
      chunks << Chunk.new(index: chunks.length, text: piece, tokens: tok(piece)) unless piece.empty?
      break if end_ >= clean.length

      start = [end_ - overlap_chars, start + 1].max
    end
    chunks
  end

  # Group whole sentences up to the token target; boundaries never split a
  # sentence. A single sentence larger than the target becomes its own chunk.
  def chunk_by_sentences(text, opts)
    raise ArgumentError, 'sizeTokens must be > 0' if opts.size_tokens <= 0

    sentences = split_sentences(text)
    return [] if sentences.empty?

    chunks = []
    current = []
    current_tokens = 0
    flush = lambda do
      return if current.empty?

      piece = current.join(' ')
      chunks << Chunk.new(index: chunks.length, text: piece, tokens: tok(piece))
      current = []
      current_tokens = 0
    end
    sentences.each do |sentence|
      t = tok(sentence)
      flush.call if current_tokens.positive? && current_tokens + t > opts.size_tokens
      current << sentence
      current_tokens += t
    end
    flush.call
    chunks
  end

  # 1-6 '#' then whitespace then heading text (a backslash keeps '#' out of
  # interpolation inside a Ruby regex literal).
  HEADING_LINE = /\A(\#{1,6})\s+(.*)\z/.freeze

  # Split on markdown headings; oversized sections fall back to sentence
  # grouping, and every chunk carries the section heading.
  def chunk_markdown(text, opts)
    raise ArgumentError, 'sizeTokens must be > 0' if opts.size_tokens <= 0

    sections = []
    pending_heading = nil
    pending_body = []
    text.split("\n", -1).each do |line|
      m = HEADING_LINE.match(line)
      if m
        sections << [pending_heading, pending_body] unless pending_body.empty?
        pending_heading = m[2].strip
        pending_body = []
      else
        pending_body << line
      end
    end
    sections << [pending_heading, pending_body] unless pending_body.empty?

    chunks = []
    sections.each do |heading, body|
      clean = body.join("\n").strip
      next if clean.empty?

      whole = heading ? "# #{heading}\n#{clean}" : clean
      if tok(whole) <= opts.size_tokens
        chunks << Chunk.new(index: chunks.length, text: whole, tokens: tok(whole), heading: heading)
        next
      end
      chunk_by_sentences(clean, opts).each do |c|
        chunks << Chunk.new(index: chunks.length, text: c.text, tokens: c.tokens, heading: heading)
      end
    end
    chunks
  end

  def stats_for(strategy, chunks)
    count = chunks.length
    sizes = chunks.map(&:tokens)
    boundaries = chunks[0...-1].map { |c| ends_sentence(c.text) }
    Stats.new(
      count: count,
      min_tokens: count.positive? ? sizes.min : 0,
      max_tokens: count.positive? ? sizes.max : 0,
      avg_tokens: count.positive? ? (sizes.sum.to_f / count).round : 0,
      sentence_boundary_share: boundaries.empty? ? 1.0 : boundaries.count(true).to_f / boundaries.length
    )
  end

  # Run all three strategies over one document and report comparable stats.
  def compare_strategies(text, opts)
    %w[fixed sentence markdown].each_with_object({}) do |strategy, out|
      chunks =
        case strategy
        when 'fixed' then chunk_fixed(text, opts)
        when 'sentence' then chunk_by_sentences(text, opts)
        else chunk_markdown(text, opts)
        end
      out[strategy] = Result.new(
        strategy: strategy, chunks: chunks, stats: stats_for(strategy, chunks)
      )
    end
  end
end

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →