Skip to content

RAG Chunk Comparator — C# source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.

// RAG Chunk Comparator — chunk one document three ways and report the stats
// that matter for retrieval.
//
// Language: C# (.NET 8, zero dependencies)
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator

using System;
using System.Collections.Generic;
using System.Linq;
using System.Text;
using System.Text.RegularExpressions;

namespace CosmoDev.RagChunkComparator;

public enum ChunkStrategy { Fixed, Sentence, Markdown }

public sealed record ChunkOptions(
    int SizeTokens,
    int OverlapTokens = 0);

public sealed record Chunk(
    int Index,
    string Text,
    int Tokens,
    string? Heading = null);

public sealed record StrategyStats(
    int Count,
    int MinTokens,
    int MaxTokens,
    int AvgTokens,
    double SentenceBoundaryShare);

public sealed record StrategyResult(
    ChunkStrategy Strategy,
    IReadOnlyList<Chunk> Chunks,
    StrategyStats Stats);

public static class RagChunkComparator
{
    // The `type: 'prose'` path of the tokenEstimator, inlined: every
    // non-empty line costs max(1, round(length / 4)) tokens; empty text is 0.
    private static int Tok(string s)
    {
        if (string.IsNullOrEmpty(s)) return 0;
        int tokens = 0;
        foreach (var line in s.Split('\n'))
        {
            if (line.Length > 0)
            {
                tokens += Math.Max(1, (int)Math.Round(line.Length / 4.0, MidpointRounding.AwayFromZero));
            }
        }
        return tokens;
    }

    /// <summary>Split on sentence enders followed by whitespace or end of text.</summary>
    public static List<string> SplitSentences(string text)
    {
        var collapsed = Regex.Replace(text, @"\s+", " ").Trim();
        return Regex.Split(collapsed, @"(?<=[.!?]) +")
            .Where(s => s.Length > 0)
            .ToList();
    }

    private static bool EndsSentence(string s) =>
        Regex.IsMatch(s.Trim(), @"[.!?][""')\]]?$");

    /// <summary>
    /// Greedy character accumulation to a token target (overlapping allowed).
    /// Throws <see cref="ArgumentOutOfRangeException"/> on impossible options
    /// (the TS RangeError contract).
    /// </summary>
    public static List<Chunk> ChunkFixed(string text, ChunkOptions opts)
    {
        if (opts.SizeTokens <= 0) throw new ArgumentOutOfRangeException(nameof(opts), "sizeTokens must be > 0");
        if (opts.OverlapTokens < 0 || opts.OverlapTokens >= opts.SizeTokens)
            throw new ArgumentOutOfRangeException(nameof(opts), "overlapTokens must be in [0, sizeTokens)");
        var clean = text.Trim();
        if (clean.Length == 0) return new List<Chunk>();
        // ~4 chars per prose token: step by tokens, verify with the estimator.
        int charStep = Math.Max(1, (int)Math.Round(opts.SizeTokens * 4.0, MidpointRounding.AwayFromZero));
        int overlapChars = (int)Math.Round(opts.OverlapTokens * 4.0, MidpointRounding.AwayFromZero);
        var chunks = new List<Chunk>();
        int start = 0;
        while (start < clean.Length)
        {
            int end = Math.Min(start + charStep, clean.Length);
            // Prefer cutting at whitespace near the target.
            if (end < clean.Length)
            {
                int cut = clean.LastIndexOf(' ', end);
                if (cut > start) end = cut;
            }
            var piece = clean[start..end].Trim();
            if (piece.Length > 0) chunks.Add(new Chunk(chunks.Count, piece, Tok(piece)));
            if (end >= clean.Length) break;
            start = Math.Max(end - overlapChars, start + 1);
        }
        return chunks;
    }

    /// <summary>
    /// Group whole sentences up to the token target; boundaries never split a
    /// sentence. A single sentence larger than the target becomes its own chunk.
    /// </summary>
    public static List<Chunk> ChunkBySentences(string text, ChunkOptions opts)
    {
        if (opts.SizeTokens <= 0) throw new ArgumentOutOfRangeException(nameof(opts), "sizeTokens must be > 0");
        var sentences = SplitSentences(text);
        if (sentences.Count == 0) return new List<Chunk>();
        var chunks = new List<Chunk>();
        var current = new List<string>();
        int currentTokens = 0;
        void Flush()
        {
            if (current.Count == 0) return;
            var piece = string.Join(" ", current);
            chunks.Add(new Chunk(chunks.Count, piece, Tok(piece)));
            current = new List<string>();
            currentTokens = 0;
        }
        foreach (var sentence in sentences)
        {
            int t = Tok(sentence);
            if (currentTokens > 0 && currentTokens + t > opts.SizeTokens) Flush();
            current.Add(sentence);
            currentTokens += t;
        }
        Flush();
        return chunks;
    }

    private static readonly Regex HeadingRx = new(@"^(#{1,6})\s+(.*)$", RegexOptions.Compiled);

    /// <summary>
    /// Split on markdown headings; oversized sections fall back to sentence
    /// grouping, and every chunk carries the section heading.
    /// </summary>
    public static List<Chunk> ChunkMarkdown(string text, ChunkOptions opts)
    {
        if (opts.SizeTokens <= 0) throw new ArgumentOutOfRangeException(nameof(opts), "sizeTokens must be > 0");
        var lines = text.Split('\n');
        var sections = new List<(string? Heading, List<string> Body)>();
        var current = (Heading: (string?)null, Body: new List<string>());
        foreach (var line in lines)
        {
            var m = HeadingRx.Match(line);
            if (m.Success)
            {
                if (current.Body.Count > 0) sections.Add(current);
                current = (m.Groups[2].Value.Trim(), new List<string>());
            }
            else
            {
                current.Body.Add(line);
            }
        }
        if (current.Body.Count > 0) sections.Add(current);

        var chunks = new List<Chunk>();
        foreach (var (heading, body) in sections)
        {
            var clean = string.Join("\n", body).Trim();
            if (clean.Length == 0) continue;
            var whole = heading != null ? $"# {heading}\n{clean}" : clean;
            if (Tok(whole) <= opts.SizeTokens)
            {
                chunks.Add(new Chunk(chunks.Count, whole, Tok(whole), heading));
                continue;
            }
            foreach (var c in ChunkBySentences(clean, opts))
            {
                chunks.Add(new Chunk(chunks.Count, c.Text, c.Tokens, heading));
            }
        }
        return chunks;
    }

    private static StrategyResult StatsFor(ChunkStrategy strategy, IReadOnlyList<Chunk> chunks)
    {
        int count = chunks.Count;
        int min = count > 0 ? chunks.Min(c => c.Tokens) : 0;
        int max = count > 0 ? chunks.Max(c => c.Tokens) : 0;
        int avg = count > 0 ? (int)Math.Round(chunks.Sum(c => (double)c.Tokens) / count, MidpointRounding.AwayFromZero) : 0;
        var boundaries = chunks.Take(chunks.Count - 1).Select(c => EndsSentence(c.Text)).ToList();
        double share = boundaries.Count > 0
            ? boundaries.Count(b => b) / (double)boundaries.Count
            : 1.0; // a single chunk has no internal boundaries to botch
        return new StrategyResult(strategy, chunks, new StrategyStats(count, min, max, avg, share));
    }

    /// <summary>Run all three strategies over one document and report comparable stats.</summary>
    public static IReadOnlyDictionary<ChunkStrategy, StrategyResult> CompareStrategies(
        string text, ChunkOptions opts) =>
        new Dictionary<ChunkStrategy, StrategyResult>
        {
            [ChunkStrategy.Fixed] = StatsFor(ChunkStrategy.Fixed, ChunkFixed(text, opts)),
            [ChunkStrategy.Sentence] = StatsFor(ChunkStrategy.Sentence, ChunkBySentences(text, opts)),
            [ChunkStrategy.Markdown] = StatsFor(ChunkStrategy.Markdown, ChunkMarkdown(text, opts)),
        };
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →