RAG Chunk Comparator — C# source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.
// RAG Chunk Comparator — chunk one document three ways and report the stats
// that matter for retrieval.
//
// Language: C# (.NET 8, zero dependencies)
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
using System;
using System.Collections.Generic;
using System.Linq;
using System.Text;
using System.Text.RegularExpressions;
namespace CosmoDev.RagChunkComparator;
public enum ChunkStrategy { Fixed, Sentence, Markdown }
public sealed record ChunkOptions(
int SizeTokens,
int OverlapTokens = 0);
public sealed record Chunk(
int Index,
string Text,
int Tokens,
string? Heading = null);
public sealed record StrategyStats(
int Count,
int MinTokens,
int MaxTokens,
int AvgTokens,
double SentenceBoundaryShare);
public sealed record StrategyResult(
ChunkStrategy Strategy,
IReadOnlyList<Chunk> Chunks,
StrategyStats Stats);
public static class RagChunkComparator
{
// The `type: 'prose'` path of the tokenEstimator, inlined: every
// non-empty line costs max(1, round(length / 4)) tokens; empty text is 0.
private static int Tok(string s)
{
if (string.IsNullOrEmpty(s)) return 0;
int tokens = 0;
foreach (var line in s.Split('\n'))
{
if (line.Length > 0)
{
tokens += Math.Max(1, (int)Math.Round(line.Length / 4.0, MidpointRounding.AwayFromZero));
}
}
return tokens;
}
/// <summary>Split on sentence enders followed by whitespace or end of text.</summary>
public static List<string> SplitSentences(string text)
{
var collapsed = Regex.Replace(text, @"\s+", " ").Trim();
return Regex.Split(collapsed, @"(?<=[.!?]) +")
.Where(s => s.Length > 0)
.ToList();
}
private static bool EndsSentence(string s) =>
Regex.IsMatch(s.Trim(), @"[.!?][""')\]]?$");
/// <summary>
/// Greedy character accumulation to a token target (overlapping allowed).
/// Throws <see cref="ArgumentOutOfRangeException"/> on impossible options
/// (the TS RangeError contract).
/// </summary>
public static List<Chunk> ChunkFixed(string text, ChunkOptions opts)
{
if (opts.SizeTokens <= 0) throw new ArgumentOutOfRangeException(nameof(opts), "sizeTokens must be > 0");
if (opts.OverlapTokens < 0 || opts.OverlapTokens >= opts.SizeTokens)
throw new ArgumentOutOfRangeException(nameof(opts), "overlapTokens must be in [0, sizeTokens)");
var clean = text.Trim();
if (clean.Length == 0) return new List<Chunk>();
// ~4 chars per prose token: step by tokens, verify with the estimator.
int charStep = Math.Max(1, (int)Math.Round(opts.SizeTokens * 4.0, MidpointRounding.AwayFromZero));
int overlapChars = (int)Math.Round(opts.OverlapTokens * 4.0, MidpointRounding.AwayFromZero);
var chunks = new List<Chunk>();
int start = 0;
while (start < clean.Length)
{
int end = Math.Min(start + charStep, clean.Length);
// Prefer cutting at whitespace near the target.
if (end < clean.Length)
{
int cut = clean.LastIndexOf(' ', end);
if (cut > start) end = cut;
}
var piece = clean[start..end].Trim();
if (piece.Length > 0) chunks.Add(new Chunk(chunks.Count, piece, Tok(piece)));
if (end >= clean.Length) break;
start = Math.Max(end - overlapChars, start + 1);
}
return chunks;
}
/// <summary>
/// Group whole sentences up to the token target; boundaries never split a
/// sentence. A single sentence larger than the target becomes its own chunk.
/// </summary>
public static List<Chunk> ChunkBySentences(string text, ChunkOptions opts)
{
if (opts.SizeTokens <= 0) throw new ArgumentOutOfRangeException(nameof(opts), "sizeTokens must be > 0");
var sentences = SplitSentences(text);
if (sentences.Count == 0) return new List<Chunk>();
var chunks = new List<Chunk>();
var current = new List<string>();
int currentTokens = 0;
void Flush()
{
if (current.Count == 0) return;
var piece = string.Join(" ", current);
chunks.Add(new Chunk(chunks.Count, piece, Tok(piece)));
current = new List<string>();
currentTokens = 0;
}
foreach (var sentence in sentences)
{
int t = Tok(sentence);
if (currentTokens > 0 && currentTokens + t > opts.SizeTokens) Flush();
current.Add(sentence);
currentTokens += t;
}
Flush();
return chunks;
}
private static readonly Regex HeadingRx = new(@"^(#{1,6})\s+(.*)$", RegexOptions.Compiled);
/// <summary>
/// Split on markdown headings; oversized sections fall back to sentence
/// grouping, and every chunk carries the section heading.
/// </summary>
public static List<Chunk> ChunkMarkdown(string text, ChunkOptions opts)
{
if (opts.SizeTokens <= 0) throw new ArgumentOutOfRangeException(nameof(opts), "sizeTokens must be > 0");
var lines = text.Split('\n');
var sections = new List<(string? Heading, List<string> Body)>();
var current = (Heading: (string?)null, Body: new List<string>());
foreach (var line in lines)
{
var m = HeadingRx.Match(line);
if (m.Success)
{
if (current.Body.Count > 0) sections.Add(current);
current = (m.Groups[2].Value.Trim(), new List<string>());
}
else
{
current.Body.Add(line);
}
}
if (current.Body.Count > 0) sections.Add(current);
var chunks = new List<Chunk>();
foreach (var (heading, body) in sections)
{
var clean = string.Join("\n", body).Trim();
if (clean.Length == 0) continue;
var whole = heading != null ? $"# {heading}\n{clean}" : clean;
if (Tok(whole) <= opts.SizeTokens)
{
chunks.Add(new Chunk(chunks.Count, whole, Tok(whole), heading));
continue;
}
foreach (var c in ChunkBySentences(clean, opts))
{
chunks.Add(new Chunk(chunks.Count, c.Text, c.Tokens, heading));
}
}
return chunks;
}
private static StrategyResult StatsFor(ChunkStrategy strategy, IReadOnlyList<Chunk> chunks)
{
int count = chunks.Count;
int min = count > 0 ? chunks.Min(c => c.Tokens) : 0;
int max = count > 0 ? chunks.Max(c => c.Tokens) : 0;
int avg = count > 0 ? (int)Math.Round(chunks.Sum(c => (double)c.Tokens) / count, MidpointRounding.AwayFromZero) : 0;
var boundaries = chunks.Take(chunks.Count - 1).Select(c => EndsSentence(c.Text)).ToList();
double share = boundaries.Count > 0
? boundaries.Count(b => b) / (double)boundaries.Count
: 1.0; // a single chunk has no internal boundaries to botch
return new StrategyResult(strategy, chunks, new StrategyStats(count, min, max, avg, share));
}
/// <summary>Run all three strategies over one document and report comparable stats.</summary>
public static IReadOnlyDictionary<ChunkStrategy, StrategyResult> CompareStrategies(
string text, ChunkOptions opts) =>
new Dictionary<ChunkStrategy, StrategyResult>
{
[ChunkStrategy.Fixed] = StatsFor(ChunkStrategy.Fixed, ChunkFixed(text, opts)),
[ChunkStrategy.Sentence] = StatsFor(ChunkStrategy.Sentence, ChunkBySentences(text, opts)),
[ChunkStrategy.Markdown] = StatsFor(ChunkStrategy.Markdown, ChunkMarkdown(text, opts)),
};
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →