RAG Chunk Comparator — Zig source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
// RAG Chunk Comparator — chunk one document three ways and report the stats
// that matter for retrieval.
//
// Language: Zig (0.13+, stdlib only)
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
// implementation). All output is allocated from the caller's allocator;
// chunk text is COPIED (free with freeChunks / freeComparison). The TS
// string operations (regex sentence split, heading match) become manual
// scans. RangeError maps to error unions.
//
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
const std = @import("std");
pub const Strategy = enum { fixed, sentence, markdown };
/// Target chunk size in tokens + fixed-strategy overlap.
pub const Options = struct {
size_tokens: i64,
overlap_tokens: i64 = 0,
};
pub const Chunk = struct {
index: i64,
text: []const u8, // allocated
tokens: i64,
/// Nearest markdown heading; empty slice when absent.
heading: []const u8 = &.{},
};
pub const Stats = struct {
count: i64,
min_tokens: i64,
max_tokens: i64,
avg_tokens: i64,
/// Share of chunk boundaries that fall on a sentence end (0..1).
sentence_boundary_share: f64,
};
pub const Result = struct {
strategy: Strategy,
chunks: []Chunk,
stats: Stats,
};
pub const Comparison = struct {
fixed: Result,
sentence: Result,
markdown: Result,
};
pub const ChunkError = error{
SizeTokens,
OverlapTokens,
OutOfMemory,
};
/// The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
/// line costs max(1, round(length / 4)) tokens; empty text is 0.
fn tok(text: []const u8) i64 {
if (text.len == 0) return 0;
var tokens: i64 = 0;
var start: usize = 0;
while (start <= text.len) {
const end = std.mem.indexOfScalarPos(u8, text, start, '\n') orelse text.len;
const len = end - start;
if (len > 0) {
const per: i64 = @intFromFloat(@as(f64, @floatFromInt(len)) / 4.0 + 0.5);
tokens += @max(1, per);
}
if (end == text.len) break;
start = end + 1;
}
return tokens;
}
fn trim(s: []const u8) []const u8 {
return std.mem.trim(u8, s, " \t\r\n");
}
/// True when the text ends a sentence: [.!?] optionally wrapped by " ' ) ].
fn endsSentence(s: []const u8) bool {
var t = trim(s);
if (t.len == 0) return false;
const wrappers = "\"')]";
if (std.mem.indexOfScalar(u8, wrappers, t[t.len - 1]) != null) {
if (t.len < 2) return false;
t = t[0 .. t.len - 1];
}
if (t.len == 0) return false;
const last = t[t.len - 1];
return last == '.' or last == '!' or last == '?';
}
const Builder = struct {
alloc: std.mem.Allocator,
chunks: std.ArrayList(Chunk),
fn push(self: *Builder, text: []const u8, heading: []const u8) !void {
const piece = trim(text);
if (piece.len == 0) return;
try self.chunks.append(.{
.index = @intCast(self.chunks.items.len),
.text = try self.alloc.dupe(u8, piece),
.tokens = tok(piece),
.heading = if (heading.len > 0) try self.alloc.dupe(u8, heading) else &.{},
});
}
};
/// Split on sentence enders followed by whitespace or end of text, with
/// whitespace collapsed to single spaces. Output strings are allocated.
fn splitSentences(alloc: std.mem.Allocator, text: []const u8, out: *std.ArrayList([]const u8)) !void {
// pass 1: collapse whitespace
var collapsed: std.ArrayList(u8) = .init(alloc);
defer collapsed.deinit();
for (text) |c| {
const ws = c == ' ' or c == '\t' or c == '\n' or c == '\r';
if (ws) {
if (collapsed.items.len > 0 and collapsed.items[collapsed.items.len - 1] != ' ') {
try collapsed.append(' ');
}
} else {
try collapsed.append(c);
}
}
const src = trim(collapsed.items);
// pass 2: cut after [.!?] followed by a space (or the end)
var start: usize = 0;
var i: usize = 0;
while (i < src.len) : (i += 1) {
const ender = src[i] == '.' or src[i] == '!' or src[i] == '?';
const next_is_space = i + 1 < src.len and src[i + 1] == ' ';
if (ender and next_is_space) {
const piece = trim(src[start .. i + 1]);
if (piece.len > 0) try out.append(try alloc.dupe(u8, piece));
start = i + 1;
while (start < src.len and src[start] == ' ') start += 1;
i = start - 1;
}
}
const tail = trim(src[start..]);
if (tail.len > 0) try out.append(try alloc.dupe(u8, tail));
}
/// Greedy character accumulation to a token target (overlapping allowed).
pub fn chunkFixed(alloc: std.mem.Allocator, text: []const u8, opts: Options) ChunkError![]Chunk {
if (opts.size_tokens <= 0) return error.SizeTokens;
if (opts.overlap_tokens < 0 or opts.overlap_tokens >= opts.size_tokens) {
return error.OverlapTokens;
}
const clean = trim(text);
if (clean.len == 0) return &.{};
// ~4 chars per prose token: step by tokens, verify with the estimator.
const step: usize = @intCast(@max(1, opts.size_tokens * 4));
const overlap: usize = @intCast(@max(0, opts.overlap_tokens * 4));
var b: Builder = .{ .alloc = alloc, .chunks = .init(alloc) };
errdefer freeChunks(alloc, b.chunks.items);
var start: usize = 0;
while (start < clean.len) {
var end = @min(start + step, clean.len);
// Prefer cutting at whitespace near the target.
if (end < clean.len) {
var cut: ?usize = null;
var i: usize = end;
while (i > start) {
i -= 1;
if (clean[i] == ' ') {
cut = i;
break;
}
}
if (cut) |c| {
if (c > start) end = c;
}
}
try b.push(clean[start..end], &.{});
if (end >= clean.len) break;
const next = if (end > overlap) end - overlap else start + 1;
start = @max(next, start + 1);
}
return b.chunks.toOwnedSlice() catch return error.OutOfMemory;
}
/// Group whole sentences up to the token target; boundaries never split a
/// sentence. A single sentence larger than the target becomes its own chunk.
pub fn chunkBySentences(alloc: std.mem.Allocator, text: []const u8, opts: Options) ChunkError![]Chunk {
if (opts.size_tokens <= 0) return error.SizeTokens;
var sentences: std.ArrayList([]const u8) = .init(alloc);
defer {
for (sentences.items) |s| alloc.free(s);
sentences.deinit();
}
try splitSentences(alloc, text, &sentences);
if (sentences.items.len == 0) return &.{};
var b: Builder = .{ .alloc = alloc, .chunks = .init(alloc) };
errdefer freeChunks(alloc, b.chunks.items);
var current: std.ArrayList(u8) = .init(alloc);
defer current.deinit();
var current_tokens: i64 = 0;
for (sentences.items) |sentence| {
const t = tok(sentence);
if (current_tokens > 0 and current_tokens + t > opts.size_tokens) {
try b.push(current.items, &.{});
current.clearRetainingCapacity();
current_tokens = 0;
}
if (current.items.len > 0) try current.append(' ');
try current.appendSlice(sentence);
current_tokens += t;
}
try b.push(current.items, &.{});
return b.chunks.toOwnedSlice() catch return error.OutOfMemory;
}
/// One line's heading text, when the line is 1-6 '#' + whitespace + text.
fn headingText(line: []const u8) ?[]const u8 {
var hashes: usize = 0;
while (hashes < line.len and line[hashes] == '#') hashes += 1;
if (hashes < 1 or hashes > 6) return null;
var i = hashes;
if (i >= line.len or (line[i] != ' ' and line[i] != '\t')) return null;
while (i < line.len and (line[i] == ' ' or line[i] == '\t')) i += 1;
const rest = trim(line[i..]);
if (rest.len == 0) return null;
return rest;
}
/// Split on markdown headings; oversized sections fall back to sentence
/// grouping, and every chunk carries the section heading.
pub fn chunkMarkdown(alloc: std.mem.Allocator, text: []const u8, opts: Options) ChunkError![]Chunk {
if (opts.size_tokens <= 0) return error.SizeTokens;
// Pass 1: split into sections at heading lines (bodies joined by '\n').
const Section = struct { heading: []const u8, body: []const u8 };
var sections: std.ArrayList(Section) = .init(alloc);
defer {
for (sections.items) |s| {
alloc.free(s.heading);
alloc.free(s.body);
}
sections.deinit();
}
{
var pending_heading: []const u8 = &.{};
var pending_body: std.ArrayList(u8) = .init(alloc);
defer pending_body.deinit();
var it = std.mem.splitScalar(u8, text, '\n');
while (it.next()) |line| {
if (headingText(line)) |h| {
if (pending_body.items.len > 0) {
try sections.append(.{
.heading = try alloc.dupe(u8, pending_heading),
.body = try alloc.dupe(u8, pending_body.items),
});
}
pending_heading = h;
pending_body.clearRetainingCapacity();
} else {
if (pending_body.items.len > 0) try pending_body.append('\n');
try pending_body.appendSlice(line);
}
}
if (pending_body.items.len > 0) {
try sections.append(.{
.heading = if (pending_heading.len > 0) try alloc.dupe(u8, pending_heading) else &.{},
.body = try alloc.dupe(u8, pending_body.items),
});
}
}
// Pass 2: emit each section whole or sentence-grouped.
var b: Builder = .{ .alloc = alloc, .chunks = .init(alloc) };
errdefer freeChunks(alloc, b.chunks.items);
for (sections.items) |section| {
const clean = trim(section.body);
if (clean.len == 0) continue;
var whole: std.ArrayList(u8) = .init(alloc);
defer whole.deinit();
if (section.heading.len > 0) {
try whole.writer().print("# {s}\n{s}", .{ section.heading, clean });
} else {
try whole.appendSlice(clean);
}
if (tok(whole.items) <= opts.size_tokens) {
try b.push(whole.items, section.heading);
continue;
}
const sub = try chunkBySentences(alloc, clean, opts);
defer freeChunks(alloc, sub);
for (sub) |c| {
try b.push(c.text, section.heading);
}
}
return b.chunks.toOwnedSlice() catch return error.OutOfMemory;
}
fn statsFor(strategy: Strategy, chunks: []Chunk) Result {
var min: i64 = 0;
var max: i64 = 0;
var sum: i64 = 0;
if (chunks.len > 0) {
min = chunks[0].tokens;
max = chunks[0].tokens;
for (chunks) |c| {
min = @min(min, c.tokens);
max = @max(max, c.tokens);
sum += c.tokens;
}
}
var boundaries: usize = 0;
var ending: usize = 0;
for (chunks[0 .. chunks.len -| 1]) |c| {
boundaries += 1;
if (endsSentence(c.text)) ending += 1;
}
return .{
.strategy = strategy,
.chunks = chunks,
.stats = .{
.count = @intCast(chunks.len),
.min_tokens = min,
.max_tokens = max,
.avg_tokens = if (chunks.len > 0) @divTrunc(sum, @as(i64, @intCast(chunks.len))) else 0,
.sentence_boundary_share = if (boundaries > 0)
@as(f64, @floatFromInt(ending)) / @as(f64, @floatFromInt(boundaries))
else
1.0, // a single chunk has no internal boundaries to botch
},
};
}
/// Run all three strategies over one document and report comparable stats.
pub fn compareStrategies(
alloc: std.mem.Allocator,
text: []const u8,
opts: Options,
) ChunkError!Comparison {
const fixed = try chunkFixed(alloc, text, opts);
const sentence = try chunkBySentences(alloc, text, opts);
const markdown = try chunkMarkdown(alloc, text, opts);
return .{
.fixed = statsFor(.fixed, fixed),
.sentence = statsFor(.sentence, sentence),
.markdown = statsFor(.markdown, markdown),
};
}
/// Free one strategy's chunks (each text/heading is a separate allocation).
pub fn freeChunks(alloc: std.mem.Allocator, chunks: []Chunk) void {
for (chunks) |c| {
alloc.free(c.text);
if (c.heading.len > 0) alloc.free(c.heading);
}
alloc.free(chunks);
}
/// Free a whole comparison.
pub fn freeComparison(alloc: std.mem.Allocator, cmp: Comparison) void {
freeChunks(alloc, cmp.fixed.chunks);
freeChunks(alloc, cmp.sentence.chunks);
freeChunks(alloc, cmp.markdown.chunks);
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →