Skip to content

RAG Chunk Comparator — Zig source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// RAG Chunk Comparator — chunk one document three ways and report the stats
// that matter for retrieval.
//
// Language: Zig (0.13+, stdlib only)
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
// implementation). All output is allocated from the caller's allocator;
// chunk text is COPIED (free with freeChunks / freeComparison). The TS
// string operations (regex sentence split, heading match) become manual
// scans. RangeError maps to error unions.
//
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator

const std = @import("std");

pub const Strategy = enum { fixed, sentence, markdown };

/// Target chunk size in tokens + fixed-strategy overlap.
pub const Options = struct {
    size_tokens: i64,
    overlap_tokens: i64 = 0,
};

pub const Chunk = struct {
    index: i64,
    text: []const u8, // allocated
    tokens: i64,
    /// Nearest markdown heading; empty slice when absent.
    heading: []const u8 = &.{},
};

pub const Stats = struct {
    count: i64,
    min_tokens: i64,
    max_tokens: i64,
    avg_tokens: i64,
    /// Share of chunk boundaries that fall on a sentence end (0..1).
    sentence_boundary_share: f64,
};

pub const Result = struct {
    strategy: Strategy,
    chunks: []Chunk,
    stats: Stats,
};

pub const Comparison = struct {
    fixed: Result,
    sentence: Result,
    markdown: Result,
};

pub const ChunkError = error{
    SizeTokens,
    OverlapTokens,
    OutOfMemory,
};

/// The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
/// line costs max(1, round(length / 4)) tokens; empty text is 0.
fn tok(text: []const u8) i64 {
    if (text.len == 0) return 0;
    var tokens: i64 = 0;
    var start: usize = 0;
    while (start <= text.len) {
        const end = std.mem.indexOfScalarPos(u8, text, start, '\n') orelse text.len;
        const len = end - start;
        if (len > 0) {
            const per: i64 = @intFromFloat(@as(f64, @floatFromInt(len)) / 4.0 + 0.5);
            tokens += @max(1, per);
        }
        if (end == text.len) break;
        start = end + 1;
    }
    return tokens;
}

fn trim(s: []const u8) []const u8 {
    return std.mem.trim(u8, s, " \t\r\n");
}

/// True when the text ends a sentence: [.!?] optionally wrapped by " ' ) ].
fn endsSentence(s: []const u8) bool {
    var t = trim(s);
    if (t.len == 0) return false;
    const wrappers = "\"')]";
    if (std.mem.indexOfScalar(u8, wrappers, t[t.len - 1]) != null) {
        if (t.len < 2) return false;
        t = t[0 .. t.len - 1];
    }
    if (t.len == 0) return false;
    const last = t[t.len - 1];
    return last == '.' or last == '!' or last == '?';
}

const Builder = struct {
    alloc: std.mem.Allocator,
    chunks: std.ArrayList(Chunk),

    fn push(self: *Builder, text: []const u8, heading: []const u8) !void {
        const piece = trim(text);
        if (piece.len == 0) return;
        try self.chunks.append(.{
            .index = @intCast(self.chunks.items.len),
            .text = try self.alloc.dupe(u8, piece),
            .tokens = tok(piece),
            .heading = if (heading.len > 0) try self.alloc.dupe(u8, heading) else &.{},
        });
    }
};

/// Split on sentence enders followed by whitespace or end of text, with
/// whitespace collapsed to single spaces. Output strings are allocated.
fn splitSentences(alloc: std.mem.Allocator, text: []const u8, out: *std.ArrayList([]const u8)) !void {
    // pass 1: collapse whitespace
    var collapsed: std.ArrayList(u8) = .init(alloc);
    defer collapsed.deinit();
    for (text) |c| {
        const ws = c == ' ' or c == '\t' or c == '\n' or c == '\r';
        if (ws) {
            if (collapsed.items.len > 0 and collapsed.items[collapsed.items.len - 1] != ' ') {
                try collapsed.append(' ');
            }
        } else {
            try collapsed.append(c);
        }
    }
    const src = trim(collapsed.items);
    // pass 2: cut after [.!?] followed by a space (or the end)
    var start: usize = 0;
    var i: usize = 0;
    while (i < src.len) : (i += 1) {
        const ender = src[i] == '.' or src[i] == '!' or src[i] == '?';
        const next_is_space = i + 1 < src.len and src[i + 1] == ' ';
        if (ender and next_is_space) {
            const piece = trim(src[start .. i + 1]);
            if (piece.len > 0) try out.append(try alloc.dupe(u8, piece));
            start = i + 1;
            while (start < src.len and src[start] == ' ') start += 1;
            i = start - 1;
        }
    }
    const tail = trim(src[start..]);
    if (tail.len > 0) try out.append(try alloc.dupe(u8, tail));
}

/// Greedy character accumulation to a token target (overlapping allowed).
pub fn chunkFixed(alloc: std.mem.Allocator, text: []const u8, opts: Options) ChunkError![]Chunk {
    if (opts.size_tokens <= 0) return error.SizeTokens;
    if (opts.overlap_tokens < 0 or opts.overlap_tokens >= opts.size_tokens) {
        return error.OverlapTokens;
    }
    const clean = trim(text);
    if (clean.len == 0) return &.{};

    // ~4 chars per prose token: step by tokens, verify with the estimator.
    const step: usize = @intCast(@max(1, opts.size_tokens * 4));
    const overlap: usize = @intCast(@max(0, opts.overlap_tokens * 4));

    var b: Builder = .{ .alloc = alloc, .chunks = .init(alloc) };
    errdefer freeChunks(alloc, b.chunks.items);
    var start: usize = 0;
    while (start < clean.len) {
        var end = @min(start + step, clean.len);
        // Prefer cutting at whitespace near the target.
        if (end < clean.len) {
            var cut: ?usize = null;
            var i: usize = end;
            while (i > start) {
                i -= 1;
                if (clean[i] == ' ') {
                    cut = i;
                    break;
                }
            }
            if (cut) |c| {
                if (c > start) end = c;
            }
        }
        try b.push(clean[start..end], &.{});
        if (end >= clean.len) break;
        const next = if (end > overlap) end - overlap else start + 1;
        start = @max(next, start + 1);
    }
    return b.chunks.toOwnedSlice() catch return error.OutOfMemory;
}

/// Group whole sentences up to the token target; boundaries never split a
/// sentence. A single sentence larger than the target becomes its own chunk.
pub fn chunkBySentences(alloc: std.mem.Allocator, text: []const u8, opts: Options) ChunkError![]Chunk {
    if (opts.size_tokens <= 0) return error.SizeTokens;
    var sentences: std.ArrayList([]const u8) = .init(alloc);
    defer {
        for (sentences.items) |s| alloc.free(s);
        sentences.deinit();
    }
    try splitSentences(alloc, text, &sentences);
    if (sentences.items.len == 0) return &.{};

    var b: Builder = .{ .alloc = alloc, .chunks = .init(alloc) };
    errdefer freeChunks(alloc, b.chunks.items);
    var current: std.ArrayList(u8) = .init(alloc);
    defer current.deinit();
    var current_tokens: i64 = 0;
    for (sentences.items) |sentence| {
        const t = tok(sentence);
        if (current_tokens > 0 and current_tokens + t > opts.size_tokens) {
            try b.push(current.items, &.{});
            current.clearRetainingCapacity();
            current_tokens = 0;
        }
        if (current.items.len > 0) try current.append(' ');
        try current.appendSlice(sentence);
        current_tokens += t;
    }
    try b.push(current.items, &.{});
    return b.chunks.toOwnedSlice() catch return error.OutOfMemory;
}

/// One line's heading text, when the line is 1-6 '#' + whitespace + text.
fn headingText(line: []const u8) ?[]const u8 {
    var hashes: usize = 0;
    while (hashes < line.len and line[hashes] == '#') hashes += 1;
    if (hashes < 1 or hashes > 6) return null;
    var i = hashes;
    if (i >= line.len or (line[i] != ' ' and line[i] != '\t')) return null;
    while (i < line.len and (line[i] == ' ' or line[i] == '\t')) i += 1;
    const rest = trim(line[i..]);
    if (rest.len == 0) return null;
    return rest;
}

/// Split on markdown headings; oversized sections fall back to sentence
/// grouping, and every chunk carries the section heading.
pub fn chunkMarkdown(alloc: std.mem.Allocator, text: []const u8, opts: Options) ChunkError![]Chunk {
    if (opts.size_tokens <= 0) return error.SizeTokens;

    // Pass 1: split into sections at heading lines (bodies joined by '\n').
    const Section = struct { heading: []const u8, body: []const u8 };
    var sections: std.ArrayList(Section) = .init(alloc);
    defer {
        for (sections.items) |s| {
            alloc.free(s.heading);
            alloc.free(s.body);
        }
        sections.deinit();
    }
    {
        var pending_heading: []const u8 = &.{};
        var pending_body: std.ArrayList(u8) = .init(alloc);
        defer pending_body.deinit();
        var it = std.mem.splitScalar(u8, text, '\n');
        while (it.next()) |line| {
            if (headingText(line)) |h| {
                if (pending_body.items.len > 0) {
                    try sections.append(.{
                        .heading = try alloc.dupe(u8, pending_heading),
                        .body = try alloc.dupe(u8, pending_body.items),
                    });
                }
                pending_heading = h;
                pending_body.clearRetainingCapacity();
            } else {
                if (pending_body.items.len > 0) try pending_body.append('\n');
                try pending_body.appendSlice(line);
            }
        }
        if (pending_body.items.len > 0) {
            try sections.append(.{
                .heading = if (pending_heading.len > 0) try alloc.dupe(u8, pending_heading) else &.{},
                .body = try alloc.dupe(u8, pending_body.items),
            });
        }
    }

    // Pass 2: emit each section whole or sentence-grouped.
    var b: Builder = .{ .alloc = alloc, .chunks = .init(alloc) };
    errdefer freeChunks(alloc, b.chunks.items);
    for (sections.items) |section| {
        const clean = trim(section.body);
        if (clean.len == 0) continue;
        var whole: std.ArrayList(u8) = .init(alloc);
        defer whole.deinit();
        if (section.heading.len > 0) {
            try whole.writer().print("# {s}\n{s}", .{ section.heading, clean });
        } else {
            try whole.appendSlice(clean);
        }
        if (tok(whole.items) <= opts.size_tokens) {
            try b.push(whole.items, section.heading);
            continue;
        }
        const sub = try chunkBySentences(alloc, clean, opts);
        defer freeChunks(alloc, sub);
        for (sub) |c| {
            try b.push(c.text, section.heading);
        }
    }
    return b.chunks.toOwnedSlice() catch return error.OutOfMemory;
}

fn statsFor(strategy: Strategy, chunks: []Chunk) Result {
    var min: i64 = 0;
    var max: i64 = 0;
    var sum: i64 = 0;
    if (chunks.len > 0) {
        min = chunks[0].tokens;
        max = chunks[0].tokens;
        for (chunks) |c| {
            min = @min(min, c.tokens);
            max = @max(max, c.tokens);
            sum += c.tokens;
        }
    }
    var boundaries: usize = 0;
    var ending: usize = 0;
    for (chunks[0 .. chunks.len -| 1]) |c| {
        boundaries += 1;
        if (endsSentence(c.text)) ending += 1;
    }
    return .{
        .strategy = strategy,
        .chunks = chunks,
        .stats = .{
            .count = @intCast(chunks.len),
            .min_tokens = min,
            .max_tokens = max,
            .avg_tokens = if (chunks.len > 0) @divTrunc(sum, @as(i64, @intCast(chunks.len))) else 0,
            .sentence_boundary_share = if (boundaries > 0)
                @as(f64, @floatFromInt(ending)) / @as(f64, @floatFromInt(boundaries))
            else
                1.0, // a single chunk has no internal boundaries to botch
        },
    };
}

/// Run all three strategies over one document and report comparable stats.
pub fn compareStrategies(
    alloc: std.mem.Allocator,
    text: []const u8,
    opts: Options,
) ChunkError!Comparison {
    const fixed = try chunkFixed(alloc, text, opts);
    const sentence = try chunkBySentences(alloc, text, opts);
    const markdown = try chunkMarkdown(alloc, text, opts);
    return .{
        .fixed = statsFor(.fixed, fixed),
        .sentence = statsFor(.sentence, sentence),
        .markdown = statsFor(.markdown, markdown),
    };
}

/// Free one strategy's chunks (each text/heading is a separate allocation).
pub fn freeChunks(alloc: std.mem.Allocator, chunks: []Chunk) void {
    for (chunks) |c| {
        alloc.free(c.text);
        if (c.heading.len > 0) alloc.free(c.heading);
    }
    alloc.free(chunks);
}

/// Free a whole comparison.
pub fn freeComparison(alloc: std.mem.Allocator, cmp: Comparison) void {
    freeChunks(alloc, cmp.fixed.chunks);
    freeChunks(alloc, cmp.sentence.chunks);
    freeChunks(alloc, cmp.markdown.chunks);
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →