Skip to content

RAG Chunk Comparator — C++ source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// RAG Chunk Comparator — chunk one document three ways (fixed-size,
// sentence-aware, markdown-heading-aware) and compare retrieval stats.
//
// Language: C++ (C++17, standard library only)
// Source:   CosmoDev polyglot showcase port of the RAG Chunk Comparator
//           tool (slug: rag-chunk-comparator).
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
//           implementation).
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Token sizes inline the tokenEstimator prose heuristic (~4 chars per
// token, per non-empty line, minimum one token per line) so this file is
// self-contained; lengths are byte counts, exact for ASCII text. The
// TypeScript RangeError becomes std::invalid_argument.

#include <algorithm>
#include <cmath>
#include <cctype>
#include <optional>
#include <stdexcept>
#include <string>
#include <vector>

/** Target chunk size in tokens; overlap between consecutive fixed chunks. */
struct ChunkOptions {
    long sizeTokens;
    long overlapTokens = 0;  // fixed strategy only
};

enum class ChunkStrategy { Fixed, Sentence, Markdown };

/** One chunk: text, prose-heuristic token count, optional nearest heading. */
struct Chunk {
    std::size_t index;
    std::string text;
    long tokens;
    std::optional<std::string> heading;  // markdown chunks only
};

struct StrategyStats {
    std::size_t count;
    long minTokens;
    long maxTokens;
    long avgTokens;
    double sentenceBoundaryShare;  // share of boundaries on a sentence end (0-1)
};

struct StrategyResult {
    ChunkStrategy strategy;
    std::vector<Chunk> chunks;
    StrategyStats stats;
};

namespace {

/** Prose token estimate: chars/4 per non-empty line, min 1 per line. */
long tok(const std::string& s) {
    long total = 0;
    std::size_t start = 0;
    while (start <= s.size()) {
        std::size_t nl = s.find('\n', start);
        std::size_t end = (nl == std::string::npos) ? s.size() : nl;
        std::string line = s.substr(start, end - start);
        if (!line.empty() && line.back() == '\r') line.pop_back();
        bool nonEmpty = false;
        for (unsigned char c : line) {
            if (std::isspace(c) == 0) { nonEmpty = true; break; }
        }
        if (nonEmpty) {
            total += std::max(1L, static_cast<long>(std::floor(line.size() / 4.0 + 0.5)));
        }
        if (nl == std::string::npos) break;
        start = nl + 1;
    }
    return total;
}

bool endsSentence(const std::string& s) {
    std::size_t b = 0, e = s.size();
    while (b < e && std::isspace(static_cast<unsigned char>(s[b])) != 0) ++b;
    while (e > b && std::isspace(static_cast<unsigned char>(s[e - 1])) != 0) --e;
    if (e == b) return false;
    char last = s[e - 1];
    auto isEnder = [](char c) { return c == '.' || c == '!' || c == '?'; };
    if (isEnder(last)) return true;
    if ((last == '"' || last == '\'' || last == ')' || last == ']') && e - b >= 2) {
        return isEnder(s[e - 2]);
    }
    return false;
}

/** Match ^(#{1,6})\s+(.*)$ and return the trimmed heading text. */
std::optional<std::string> parseHeading(const std::string& line) {
    std::size_t hashes = 0;
    while (hashes < line.size() && line[hashes] == '#') ++hashes;
    if (hashes < 1 || hashes > 6) return std::nullopt;
    std::size_t i = hashes;
    if (i >= line.size() || std::isspace(static_cast<unsigned char>(line[i])) == 0) {
        return std::nullopt;
    }
    while (i < line.size() && std::isspace(static_cast<unsigned char>(line[i])) != 0) ++i;
    std::string rest = line.substr(i);
    // trim
    std::size_t b = rest.find_first_not_of(" \t\r\n\f\v");
    if (b == std::string::npos) return std::string{};
    std::size_t e = rest.find_last_not_of(" \t\r\n\f\v");
    return rest.substr(b, e - b + 1);
}

StrategyResult statsFor(ChunkStrategy strategy, std::vector<Chunk> chunks) {
    const std::size_t count = chunks.size();
    long minTokens = 0, maxTokens = 0, sum = 0;
    for (const Chunk& c : chunks) {
        sum += c.tokens;
        minTokens = (minTokens == 0) ? c.tokens : std::min(minTokens, c.tokens);
        maxTokens = std::max(maxTokens, c.tokens);
    }
    const long avgTokens = count ? static_cast<long>(std::floor(sum / static_cast<double>(count) + 0.5)) : 0;
    std::size_t boundaryCount = 0, boundaryEnds = 0;
    for (std::size_t i = 0; i + 1 < count; ++i) {
        ++boundaryCount;
        if (endsSentence(chunks[i].text)) ++boundaryEnds;
    }
    // A single chunk has no internal boundaries to botch.
    const double share = boundaryCount ? static_cast<double>(boundaryEnds) / boundaryCount : 1.0;
    return StrategyResult{strategy, std::move(chunks),
                          StrategyStats{count, minTokens, maxTokens, avgTokens, share}};
}

}  // namespace

/** Split on sentence enders followed by whitespace or end of text. */
std::vector<std::string> splitSentences(const std::string& text) {
    // Collapse all whitespace runs to single spaces, then trim.
    std::string norm;
    bool pendingSpace = false;
    for (unsigned char c : text) {
        if (std::isspace(c) != 0) {
            pendingSpace = !norm.empty();
        } else {
            if (pendingSpace) norm += ' ';
            pendingSpace = false;
            norm += static_cast<char>(c);
        }
    }
    std::vector<std::string> out;
    std::string cur;
    for (std::size_t i = 0; i < norm.size(); ++i) {
        const char c = norm[i];
        if ((c == '.' || c == '!' || c == '?') && i + 1 < norm.size() && norm[i + 1] == ' ') {
            cur += c;
            out.push_back(cur);
            cur.clear();
            while (i + 1 < norm.size() && norm[i + 1] == ' ') ++i;  // skip the space run
            continue;
        }
        cur += c;
    }
    if (!cur.empty()) out.push_back(cur);
    out.erase(std::remove_if(out.begin(), out.end(),
                             [](const std::string& s) { return s.empty(); }),
              out.end());
    return out;
}

/** Greedy character accumulation to a token target (overlapping allowed). */
std::vector<Chunk> chunkFixed(const std::string& text, const ChunkOptions& opts) {
    const long sizeTokens = opts.sizeTokens;
    const long overlapTokens = opts.overlapTokens;
    if (sizeTokens <= 0) throw std::invalid_argument("sizeTokens must be > 0");
    if (overlapTokens < 0 || overlapTokens >= sizeTokens) {
        throw std::invalid_argument("overlapTokens must be in [0, sizeTokens)");
    }
    // trim
    std::size_t tb = text.find_first_not_of(" \t\r\n\f\v");
    if (tb == std::string::npos) return {};
    std::size_t te = text.find_last_not_of(" \t\r\n\f\v");
    const std::string clean = text.substr(tb, te - tb + 1);
    // ~4 chars per prose token: step by tokens, verify with the estimator.
    const std::size_t charStep = static_cast<std::size_t>(std::max(1L, sizeTokens * 4));
    const std::size_t overlapChars = static_cast<std::size_t>(overlapTokens * 4);
    std::vector<Chunk> chunks;
    std::size_t start = 0;
    while (start < clean.size()) {
        std::size_t end = std::min(start + charStep, clean.size());
        // Prefer cutting at whitespace near the target — but never trim the
        // document's final piece back to a word when it already fits.
        if (end < clean.size()) {
            std::size_t cut = clean.rfind(' ', end);
            if (cut != std::string::npos && cut > start) end = cut;
        }
        std::size_t pb = start, pe = end;
        while (pb < pe && std::isspace(static_cast<unsigned char>(clean[pb])) != 0) ++pb;
        while (pe > pb && std::isspace(static_cast<unsigned char>(clean[pe - 1])) != 0) --pe;
        if (pb < pe) {
            std::string piece = clean.substr(pb, pe - pb);
            chunks.push_back(Chunk{chunks.size(), piece, tok(piece), std::nullopt});
        }
        if (end >= clean.size()) break;
        start = std::max(end - overlapChars, start + 1);
    }
    return chunks;
}

/** Group whole sentences up to the token target; boundaries never split a sentence. */
std::vector<Chunk> chunkBySentences(const std::string& text, const ChunkOptions& opts) {
    const long sizeTokens = opts.sizeTokens;
    if (sizeTokens <= 0) throw std::invalid_argument("sizeTokens must be > 0");
    const std::vector<std::string> sentences = splitSentences(text);
    if (sentences.empty()) return {};
    std::vector<Chunk> chunks;
    std::vector<std::string> current;
    long currentTokens = 0;
    for (const std::string& sentence : sentences) {
        const long t = tok(sentence);
        if (currentTokens > 0 && currentTokens + t > sizeTokens) {
            std::string piece;
            for (std::size_t i = 0; i < current.size(); ++i) {
                if (i) piece += ' ';
                piece += current[i];
            }
            chunks.push_back(Chunk{chunks.size(), piece, tok(piece), std::nullopt});
            current.clear();
            currentTokens = 0;
        }
        current.push_back(sentence);
        currentTokens += t;
        // A single sentence larger than the target becomes its own chunk.
    }
    if (!current.empty()) {
        std::string piece;
        for (std::size_t i = 0; i < current.size(); ++i) {
            if (i) piece += ' ';
            piece += current[i];
        }
        chunks.push_back(Chunk{chunks.size(), piece, tok(piece), std::nullopt});
    }
    return chunks;
}

/** Split on markdown headings; oversized sections fall back to sentence grouping. */
std::vector<Chunk> chunkMarkdown(const std::string& text, const ChunkOptions& opts) {
    const long sizeTokens = opts.sizeTokens;
    if (sizeTokens <= 0) throw std::invalid_argument("sizeTokens must be > 0");
    struct Section {
        std::optional<std::string> heading;
        std::vector<std::string> body;
    };
    std::vector<Section> sections;
    Section current;
    std::size_t start = 0;
    while (start <= text.size()) {
        std::size_t nl = text.find('\n', start);
        std::size_t end = (nl == std::string::npos) ? text.size() : nl;
        std::string line = text.substr(start, end - start);
        if (std::optional<std::string> h = parseHeading(line)) {
            if (!current.body.empty()) sections.push_back(std::move(current));
            current = Section{std::move(h), {}};
        } else {
            current.body.push_back(std::move(line));
        }
        if (nl == std::string::npos) break;
        start = nl + 1;
    }
    if (!current.body.empty()) sections.push_back(std::move(current));

    std::vector<Chunk> chunks;
    for (Section& section : sections) {
        std::string joined;
        for (std::size_t i = 0; i < section.body.size(); ++i) {
            if (i) joined += '\n';
            joined += section.body[i];
        }
        std::size_t bb = joined.find_first_not_of(" \t\r\n\f\v");
        if (bb == std::string::npos) continue;
        std::size_t be = joined.find_last_not_of(" \t\r\n\f\v");
        const std::string body = joined.substr(bb, be - bb + 1);
        const std::string whole = section.heading
                                      ? "# " + *section.heading + "\n" + body
                                      : body;
        if (tok(whole) <= sizeTokens) {
            chunks.push_back(Chunk{chunks.size(), whole, tok(whole), section.heading});
            continue;
        }
        // Oversized section: sentence-group the body, stamp every chunk with
        // the heading.
        for (Chunk& c : chunkBySentences(body, opts)) {
            chunks.push_back(Chunk{chunks.size(), std::move(c.text), c.tokens, section.heading});
        }
    }
    return chunks;
}

/** All three strategies over one document, as comparable stats. */
struct CompareResults {
    StrategyResult fixed_;
    StrategyResult sentence;
    StrategyResult markdown;
};

/** Run all three strategies over one document and report comparable stats. */
CompareResults compareStrategies(const std::string& text, const ChunkOptions& opts) {
    return CompareResults{
        statsFor(ChunkStrategy::Fixed, chunkFixed(text, opts)),
        statsFor(ChunkStrategy::Sentence, chunkBySentences(text, opts)),
        statsFor(ChunkStrategy::Markdown, chunkMarkdown(text, opts)),
    };
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →