RAG Chunk Comparator — C++ source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// RAG Chunk Comparator — chunk one document three ways (fixed-size,
// sentence-aware, markdown-heading-aware) and compare retrieval stats.
//
// Language: C++ (C++17, standard library only)
// Source: CosmoDev polyglot showcase port of the RAG Chunk Comparator
// tool (slug: rag-chunk-comparator).
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
// implementation).
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Token sizes inline the tokenEstimator prose heuristic (~4 chars per
// token, per non-empty line, minimum one token per line) so this file is
// self-contained; lengths are byte counts, exact for ASCII text. The
// TypeScript RangeError becomes std::invalid_argument.
#include <algorithm>
#include <cmath>
#include <cctype>
#include <optional>
#include <stdexcept>
#include <string>
#include <vector>
/** Target chunk size in tokens; overlap between consecutive fixed chunks. */
struct ChunkOptions {
long sizeTokens;
long overlapTokens = 0; // fixed strategy only
};
enum class ChunkStrategy { Fixed, Sentence, Markdown };
/** One chunk: text, prose-heuristic token count, optional nearest heading. */
struct Chunk {
std::size_t index;
std::string text;
long tokens;
std::optional<std::string> heading; // markdown chunks only
};
struct StrategyStats {
std::size_t count;
long minTokens;
long maxTokens;
long avgTokens;
double sentenceBoundaryShare; // share of boundaries on a sentence end (0-1)
};
struct StrategyResult {
ChunkStrategy strategy;
std::vector<Chunk> chunks;
StrategyStats stats;
};
namespace {
/** Prose token estimate: chars/4 per non-empty line, min 1 per line. */
long tok(const std::string& s) {
long total = 0;
std::size_t start = 0;
while (start <= s.size()) {
std::size_t nl = s.find('\n', start);
std::size_t end = (nl == std::string::npos) ? s.size() : nl;
std::string line = s.substr(start, end - start);
if (!line.empty() && line.back() == '\r') line.pop_back();
bool nonEmpty = false;
for (unsigned char c : line) {
if (std::isspace(c) == 0) { nonEmpty = true; break; }
}
if (nonEmpty) {
total += std::max(1L, static_cast<long>(std::floor(line.size() / 4.0 + 0.5)));
}
if (nl == std::string::npos) break;
start = nl + 1;
}
return total;
}
bool endsSentence(const std::string& s) {
std::size_t b = 0, e = s.size();
while (b < e && std::isspace(static_cast<unsigned char>(s[b])) != 0) ++b;
while (e > b && std::isspace(static_cast<unsigned char>(s[e - 1])) != 0) --e;
if (e == b) return false;
char last = s[e - 1];
auto isEnder = [](char c) { return c == '.' || c == '!' || c == '?'; };
if (isEnder(last)) return true;
if ((last == '"' || last == '\'' || last == ')' || last == ']') && e - b >= 2) {
return isEnder(s[e - 2]);
}
return false;
}
/** Match ^(#{1,6})\s+(.*)$ and return the trimmed heading text. */
std::optional<std::string> parseHeading(const std::string& line) {
std::size_t hashes = 0;
while (hashes < line.size() && line[hashes] == '#') ++hashes;
if (hashes < 1 || hashes > 6) return std::nullopt;
std::size_t i = hashes;
if (i >= line.size() || std::isspace(static_cast<unsigned char>(line[i])) == 0) {
return std::nullopt;
}
while (i < line.size() && std::isspace(static_cast<unsigned char>(line[i])) != 0) ++i;
std::string rest = line.substr(i);
// trim
std::size_t b = rest.find_first_not_of(" \t\r\n\f\v");
if (b == std::string::npos) return std::string{};
std::size_t e = rest.find_last_not_of(" \t\r\n\f\v");
return rest.substr(b, e - b + 1);
}
StrategyResult statsFor(ChunkStrategy strategy, std::vector<Chunk> chunks) {
const std::size_t count = chunks.size();
long minTokens = 0, maxTokens = 0, sum = 0;
for (const Chunk& c : chunks) {
sum += c.tokens;
minTokens = (minTokens == 0) ? c.tokens : std::min(minTokens, c.tokens);
maxTokens = std::max(maxTokens, c.tokens);
}
const long avgTokens = count ? static_cast<long>(std::floor(sum / static_cast<double>(count) + 0.5)) : 0;
std::size_t boundaryCount = 0, boundaryEnds = 0;
for (std::size_t i = 0; i + 1 < count; ++i) {
++boundaryCount;
if (endsSentence(chunks[i].text)) ++boundaryEnds;
}
// A single chunk has no internal boundaries to botch.
const double share = boundaryCount ? static_cast<double>(boundaryEnds) / boundaryCount : 1.0;
return StrategyResult{strategy, std::move(chunks),
StrategyStats{count, minTokens, maxTokens, avgTokens, share}};
}
} // namespace
/** Split on sentence enders followed by whitespace or end of text. */
std::vector<std::string> splitSentences(const std::string& text) {
// Collapse all whitespace runs to single spaces, then trim.
std::string norm;
bool pendingSpace = false;
for (unsigned char c : text) {
if (std::isspace(c) != 0) {
pendingSpace = !norm.empty();
} else {
if (pendingSpace) norm += ' ';
pendingSpace = false;
norm += static_cast<char>(c);
}
}
std::vector<std::string> out;
std::string cur;
for (std::size_t i = 0; i < norm.size(); ++i) {
const char c = norm[i];
if ((c == '.' || c == '!' || c == '?') && i + 1 < norm.size() && norm[i + 1] == ' ') {
cur += c;
out.push_back(cur);
cur.clear();
while (i + 1 < norm.size() && norm[i + 1] == ' ') ++i; // skip the space run
continue;
}
cur += c;
}
if (!cur.empty()) out.push_back(cur);
out.erase(std::remove_if(out.begin(), out.end(),
[](const std::string& s) { return s.empty(); }),
out.end());
return out;
}
/** Greedy character accumulation to a token target (overlapping allowed). */
std::vector<Chunk> chunkFixed(const std::string& text, const ChunkOptions& opts) {
const long sizeTokens = opts.sizeTokens;
const long overlapTokens = opts.overlapTokens;
if (sizeTokens <= 0) throw std::invalid_argument("sizeTokens must be > 0");
if (overlapTokens < 0 || overlapTokens >= sizeTokens) {
throw std::invalid_argument("overlapTokens must be in [0, sizeTokens)");
}
// trim
std::size_t tb = text.find_first_not_of(" \t\r\n\f\v");
if (tb == std::string::npos) return {};
std::size_t te = text.find_last_not_of(" \t\r\n\f\v");
const std::string clean = text.substr(tb, te - tb + 1);
// ~4 chars per prose token: step by tokens, verify with the estimator.
const std::size_t charStep = static_cast<std::size_t>(std::max(1L, sizeTokens * 4));
const std::size_t overlapChars = static_cast<std::size_t>(overlapTokens * 4);
std::vector<Chunk> chunks;
std::size_t start = 0;
while (start < clean.size()) {
std::size_t end = std::min(start + charStep, clean.size());
// Prefer cutting at whitespace near the target — but never trim the
// document's final piece back to a word when it already fits.
if (end < clean.size()) {
std::size_t cut = clean.rfind(' ', end);
if (cut != std::string::npos && cut > start) end = cut;
}
std::size_t pb = start, pe = end;
while (pb < pe && std::isspace(static_cast<unsigned char>(clean[pb])) != 0) ++pb;
while (pe > pb && std::isspace(static_cast<unsigned char>(clean[pe - 1])) != 0) --pe;
if (pb < pe) {
std::string piece = clean.substr(pb, pe - pb);
chunks.push_back(Chunk{chunks.size(), piece, tok(piece), std::nullopt});
}
if (end >= clean.size()) break;
start = std::max(end - overlapChars, start + 1);
}
return chunks;
}
/** Group whole sentences up to the token target; boundaries never split a sentence. */
std::vector<Chunk> chunkBySentences(const std::string& text, const ChunkOptions& opts) {
const long sizeTokens = opts.sizeTokens;
if (sizeTokens <= 0) throw std::invalid_argument("sizeTokens must be > 0");
const std::vector<std::string> sentences = splitSentences(text);
if (sentences.empty()) return {};
std::vector<Chunk> chunks;
std::vector<std::string> current;
long currentTokens = 0;
for (const std::string& sentence : sentences) {
const long t = tok(sentence);
if (currentTokens > 0 && currentTokens + t > sizeTokens) {
std::string piece;
for (std::size_t i = 0; i < current.size(); ++i) {
if (i) piece += ' ';
piece += current[i];
}
chunks.push_back(Chunk{chunks.size(), piece, tok(piece), std::nullopt});
current.clear();
currentTokens = 0;
}
current.push_back(sentence);
currentTokens += t;
// A single sentence larger than the target becomes its own chunk.
}
if (!current.empty()) {
std::string piece;
for (std::size_t i = 0; i < current.size(); ++i) {
if (i) piece += ' ';
piece += current[i];
}
chunks.push_back(Chunk{chunks.size(), piece, tok(piece), std::nullopt});
}
return chunks;
}
/** Split on markdown headings; oversized sections fall back to sentence grouping. */
std::vector<Chunk> chunkMarkdown(const std::string& text, const ChunkOptions& opts) {
const long sizeTokens = opts.sizeTokens;
if (sizeTokens <= 0) throw std::invalid_argument("sizeTokens must be > 0");
struct Section {
std::optional<std::string> heading;
std::vector<std::string> body;
};
std::vector<Section> sections;
Section current;
std::size_t start = 0;
while (start <= text.size()) {
std::size_t nl = text.find('\n', start);
std::size_t end = (nl == std::string::npos) ? text.size() : nl;
std::string line = text.substr(start, end - start);
if (std::optional<std::string> h = parseHeading(line)) {
if (!current.body.empty()) sections.push_back(std::move(current));
current = Section{std::move(h), {}};
} else {
current.body.push_back(std::move(line));
}
if (nl == std::string::npos) break;
start = nl + 1;
}
if (!current.body.empty()) sections.push_back(std::move(current));
std::vector<Chunk> chunks;
for (Section& section : sections) {
std::string joined;
for (std::size_t i = 0; i < section.body.size(); ++i) {
if (i) joined += '\n';
joined += section.body[i];
}
std::size_t bb = joined.find_first_not_of(" \t\r\n\f\v");
if (bb == std::string::npos) continue;
std::size_t be = joined.find_last_not_of(" \t\r\n\f\v");
const std::string body = joined.substr(bb, be - bb + 1);
const std::string whole = section.heading
? "# " + *section.heading + "\n" + body
: body;
if (tok(whole) <= sizeTokens) {
chunks.push_back(Chunk{chunks.size(), whole, tok(whole), section.heading});
continue;
}
// Oversized section: sentence-group the body, stamp every chunk with
// the heading.
for (Chunk& c : chunkBySentences(body, opts)) {
chunks.push_back(Chunk{chunks.size(), std::move(c.text), c.tokens, section.heading});
}
}
return chunks;
}
/** All three strategies over one document, as comparable stats. */
struct CompareResults {
StrategyResult fixed_;
StrategyResult sentence;
StrategyResult markdown;
};
/** Run all three strategies over one document and report comparable stats. */
CompareResults compareStrategies(const std::string& text, const ChunkOptions& opts) {
return CompareResults{
statsFor(ChunkStrategy::Fixed, chunkFixed(text, opts)),
statsFor(ChunkStrategy::Sentence, chunkBySentences(text, opts)),
statsFor(ChunkStrategy::Markdown, chunkMarkdown(text, opts)),
};
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →