RAG Chunk Comparator — C source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/*
* RAG Chunk Comparator — chunk one document three ways and report the stats
* that matter for retrieval.
*
* Language: C (C11, standard library only)
* Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
* implementation). The sibling javascript.js carries the same port.
* Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
*
* C shape: fixed-capacity output; chunk text is COPIED into caller-owned
* buffers carved from one arena (a single big malloc). The TS string
* operations (regex sentence split, heading match) become manual scans.
* RangeError maps to a negative return code; 0 is success.
*/
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#define RAG_MAX_CHUNKS 128
#define RAG_CHUNK_TEXT 4096
#define RAG_HEADING 256
typedef enum { RAG_FIXED = 0, RAG_SENTENCE = 1, RAG_MARKDOWN = 2 } rag_strategy;
typedef struct {
long size_tokens;
long overlap_tokens; /* fixed strategy only; 0 default */
} rag_options;
typedef struct {
long index;
char text[RAG_CHUNK_TEXT];
long tokens;
int has_heading;
char heading[RAG_HEADING];
} rag_chunk;
typedef struct {
long count;
long min_tokens;
long max_tokens;
long avg_tokens;
double sentence_boundary_share; /* 0..1; 1 when there are no boundaries */
} rag_stats;
typedef struct {
rag_chunk chunks[RAG_MAX_CHUNKS];
size_t chunk_count;
rag_stats stats;
} rag_result;
typedef struct {
rag_result fixed;
rag_result sentence;
rag_result markdown;
} rag_comparison;
/* Error codes mirroring the TS RangeError contract. */
#define RAG_ERR_SIZE_TOKENS (-1) /* sizeTokens must be > 0 */
#define RAG_ERR_OVERLAP (-2) /* overlapTokens must be in [0, sizeTokens) */
#define RAG_ERR_ARENA (-3) /* arena exhausted */
/*
* The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
* line costs max(1, round(length / 4)) tokens; empty text is 0.
*/
static long rag_tok(const char *text) {
if (text == NULL || text[0] == '\0') return 0;
long tokens = 0;
const char *line = text;
for (const char *p = text;; p++) {
if (*p == '\n' || *p == '\0') {
size_t len = (size_t)(p - line);
if (len > 0) {
long per = (long)(((double)len / 4.0) + 0.5);
tokens += per < 1 ? 1 : per;
}
if (*p == '\0') break;
line = p + 1;
}
}
return tokens;
}
static void rag_trim(const char *src, char *dst, size_t cap) {
size_t n = strlen(src);
size_t start = 0;
while (start < n && (src[start] == ' ' || src[start] == '\t' || src[start] == '\n' || src[start] == '\r')) start++;
size_t end = n;
while (end > start && (src[end - 1] == ' ' || src[end - 1] == '\t' || src[end - 1] == '\n' || src[end - 1] == '\r')) end--;
size_t len = end - start;
if (len >= cap) len = cap - 1;
memcpy(dst, src + start, len);
dst[len] = '\0';
}
/* True when the text ends a sentence: [.!?] optionally wrapped by " ' ) ]. */
static int rag_ends_sentence(const char *s) {
size_t n = strlen(s);
while (n > 0 && (s[n - 1] == ' ' || s[n - 1] == '\t' || s[n - 1] == '\n')) n--;
if (n == 0) return 0;
char last = s[n - 1];
if (last == '"' || last == '\'' || last == ')' || last == ']') {
if (n < 2) return 0;
last = s[n - 2];
}
return last == '.' || last == '!' || last == '?';
}
static rag_stats rag_stats_for(const rag_chunk *chunks, size_t count) {
rag_stats st = {0, 0, 0, 0, 1.0};
st.count = (long)count;
if (count == 0) return st;
long sum = 0;
st.min_tokens = chunks[0].tokens;
st.max_tokens = chunks[0].tokens;
for (size_t i = 0; i < count; i++) {
if (chunks[i].tokens < st.min_tokens) st.min_tokens = chunks[i].tokens;
if (chunks[i].tokens > st.max_tokens) st.max_tokens = chunks[i].tokens;
sum += chunks[i].tokens;
}
st.avg_tokens = sum / (long)count;
size_t boundaries = 0, ending = 0;
for (size_t i = 0; i + 1 < count; i++) {
boundaries++;
if (rag_ends_sentence(chunks[i].text)) ending++;
}
st.sentence_boundary_share = boundaries > 0
? (double)ending / (double)boundaries
: 1.0; /* a single chunk has no internal boundaries to botch */
return st;
}
/* Split on sentence enders followed by whitespace or end of text. The
* whitespace run itself is collapsed: output sentences are trimmed. */
static size_t rag_split_sentences(const char *text, char out[][RAG_CHUNK_TEXT], size_t max) {
/* pass 1: collapse whitespace to single spaces */
char collapsed[RAG_CHUNK_TEXT * RAG_MAX_CHUNKS / 4];
size_t cn = 0;
for (const char *p = text; *p; p++) {
if (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') {
if (cn > 0 && collapsed[cn - 1] != ' ') collapsed[cn++] = ' ';
} else if (cn < sizeof(collapsed) - 1) {
collapsed[cn++] = *p;
}
}
while (cn > 0 && collapsed[cn - 1] == ' ') cn--;
collapsed[cn] = '\0';
/* pass 2: cut after [.!?] followed by a space */
size_t count = 0;
size_t start = 0;
for (size_t i = 0; i <= cn; i++) {
char c = collapsed[i];
int ender = (c == '.' || c == '!' || c == '?');
if ((ender && collapsed[i + 1] == ' ') || c == '\0') {
size_t len = i + 1 - start;
if (ender && collapsed[i + 1] == ' ') {
len = i + 1 - start; /* keep the ender */
}
if (len > 0 && count < max) {
if (len >= RAG_CHUNK_TEXT) len = RAG_CHUNK_TEXT - 1;
memcpy(out[count], collapsed + start, len);
out[count][len] = '\0';
count++;
}
if (c == '\0') break;
start = i + 1;
while (collapsed[start] == ' ') start++;
i = start - 1;
}
}
return count;
}
static int rag_push_chunk(rag_result *r, const char *text, const char *heading) {
if (r->chunk_count >= RAG_MAX_CHUNKS) return RAG_ERR_ARENA;
rag_chunk *c = &r->chunks[r->chunk_count];
c->index = (long)r->chunk_count;
rag_trim(text, c->text, sizeof(c->text));
c->tokens = rag_tok(c->text);
if (heading != NULL) {
c->has_heading = 1;
snprintf(c->heading, sizeof(c->heading), "%s", heading);
} else {
c->has_heading = 0;
c->heading[0] = '\0';
}
r->chunk_count++;
return 0;
}
/* Greedy character accumulation to a token target (overlapping allowed). */
static int rag_chunk_fixed(const char *text, rag_options opts, rag_result *out) {
memset(out, 0, sizeof(*out));
if (opts.size_tokens <= 0) return RAG_ERR_SIZE_TOKENS;
if (opts.overlap_tokens < 0 || opts.overlap_tokens >= opts.size_tokens) {
return RAG_ERR_OVERLAP;
}
char clean[strlen(text) + 1];
rag_trim(text, clean, sizeof(clean));
if (clean[0] == '\0') return 0;
/* ~4 chars per prose token: step by tokens, verify with the estimator. */
long char_step = opts.size_tokens * 4;
if (char_step < 1) char_step = 1;
long overlap_chars = opts.overlap_tokens * 4;
size_t len = strlen(clean);
size_t start = 0;
while (start < len) {
size_t end = start + (size_t)char_step;
if (end > len) end = len;
/* Prefer cutting at whitespace near the target. */
if (end < len) {
long cut = -1;
for (size_t i = end; i > start; i--) {
if (clean[i - 1] == ' ') { cut = (long)(i - 1); break; }
}
if (cut > (long)start) end = (size_t)cut;
}
char piece[RAG_CHUNK_TEXT];
rag_trim(clean + start, piece, sizeof(piece));
if (piece[0] != '\0') {
int rc = rag_push_chunk(out, piece, NULL);
if (rc != 0) return rc;
}
if (end >= len) break;
size_t next = (size_t)((long)end - overlap_chars);
if (next <= start) next = start + 1;
start = next;
}
out->stats = rag_stats_for(out->chunks, out->chunk_count);
return 0;
}
/* Group whole sentences up to the token target; boundaries never split one. */
static int rag_chunk_sentences(const char *text, rag_options opts, rag_result *out) {
memset(out, 0, sizeof(*out));
if (opts.size_tokens <= 0) return RAG_ERR_SIZE_TOKENS;
char sentences[RAG_MAX_CHUNKS][RAG_CHUNK_TEXT];
size_t n = rag_split_sentences(text, sentences, RAG_MAX_CHUNKS);
if (n == 0) return 0;
char current[RAG_CHUNK_TEXT * 8] = "";
long current_tokens = 0;
for (size_t i = 0; i < n; i++) {
long t = rag_tok(sentences[i]);
if (current_tokens > 0 && current_tokens + t > opts.size_tokens) {
int rc = rag_push_chunk(out, current, NULL);
if (rc != 0) return rc;
current[0] = '\0';
current_tokens = 0;
}
if (current[0] != '\0') strncat(current, " ", sizeof(current) - strlen(current) - 1);
strncat(current, sentences[i], sizeof(current) - strlen(current) - 1);
current_tokens += t;
/* A single sentence larger than the target becomes its own chunk. */
}
if (current[0] != '\0') {
int rc = rag_push_chunk(out, current, NULL);
if (rc != 0) return rc;
}
out->stats = rag_stats_for(out->chunks, out->chunk_count);
return 0;
}
/* Split on markdown headings; oversized sections fall back to sentence
* grouping. A heading is 1-6 '#' followed by whitespace + text. */
static int heading_of(const char *line, char *out, size_t cap) {
int hashes = 0;
while (line[hashes] == '#') hashes++;
if (hashes < 1 || hashes > 6) return 0;
const char *p = line + hashes;
if (*p != ' ' && *p != '\t') return 0;
while (*p == ' ' || *p == '\t') p++;
if (*p == '\0' || *p == '\n') return 0;
size_t len = strlen(p);
while (len > 0 && (p[len - 1] == '\n' || p[len - 1] == ' ')) len--;
if (len >= cap) len = cap - 1;
memcpy(out, p, len);
out[len] = '\0';
return 1;
}
static int rag_chunk_markdown(const char *text, rag_options opts, rag_result *out) {
memset(out, 0, sizeof(*out));
if (opts.size_tokens <= 0) return RAG_ERR_SIZE_TOKENS;
/* Pass 1: split into sections at heading lines. */
char bodies[RAG_MAX_CHUNKS][RAG_CHUNK_TEXT * 2];
char headings[RAG_MAX_CHUNKS][RAG_HEADING];
int has_heading[RAG_MAX_CHUNKS];
size_t sections = 0;
{
char body[RAG_CHUNK_TEXT * 2] = "";
char heading[RAG_HEADING] = "";
int in_heading = 0;
const char *line = text;
while (line != NULL && *line != '\0') {
const char *nl = strchr(line, '\n');
size_t ll = nl ? (size_t)(nl - line) : strlen(line);
char buf[RAG_CHUNK_TEXT];
if (ll >= sizeof(buf)) ll = sizeof(buf) - 1;
memcpy(buf, line, ll);
buf[ll] = '\0';
char h[RAG_HEADING];
if (heading_of(buf, h, sizeof(h))) {
if (body[0] != '\0' && sections < RAG_MAX_CHUNKS) {
strcpy(bodies[sections], body);
strcpy(headings[sections], heading);
has_heading[sections] = in_heading;
sections++;
}
strcpy(heading, h);
in_heading = 1;
body[0] = '\0';
} else {
if (body[0] != '\0') strncat(body, "\n", sizeof(body) - strlen(body) - 1);
strncat(body, buf, sizeof(body) - strlen(body) - 1);
}
line = nl ? nl + 1 : NULL;
}
if (body[0] != '\0' && sections < RAG_MAX_CHUNKS) {
strcpy(bodies[sections], body);
strcpy(headings[sections], heading);
has_heading[sections] = in_heading;
sections++;
}
}
/* Pass 2: emit each section whole or sentence-grouped. */
for (size_t i = 0; i < sections; i++) {
char clean[RAG_CHUNK_TEXT * 2];
rag_trim(bodies[i], clean, sizeof(clean));
if (clean[0] == '\0') continue;
char whole[RAG_CHUNK_TEXT * 2];
if (has_heading[i]) {
snprintf(whole, sizeof(whole), "# %s\n%s", headings[i], clean);
} else {
snprintf(whole, sizeof(whole), "%s", clean);
}
if (rag_tok(whole) <= opts.size_tokens) {
int rc = rag_push_chunk(out, whole, has_heading[i] ? headings[i] : NULL);
if (rc != 0) return rc;
continue;
}
/* Oversized section: sentence-group the body, stamp every chunk. */
rag_result sub;
int rc = rag_chunk_sentences(clean, opts, &sub);
if (rc != 0) return rc;
for (size_t j = 0; j < sub.chunk_count; j++) {
rc = rag_push_chunk(out, sub.chunks[j].text, has_heading[i] ? headings[i] : NULL);
if (rc != 0) return rc;
}
}
out->stats = rag_stats_for(out->chunks, out->chunk_count);
return 0;
}
/*
* Run all three strategies over one document and report comparable stats.
* Returns 0, or a negative RAG_ERR_* code (the plan is zeroed on error).
*/
int rag_compare_strategies(const char *text, rag_options opts, rag_comparison *out) {
memset(out, 0, sizeof(*out));
int rc = rag_chunk_fixed(text, opts, &out->fixed);
if (rc != 0) return rc;
rc = rag_chunk_sentences(text, opts, &out->sentence);
if (rc != 0) return rc;
rc = rag_chunk_markdown(text, opts, &out->markdown);
if (rc != 0) return rc;
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →