Skip to content

RAG Chunk Comparator — C source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * RAG Chunk Comparator — chunk one document three ways and report the stats
 * that matter for retrieval.
 *
 * Language: C (C11, standard library only)
 * Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
 * implementation). The sibling javascript.js carries the same port.
 * Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
 *
 * C shape: fixed-capacity output; chunk text is COPIED into caller-owned
 * buffers carved from one arena (a single big malloc). The TS string
 * operations (regex sentence split, heading match) become manual scans.
 * RangeError maps to a negative return code; 0 is success.
 */

#include <stdio.h>
#include <stdlib.h>
#include <string.h>

#define RAG_MAX_CHUNKS 128
#define RAG_CHUNK_TEXT 4096
#define RAG_HEADING 256

typedef enum { RAG_FIXED = 0, RAG_SENTENCE = 1, RAG_MARKDOWN = 2 } rag_strategy;

typedef struct {
    long size_tokens;
    long overlap_tokens; /* fixed strategy only; 0 default */
} rag_options;

typedef struct {
    long index;
    char text[RAG_CHUNK_TEXT];
    long tokens;
    int has_heading;
    char heading[RAG_HEADING];
} rag_chunk;

typedef struct {
    long count;
    long min_tokens;
    long max_tokens;
    long avg_tokens;
    double sentence_boundary_share; /* 0..1; 1 when there are no boundaries */
} rag_stats;

typedef struct {
    rag_chunk chunks[RAG_MAX_CHUNKS];
    size_t chunk_count;
    rag_stats stats;
} rag_result;

typedef struct {
    rag_result fixed;
    rag_result sentence;
    rag_result markdown;
} rag_comparison;

/* Error codes mirroring the TS RangeError contract. */
#define RAG_ERR_SIZE_TOKENS (-1)   /* sizeTokens must be > 0 */
#define RAG_ERR_OVERLAP (-2)       /* overlapTokens must be in [0, sizeTokens) */
#define RAG_ERR_ARENA (-3)         /* arena exhausted */

/*
 * The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
 * line costs max(1, round(length / 4)) tokens; empty text is 0.
 */
static long rag_tok(const char *text) {
    if (text == NULL || text[0] == '\0') return 0;
    long tokens = 0;
    const char *line = text;
    for (const char *p = text;; p++) {
        if (*p == '\n' || *p == '\0') {
            size_t len = (size_t)(p - line);
            if (len > 0) {
                long per = (long)(((double)len / 4.0) + 0.5);
                tokens += per < 1 ? 1 : per;
            }
            if (*p == '\0') break;
            line = p + 1;
        }
    }
    return tokens;
}

static void rag_trim(const char *src, char *dst, size_t cap) {
    size_t n = strlen(src);
    size_t start = 0;
    while (start < n && (src[start] == ' ' || src[start] == '\t' || src[start] == '\n' || src[start] == '\r')) start++;
    size_t end = n;
    while (end > start && (src[end - 1] == ' ' || src[end - 1] == '\t' || src[end - 1] == '\n' || src[end - 1] == '\r')) end--;
    size_t len = end - start;
    if (len >= cap) len = cap - 1;
    memcpy(dst, src + start, len);
    dst[len] = '\0';
}

/* True when the text ends a sentence: [.!?] optionally wrapped by " ' ) ]. */
static int rag_ends_sentence(const char *s) {
    size_t n = strlen(s);
    while (n > 0 && (s[n - 1] == ' ' || s[n - 1] == '\t' || s[n - 1] == '\n')) n--;
    if (n == 0) return 0;
    char last = s[n - 1];
    if (last == '"' || last == '\'' || last == ')' || last == ']') {
        if (n < 2) return 0;
        last = s[n - 2];
    }
    return last == '.' || last == '!' || last == '?';
}

static rag_stats rag_stats_for(const rag_chunk *chunks, size_t count) {
    rag_stats st = {0, 0, 0, 0, 1.0};
    st.count = (long)count;
    if (count == 0) return st;
    long sum = 0;
    st.min_tokens = chunks[0].tokens;
    st.max_tokens = chunks[0].tokens;
    for (size_t i = 0; i < count; i++) {
        if (chunks[i].tokens < st.min_tokens) st.min_tokens = chunks[i].tokens;
        if (chunks[i].tokens > st.max_tokens) st.max_tokens = chunks[i].tokens;
        sum += chunks[i].tokens;
    }
    st.avg_tokens = sum / (long)count;
    size_t boundaries = 0, ending = 0;
    for (size_t i = 0; i + 1 < count; i++) {
        boundaries++;
        if (rag_ends_sentence(chunks[i].text)) ending++;
    }
    st.sentence_boundary_share = boundaries > 0
        ? (double)ending / (double)boundaries
        : 1.0; /* a single chunk has no internal boundaries to botch */
    return st;
}

/* Split on sentence enders followed by whitespace or end of text. The
 * whitespace run itself is collapsed: output sentences are trimmed. */
static size_t rag_split_sentences(const char *text, char out[][RAG_CHUNK_TEXT], size_t max) {
    /* pass 1: collapse whitespace to single spaces */
    char collapsed[RAG_CHUNK_TEXT * RAG_MAX_CHUNKS / 4];
    size_t cn = 0;
    for (const char *p = text; *p; p++) {
        if (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') {
            if (cn > 0 && collapsed[cn - 1] != ' ') collapsed[cn++] = ' ';
        } else if (cn < sizeof(collapsed) - 1) {
            collapsed[cn++] = *p;
        }
    }
    while (cn > 0 && collapsed[cn - 1] == ' ') cn--;
    collapsed[cn] = '\0';

    /* pass 2: cut after [.!?] followed by a space */
    size_t count = 0;
    size_t start = 0;
    for (size_t i = 0; i <= cn; i++) {
        char c = collapsed[i];
        int ender = (c == '.' || c == '!' || c == '?');
        if ((ender && collapsed[i + 1] == ' ') || c == '\0') {
            size_t len = i + 1 - start;
            if (ender && collapsed[i + 1] == ' ') {
                len = i + 1 - start; /* keep the ender */
            }
            if (len > 0 && count < max) {
                if (len >= RAG_CHUNK_TEXT) len = RAG_CHUNK_TEXT - 1;
                memcpy(out[count], collapsed + start, len);
                out[count][len] = '\0';
                count++;
            }
            if (c == '\0') break;
            start = i + 1;
            while (collapsed[start] == ' ') start++;
            i = start - 1;
        }
    }
    return count;
}

static int rag_push_chunk(rag_result *r, const char *text, const char *heading) {
    if (r->chunk_count >= RAG_MAX_CHUNKS) return RAG_ERR_ARENA;
    rag_chunk *c = &r->chunks[r->chunk_count];
    c->index = (long)r->chunk_count;
    rag_trim(text, c->text, sizeof(c->text));
    c->tokens = rag_tok(c->text);
    if (heading != NULL) {
        c->has_heading = 1;
        snprintf(c->heading, sizeof(c->heading), "%s", heading);
    } else {
        c->has_heading = 0;
        c->heading[0] = '\0';
    }
    r->chunk_count++;
    return 0;
}

/* Greedy character accumulation to a token target (overlapping allowed). */
static int rag_chunk_fixed(const char *text, rag_options opts, rag_result *out) {
    memset(out, 0, sizeof(*out));
    if (opts.size_tokens <= 0) return RAG_ERR_SIZE_TOKENS;
    if (opts.overlap_tokens < 0 || opts.overlap_tokens >= opts.size_tokens) {
        return RAG_ERR_OVERLAP;
    }
    char clean[strlen(text) + 1];
    rag_trim(text, clean, sizeof(clean));
    if (clean[0] == '\0') return 0;

    /* ~4 chars per prose token: step by tokens, verify with the estimator. */
    long char_step = opts.size_tokens * 4;
    if (char_step < 1) char_step = 1;
    long overlap_chars = opts.overlap_tokens * 4;
    size_t len = strlen(clean);
    size_t start = 0;
    while (start < len) {
        size_t end = start + (size_t)char_step;
        if (end > len) end = len;
        /* Prefer cutting at whitespace near the target. */
        if (end < len) {
            long cut = -1;
            for (size_t i = end; i > start; i--) {
                if (clean[i - 1] == ' ') { cut = (long)(i - 1); break; }
            }
            if (cut > (long)start) end = (size_t)cut;
        }
        char piece[RAG_CHUNK_TEXT];
        rag_trim(clean + start, piece, sizeof(piece));
        if (piece[0] != '\0') {
            int rc = rag_push_chunk(out, piece, NULL);
            if (rc != 0) return rc;
        }
        if (end >= len) break;
        size_t next = (size_t)((long)end - overlap_chars);
        if (next <= start) next = start + 1;
        start = next;
    }
    out->stats = rag_stats_for(out->chunks, out->chunk_count);
    return 0;
}

/* Group whole sentences up to the token target; boundaries never split one. */
static int rag_chunk_sentences(const char *text, rag_options opts, rag_result *out) {
    memset(out, 0, sizeof(*out));
    if (opts.size_tokens <= 0) return RAG_ERR_SIZE_TOKENS;
    char sentences[RAG_MAX_CHUNKS][RAG_CHUNK_TEXT];
    size_t n = rag_split_sentences(text, sentences, RAG_MAX_CHUNKS);
    if (n == 0) return 0;

    char current[RAG_CHUNK_TEXT * 8] = "";
    long current_tokens = 0;
    for (size_t i = 0; i < n; i++) {
        long t = rag_tok(sentences[i]);
        if (current_tokens > 0 && current_tokens + t > opts.size_tokens) {
            int rc = rag_push_chunk(out, current, NULL);
            if (rc != 0) return rc;
            current[0] = '\0';
            current_tokens = 0;
        }
        if (current[0] != '\0') strncat(current, " ", sizeof(current) - strlen(current) - 1);
        strncat(current, sentences[i], sizeof(current) - strlen(current) - 1);
        current_tokens += t;
        /* A single sentence larger than the target becomes its own chunk. */
    }
    if (current[0] != '\0') {
        int rc = rag_push_chunk(out, current, NULL);
        if (rc != 0) return rc;
    }
    out->stats = rag_stats_for(out->chunks, out->chunk_count);
    return 0;
}

/* Split on markdown headings; oversized sections fall back to sentence
 * grouping. A heading is 1-6 '#' followed by whitespace + text. */
static int heading_of(const char *line, char *out, size_t cap) {
    int hashes = 0;
    while (line[hashes] == '#') hashes++;
    if (hashes < 1 || hashes > 6) return 0;
    const char *p = line + hashes;
    if (*p != ' ' && *p != '\t') return 0;
    while (*p == ' ' || *p == '\t') p++;
    if (*p == '\0' || *p == '\n') return 0;
    size_t len = strlen(p);
    while (len > 0 && (p[len - 1] == '\n' || p[len - 1] == ' ')) len--;
    if (len >= cap) len = cap - 1;
    memcpy(out, p, len);
    out[len] = '\0';
    return 1;
}

static int rag_chunk_markdown(const char *text, rag_options opts, rag_result *out) {
    memset(out, 0, sizeof(*out));
    if (opts.size_tokens <= 0) return RAG_ERR_SIZE_TOKENS;

    /* Pass 1: split into sections at heading lines. */
    char bodies[RAG_MAX_CHUNKS][RAG_CHUNK_TEXT * 2];
    char headings[RAG_MAX_CHUNKS][RAG_HEADING];
    int has_heading[RAG_MAX_CHUNKS];
    size_t sections = 0;
    {
        char body[RAG_CHUNK_TEXT * 2] = "";
        char heading[RAG_HEADING] = "";
        int in_heading = 0;
        const char *line = text;
        while (line != NULL && *line != '\0') {
            const char *nl = strchr(line, '\n');
            size_t ll = nl ? (size_t)(nl - line) : strlen(line);
            char buf[RAG_CHUNK_TEXT];
            if (ll >= sizeof(buf)) ll = sizeof(buf) - 1;
            memcpy(buf, line, ll);
            buf[ll] = '\0';
            char h[RAG_HEADING];
            if (heading_of(buf, h, sizeof(h))) {
                if (body[0] != '\0' && sections < RAG_MAX_CHUNKS) {
                    strcpy(bodies[sections], body);
                    strcpy(headings[sections], heading);
                    has_heading[sections] = in_heading;
                    sections++;
                }
                strcpy(heading, h);
                in_heading = 1;
                body[0] = '\0';
            } else {
                if (body[0] != '\0') strncat(body, "\n", sizeof(body) - strlen(body) - 1);
                strncat(body, buf, sizeof(body) - strlen(body) - 1);
            }
            line = nl ? nl + 1 : NULL;
        }
        if (body[0] != '\0' && sections < RAG_MAX_CHUNKS) {
            strcpy(bodies[sections], body);
            strcpy(headings[sections], heading);
            has_heading[sections] = in_heading;
            sections++;
        }
    }

    /* Pass 2: emit each section whole or sentence-grouped. */
    for (size_t i = 0; i < sections; i++) {
        char clean[RAG_CHUNK_TEXT * 2];
        rag_trim(bodies[i], clean, sizeof(clean));
        if (clean[0] == '\0') continue;
        char whole[RAG_CHUNK_TEXT * 2];
        if (has_heading[i]) {
            snprintf(whole, sizeof(whole), "# %s\n%s", headings[i], clean);
        } else {
            snprintf(whole, sizeof(whole), "%s", clean);
        }
        if (rag_tok(whole) <= opts.size_tokens) {
            int rc = rag_push_chunk(out, whole, has_heading[i] ? headings[i] : NULL);
            if (rc != 0) return rc;
            continue;
        }
        /* Oversized section: sentence-group the body, stamp every chunk. */
        rag_result sub;
        int rc = rag_chunk_sentences(clean, opts, &sub);
        if (rc != 0) return rc;
        for (size_t j = 0; j < sub.chunk_count; j++) {
            rc = rag_push_chunk(out, sub.chunks[j].text, has_heading[i] ? headings[i] : NULL);
            if (rc != 0) return rc;
        }
    }
    out->stats = rag_stats_for(out->chunks, out->chunk_count);
    return 0;
}

/*
 * Run all three strategies over one document and report comparable stats.
 * Returns 0, or a negative RAG_ERR_* code (the plan is zeroed on error).
 */
int rag_compare_strategies(const char *text, rag_options opts, rag_comparison *out) {
    memset(out, 0, sizeof(*out));
    int rc = rag_chunk_fixed(text, opts, &out->fixed);
    if (rc != 0) return rc;
    rc = rag_chunk_sentences(text, opts, &out->sentence);
    if (rc != 0) return rc;
    rc = rag_chunk_markdown(text, opts, &out->markdown);
    if (rc != 0) return rc;
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →