Skip to content

Embedding Chunk Planner — C source

Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * Embedding Chunk Planner — pure chunking math for RAG pipelines.
 *
 * Language: C (C11, standard library only)
 * Source:   CosmoDev polyglot showcase port of the Embedding Chunk Planner
 *           tool, ported from src/lib/embeddingPlanner.ts (the canonical
 *           TypeScript implementation).
 * Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
 * License:  display source — part of CosmoDev's polyglot tool pages.
 *
 * Design goals:
 *   - Pure + deterministic; no undefined behavior for any input.
 *   - Functionally equivalent to the TS reference: same inputs -> same outputs.
 *   - Self-contained: stdlib only (no third-party headers). The model price
 *     table is inlined below, mirrored from src/lib/ai/embeddings.ts — prices
 *     NEVER live in the planner itself.
 *
 * Behavior (mirrors the TS source exactly):
 *   - chunk_size <= 0 or total_tokens <= 0 -> the zero plan (nothing to embed).
 *   - Negative overlap is treated as 0; overlap then clamps to at most
 *     chunk_size / 2 so consecutive chunks always advance.
 *   - chunks = max(1, ceil((total_tokens - overlap) / (chunk_size - overlap)))
 *     — a tiny document still yields one chunk.
 */

#include <assert.h>
#include <math.h>
#include <stdbool.h>
#include <stddef.h>
#include <string.h>

/* One embedding model's offered dimensions (ascending, Matryoshka shortening
 * included) and pricing: USD per 1M input tokens. */
typedef struct {
    const char *id;
    const char *vendor;
    const long *dims;
    size_t dims_len;
    double input_per_m;
} EmbeddingModel;

/* Embedding model price table — the SSOT for pricing, mirrored from
 * src/lib/ai/embeddings.ts. Refresh both files together. */
static const long DIMS_3_SMALL[] = {512, 1536};
static const long DIMS_3_LARGE[] = {256, 1024, 3072};
static const long DIMS_ENGLISH_V3[] = {512, 1024, 1536};
static const long DIMS_VOYAGE_LITE[] = {512, 1024};

static const EmbeddingModel EMBEDDING_MODELS[] = {
    {"text-embedding-3-small", "OpenAI", DIMS_3_SMALL, 2, 0.02},
    {"text-embedding-3-large", "OpenAI", DIMS_3_LARGE, 3, 0.13},
    {"embed-english-v3.0", "Cohere", DIMS_ENGLISH_V3, 3, 0.1},
    {"voyage-3-lite", "Voyage AI", DIMS_VOYAGE_LITE, 2, 0.02},
};
#define EMBEDDING_MODEL_COUNT (sizeof EMBEDDING_MODELS / sizeof EMBEDDING_MODELS[0])

/* Default knobs: 512-token chunks, 64-token overlap (TS DEFAULT_CHUNK_OPTIONS). */
enum { DEFAULT_CHUNK_SIZE = 512, DEFAULT_OVERLAP = 64 };

/* Chunking knobs, in tokens. Mirrors the TS `Partial<ChunkOptions>`: each
 * field is independently optional — has_* == false falls back to the 512 / 64
 * default, and setting one leaves the other at its default. */
typedef struct {
    long chunk_size;
    bool has_chunk_size;
    long overlap;
    bool has_overlap;
} ChunkOptions;

/* How a document splits into overlapping chunks. */
typedef struct {
    long chunks;
    long total_tokens_with_overlap;
    long overhead_tokens;
} ChunkPlan;

/* The "nothing to embed" plan the TS source returns for zero/negative input
 * or a non-positive chunk size. */
static const ChunkPlan CHUNK_PLAN_ZERO = {0, 0, 0};

static long clamp_long(long v, long lo, long hi) {
    return v < lo ? lo : (v > hi ? hi : v);
}

static long max_long(long a, long b) {
    return a > b ? a : b;
}

/* Look up an embedding model by id. Returns NULL for unknown ids. */
const EmbeddingModel *get_embedding_model(const char *id) {
    for (size_t i = 0; i < EMBEDDING_MODEL_COUNT; i++) {
        if (strcmp(EMBEDDING_MODELS[i].id, id) == 0) {
            return &EMBEDDING_MODELS[i];
        }
    }
    return NULL;
}

/* Plan how `total_tokens` split into overlapping chunks. `opts` may be NULL
 * (both defaults), mirroring the TS optional parameter. */
ChunkPlan plan_chunks(long total_tokens, const ChunkOptions *opts) {
    long chunk_size = (opts != NULL && opts->has_chunk_size) ? opts->chunk_size : DEFAULT_CHUNK_SIZE;
    long overlap_raw = (opts != NULL && opts->has_overlap) ? opts->overlap : DEFAULT_OVERLAP;

    if (chunk_size <= 0 || total_tokens <= 0) {
        return CHUNK_PLAN_ZERO;
    }

    /* min(max(overlap, 0), chunk_size / 2) — the TS clamp. Overlap that large
     * would never advance, so consecutive chunks always gain at least half a
     * chunk. (chunk_size >= 1 here, so chunk_size - overlap is never zero.) */
    long overlap = clamp_long(overlap_raw, 0, chunk_size / 2);

    /* Float division + ceil mirrors TS's Math.ceil exactly (a tiny document
     * lands the quotient just below zero; ceil brings it to 0 and max(1, ..)
     * lifts it back to one chunk). */
    long chunks = max_long(1, (long)ceil((double)(total_tokens - overlap) / (double)(chunk_size - overlap)));

    long total_tokens_with_overlap = total_tokens + (chunks - 1) * overlap;
    ChunkPlan plan = {chunks, total_tokens_with_overlap, total_tokens_with_overlap - total_tokens};
    return plan;
}

/* Chunk plan plus pricing for one embedding call. The three chunk fields are
 * flattened in (the TS `...plan` spread) so the struct reads like the TS
 * `EmbeddingPlan extends ChunkPlan`. */
typedef struct {
    long chunks;
    long total_tokens_with_overlap;
    long overhead_tokens;
    const EmbeddingModel *model;
    long vectors; /* one vector per chunk */
    double cost;  /* USD: total_tokens_with_overlap / 1e6 * model->input_per_m */
} EmbeddingPlan;

/* Chunk a document AND price its embedding for `model_id` at `dims`
 * dimensions. Unknown model, or dims the model does not offer -> false
 * (*out untouched); success -> true with *out filled in. */
bool plan_embedding(long total_tokens, const char *model_id, long dims,
                    const ChunkOptions *opts, EmbeddingPlan *out) {
    const EmbeddingModel *model = get_embedding_model(model_id);
    if (model == NULL) {
        return false;
    }
    bool offered = false;
    for (size_t i = 0; i < model->dims_len; i++) {
        if (model->dims[i] == dims) {
            offered = true;
            break;
        }
    }
    if (!offered) {
        return false;
    }
    ChunkPlan plan = plan_chunks(total_tokens, opts);
    EmbeddingPlan priced = {
        plan.chunks,
        plan.total_tokens_with_overlap,
        plan.overhead_tokens,
        model,
        plan.chunks,
        (double)plan.total_tokens_with_overlap / 1e6 * model->input_per_m,
    };
    if (out != NULL) {
        *out = priced;
    }
    return true;
}

/* ---------- showcase examples (the canonical suite lives in src/lib) ---------- */

static bool plan_equals(ChunkPlan a, ChunkPlan b) {
    return a.chunks == b.chunks &&
           a.total_tokens_with_overlap == b.total_tokens_with_overlap &&
           a.overhead_tokens == b.overhead_tokens;
}

int main(void) {
    /* 1,000 tokens: ceil((1000-64)/(512-64)) = 3 chunks, 2 seams x 64. */
    ChunkPlan a = plan_chunks(1000, NULL);
    assert(plan_equals(a, (ChunkPlan){3, 1128, 128}));

    /* A document that fits one chunk has no seam overhead. */
    assert(plan_equals(plan_chunks(512, NULL), (ChunkPlan){1, 512, 0}));

    /* Zero/negative input or non-positive chunk size -> the zero plan. */
    assert(plan_equals(plan_chunks(0, NULL), CHUNK_PLAN_ZERO));
    assert(plan_equals(plan_chunks(-100, NULL), CHUNK_PLAN_ZERO));
    assert(plan_equals(plan_chunks(1000, &(ChunkOptions){.chunk_size = 0, .has_chunk_size = true}), CHUNK_PLAN_ZERO));
    assert(plan_equals(plan_chunks(1000, &(ChunkOptions){.chunk_size = -8, .has_chunk_size = true}), CHUNK_PLAN_ZERO));

    /* overlap 600 > floor(512/2) = 256 -> clamped to 256. */
    ChunkOptions big_overlap = {.overlap = 600, .has_overlap = true};
    assert(plan_equals(plan_chunks(1000, &big_overlap), (ChunkPlan){3, 1512, 512}));

    /* Negative overlap clamps to 0: 1000 tokens -> ceil(1000/512) = 2 chunks. */
    ChunkOptions neg_overlap = {.overlap = -5, .has_overlap = true};
    assert(plan_equals(plan_chunks(1000, &neg_overlap), (ChunkPlan){2, 1000, 0}));

    /* chunk_size without overlap: ceil((1000-64)/192) = 5 chunks. */
    ChunkOptions small_chunks = {.chunk_size = 256, .has_chunk_size = true};
    assert(plan_equals(plan_chunks(1000, &small_chunks), (ChunkPlan){5, 1256, 256}));

    /* Shorter than the overlap still yields one chunk. */
    ChunkOptions std_overlap = {.overlap = 64, .has_overlap = true};
    assert(plan_equals(plan_chunks(50, &std_overlap), (ChunkPlan){1, 50, 0}));

    /* chunk_size of 1 clamps overlap to 0: ceil(3/1) = 3 chunks. */
    ChunkOptions tiny = {.chunk_size = 1, .has_chunk_size = true};
    assert(plan_equals(plan_chunks(3, &tiny), (ChunkPlan){3, 3, 0}));

    /* Pricing: 1,000 tokens on text-embedding-3-small @ 1536 dims. */
    EmbeddingPlan priced;
    assert(plan_embedding(1000, "text-embedding-3-small", 1536, NULL, &priced));
    assert(priced.chunks == 3 && priced.total_tokens_with_overlap == 1128 && priced.vectors == 3);
    assert(fabs(priced.cost - 0.00002256) < 1e-12); /* 1128 / 1e6 * $0.02 */

    /* A single-chunk document on voyage-3-lite @ 512 dims. */
    EmbeddingPlan single;
    assert(plan_embedding(512, "voyage-3-lite", 512, NULL, &single));
    assert(single.vectors == 1); /* one vector for one chunk */
    assert(fabs(single.cost - 0.00001024) < 1e-12); /* 512 / 1e6 * $0.02 */

    /* Unknown model or unoffered dims -> false. */
    EmbeddingPlan ignored;
    assert(!plan_embedding(1000, "text-embedding-3-small", 999, NULL, &ignored));
    assert(!plan_embedding(1000, "ghost", 1536, NULL, &ignored));

    /* Zero tokens price out to a zero-cost plan. */
    EmbeddingPlan zero;
    assert(plan_embedding(0, "text-embedding-3-small", 1536, NULL, &zero));
    assert(zero.chunks == 0 && zero.total_tokens_with_overlap == 0 && zero.cost == 0.0);

    return 0;
}

Also available in 12 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →