Embedding Chunk Planner — C source
Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/*
* Embedding Chunk Planner — pure chunking math for RAG pipelines.
*
* Language: C (C11, standard library only)
* Source: CosmoDev polyglot showcase port of the Embedding Chunk Planner
* tool, ported from src/lib/embeddingPlanner.ts (the canonical
* TypeScript implementation).
* Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; no undefined behavior for any input.
* - Functionally equivalent to the TS reference: same inputs -> same outputs.
* - Self-contained: stdlib only (no third-party headers). The model price
* table is inlined below, mirrored from src/lib/ai/embeddings.ts — prices
* NEVER live in the planner itself.
*
* Behavior (mirrors the TS source exactly):
* - chunk_size <= 0 or total_tokens <= 0 -> the zero plan (nothing to embed).
* - Negative overlap is treated as 0; overlap then clamps to at most
* chunk_size / 2 so consecutive chunks always advance.
* - chunks = max(1, ceil((total_tokens - overlap) / (chunk_size - overlap)))
* — a tiny document still yields one chunk.
*/
#include <assert.h>
#include <math.h>
#include <stdbool.h>
#include <stddef.h>
#include <string.h>
/* One embedding model's offered dimensions (ascending, Matryoshka shortening
* included) and pricing: USD per 1M input tokens. */
typedef struct {
const char *id;
const char *vendor;
const long *dims;
size_t dims_len;
double input_per_m;
} EmbeddingModel;
/* Embedding model price table — the SSOT for pricing, mirrored from
* src/lib/ai/embeddings.ts. Refresh both files together. */
static const long DIMS_3_SMALL[] = {512, 1536};
static const long DIMS_3_LARGE[] = {256, 1024, 3072};
static const long DIMS_ENGLISH_V3[] = {512, 1024, 1536};
static const long DIMS_VOYAGE_LITE[] = {512, 1024};
static const EmbeddingModel EMBEDDING_MODELS[] = {
{"text-embedding-3-small", "OpenAI", DIMS_3_SMALL, 2, 0.02},
{"text-embedding-3-large", "OpenAI", DIMS_3_LARGE, 3, 0.13},
{"embed-english-v3.0", "Cohere", DIMS_ENGLISH_V3, 3, 0.1},
{"voyage-3-lite", "Voyage AI", DIMS_VOYAGE_LITE, 2, 0.02},
};
#define EMBEDDING_MODEL_COUNT (sizeof EMBEDDING_MODELS / sizeof EMBEDDING_MODELS[0])
/* Default knobs: 512-token chunks, 64-token overlap (TS DEFAULT_CHUNK_OPTIONS). */
enum { DEFAULT_CHUNK_SIZE = 512, DEFAULT_OVERLAP = 64 };
/* Chunking knobs, in tokens. Mirrors the TS `Partial<ChunkOptions>`: each
* field is independently optional — has_* == false falls back to the 512 / 64
* default, and setting one leaves the other at its default. */
typedef struct {
long chunk_size;
bool has_chunk_size;
long overlap;
bool has_overlap;
} ChunkOptions;
/* How a document splits into overlapping chunks. */
typedef struct {
long chunks;
long total_tokens_with_overlap;
long overhead_tokens;
} ChunkPlan;
/* The "nothing to embed" plan the TS source returns for zero/negative input
* or a non-positive chunk size. */
static const ChunkPlan CHUNK_PLAN_ZERO = {0, 0, 0};
static long clamp_long(long v, long lo, long hi) {
return v < lo ? lo : (v > hi ? hi : v);
}
static long max_long(long a, long b) {
return a > b ? a : b;
}
/* Look up an embedding model by id. Returns NULL for unknown ids. */
const EmbeddingModel *get_embedding_model(const char *id) {
for (size_t i = 0; i < EMBEDDING_MODEL_COUNT; i++) {
if (strcmp(EMBEDDING_MODELS[i].id, id) == 0) {
return &EMBEDDING_MODELS[i];
}
}
return NULL;
}
/* Plan how `total_tokens` split into overlapping chunks. `opts` may be NULL
* (both defaults), mirroring the TS optional parameter. */
ChunkPlan plan_chunks(long total_tokens, const ChunkOptions *opts) {
long chunk_size = (opts != NULL && opts->has_chunk_size) ? opts->chunk_size : DEFAULT_CHUNK_SIZE;
long overlap_raw = (opts != NULL && opts->has_overlap) ? opts->overlap : DEFAULT_OVERLAP;
if (chunk_size <= 0 || total_tokens <= 0) {
return CHUNK_PLAN_ZERO;
}
/* min(max(overlap, 0), chunk_size / 2) — the TS clamp. Overlap that large
* would never advance, so consecutive chunks always gain at least half a
* chunk. (chunk_size >= 1 here, so chunk_size - overlap is never zero.) */
long overlap = clamp_long(overlap_raw, 0, chunk_size / 2);
/* Float division + ceil mirrors TS's Math.ceil exactly (a tiny document
* lands the quotient just below zero; ceil brings it to 0 and max(1, ..)
* lifts it back to one chunk). */
long chunks = max_long(1, (long)ceil((double)(total_tokens - overlap) / (double)(chunk_size - overlap)));
long total_tokens_with_overlap = total_tokens + (chunks - 1) * overlap;
ChunkPlan plan = {chunks, total_tokens_with_overlap, total_tokens_with_overlap - total_tokens};
return plan;
}
/* Chunk plan plus pricing for one embedding call. The three chunk fields are
* flattened in (the TS `...plan` spread) so the struct reads like the TS
* `EmbeddingPlan extends ChunkPlan`. */
typedef struct {
long chunks;
long total_tokens_with_overlap;
long overhead_tokens;
const EmbeddingModel *model;
long vectors; /* one vector per chunk */
double cost; /* USD: total_tokens_with_overlap / 1e6 * model->input_per_m */
} EmbeddingPlan;
/* Chunk a document AND price its embedding for `model_id` at `dims`
* dimensions. Unknown model, or dims the model does not offer -> false
* (*out untouched); success -> true with *out filled in. */
bool plan_embedding(long total_tokens, const char *model_id, long dims,
const ChunkOptions *opts, EmbeddingPlan *out) {
const EmbeddingModel *model = get_embedding_model(model_id);
if (model == NULL) {
return false;
}
bool offered = false;
for (size_t i = 0; i < model->dims_len; i++) {
if (model->dims[i] == dims) {
offered = true;
break;
}
}
if (!offered) {
return false;
}
ChunkPlan plan = plan_chunks(total_tokens, opts);
EmbeddingPlan priced = {
plan.chunks,
plan.total_tokens_with_overlap,
plan.overhead_tokens,
model,
plan.chunks,
(double)plan.total_tokens_with_overlap / 1e6 * model->input_per_m,
};
if (out != NULL) {
*out = priced;
}
return true;
}
/* ---------- showcase examples (the canonical suite lives in src/lib) ---------- */
static bool plan_equals(ChunkPlan a, ChunkPlan b) {
return a.chunks == b.chunks &&
a.total_tokens_with_overlap == b.total_tokens_with_overlap &&
a.overhead_tokens == b.overhead_tokens;
}
int main(void) {
/* 1,000 tokens: ceil((1000-64)/(512-64)) = 3 chunks, 2 seams x 64. */
ChunkPlan a = plan_chunks(1000, NULL);
assert(plan_equals(a, (ChunkPlan){3, 1128, 128}));
/* A document that fits one chunk has no seam overhead. */
assert(plan_equals(plan_chunks(512, NULL), (ChunkPlan){1, 512, 0}));
/* Zero/negative input or non-positive chunk size -> the zero plan. */
assert(plan_equals(plan_chunks(0, NULL), CHUNK_PLAN_ZERO));
assert(plan_equals(plan_chunks(-100, NULL), CHUNK_PLAN_ZERO));
assert(plan_equals(plan_chunks(1000, &(ChunkOptions){.chunk_size = 0, .has_chunk_size = true}), CHUNK_PLAN_ZERO));
assert(plan_equals(plan_chunks(1000, &(ChunkOptions){.chunk_size = -8, .has_chunk_size = true}), CHUNK_PLAN_ZERO));
/* overlap 600 > floor(512/2) = 256 -> clamped to 256. */
ChunkOptions big_overlap = {.overlap = 600, .has_overlap = true};
assert(plan_equals(plan_chunks(1000, &big_overlap), (ChunkPlan){3, 1512, 512}));
/* Negative overlap clamps to 0: 1000 tokens -> ceil(1000/512) = 2 chunks. */
ChunkOptions neg_overlap = {.overlap = -5, .has_overlap = true};
assert(plan_equals(plan_chunks(1000, &neg_overlap), (ChunkPlan){2, 1000, 0}));
/* chunk_size without overlap: ceil((1000-64)/192) = 5 chunks. */
ChunkOptions small_chunks = {.chunk_size = 256, .has_chunk_size = true};
assert(plan_equals(plan_chunks(1000, &small_chunks), (ChunkPlan){5, 1256, 256}));
/* Shorter than the overlap still yields one chunk. */
ChunkOptions std_overlap = {.overlap = 64, .has_overlap = true};
assert(plan_equals(plan_chunks(50, &std_overlap), (ChunkPlan){1, 50, 0}));
/* chunk_size of 1 clamps overlap to 0: ceil(3/1) = 3 chunks. */
ChunkOptions tiny = {.chunk_size = 1, .has_chunk_size = true};
assert(plan_equals(plan_chunks(3, &tiny), (ChunkPlan){3, 3, 0}));
/* Pricing: 1,000 tokens on text-embedding-3-small @ 1536 dims. */
EmbeddingPlan priced;
assert(plan_embedding(1000, "text-embedding-3-small", 1536, NULL, &priced));
assert(priced.chunks == 3 && priced.total_tokens_with_overlap == 1128 && priced.vectors == 3);
assert(fabs(priced.cost - 0.00002256) < 1e-12); /* 1128 / 1e6 * $0.02 */
/* A single-chunk document on voyage-3-lite @ 512 dims. */
EmbeddingPlan single;
assert(plan_embedding(512, "voyage-3-lite", 512, NULL, &single));
assert(single.vectors == 1); /* one vector for one chunk */
assert(fabs(single.cost - 0.00001024) < 1e-12); /* 512 / 1e6 * $0.02 */
/* Unknown model or unoffered dims -> false. */
EmbeddingPlan ignored;
assert(!plan_embedding(1000, "text-embedding-3-small", 999, NULL, &ignored));
assert(!plan_embedding(1000, "ghost", 1536, NULL, &ignored));
/* Zero tokens price out to a zero-cost plan. */
EmbeddingPlan zero;
assert(plan_embedding(0, "text-embedding-3-small", 1536, NULL, &zero));
assert(zero.chunks == 0 && zero.total_tokens_with_overlap == 0 && zero.cost == 0.0);
return 0;
}
Also available in 12 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →