Embedding Chunk Planner — Zig source
Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! Embedding Chunk Planner — pure chunking math for RAG pipelines.
//!
//! Language: Zig (Zig 0.13, standard library only)
//! Source: CosmoDev polyglot showcase port of the Embedding Chunk Planner
//! tool, ported from src/lib/embeddingPlanner.ts (the canonical
//! TypeScript implementation).
//! Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; no allocator needed, no unreachable for any input.
//! - Functionally equivalent to the TS reference: same inputs -> same outputs.
//! - Self-contained: std only (no third-party packages). The model price
//! table is inlined below, mirrored from src/lib/ai/embeddings.ts — prices
//! NEVER live in the planner itself.
//!
//! Behavior (mirrors the TS source exactly):
//! - chunk_size <= 0 or total_tokens <= 0 -> ChunkPlan.zero (nothing to
//! embed).
//! - Negative overlap is treated as 0; overlap then clamps to at most
//! chunk_size / 2 so consecutive chunks always advance.
//! - chunks = max(1, ceil((total_tokens - overlap) / (chunk_size - overlap)))
//! — a tiny document still yields one chunk.
const std = @import("std");
/// One embedding model's offered dimensions (ascending, Matryoshka shortening
/// included) and pricing: USD per 1M input tokens.
pub const EmbeddingModel = struct {
id: []const u8,
vendor: []const u8,
dims: []const i64,
input_per_m: f64,
};
/// Embedding model price table — the SSOT for pricing, mirrored from
/// src/lib/ai/embeddings.ts. Refresh both files together.
pub const embedding_models = [_]EmbeddingModel{
.{ .id = "text-embedding-3-small", .vendor = "OpenAI", .dims = &[_]i64{ 512, 1536 }, .input_per_m = 0.02 },
.{ .id = "text-embedding-3-large", .vendor = "OpenAI", .dims = &[_]i64{ 256, 1024, 3072 }, .input_per_m = 0.13 },
.{ .id = "embed-english-v3.0", .vendor = "Cohere", .dims = &[_]i64{ 512, 1024, 1536 }, .input_per_m = 0.1 },
.{ .id = "voyage-3-lite", .vendor = "Voyage AI", .dims = &[_]i64{ 512, 1024 }, .input_per_m = 0.02 },
};
/// Default knobs: 512-token chunks, 64-token overlap (TS DEFAULT_CHUNK_OPTIONS).
pub const default_chunk_size: i64 = 512;
pub const default_overlap: i64 = 64;
/// Chunking knobs, in tokens. Mirrors the TS `Partial<ChunkOptions>`: each
/// field is independently optional — null falls back to the 512 / 64 default,
/// and setting one leaves the other at its default.
pub const ChunkOptions = struct {
chunk_size: ?i64 = null,
overlap: ?i64 = null,
};
/// How a document splits into overlapping chunks.
pub const ChunkPlan = struct {
chunks: i64,
total_tokens_with_overlap: i64,
overhead_tokens: i64,
/// The "nothing to embed" plan the TS source returns for zero/negative
/// input or a non-positive chunk size.
pub const zero = ChunkPlan{ .chunks = 0, .total_tokens_with_overlap = 0, .overhead_tokens = 0 };
};
/// Chunk plan plus pricing for one embedding call. The three chunk fields are
/// flattened in (the TS `...plan` spread) so the struct reads like the TS
/// `EmbeddingPlan extends ChunkPlan`. `vectors` is one per chunk; `cost` is
/// USD: total_tokens_with_overlap / 1e6 * model.input_per_m.
pub const EmbeddingPlan = struct {
chunks: i64,
total_tokens_with_overlap: i64,
overhead_tokens: i64,
model: *const EmbeddingModel,
vectors: i64,
cost: f64,
};
/// Look up an embedding model by id. Returns null for unknown ids.
pub fn getEmbeddingModel(id: []const u8) ?*const EmbeddingModel {
for (&embedding_models) |*model| {
if (std.mem.eql(u8, model.id, id)) {
return model;
}
}
return null;
}
/// Plan how `total_tokens` split into overlapping chunks. `opts` may be null
/// (both defaults), mirroring the TS optional parameter; each null field falls
/// back to the 512 / 64 default independently.
pub fn planChunks(total_tokens: i64, opts: ?ChunkOptions) ChunkPlan {
const chunk_size = (if (opts) |o| o.chunk_size else null) orelse default_chunk_size;
const overlap_raw = (if (opts) |o| o.overlap else null) orelse default_overlap;
if (chunk_size <= 0 or total_tokens <= 0) {
return ChunkPlan.zero;
}
// @min(@max(overlap_raw, 0), chunk_size / 2) — the TS clamp. Overlap that
// large would never advance, so consecutive chunks always gain at least
// half a chunk. (chunk_size >= 1 here, so chunk_size - overlap is never
// zero; both operands are positive so `/` truncates like TS's floor.)
const overlap = @min(@max(overlap_raw, 0), chunk_size / 2);
// Float division + ceil mirrors TS's Math.ceil exactly (a tiny document
// lands the quotient just below zero; ceil brings it to 0 and @max(1, ..)
// lifts it back to one chunk).
const quotient = @as(f64, @floatFromInt(total_tokens - overlap)) /
@as(f64, @floatFromInt(chunk_size - overlap));
const chunks = @max(@as(i64, 1), @as(i64, @intFromFloat(@ceil(quotient))));
const total_tokens_with_overlap = total_tokens + (chunks - 1) * overlap;
return .{
.chunks = chunks,
.total_tokens_with_overlap = total_tokens_with_overlap,
.overhead_tokens = total_tokens_with_overlap - total_tokens,
};
}
/// Chunk a document AND price its embedding for `model_id` at `dims`
/// dimensions. Unknown model, or dims the model does not offer -> null.
pub fn planEmbedding(
total_tokens: i64,
model_id: []const u8,
dims: i64,
opts: ?ChunkOptions,
) ?EmbeddingPlan {
const model = getEmbeddingModel(model_id) orelse return null;
var offered = false;
for (model.dims) |d| {
if (d == dims) {
offered = true;
break;
}
}
if (!offered) {
return null;
}
const plan = planChunks(total_tokens, opts);
return .{
.chunks = plan.chunks,
.total_tokens_with_overlap = plan.total_tokens_with_overlap,
.overhead_tokens = plan.overhead_tokens,
.model = model,
.vectors = plan.chunks,
.cost = @as(f64, @floatFromInt(plan.total_tokens_with_overlap)) / 1e6 * model.input_per_m,
};
}
// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
const testing = std.testing;
test "uses 512/64 defaults" {
// 1,000 tokens: ceil((1000-64)/(512-64)) = 3 chunks, 2 seams x 64.
try testing.expectEqual(ChunkPlan{
.chunks = 3,
.total_tokens_with_overlap = 1128,
.overhead_tokens = 128,
}, planChunks(1000, null));
}
test "single chunk when the document fits" {
try testing.expectEqual(ChunkPlan{
.chunks = 1,
.total_tokens_with_overlap = 512,
.overhead_tokens = 0,
}, planChunks(512, null));
}
test "zeros for degenerate input" {
try testing.expectEqual(ChunkPlan.zero, planChunks(0, null));
try testing.expectEqual(ChunkPlan.zero, planChunks(-100, null));
try testing.expectEqual(ChunkPlan.zero, planChunks(1000, .{ .chunk_size = 0 }));
try testing.expectEqual(ChunkPlan.zero, planChunks(1000, .{ .chunk_size = -8 }));
}
test "clamps overlap to half the chunk size" {
// overlap 600 > floor(512/2) = 256 -> clamped to 256.
try testing.expectEqual(ChunkPlan{
.chunks = 3,
.total_tokens_with_overlap = 1512,
.overhead_tokens = 512,
}, planChunks(1000, .{ .overlap = 600 }));
}
test "negative overlap is zero and partial options merge" {
try testing.expectEqual(ChunkPlan{
.chunks = 2,
.total_tokens_with_overlap = 1000,
.overhead_tokens = 0,
}, planChunks(1000, .{ .overlap = -5 }));
// chunk_size without overlap: ceil((1000-64)/192) = 5 chunks.
try testing.expectEqual(ChunkPlan{
.chunks = 5,
.total_tokens_with_overlap = 1256,
.overhead_tokens = 256,
}, planChunks(1000, .{ .chunk_size = 256 }));
}
test "one chunk when shorter than the overlap" {
// ceil((50-64)/448) = 0 -> clamped up to 1 chunk.
try testing.expectEqual(ChunkPlan{
.chunks = 1,
.total_tokens_with_overlap = 50,
.overhead_tokens = 0,
}, planChunks(50, .{ .overlap = 64 }));
}
test "chunk size of one clamps overlap to zero" {
try testing.expectEqual(ChunkPlan{
.chunks = 3,
.total_tokens_with_overlap = 3,
.overhead_tokens = 0,
}, planChunks(3, .{ .chunk_size = 1 }));
}
test "prices a known plan" {
const plan = (planEmbedding(1000, "text-embedding-3-small", 1536, null)).?;
try testing.expectEqual(@as(i64, 3), plan.chunks);
try testing.expectEqual(@as(i64, 1128), plan.total_tokens_with_overlap);
try testing.expectEqual(@as(i64, 3), plan.vectors);
// 1128 / 1e6 * $0.02
try testing.expectApproxEqAbs(@as(f64, 0.00002256), plan.cost, 1e-12);
try testing.expectEqualStrings("text-embedding-3-small", plan.model.id);
}
test "prices a single chunk document" {
const plan = (planEmbedding(512, "voyage-3-lite", 512, null)).?;
try testing.expectEqual(@as(i64, 1), plan.vectors);
// 512 / 1e6 * $0.02
try testing.expectApproxEqAbs(@as(f64, 0.00001024), plan.cost, 1e-12);
}
test "unknown model or dims is null" {
try testing.expect(planEmbedding(1000, "text-embedding-3-small", 999, null) == null);
try testing.expect(planEmbedding(1000, "ghost", 1536, null) == null);
}
test "zero tokens is a zero cost plan" {
const plan = (planEmbedding(0, "text-embedding-3-small", 1536, null)).?;
try testing.expectEqual(@as(i64, 0), plan.chunks);
try testing.expectEqual(@as(i64, 0), plan.vectors);
try testing.expectEqual(@as(f64, 0.0), plan.cost);
}
Also available in 12 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →