Skip to content

Embedding Chunk Planner — Zig source

Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! Embedding Chunk Planner — pure chunking math for RAG pipelines.
//!
//! Language: Zig (Zig 0.13, standard library only)
//! Source:   CosmoDev polyglot showcase port of the Embedding Chunk Planner
//!           tool, ported from src/lib/embeddingPlanner.ts (the canonical
//!           TypeScript implementation).
//! Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; no allocator needed, no unreachable for any input.
//!   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//!   - Self-contained: std only (no third-party packages). The model price
//!     table is inlined below, mirrored from src/lib/ai/embeddings.ts — prices
//!     NEVER live in the planner itself.
//!
//! Behavior (mirrors the TS source exactly):
//!   - chunk_size <= 0 or total_tokens <= 0 -> ChunkPlan.zero (nothing to
//!     embed).
//!   - Negative overlap is treated as 0; overlap then clamps to at most
//!     chunk_size / 2 so consecutive chunks always advance.
//!   - chunks = max(1, ceil((total_tokens - overlap) / (chunk_size - overlap)))
//!     — a tiny document still yields one chunk.

const std = @import("std");

/// One embedding model's offered dimensions (ascending, Matryoshka shortening
/// included) and pricing: USD per 1M input tokens.
pub const EmbeddingModel = struct {
    id: []const u8,
    vendor: []const u8,
    dims: []const i64,
    input_per_m: f64,
};

/// Embedding model price table — the SSOT for pricing, mirrored from
/// src/lib/ai/embeddings.ts. Refresh both files together.
pub const embedding_models = [_]EmbeddingModel{
    .{ .id = "text-embedding-3-small", .vendor = "OpenAI", .dims = &[_]i64{ 512, 1536 }, .input_per_m = 0.02 },
    .{ .id = "text-embedding-3-large", .vendor = "OpenAI", .dims = &[_]i64{ 256, 1024, 3072 }, .input_per_m = 0.13 },
    .{ .id = "embed-english-v3.0", .vendor = "Cohere", .dims = &[_]i64{ 512, 1024, 1536 }, .input_per_m = 0.1 },
    .{ .id = "voyage-3-lite", .vendor = "Voyage AI", .dims = &[_]i64{ 512, 1024 }, .input_per_m = 0.02 },
};

/// Default knobs: 512-token chunks, 64-token overlap (TS DEFAULT_CHUNK_OPTIONS).
pub const default_chunk_size: i64 = 512;
pub const default_overlap: i64 = 64;

/// Chunking knobs, in tokens. Mirrors the TS `Partial<ChunkOptions>`: each
/// field is independently optional — null falls back to the 512 / 64 default,
/// and setting one leaves the other at its default.
pub const ChunkOptions = struct {
    chunk_size: ?i64 = null,
    overlap: ?i64 = null,
};

/// How a document splits into overlapping chunks.
pub const ChunkPlan = struct {
    chunks: i64,
    total_tokens_with_overlap: i64,
    overhead_tokens: i64,

    /// The "nothing to embed" plan the TS source returns for zero/negative
    /// input or a non-positive chunk size.
    pub const zero = ChunkPlan{ .chunks = 0, .total_tokens_with_overlap = 0, .overhead_tokens = 0 };
};

/// Chunk plan plus pricing for one embedding call. The three chunk fields are
/// flattened in (the TS `...plan` spread) so the struct reads like the TS
/// `EmbeddingPlan extends ChunkPlan`. `vectors` is one per chunk; `cost` is
/// USD: total_tokens_with_overlap / 1e6 * model.input_per_m.
pub const EmbeddingPlan = struct {
    chunks: i64,
    total_tokens_with_overlap: i64,
    overhead_tokens: i64,
    model: *const EmbeddingModel,
    vectors: i64,
    cost: f64,
};

/// Look up an embedding model by id. Returns null for unknown ids.
pub fn getEmbeddingModel(id: []const u8) ?*const EmbeddingModel {
    for (&embedding_models) |*model| {
        if (std.mem.eql(u8, model.id, id)) {
            return model;
        }
    }
    return null;
}

/// Plan how `total_tokens` split into overlapping chunks. `opts` may be null
/// (both defaults), mirroring the TS optional parameter; each null field falls
/// back to the 512 / 64 default independently.
pub fn planChunks(total_tokens: i64, opts: ?ChunkOptions) ChunkPlan {
    const chunk_size = (if (opts) |o| o.chunk_size else null) orelse default_chunk_size;
    const overlap_raw = (if (opts) |o| o.overlap else null) orelse default_overlap;

    if (chunk_size <= 0 or total_tokens <= 0) {
        return ChunkPlan.zero;
    }

    // @min(@max(overlap_raw, 0), chunk_size / 2) — the TS clamp. Overlap that
    // large would never advance, so consecutive chunks always gain at least
    // half a chunk. (chunk_size >= 1 here, so chunk_size - overlap is never
    // zero; both operands are positive so `/` truncates like TS's floor.)
    const overlap = @min(@max(overlap_raw, 0), chunk_size / 2);

    // Float division + ceil mirrors TS's Math.ceil exactly (a tiny document
    // lands the quotient just below zero; ceil brings it to 0 and @max(1, ..)
    // lifts it back to one chunk).
    const quotient = @as(f64, @floatFromInt(total_tokens - overlap)) /
        @as(f64, @floatFromInt(chunk_size - overlap));
    const chunks = @max(@as(i64, 1), @as(i64, @intFromFloat(@ceil(quotient))));

    const total_tokens_with_overlap = total_tokens + (chunks - 1) * overlap;
    return .{
        .chunks = chunks,
        .total_tokens_with_overlap = total_tokens_with_overlap,
        .overhead_tokens = total_tokens_with_overlap - total_tokens,
    };
}

/// Chunk a document AND price its embedding for `model_id` at `dims`
/// dimensions. Unknown model, or dims the model does not offer -> null.
pub fn planEmbedding(
    total_tokens: i64,
    model_id: []const u8,
    dims: i64,
    opts: ?ChunkOptions,
) ?EmbeddingPlan {
    const model = getEmbeddingModel(model_id) orelse return null;
    var offered = false;
    for (model.dims) |d| {
        if (d == dims) {
            offered = true;
            break;
        }
    }
    if (!offered) {
        return null;
    }

    const plan = planChunks(total_tokens, opts);
    return .{
        .chunks = plan.chunks,
        .total_tokens_with_overlap = plan.total_tokens_with_overlap,
        .overhead_tokens = plan.overhead_tokens,
        .model = model,
        .vectors = plan.chunks,
        .cost = @as(f64, @floatFromInt(plan.total_tokens_with_overlap)) / 1e6 * model.input_per_m,
    };
}

// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
const testing = std.testing;

test "uses 512/64 defaults" {
    // 1,000 tokens: ceil((1000-64)/(512-64)) = 3 chunks, 2 seams x 64.
    try testing.expectEqual(ChunkPlan{
        .chunks = 3,
        .total_tokens_with_overlap = 1128,
        .overhead_tokens = 128,
    }, planChunks(1000, null));
}

test "single chunk when the document fits" {
    try testing.expectEqual(ChunkPlan{
        .chunks = 1,
        .total_tokens_with_overlap = 512,
        .overhead_tokens = 0,
    }, planChunks(512, null));
}

test "zeros for degenerate input" {
    try testing.expectEqual(ChunkPlan.zero, planChunks(0, null));
    try testing.expectEqual(ChunkPlan.zero, planChunks(-100, null));
    try testing.expectEqual(ChunkPlan.zero, planChunks(1000, .{ .chunk_size = 0 }));
    try testing.expectEqual(ChunkPlan.zero, planChunks(1000, .{ .chunk_size = -8 }));
}

test "clamps overlap to half the chunk size" {
    // overlap 600 > floor(512/2) = 256 -> clamped to 256.
    try testing.expectEqual(ChunkPlan{
        .chunks = 3,
        .total_tokens_with_overlap = 1512,
        .overhead_tokens = 512,
    }, planChunks(1000, .{ .overlap = 600 }));
}

test "negative overlap is zero and partial options merge" {
    try testing.expectEqual(ChunkPlan{
        .chunks = 2,
        .total_tokens_with_overlap = 1000,
        .overhead_tokens = 0,
    }, planChunks(1000, .{ .overlap = -5 }));
    // chunk_size without overlap: ceil((1000-64)/192) = 5 chunks.
    try testing.expectEqual(ChunkPlan{
        .chunks = 5,
        .total_tokens_with_overlap = 1256,
        .overhead_tokens = 256,
    }, planChunks(1000, .{ .chunk_size = 256 }));
}

test "one chunk when shorter than the overlap" {
    // ceil((50-64)/448) = 0 -> clamped up to 1 chunk.
    try testing.expectEqual(ChunkPlan{
        .chunks = 1,
        .total_tokens_with_overlap = 50,
        .overhead_tokens = 0,
    }, planChunks(50, .{ .overlap = 64 }));
}

test "chunk size of one clamps overlap to zero" {
    try testing.expectEqual(ChunkPlan{
        .chunks = 3,
        .total_tokens_with_overlap = 3,
        .overhead_tokens = 0,
    }, planChunks(3, .{ .chunk_size = 1 }));
}

test "prices a known plan" {
    const plan = (planEmbedding(1000, "text-embedding-3-small", 1536, null)).?;
    try testing.expectEqual(@as(i64, 3), plan.chunks);
    try testing.expectEqual(@as(i64, 1128), plan.total_tokens_with_overlap);
    try testing.expectEqual(@as(i64, 3), plan.vectors);
    // 1128 / 1e6 * $0.02
    try testing.expectApproxEqAbs(@as(f64, 0.00002256), plan.cost, 1e-12);
    try testing.expectEqualStrings("text-embedding-3-small", plan.model.id);
}

test "prices a single chunk document" {
    const plan = (planEmbedding(512, "voyage-3-lite", 512, null)).?;
    try testing.expectEqual(@as(i64, 1), plan.vectors);
    // 512 / 1e6 * $0.02
    try testing.expectApproxEqAbs(@as(f64, 0.00001024), plan.cost, 1e-12);
}

test "unknown model or dims is null" {
    try testing.expect(planEmbedding(1000, "text-embedding-3-small", 999, null) == null);
    try testing.expect(planEmbedding(1000, "ghost", 1536, null) == null);
}

test "zero tokens is a zero cost plan" {
    const plan = (planEmbedding(0, "text-embedding-3-small", 1536, null)).?;
    try testing.expectEqual(@as(i64, 0), plan.chunks);
    try testing.expectEqual(@as(i64, 0), plan.vectors);
    try testing.expectEqual(@as(f64, 0.0), plan.cost);
}

Also available in 12 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →