Skip to content

Embedding Chunk Planner — C++ source

Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Embedding Chunk Planner — pure chunking math for RAG pipelines.
//
// Language: C++ (C++17, standard library only)
// Source:   CosmoDev polyglot showcase port of the Embedding Chunk Planner
//           tool, ported from src/lib/embeddingPlanner.ts (the canonical
//           TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; never throws.
//   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//   - Self-contained: std only (no Boost / crates needed). The model price
//     table is inlined below, mirrored from src/lib/ai/embeddings.ts — prices
//     NEVER live in the planner itself.
//
// Behavior (mirrors the TS source exactly):
//   - chunk_size <= 0 or total_tokens <= 0 -> ZERO_PLAN (nothing to embed).
//   - Negative overlap is treated as 0; overlap then clamps to at most
//     chunk_size / 2 so consecutive chunks always advance.
//   - chunks = max(1, ceil((total_tokens - overlap) / (chunk_size - overlap)))
//     — a tiny document still yields one chunk.

#include <algorithm>
#include <cassert>
#include <cmath>
#include <optional>
#include <string_view>
#include <vector>

/// One embedding model's offered dimensions (ascending, Matryoshka shortening
/// included) and pricing: USD per 1M input tokens.
struct EmbeddingModel {
    std::string_view id;
    std::string_view vendor;
    std::vector<long long> dims;
    double input_per_m;
};

/// Embedding model price table — the SSOT for pricing, mirrored from
/// src/lib/ai/embeddings.ts. Refresh both files together.
const std::vector<EmbeddingModel> EMBEDDING_MODELS{
    {"text-embedding-3-small", "OpenAI", {512, 1536}, 0.02},
    {"text-embedding-3-large", "OpenAI", {256, 1024, 3072}, 0.13},
    {"embed-english-v3.0", "Cohere", {512, 1024, 1536}, 0.1},
    {"voyage-3-lite", "Voyage AI", {512, 1024}, 0.02},
};

/// Default knobs: 512-token chunks, 64-token overlap (TS DEFAULT_CHUNK_OPTIONS).
constexpr long long DEFAULT_CHUNK_SIZE = 512;
constexpr long long DEFAULT_OVERLAP = 64;

/// Look up an embedding model by id. Returns std::nullopt for unknown ids.
std::optional<const EmbeddingModel *> get_embedding_model(std::string_view id) {
    for (const auto &model : EMBEDDING_MODELS) {
        if (model.id == id) {
            return &model;
        }
    }
    return std::nullopt;
}

/// Chunking knobs, in tokens. Mirrors the TS `Partial<ChunkOptions>`: each
/// field is independently optional — std::nullopt falls back to the 512 / 64
/// default, and setting one leaves the other at its default.
struct ChunkOptions {
    std::optional<long long> chunk_size;
    std::optional<long long> overlap;
};

/// How a document splits into overlapping chunks.
struct ChunkPlan {
    long long chunks;
    long long total_tokens_with_overlap;
    long long overhead_tokens;

    constexpr bool operator==(const ChunkPlan &other) const {
        return chunks == other.chunks &&
               total_tokens_with_overlap == other.total_tokens_with_overlap &&
               overhead_tokens == other.overhead_tokens;
    }
};

/// The "nothing to embed" plan the TS source returns for zero/negative input
/// or a non-positive chunk size.
constexpr ChunkPlan ZERO_PLAN{0, 0, 0};

/// Plan how `total_tokens` split into overlapping chunks. `opts` may be
/// std::nullopt (both defaults), mirroring the TS optional parameter.
ChunkPlan plan_chunks(long long total_tokens, std::optional<ChunkOptions> opts = std::nullopt) {
    long long chunk_size = (opts && opts->chunk_size) ? *opts->chunk_size : DEFAULT_CHUNK_SIZE;
    long long overlap_raw = (opts && opts->overlap) ? *opts->overlap : DEFAULT_OVERLAP;

    if (chunk_size <= 0 || total_tokens <= 0) {
        return ZERO_PLAN;
    }

    // min(max(overlap, 0), chunk_size / 2) — the TS clamp. Overlap that large
    // would never advance, so consecutive chunks always gain at least half a
    // chunk. (chunk_size >= 1 here, so chunk_size - overlap is never zero.)
    long long overlap = std::clamp(overlap_raw, 0LL, chunk_size / 2);

    // Float division + ceil mirrors TS's Math.ceil exactly (a tiny document
    // lands the quotient just below zero; ceil brings it to 0 and max(1, ..)
    // lifts it back to one chunk).
    long long chunks = std::max(1LL, static_cast<long long>(
                                         std::ceil(static_cast<double>(total_tokens - overlap) /
                                                   static_cast<double>(chunk_size - overlap))));

    long long total_tokens_with_overlap = total_tokens + (chunks - 1) * overlap;
    return ChunkPlan{chunks, total_tokens_with_overlap, total_tokens_with_overlap - total_tokens};
}

/// Chunk plan plus pricing for one embedding call. The three chunk fields are
/// flattened in (the TS `...plan` spread) so the struct reads like the TS
/// `EmbeddingPlan extends ChunkPlan`.
struct EmbeddingPlan {
    long long chunks;
    long long total_tokens_with_overlap;
    long long overhead_tokens;
    const EmbeddingModel *model;
    long long vectors; ///< one vector per chunk
    double cost;       ///< USD: total_tokens_with_overlap / 1e6 * model->input_per_m
};

/// Chunk a document AND price its embedding for `model_id` at `dims`
/// dimensions. Unknown model, or dims the model does not offer ->
/// std::nullopt.
std::optional<EmbeddingPlan> plan_embedding(long long total_tokens, std::string_view model_id,
                                            long long dims,
                                            std::optional<ChunkOptions> opts = std::nullopt) {
    const EmbeddingModel *model = get_embedding_model(model_id).value_or(nullptr);
    if (model == nullptr) {
        return std::nullopt;
    }
    if (std::find(model->dims.begin(), model->dims.end(), dims) == model->dims.end()) {
        return std::nullopt;
    }
    ChunkPlan plan = plan_chunks(total_tokens, opts);
    return EmbeddingPlan{
        plan.chunks,
        plan.total_tokens_with_overlap,
        plan.overhead_tokens,
        model,
        plan.chunks,
        static_cast<double>(plan.total_tokens_with_overlap) / 1e6 * model->input_per_m,
    };
}

// ---------- showcase examples (the canonical suite lives in src/lib) ----------
int main() {
    const ChunkOptions zero_chunk_size{0, std::nullopt};
    const ChunkOptions negative_chunk_size{-8, std::nullopt};
    const ChunkOptions big_overlap{std::nullopt, 600};
    const ChunkOptions negative_overlap{std::nullopt, -5};
    const ChunkOptions small_chunks{256, std::nullopt};
    const ChunkOptions std_overlap{std::nullopt, 64};
    const ChunkOptions tiny_chunks{1, std::nullopt};

    // 1,000 tokens: ceil((1000-64)/(512-64)) = 3 chunks, 2 seams x 64.
    assert((plan_chunks(1000) == ChunkPlan{3, 1128, 128}));

    // A document that fits one chunk has no seam overhead.
    assert((plan_chunks(512) == ChunkPlan{1, 512, 0}));

    // Zero/negative input or non-positive chunk size -> the zero plan.
    assert(plan_chunks(0) == ZERO_PLAN);
    assert(plan_chunks(-100) == ZERO_PLAN);
    assert(plan_chunks(1000, zero_chunk_size) == ZERO_PLAN);
    assert(plan_chunks(1000, negative_chunk_size) == ZERO_PLAN);

    // overlap 600 > floor(512/2) = 256 -> clamped to 256.
    assert((plan_chunks(1000, big_overlap) == ChunkPlan{3, 1512, 512}));

    // Negative overlap clamps to 0: 1000 tokens -> ceil(1000/512) = 2 chunks.
    assert((plan_chunks(1000, negative_overlap) == ChunkPlan{2, 1000, 0}));

    // chunk_size without overlap: ceil((1000-64)/192) = 5 chunks.
    assert((plan_chunks(1000, small_chunks) == ChunkPlan{5, 1256, 256}));

    // Shorter than the overlap still yields one chunk.
    assert((plan_chunks(50, std_overlap) == ChunkPlan{1, 50, 0}));

    // chunk_size of 1 clamps overlap to 0: ceil(3/1) = 3 chunks.
    assert((plan_chunks(3, tiny_chunks) == ChunkPlan{3, 3, 0}));

    // Pricing: 1,000 tokens on text-embedding-3-small @ 1536 dims.
    auto priced = plan_embedding(1000, "text-embedding-3-small", 1536);
    assert(priced && priced->chunks == 3 && priced->total_tokens_with_overlap == 1128 && priced->vectors == 3);
    assert(std::abs(priced->cost - 0.00002256) < 1e-12); // 1128 / 1e6 * $0.02

    // A single-chunk document on voyage-3-lite @ 512 dims.
    auto single = plan_embedding(512, "voyage-3-lite", 512);
    assert(single && single->vectors == 1);
    assert(std::abs(single->cost - 0.00001024) < 1e-12); // 512 / 1e6 * $0.02

    // Unknown model or unoffered dims -> std::nullopt.
    assert(!plan_embedding(1000, "text-embedding-3-small", 999).has_value());
    assert(!plan_embedding(1000, "ghost", 1536).has_value());

    // Zero tokens price out to a zero-cost plan.
    auto zero = plan_embedding(0, "text-embedding-3-small", 1536);
    assert(zero && zero->chunks == 0 && zero->vectors == 0 && zero->cost == 0.0);

    return 0;
}

Also available in 12 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →