Skip to content

Embedding Chunk Planner — Rust source

Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! Embedding Chunk Planner — pure chunking math for RAG pipelines.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source:   CosmoDev polyglot showcase port of the Embedding Chunk Planner
//!           tool, ported from src/lib/embeddingPlanner.ts (the canonical
//!           TypeScript implementation).
//! Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; never panics.
//!   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//!   - Self-contained: std only (no crates.io dependencies). The model price
//!     table is inlined below, mirrored from src/lib/ai/embeddings.ts — prices
//!     NEVER live in the planner itself.
//!
//! Behavior (mirrors the TS source exactly):
//!   - chunk_size <= 0 or total_tokens <= 0 -> ChunkPlan::ZERO (nothing to
//!     embed).
//!   - Negative overlap is treated as 0; overlap then clamps to at most
//!     chunk_size / 2 so consecutive chunks always advance.
//!   - chunks = max(1, ceil((total_tokens - overlap) / (chunk_size - overlap)))
//!     — a tiny document still yields one chunk.

/// One embedding model's offered dimensions (ascending, Matryoshka shortening
/// included) and pricing: USD per 1M input tokens.
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct EmbeddingModel {
    pub id: &'static str,
    pub vendor: &'static str,
    pub dims: &'static [i64],
    pub input_per_m: f64,
}

/// Embedding model price table — the SSOT for pricing, mirrored from
/// src/lib/ai/embeddings.ts. Refresh both files together.
pub const EMBEDDING_MODELS: &[EmbeddingModel] = &[
    EmbeddingModel { id: "text-embedding-3-small", vendor: "OpenAI", dims: &[512, 1536], input_per_m: 0.02 },
    EmbeddingModel { id: "text-embedding-3-large", vendor: "OpenAI", dims: &[256, 1024, 3072], input_per_m: 0.13 },
    EmbeddingModel { id: "embed-english-v3.0", vendor: "Cohere", dims: &[512, 1024, 1536], input_per_m: 0.1 },
    EmbeddingModel { id: "voyage-3-lite", vendor: "Voyage AI", dims: &[512, 1024], input_per_m: 0.02 },
];

/// Default knobs: 512-token chunks, 64-token overlap (TS DEFAULT_CHUNK_OPTIONS).
pub const DEFAULT_CHUNK_SIZE: i64 = 512;
pub const DEFAULT_OVERLAP: i64 = 64;

/// Look up an embedding model by id. Returns `None` for unknown ids.
pub fn get_embedding_model(id: &str) -> Option<&'static EmbeddingModel> {
    EMBEDDING_MODELS.iter().find(|m| m.id == id)
}

/// Chunking knobs, in tokens. Mirrors the TS `Partial<ChunkOptions>`: each
/// field is independently optional — `None` falls back to the 512 / 64
/// default, and setting one leaves the other at its default. Construct with
/// `..Default::default()`.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct ChunkOptions {
    pub chunk_size: Option<i64>,
    pub overlap: Option<i64>,
}

/// How a document splits into overlapping chunks.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct ChunkPlan {
    pub chunks: i64,
    pub total_tokens_with_overlap: i64,
    pub overhead_tokens: i64,
}

impl ChunkPlan {
    /// The "nothing to embed" plan the TS source returns for zero/negative
    /// input or a non-positive chunk size.
    pub const ZERO: ChunkPlan = ChunkPlan {
        chunks: 0,
        total_tokens_with_overlap: 0,
        overhead_tokens: 0,
    };
}

/// Plan how `total_tokens` split into overlapping chunks.
///
/// `opts` is optional wholesale (`None` = both defaults), mirroring the TS
/// optional parameter.
pub fn plan_chunks(total_tokens: i64, opts: Option<&ChunkOptions>) -> ChunkPlan {
    let chunk_size = opts.and_then(|o| o.chunk_size).unwrap_or(DEFAULT_CHUNK_SIZE);
    let overlap_raw = opts.and_then(|o| o.overlap).unwrap_or(DEFAULT_OVERLAP);

    if chunk_size <= 0 || total_tokens <= 0 {
        return ChunkPlan::ZERO;
    }

    // min(max(overlap, 0), chunk_size / 2) — the TS clamp. Overlap that large
    // would never advance, so consecutive chunks always gain at least half a
    // chunk. (chunk_size >= 1 here, so chunk_size - overlap is never zero.)
    let overlap = overlap_raw.clamp(0, chunk_size / 2);

    // Float division + ceil mirrors TS's Math.ceil exactly (a tiny document
    // lands the quotient just below zero; ceil brings it to 0 and max(1, ..)
    // lifts it back to one chunk).
    let chunks = (((total_tokens - overlap) as f64 / (chunk_size - overlap) as f64).ceil() as i64).max(1);

    let total_tokens_with_overlap = total_tokens + (chunks - 1) * overlap;
    ChunkPlan {
        chunks,
        total_tokens_with_overlap,
        overhead_tokens: total_tokens_with_overlap - total_tokens,
    }
}

/// Chunk plan plus pricing for one embedding call. The three chunk fields are
/// flattened in (the TS `...plan` spread) so the struct reads like the TS
/// `EmbeddingPlan extends ChunkPlan`.
#[derive(Debug, Clone, PartialEq)]
pub struct EmbeddingPlan {
    pub chunks: i64,
    pub total_tokens_with_overlap: i64,
    pub overhead_tokens: i64,
    pub model: &'static EmbeddingModel,
    /// One vector per chunk.
    pub vectors: i64,
    /// USD: total_tokens_with_overlap / 1e6 * model.input_per_m.
    pub cost: f64,
}

/// Chunk a document AND price its embedding for `model_id` at `dims`
/// dimensions. Unknown model, or dims the model does not offer -> `None`.
pub fn plan_embedding(
    total_tokens: i64,
    model_id: &str,
    dims: i64,
    opts: Option<&ChunkOptions>,
) -> Option<EmbeddingPlan> {
    let model = get_embedding_model(model_id)?;
    if !model.dims.contains(&dims) {
        return None;
    }
    let plan = plan_chunks(total_tokens, opts);
    Some(EmbeddingPlan {
        chunks: plan.chunks,
        total_tokens_with_overlap: plan.total_tokens_with_overlap,
        overhead_tokens: plan.overhead_tokens,
        model,
        vectors: plan.chunks,
        cost: (plan.total_tokens_with_overlap as f64 / 1e6) * model.input_per_m,
    })
}

// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn uses_512_64_defaults() {
        // 1,000 tokens: ceil((1000-64)/(512-64)) = 3 chunks, 2 seams x 64.
        assert_eq!(plan_chunks(1000, None), ChunkPlan { chunks: 3, total_tokens_with_overlap: 1128, overhead_tokens: 128 });
    }

    #[test]
    fn single_chunk_when_the_document_fits() {
        assert_eq!(plan_chunks(512, None), ChunkPlan { chunks: 1, total_tokens_with_overlap: 512, overhead_tokens: 0 });
    }

    #[test]
    fn zeros_for_degenerate_input() {
        assert_eq!(plan_chunks(0, None), ChunkPlan::ZERO);
        assert_eq!(plan_chunks(-100, None), ChunkPlan::ZERO);
        assert_eq!(plan_chunks(1000, Some(&ChunkOptions { chunk_size: Some(0), overlap: None })), ChunkPlan::ZERO);
        assert_eq!(plan_chunks(1000, Some(&ChunkOptions { chunk_size: Some(-8), overlap: None })), ChunkPlan::ZERO);
    }

    #[test]
    fn clamps_overlap_to_half_the_chunk_size() {
        // overlap 600 > floor(512/2) = 256 -> clamped to 256.
        let opts = ChunkOptions { overlap: Some(600), ..Default::default() };
        assert_eq!(plan_chunks(1000, Some(&opts)), ChunkPlan { chunks: 3, total_tokens_with_overlap: 1512, overhead_tokens: 512 });
    }

    #[test]
    fn negative_overlap_is_zero_and_partial_options_merge() {
        let neg = ChunkOptions { overlap: Some(-5), ..Default::default() };
        assert_eq!(plan_chunks(1000, Some(&neg)), ChunkPlan { chunks: 2, total_tokens_with_overlap: 1000, overhead_tokens: 0 });
        // chunkSize without overlap: ceil((1000-64)/192) = 5 chunks.
        let partial = ChunkOptions { chunk_size: Some(256), ..Default::default() };
        assert_eq!(plan_chunks(1000, Some(&partial)), ChunkPlan { chunks: 5, total_tokens_with_overlap: 1256, overhead_tokens: 256 });
    }

    #[test]
    fn one_chunk_when_shorter_than_the_overlap() {
        // ceil((50-64)/448) = 0 -> clamped up to 1 chunk.
        let opts = ChunkOptions { overlap: Some(64), ..Default::default() };
        assert_eq!(plan_chunks(50, Some(&opts)), ChunkPlan { chunks: 1, total_tokens_with_overlap: 50, overhead_tokens: 0 });
    }

    #[test]
    fn chunk_size_of_one_clamps_overlap_to_zero() {
        let opts = ChunkOptions { chunk_size: Some(1), ..Default::default() };
        assert_eq!(plan_chunks(3, Some(&opts)), ChunkPlan { chunks: 3, total_tokens_with_overlap: 3, overhead_tokens: 0 });
    }

    #[test]
    fn prices_a_known_plan() {
        let plan = plan_embedding(1000, "text-embedding-3-small", 1536, None).unwrap();
        assert_eq!(plan.chunks, 3);
        assert_eq!(plan.total_tokens_with_overlap, 1128);
        assert_eq!(plan.vectors, 3);
        // 1128 / 1e6 * $0.02
        assert!((plan.cost - 0.00002256).abs() < 1e-12);
        assert_eq!(plan.model.id, "text-embedding-3-small");
    }

    #[test]
    fn prices_a_single_chunk_document() {
        let plan = plan_embedding(512, "voyage-3-lite", 512, None).unwrap();
        assert_eq!(plan.vectors, 1);
        // 512 / 1e6 * $0.02
        assert!((plan.cost - 0.00001024).abs() < 1e-12);
    }

    #[test]
    fn unknown_model_or_dims_is_none() {
        assert!(plan_embedding(1000, "text-embedding-3-small", 999, None).is_none());
        assert!(plan_embedding(1000, "ghost", 1536, None).is_none());
    }

    #[test]
    fn zero_tokens_is_a_zero_cost_plan() {
        let plan = plan_embedding(0, "text-embedding-3-small", 1536, None).unwrap();
        assert_eq!(plan.chunks, 0);
        assert_eq!(plan.vectors, 0);
        assert_eq!(plan.cost, 0.0);
    }
}

Also available in 12 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →