Embedding Chunk Planner — Rust source
Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! Embedding Chunk Planner — pure chunking math for RAG pipelines.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source: CosmoDev polyglot showcase port of the Embedding Chunk Planner
//! tool, ported from src/lib/embeddingPlanner.ts (the canonical
//! TypeScript implementation).
//! Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; never panics.
//! - Functionally equivalent to the TS reference: same inputs -> same outputs.
//! - Self-contained: std only (no crates.io dependencies). The model price
//! table is inlined below, mirrored from src/lib/ai/embeddings.ts — prices
//! NEVER live in the planner itself.
//!
//! Behavior (mirrors the TS source exactly):
//! - chunk_size <= 0 or total_tokens <= 0 -> ChunkPlan::ZERO (nothing to
//! embed).
//! - Negative overlap is treated as 0; overlap then clamps to at most
//! chunk_size / 2 so consecutive chunks always advance.
//! - chunks = max(1, ceil((total_tokens - overlap) / (chunk_size - overlap)))
//! — a tiny document still yields one chunk.
/// One embedding model's offered dimensions (ascending, Matryoshka shortening
/// included) and pricing: USD per 1M input tokens.
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct EmbeddingModel {
pub id: &'static str,
pub vendor: &'static str,
pub dims: &'static [i64],
pub input_per_m: f64,
}
/// Embedding model price table — the SSOT for pricing, mirrored from
/// src/lib/ai/embeddings.ts. Refresh both files together.
pub const EMBEDDING_MODELS: &[EmbeddingModel] = &[
EmbeddingModel { id: "text-embedding-3-small", vendor: "OpenAI", dims: &[512, 1536], input_per_m: 0.02 },
EmbeddingModel { id: "text-embedding-3-large", vendor: "OpenAI", dims: &[256, 1024, 3072], input_per_m: 0.13 },
EmbeddingModel { id: "embed-english-v3.0", vendor: "Cohere", dims: &[512, 1024, 1536], input_per_m: 0.1 },
EmbeddingModel { id: "voyage-3-lite", vendor: "Voyage AI", dims: &[512, 1024], input_per_m: 0.02 },
];
/// Default knobs: 512-token chunks, 64-token overlap (TS DEFAULT_CHUNK_OPTIONS).
pub const DEFAULT_CHUNK_SIZE: i64 = 512;
pub const DEFAULT_OVERLAP: i64 = 64;
/// Look up an embedding model by id. Returns `None` for unknown ids.
pub fn get_embedding_model(id: &str) -> Option<&'static EmbeddingModel> {
EMBEDDING_MODELS.iter().find(|m| m.id == id)
}
/// Chunking knobs, in tokens. Mirrors the TS `Partial<ChunkOptions>`: each
/// field is independently optional — `None` falls back to the 512 / 64
/// default, and setting one leaves the other at its default. Construct with
/// `..Default::default()`.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct ChunkOptions {
pub chunk_size: Option<i64>,
pub overlap: Option<i64>,
}
/// How a document splits into overlapping chunks.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct ChunkPlan {
pub chunks: i64,
pub total_tokens_with_overlap: i64,
pub overhead_tokens: i64,
}
impl ChunkPlan {
/// The "nothing to embed" plan the TS source returns for zero/negative
/// input or a non-positive chunk size.
pub const ZERO: ChunkPlan = ChunkPlan {
chunks: 0,
total_tokens_with_overlap: 0,
overhead_tokens: 0,
};
}
/// Plan how `total_tokens` split into overlapping chunks.
///
/// `opts` is optional wholesale (`None` = both defaults), mirroring the TS
/// optional parameter.
pub fn plan_chunks(total_tokens: i64, opts: Option<&ChunkOptions>) -> ChunkPlan {
let chunk_size = opts.and_then(|o| o.chunk_size).unwrap_or(DEFAULT_CHUNK_SIZE);
let overlap_raw = opts.and_then(|o| o.overlap).unwrap_or(DEFAULT_OVERLAP);
if chunk_size <= 0 || total_tokens <= 0 {
return ChunkPlan::ZERO;
}
// min(max(overlap, 0), chunk_size / 2) — the TS clamp. Overlap that large
// would never advance, so consecutive chunks always gain at least half a
// chunk. (chunk_size >= 1 here, so chunk_size - overlap is never zero.)
let overlap = overlap_raw.clamp(0, chunk_size / 2);
// Float division + ceil mirrors TS's Math.ceil exactly (a tiny document
// lands the quotient just below zero; ceil brings it to 0 and max(1, ..)
// lifts it back to one chunk).
let chunks = (((total_tokens - overlap) as f64 / (chunk_size - overlap) as f64).ceil() as i64).max(1);
let total_tokens_with_overlap = total_tokens + (chunks - 1) * overlap;
ChunkPlan {
chunks,
total_tokens_with_overlap,
overhead_tokens: total_tokens_with_overlap - total_tokens,
}
}
/// Chunk plan plus pricing for one embedding call. The three chunk fields are
/// flattened in (the TS `...plan` spread) so the struct reads like the TS
/// `EmbeddingPlan extends ChunkPlan`.
#[derive(Debug, Clone, PartialEq)]
pub struct EmbeddingPlan {
pub chunks: i64,
pub total_tokens_with_overlap: i64,
pub overhead_tokens: i64,
pub model: &'static EmbeddingModel,
/// One vector per chunk.
pub vectors: i64,
/// USD: total_tokens_with_overlap / 1e6 * model.input_per_m.
pub cost: f64,
}
/// Chunk a document AND price its embedding for `model_id` at `dims`
/// dimensions. Unknown model, or dims the model does not offer -> `None`.
pub fn plan_embedding(
total_tokens: i64,
model_id: &str,
dims: i64,
opts: Option<&ChunkOptions>,
) -> Option<EmbeddingPlan> {
let model = get_embedding_model(model_id)?;
if !model.dims.contains(&dims) {
return None;
}
let plan = plan_chunks(total_tokens, opts);
Some(EmbeddingPlan {
chunks: plan.chunks,
total_tokens_with_overlap: plan.total_tokens_with_overlap,
overhead_tokens: plan.overhead_tokens,
model,
vectors: plan.chunks,
cost: (plan.total_tokens_with_overlap as f64 / 1e6) * model.input_per_m,
})
}
// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn uses_512_64_defaults() {
// 1,000 tokens: ceil((1000-64)/(512-64)) = 3 chunks, 2 seams x 64.
assert_eq!(plan_chunks(1000, None), ChunkPlan { chunks: 3, total_tokens_with_overlap: 1128, overhead_tokens: 128 });
}
#[test]
fn single_chunk_when_the_document_fits() {
assert_eq!(plan_chunks(512, None), ChunkPlan { chunks: 1, total_tokens_with_overlap: 512, overhead_tokens: 0 });
}
#[test]
fn zeros_for_degenerate_input() {
assert_eq!(plan_chunks(0, None), ChunkPlan::ZERO);
assert_eq!(plan_chunks(-100, None), ChunkPlan::ZERO);
assert_eq!(plan_chunks(1000, Some(&ChunkOptions { chunk_size: Some(0), overlap: None })), ChunkPlan::ZERO);
assert_eq!(plan_chunks(1000, Some(&ChunkOptions { chunk_size: Some(-8), overlap: None })), ChunkPlan::ZERO);
}
#[test]
fn clamps_overlap_to_half_the_chunk_size() {
// overlap 600 > floor(512/2) = 256 -> clamped to 256.
let opts = ChunkOptions { overlap: Some(600), ..Default::default() };
assert_eq!(plan_chunks(1000, Some(&opts)), ChunkPlan { chunks: 3, total_tokens_with_overlap: 1512, overhead_tokens: 512 });
}
#[test]
fn negative_overlap_is_zero_and_partial_options_merge() {
let neg = ChunkOptions { overlap: Some(-5), ..Default::default() };
assert_eq!(plan_chunks(1000, Some(&neg)), ChunkPlan { chunks: 2, total_tokens_with_overlap: 1000, overhead_tokens: 0 });
// chunkSize without overlap: ceil((1000-64)/192) = 5 chunks.
let partial = ChunkOptions { chunk_size: Some(256), ..Default::default() };
assert_eq!(plan_chunks(1000, Some(&partial)), ChunkPlan { chunks: 5, total_tokens_with_overlap: 1256, overhead_tokens: 256 });
}
#[test]
fn one_chunk_when_shorter_than_the_overlap() {
// ceil((50-64)/448) = 0 -> clamped up to 1 chunk.
let opts = ChunkOptions { overlap: Some(64), ..Default::default() };
assert_eq!(plan_chunks(50, Some(&opts)), ChunkPlan { chunks: 1, total_tokens_with_overlap: 50, overhead_tokens: 0 });
}
#[test]
fn chunk_size_of_one_clamps_overlap_to_zero() {
let opts = ChunkOptions { chunk_size: Some(1), ..Default::default() };
assert_eq!(plan_chunks(3, Some(&opts)), ChunkPlan { chunks: 3, total_tokens_with_overlap: 3, overhead_tokens: 0 });
}
#[test]
fn prices_a_known_plan() {
let plan = plan_embedding(1000, "text-embedding-3-small", 1536, None).unwrap();
assert_eq!(plan.chunks, 3);
assert_eq!(plan.total_tokens_with_overlap, 1128);
assert_eq!(plan.vectors, 3);
// 1128 / 1e6 * $0.02
assert!((plan.cost - 0.00002256).abs() < 1e-12);
assert_eq!(plan.model.id, "text-embedding-3-small");
}
#[test]
fn prices_a_single_chunk_document() {
let plan = plan_embedding(512, "voyage-3-lite", 512, None).unwrap();
assert_eq!(plan.vectors, 1);
// 512 / 1e6 * $0.02
assert!((plan.cost - 0.00001024).abs() < 1e-12);
}
#[test]
fn unknown_model_or_dims_is_none() {
assert!(plan_embedding(1000, "text-embedding-3-small", 999, None).is_none());
assert!(plan_embedding(1000, "ghost", 1536, None).is_none());
}
#[test]
fn zero_tokens_is_a_zero_cost_plan() {
let plan = plan_embedding(0, "text-embedding-3-small", 1536, None).unwrap();
assert_eq!(plan.chunks, 0);
assert_eq!(plan.vectors, 0);
assert_eq!(plan.cost, 0.0);
}
}
Also available in 12 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →