Embedding Chunk Planner — JavaScript source
Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.
This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
/**
* Embedding Chunk Planner - pure chunking math for RAG pipelines.
*
* Language: JavaScript (ES2022 module, runs unmodified in Node 18+ and Bun)
* Source: CosmoDev polyglot showcase port of the Embedding Chunk Planner
* tool, ported from src/lib/embeddingPlanner.ts (the canonical
* TypeScript implementation).
* Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
* License: display source - part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; never throws.
* - Functionally equivalent to the TS reference: same inputs -> same outputs.
* - Self-contained: stdlib only (no npm dependencies). The model price table
* is inlined below, mirrored from src/lib/ai/embeddings.ts - prices NEVER
* live in the planner itself.
*
* Behavior (mirrors the TS source exactly):
* - chunkSize <= 0 or totalTokens <= 0 -> { 0, 0, 0 } (nothing to embed).
* - Negative overlap is treated as 0; overlap then clamps to at most
* floor(chunkSize / 2) so consecutive chunks always advance.
* - chunks = max(1, ceil((totalTokens - overlap) / (chunkSize - overlap)))
* - a tiny document still yields one chunk.
*/
'use strict';
/**
* Chunking knobs, in tokens. Keys are both optional; defaults 512 / 64.
* @typedef {Object} ChunkOptions
* @property {number} [chunkSize]
* @property {number} [overlap]
*/
/**
* How a document splits into overlapping chunks.
* @typedef {Object} ChunkPlan
* @property {number} chunks
* @property {number} totalTokensWithOverlap
* @property {number} overheadTokens
*/
/**
* One embedding model's offered dimensions and pricing.
* @typedef {Object} EmbeddingModel
* @property {string} id
* @property {string} vendor
* @property {number[]} dims Offered dimensions, ascending.
* @property {number} inputPerM USD per 1M input tokens.
*/
/**
* Chunk plan plus pricing for one embedding call.
* @typedef {ChunkPlan & { model: EmbeddingModel, vectors: number, cost: number }} EmbeddingPlan
*/
/** @type {ChunkOptions} */
export const DEFAULT_CHUNK_OPTIONS = Object.freeze({ chunkSize: 512, overlap: 64 });
/**
* Embedding model price table - the SSOT for pricing, mirrored from
* src/lib/ai/embeddings.ts. Refresh both files together.
* @type {EmbeddingModel[]}
*/
export const EMBEDDING_MODELS = Object.freeze([
{ id: 'text-embedding-3-small', vendor: 'OpenAI', dims: [512, 1536], inputPerM: 0.02 },
{ id: 'text-embedding-3-large', vendor: 'OpenAI', dims: [256, 1024, 3072], inputPerM: 0.13 },
{ id: 'embed-english-v3.0', vendor: 'Cohere', dims: [512, 1024, 1536], inputPerM: 0.1 },
{ id: 'voyage-3-lite', vendor: 'Voyage AI', dims: [512, 1024], inputPerM: 0.02 },
]);
/**
* Look up an embedding model by id. Returns undefined for unknown ids.
* @param {string} id
* @returns {EmbeddingModel | undefined}
*/
export function getEmbeddingModel(id) {
return EMBEDDING_MODELS.find((m) => m.id === id);
}
/**
* Plan how `totalTokens` split into overlapping chunks.
* @param {number} totalTokens
* @param {ChunkOptions} [options={}]
* @returns {ChunkPlan}
*/
export function planChunks(totalTokens, options = {}) {
const chunkSize = options.chunkSize ?? DEFAULT_CHUNK_OPTIONS.chunkSize;
const overlapRaw = options.overlap ?? DEFAULT_CHUNK_OPTIONS.overlap;
if (chunkSize <= 0 || totalTokens <= 0) {
return { chunks: 0, totalTokensWithOverlap: 0, overheadTokens: 0 };
}
const overlap = Math.min(Math.max(overlapRaw, 0), Math.floor(chunkSize / 2));
const chunks = Math.max(
1,
Math.ceil((totalTokens - overlap) / (chunkSize - overlap)),
);
const totalTokensWithOverlap = totalTokens + (chunks - 1) * overlap;
return {
chunks,
totalTokensWithOverlap,
overheadTokens: totalTokensWithOverlap - totalTokens,
};
}
/**
* Chunk a document AND price its embedding for `modelId` at `dims`
* dimensions. Unknown model, or dims the model does not offer -> undefined.
* @param {number} totalTokens
* @param {string} modelId
* @param {number} dims
* @param {ChunkOptions} [options={}]
* @returns {EmbeddingPlan | undefined}
*/
export function planEmbedding(totalTokens, modelId, dims, options = {}) {
const model = getEmbeddingModel(modelId);
if (!model || !model.dims.includes(dims)) return undefined;
const plan = planChunks(totalTokens, options);
return {
...plan,
model,
vectors: plan.chunks,
cost: (plan.totalTokensWithOverlap / 1e6) * model.inputPerM,
};
}
Also available in 12 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →