Skip to content

Embedding Chunk Planner — TypeScript source

Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

/**
 * Embedding Chunk Planner — pure chunking math for RAG pipelines.
 *
 * Computes how a document splits into chunks given a chunk size and overlap
 * (both in tokens), then prices the embedding call for a given model +
 * dimensions. Prices NEVER live here — they flow from the SSOT table in
 * ./ai/embeddings (mirrored in cli/aimodels; the two test suites share
 * vectors per the lock-step contract).
 */
import { getEmbeddingModel, type EmbeddingModel } from './ai/embeddings';

/** Chunking knobs, in tokens. Defaults: 512-token chunks, 64-token overlap. */
export interface ChunkOptions {
  chunkSize: number;
  overlap: number;
}

export interface ChunkPlan {
  chunks: number;
  totalTokensWithOverlap: number;
  overheadTokens: number;
}

export const DEFAULT_CHUNK_OPTIONS: ChunkOptions = { chunkSize: 512, overlap: 64 };

/**
 * Plan how `totalTokens` split into overlapping chunks.
 *
 * - chunkSize <= 0 or totalTokens <= 0 → { 0, 0, 0 } (nothing to embed).
 * - Negative overlap is treated as 0; overlap then clamps to at most
 *   floor(chunkSize / 2) so consecutive chunks always advance by at least
 *   half a chunk (an overlap that large would never advance).
 * - chunks = max(1, ceil((totalTokens - overlap) / (chunkSize - overlap)))
 *   — a tiny document still yields one chunk.
 */
export function planChunks(totalTokens: number, opts?: Partial<ChunkOptions>): ChunkPlan {
  const chunkSize = opts?.chunkSize ?? DEFAULT_CHUNK_OPTIONS.chunkSize;
  const overlapRaw = opts?.overlap ?? DEFAULT_CHUNK_OPTIONS.overlap;
  if (chunkSize <= 0 || totalTokens <= 0) {
    return { chunks: 0, totalTokensWithOverlap: 0, overheadTokens: 0 };
  }
  const overlap = Math.min(Math.max(overlapRaw, 0), Math.floor(chunkSize / 2));
  const chunks = Math.max(
    1,
    Math.ceil((totalTokens - overlap) / (chunkSize - overlap)),
  );
  const totalTokensWithOverlap = totalTokens + (chunks - 1) * overlap;
  return {
    chunks,
    totalTokensWithOverlap,
    overheadTokens: totalTokensWithOverlap - totalTokens,
  };
}

export interface EmbeddingPlan extends ChunkPlan {
  model: EmbeddingModel;
  /** One vector per chunk. */
  vectors: number;
  /** USD: totalTokensWithOverlap / 1e6 * model.inputPerM. */
  cost: number;
}

/**
 * Chunk a document AND price its embedding for `modelId` at `dims`
 * dimensions. Unknown model, or dims the model does not offer → undefined.
 */
export function planEmbedding(
  totalTokens: number,
  modelId: string,
  dims: number,
  opts?: Partial<ChunkOptions>,
): EmbeddingPlan | undefined {
  const model = getEmbeddingModel(modelId);
  if (!model || !model.dims.includes(dims)) return undefined;
  const plan = planChunks(totalTokens, opts);
  return {
    ...plan,
    model,
    vectors: plan.chunks,
    cost: (plan.totalTokensWithOverlap / 1e6) * model.inputPerM,
  };
}

Also available in 12 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →