Skip to content

Embedding Chunk Planner — Kotlin source

Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.

This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Embedding Chunk Planner — pure chunking math for RAG pipelines.
//
// Language: Kotlin (Kotlin 1.9, standard library only)
// Source:   CosmoDev polyglot showcase port of the Embedding Chunk Planner
//           tool, ported from src/lib/embeddingPlanner.ts (the canonical
//           TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; never throws (null instead of an error).
//   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//   - Self-contained: stdlib only (no Gradle dependencies). The model price
//     table is inlined below, mirrored from src/lib/ai/embeddings.ts — prices
//     NEVER live in the planner itself.
//
// Behavior (mirrors the TS source exactly):
//   - chunkSize <= 0 or totalTokens <= 0 -> ChunkPlan.ZERO (nothing to embed).
//   - Negative overlap is treated as 0; overlap then clamps to at most
//     chunkSize / 2 so consecutive chunks always advance.
//   - chunks = max(1, ceil((totalTokens - overlap) / (chunkSize - overlap)))
//     — a tiny document still yields one chunk.

import kotlin.math.ceil
import kotlin.math.max

/**
 * One embedding model's offered dimensions (ascending, Matryoshka shortening
 * included) and pricing: USD per 1M input tokens.
 */
data class EmbeddingModel(
    val id: String,
    val vendor: String,
    val dims: List<Long>,
    val inputPerM: Double,
)

/** How a document splits into overlapping chunks. */
data class ChunkPlan(
    val chunks: Long,
    val totalTokensWithOverlap: Long,
    val overheadTokens: Long,
) {
    companion object {
        /** The "nothing to embed" plan the TS source returns for zero/negative
         *  input or a non-positive chunk size. */
        val ZERO = ChunkPlan(0, 0, 0)
    }
}

/**
 * Chunking knobs, in tokens. Mirrors the TS `Partial<ChunkOptions>`: each
 * field is independently optional — null falls back to the 512 / 64 default,
 * and setting one leaves the other at its default.
 */
data class ChunkOptions(
    val chunkSize: Long? = null,
    val overlap: Long? = null,
)

/**
 * Chunk plan plus pricing for one embedding call. The three chunk fields are
 * flattened in (the TS `...plan` spread) so the class reads like the TS
 * `EmbeddingPlan extends ChunkPlan`.
 */
data class EmbeddingPlan(
    val chunks: Long,
    val totalTokensWithOverlap: Long,
    val overheadTokens: Long,
    val model: EmbeddingModel,
    val vectors: Long, // one vector per chunk
    val cost: Double,  // USD: totalTokensWithOverlap / 1e6 * model.inputPerM
)

/** Embedding model price table — the SSOT for pricing, mirrored from
 *  src/lib/ai/embeddings.ts. Refresh both files together. */
val EMBEDDING_MODELS: List<EmbeddingModel> = listOf(
    EmbeddingModel("text-embedding-3-small", "OpenAI", listOf(512L, 1536L), 0.02),
    EmbeddingModel("text-embedding-3-large", "OpenAI", listOf(256L, 1024L, 3072L), 0.13),
    EmbeddingModel("embed-english-v3.0", "Cohere", listOf(512L, 1024L, 1536L), 0.1),
    EmbeddingModel("voyage-3-lite", "Voyage AI", listOf(512L, 1024L), 0.02),
)

/** Default knobs: 512-token chunks, 64-token overlap (TS DEFAULT_CHUNK_OPTIONS). */
const val DEFAULT_CHUNK_SIZE = 512L
const val DEFAULT_OVERLAP = 64L

/** Look up an embedding model by id. Returns null for unknown ids. */
fun getEmbeddingModel(id: String): EmbeddingModel? = EMBEDDING_MODELS.firstOrNull { it.id == id }

/**
 * Plan how [totalTokens] split into overlapping chunks. [opts] may be null
 * (both defaults), mirroring the TS optional parameter; each null field falls
 * back to the 512 / 64 default independently.
 */
fun planChunks(totalTokens: Long, opts: ChunkOptions? = null): ChunkPlan {
    val chunkSize = opts?.chunkSize ?: DEFAULT_CHUNK_SIZE
    val overlapRaw = opts?.overlap ?: DEFAULT_OVERLAP

    if (chunkSize <= 0 || totalTokens <= 0) {
        return ChunkPlan.ZERO
    }

    // overlapRaw.coerceIn(0, chunkSize / 2) — the TS clamp. Overlap that large
    // would never advance, so consecutive chunks always gain at least half a
    // chunk. (chunkSize >= 1 here, so chunkSize - overlap is never zero.)
    val overlap = overlapRaw.coerceIn(0, chunkSize / 2)

    // Float division + ceil mirrors TS's Math.ceil exactly (a tiny document
    // lands the quotient just below zero; ceil brings it to 0 and max(1, ..)
    // lifts it back to one chunk).
    val chunks = max(1L, ceil((totalTokens - overlap).toDouble() / (chunkSize - overlap)).toLong())

    val totalWithOverlap = totalTokens + (chunks - 1) * overlap
    return ChunkPlan(chunks, totalWithOverlap, totalWithOverlap - totalTokens)
}

/**
 * Chunk a document AND price its embedding for [modelId] at [dims] dimensions.
 * Unknown model, or dims the model does not offer -> null.
 */
fun planEmbedding(
    totalTokens: Long,
    modelId: String,
    dims: Long,
    opts: ChunkOptions? = null,
): EmbeddingPlan? {
    val model = getEmbeddingModel(modelId) ?: return null
    if (dims !in model.dims) return null
    val plan = planChunks(totalTokens, opts)
    return EmbeddingPlan(
        chunks = plan.chunks,
        totalTokensWithOverlap = plan.totalTokensWithOverlap,
        overheadTokens = plan.overheadTokens,
        model = model,
        vectors = plan.chunks,
        cost = plan.totalTokensWithOverlap / 1e6 * model.inputPerM,
    )
}

// ---------- showcase examples (the canonical suite lives in src/lib) ----------
fun main() {
    // 1,000 tokens: ceil((1000-64)/(512-64)) = 3 chunks, 2 seams x 64.
    check(planChunks(1000) == ChunkPlan(3, 1128, 128))

    // A document that fits one chunk has no seam overhead.
    check(planChunks(512) == ChunkPlan(1, 512, 0))

    // Zero/negative input or non-positive chunk size -> the zero plan.
    check(planChunks(0) == ChunkPlan.ZERO)
    check(planChunks(-100) == ChunkPlan.ZERO)
    check(planChunks(1000, ChunkOptions(chunkSize = 0)) == ChunkPlan.ZERO)
    check(planChunks(1000, ChunkOptions(chunkSize = -8)) == ChunkPlan.ZERO)

    // overlap 600 > floor(512/2) = 256 -> clamped to 256.
    check(planChunks(1000, ChunkOptions(overlap = 600)) == ChunkPlan(3, 1512, 512))

    // Negative overlap clamps to 0: 1000 tokens -> ceil(1000/512) = 2 chunks.
    check(planChunks(1000, ChunkOptions(overlap = -5)) == ChunkPlan(2, 1000, 0))

    // chunkSize without overlap: ceil((1000-64)/192) = 5 chunks.
    check(planChunks(1000, ChunkOptions(chunkSize = 256)) == ChunkPlan(5, 1256, 256))

    // Shorter than the overlap still yields one chunk.
    check(planChunks(50, ChunkOptions(overlap = 64)) == ChunkPlan(1, 50, 0))

    // chunkSize of 1 clamps overlap to 0: ceil(3/1) = 3 chunks.
    check(planChunks(3, ChunkOptions(chunkSize = 1)) == ChunkPlan(3, 3, 0))

    // Pricing: 1,000 tokens on text-embedding-3-small @ 1536 dims.
    val priced = planEmbedding(1000, "text-embedding-3-small", 1536)!!
    check(priced.chunks == 3L && priced.totalTokensWithOverlap == 1128L && priced.vectors == 3L)
    check(kotlin.math.abs(priced.cost - 0.00002256) < 1e-12) // 1128 / 1e6 * $0.02

    // A single-chunk document on voyage-3-lite @ 512 dims.
    val single = planEmbedding(512, "voyage-3-lite", 512)!!
    check(single.vectors == 1L)
    check(kotlin.math.abs(single.cost - 0.00001024) < 1e-12) // 512 / 1e6 * $0.02

    // Unknown model or unoffered dims -> null.
    check(planEmbedding(1000, "text-embedding-3-small", 999) == null)
    check(planEmbedding(1000, "ghost", 1536) == null)

    // Zero tokens price out to a zero-cost plan.
    val zero = planEmbedding(0, "text-embedding-3-small", 1536)!!
    check(zero.chunks == 0L && zero.vectors == 0L && zero.cost == 0.0)
}

Also available in 12 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →