Embedding Chunk Planner — Java source
Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.
This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Embedding Chunk Planner — pure chunking math for RAG pipelines.
//
// Language: Java (Java 17, standard library only)
// Source: CosmoDev polyglot showcase port of the Embedding Chunk Planner
// tool, ported from src/lib/embeddingPlanner.ts (the canonical
// TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; never throws (empty Optional instead of null).
// - Functionally equivalent to the TS reference: same inputs -> same outputs.
// - Self-contained: JDK only (no Maven/Gradle dependencies). The model price
// table is inlined below, mirrored from src/lib/ai/embeddings.ts — prices
// NEVER live in the planner itself.
//
// Behavior (mirrors the TS source exactly):
// - chunkSize <= 0 or totalTokens <= 0 -> ChunkPlan.ZERO (nothing to embed).
// - Negative overlap is treated as 0; overlap then clamps to at most
// chunkSize / 2 so consecutive chunks always advance.
// - chunks = max(1, ceil((totalTokens - overlap) / (chunkSize - overlap)))
// — a tiny document still yields one chunk.
import java.util.List;
import java.util.Optional;
// The launcher entry class comes first so `java EmbeddingChunkPlanner.java`
// single-file source mode finds main(); the supporting records follow.
public final class EmbeddingChunkPlanner {
private EmbeddingChunkPlanner() {
}
/** Embedding model price table — the SSOT for pricing, mirrored from
* src/lib/ai/embeddings.ts. Refresh both files together. */
static final List<EmbeddingModel> EMBEDDING_MODELS = List.of(
new EmbeddingModel("text-embedding-3-small", "OpenAI", List.of(512L, 1536L), 0.02),
new EmbeddingModel("text-embedding-3-large", "OpenAI", List.of(256L, 1024L, 3072L), 0.13),
new EmbeddingModel("embed-english-v3.0", "Cohere", List.of(512L, 1024L, 1536L), 0.1),
new EmbeddingModel("voyage-3-lite", "Voyage AI", List.of(512L, 1024L), 0.02));
/** Default knobs: 512-token chunks, 64-token overlap (TS DEFAULT_CHUNK_OPTIONS). */
static final long DEFAULT_CHUNK_SIZE = 512;
static final long DEFAULT_OVERLAP = 64;
/** Look up an embedding model by id. Returns {@code Optional.empty()} for unknown ids. */
static Optional<EmbeddingModel> getEmbeddingModel(String id) {
return EMBEDDING_MODELS.stream()
.filter(m -> m.id().equals(id))
.findFirst();
}
/**
* Plan how {@code totalTokens} split into overlapping chunks.
*
* <p>{@code chunkSize} / {@code overlap} are each independently optional (the TS
* {@code Partial<ChunkOptions>}); a {@code null} value falls back to the 512 / 64
* default, and passing only one leaves the other at its default.
*/
static ChunkPlan planChunks(long totalTokens, Long chunkSize, Long overlap) {
long cs = chunkSize == null ? DEFAULT_CHUNK_SIZE : chunkSize;
long overlapRaw = overlap == null ? DEFAULT_OVERLAP : overlap;
if (cs <= 0 || totalTokens <= 0) {
return ChunkPlan.ZERO;
}
// Math.min(Math.max(overlap, 0), cs / 2) — the TS clamp. Overlap that
// large would never advance, so consecutive chunks always gain at least
// half a chunk. (cs >= 1 here, so cs - overlap is never zero.)
long ov = Math.min(Math.max(overlapRaw, 0), cs / 2);
// Float division + ceil mirrors TS's Math.ceil exactly (a tiny document
// lands the quotient just below zero; ceil brings it to 0 and
// Math.max(1, ..) lifts it back to one chunk).
long chunks = Math.max(1L, (long) Math.ceil((double) (totalTokens - ov) / (cs - ov)));
long totalWithOverlap = totalTokens + (chunks - 1) * ov;
return new ChunkPlan(chunks, totalWithOverlap, totalWithOverlap - totalTokens);
}
/**
* Chunk a document AND price its embedding for {@code modelId} at {@code dims}
* dimensions. Unknown model, or dims the model does not offer ->
* {@code Optional.empty()}.
*/
static Optional<EmbeddingPlan> planEmbedding(long totalTokens, String modelId, long dims,
Long chunkSize, Long overlap) {
return getEmbeddingModel(modelId)
.filter(m -> m.dims().contains(dims))
.map(m -> {
ChunkPlan plan = planChunks(totalTokens, chunkSize, overlap);
return new EmbeddingPlan(
plan.chunks(),
plan.totalTokensWithOverlap(),
plan.overheadTokens(),
m,
plan.chunks(),
plan.totalTokensWithOverlap() / 1e6 * m.inputPerM());
});
}
// ---------- showcase examples (the canonical suite lives in src/lib) ----------
public static void main(String[] args) {
// 1,000 tokens: ceil((1000-64)/(512-64)) = 3 chunks, 2 seams x 64.
ChunkPlan a = planChunks(1000L, null, null);
check(a.chunks() == 3 && a.totalTokensWithOverlap() == 1128 && a.overheadTokens() == 128);
// A document that fits one chunk has no seam overhead.
ChunkPlan b = planChunks(512L, null, null);
check(b.chunks() == 1 && b.totalTokensWithOverlap() == 512 && b.overheadTokens() == 0);
// Zero/negative input or non-positive chunk size -> the zero plan.
check(planChunks(0L, null, null).equals(ChunkPlan.ZERO));
check(planChunks(-100L, null, null).equals(ChunkPlan.ZERO));
check(planChunks(1000L, 0L, null).equals(ChunkPlan.ZERO));
check(planChunks(1000L, -8L, null).equals(ChunkPlan.ZERO));
// overlap 600 > floor(512/2) = 256 -> clamped to 256.
ChunkPlan clamped = planChunks(1000L, null, 600L);
check(clamped.chunks() == 3 && clamped.totalTokensWithOverlap() == 1512
&& clamped.overheadTokens() == 512);
// Negative overlap clamps to 0: 1000 tokens -> ceil(1000/512) = 2 chunks.
ChunkPlan negative = planChunks(1000L, null, -5L);
check(negative.chunks() == 2 && negative.totalTokensWithOverlap() == 1000
&& negative.overheadTokens() == 0);
// chunkSize without overlap: ceil((1000-64)/192) = 5 chunks.
ChunkPlan partial = planChunks(1000L, 256L, null);
check(partial.chunks() == 5 && partial.totalTokensWithOverlap() == 1256
&& partial.overheadTokens() == 256);
// Shorter than the overlap still yields one chunk.
ChunkPlan tiny = planChunks(50L, null, 64L);
check(tiny.chunks() == 1 && tiny.totalTokensWithOverlap() == 50 && tiny.overheadTokens() == 0);
// chunkSize of 1 clamps overlap to 0: ceil(3/1) = 3 chunks.
ChunkPlan ones = planChunks(3L, 1L, null);
check(ones.chunks() == 3 && ones.totalTokensWithOverlap() == 3 && ones.overheadTokens() == 0);
// Pricing: 1,000 tokens on text-embedding-3-small @ 1536 dims.
EmbeddingPlan priced = planEmbedding(1000L, "text-embedding-3-small", 1536L, null, null).orElseThrow();
check(priced.chunks() == 3 && priced.totalTokensWithOverlap() == 1128 && priced.vectors() == 3);
check(Math.abs(priced.cost() - 0.00002256) < 1e-12); // 1128 / 1e6 * $0.02
// A single-chunk document on voyage-3-lite @ 512 dims.
EmbeddingPlan single = planEmbedding(512L, "voyage-3-lite", 512L, null, null).orElseThrow();
check(single.vectors() == 1);
check(Math.abs(single.cost() - 0.00001024) < 1e-12); // 512 / 1e6 * $0.02
// Unknown model or unoffered dims -> Optional.empty().
check(planEmbedding(1000L, "text-embedding-3-small", 999L, null, null).isEmpty());
check(planEmbedding(1000L, "ghost", 1536L, null, null).isEmpty());
// Zero tokens price out to a zero-cost plan.
EmbeddingPlan zero = planEmbedding(0L, "text-embedding-3-small", 1536L, null, null).orElseThrow();
check(zero.chunks() == 0 && zero.vectors() == 0 && zero.cost() == 0.0);
}
private static void check(boolean ok) {
if (!ok) {
throw new AssertionError("showcase example failed");
}
}
}
/**
* One embedding model's offered dimensions (ascending, Matryoshka shortening
* included) and pricing: USD per 1M input tokens.
*/
record EmbeddingModel(String id, String vendor, List<Long> dims, double inputPerM) {
}
/** How a document splits into overlapping chunks. */
record ChunkPlan(long chunks, long totalTokensWithOverlap, long overheadTokens) {
/** The "nothing to embed" plan the TS source returns for zero/negative
* input or a non-positive chunk size. */
static final ChunkPlan ZERO = new ChunkPlan(0, 0, 0);
}
/**
* Chunk plan plus pricing for one embedding call. The three chunk fields are
* flattened in (the TS {@code ...plan} spread) so the record reads like the TS
* {@code EmbeddingPlan extends ChunkPlan}.
*/
record EmbeddingPlan(long chunks, long totalTokensWithOverlap, long overheadTokens,
EmbeddingModel model, long vectors, double cost) {
}
Also available in 12 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →