Skip to content

Embedding Chunk Planner — Java source

Plan document chunking for RAG — chunk counts with overlap math, vector counts, and embedding costs per model.

This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Embedding Chunk Planner — pure chunking math for RAG pipelines.
//
// Language: Java (Java 17, standard library only)
// Source:   CosmoDev polyglot showcase port of the Embedding Chunk Planner
//           tool, ported from src/lib/embeddingPlanner.ts (the canonical
//           TypeScript implementation).
// Tool page: https://dev.cosmolabs.org/tools/embedding-chunk-planner
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; never throws (empty Optional instead of null).
//   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//   - Self-contained: JDK only (no Maven/Gradle dependencies). The model price
//     table is inlined below, mirrored from src/lib/ai/embeddings.ts — prices
//     NEVER live in the planner itself.
//
// Behavior (mirrors the TS source exactly):
//   - chunkSize <= 0 or totalTokens <= 0 -> ChunkPlan.ZERO (nothing to embed).
//   - Negative overlap is treated as 0; overlap then clamps to at most
//     chunkSize / 2 so consecutive chunks always advance.
//   - chunks = max(1, ceil((totalTokens - overlap) / (chunkSize - overlap)))
//     — a tiny document still yields one chunk.

import java.util.List;
import java.util.Optional;

// The launcher entry class comes first so `java EmbeddingChunkPlanner.java`
// single-file source mode finds main(); the supporting records follow.
public final class EmbeddingChunkPlanner {

    private EmbeddingChunkPlanner() {
    }

    /** Embedding model price table — the SSOT for pricing, mirrored from
     *  src/lib/ai/embeddings.ts. Refresh both files together. */
    static final List<EmbeddingModel> EMBEDDING_MODELS = List.of(
            new EmbeddingModel("text-embedding-3-small", "OpenAI", List.of(512L, 1536L), 0.02),
            new EmbeddingModel("text-embedding-3-large", "OpenAI", List.of(256L, 1024L, 3072L), 0.13),
            new EmbeddingModel("embed-english-v3.0", "Cohere", List.of(512L, 1024L, 1536L), 0.1),
            new EmbeddingModel("voyage-3-lite", "Voyage AI", List.of(512L, 1024L), 0.02));

    /** Default knobs: 512-token chunks, 64-token overlap (TS DEFAULT_CHUNK_OPTIONS). */
    static final long DEFAULT_CHUNK_SIZE = 512;
    static final long DEFAULT_OVERLAP = 64;

    /** Look up an embedding model by id. Returns {@code Optional.empty()} for unknown ids. */
    static Optional<EmbeddingModel> getEmbeddingModel(String id) {
        return EMBEDDING_MODELS.stream()
                .filter(m -> m.id().equals(id))
                .findFirst();
    }

    /**
     * Plan how {@code totalTokens} split into overlapping chunks.
     *
     * <p>{@code chunkSize} / {@code overlap} are each independently optional (the TS
     * {@code Partial<ChunkOptions>}); a {@code null} value falls back to the 512 / 64
     * default, and passing only one leaves the other at its default.
     */
    static ChunkPlan planChunks(long totalTokens, Long chunkSize, Long overlap) {
        long cs = chunkSize == null ? DEFAULT_CHUNK_SIZE : chunkSize;
        long overlapRaw = overlap == null ? DEFAULT_OVERLAP : overlap;

        if (cs <= 0 || totalTokens <= 0) {
            return ChunkPlan.ZERO;
        }

        // Math.min(Math.max(overlap, 0), cs / 2) — the TS clamp. Overlap that
        // large would never advance, so consecutive chunks always gain at least
        // half a chunk. (cs >= 1 here, so cs - overlap is never zero.)
        long ov = Math.min(Math.max(overlapRaw, 0), cs / 2);

        // Float division + ceil mirrors TS's Math.ceil exactly (a tiny document
        // lands the quotient just below zero; ceil brings it to 0 and
        // Math.max(1, ..) lifts it back to one chunk).
        long chunks = Math.max(1L, (long) Math.ceil((double) (totalTokens - ov) / (cs - ov)));

        long totalWithOverlap = totalTokens + (chunks - 1) * ov;
        return new ChunkPlan(chunks, totalWithOverlap, totalWithOverlap - totalTokens);
    }

    /**
     * Chunk a document AND price its embedding for {@code modelId} at {@code dims}
     * dimensions. Unknown model, or dims the model does not offer ->
     * {@code Optional.empty()}.
     */
    static Optional<EmbeddingPlan> planEmbedding(long totalTokens, String modelId, long dims,
                                                 Long chunkSize, Long overlap) {
        return getEmbeddingModel(modelId)
                .filter(m -> m.dims().contains(dims))
                .map(m -> {
                    ChunkPlan plan = planChunks(totalTokens, chunkSize, overlap);
                    return new EmbeddingPlan(
                            plan.chunks(),
                            plan.totalTokensWithOverlap(),
                            plan.overheadTokens(),
                            m,
                            plan.chunks(),
                            plan.totalTokensWithOverlap() / 1e6 * m.inputPerM());
                });
    }

    // ---------- showcase examples (the canonical suite lives in src/lib) ----------
    public static void main(String[] args) {
        // 1,000 tokens: ceil((1000-64)/(512-64)) = 3 chunks, 2 seams x 64.
        ChunkPlan a = planChunks(1000L, null, null);
        check(a.chunks() == 3 && a.totalTokensWithOverlap() == 1128 && a.overheadTokens() == 128);

        // A document that fits one chunk has no seam overhead.
        ChunkPlan b = planChunks(512L, null, null);
        check(b.chunks() == 1 && b.totalTokensWithOverlap() == 512 && b.overheadTokens() == 0);

        // Zero/negative input or non-positive chunk size -> the zero plan.
        check(planChunks(0L, null, null).equals(ChunkPlan.ZERO));
        check(planChunks(-100L, null, null).equals(ChunkPlan.ZERO));
        check(planChunks(1000L, 0L, null).equals(ChunkPlan.ZERO));
        check(planChunks(1000L, -8L, null).equals(ChunkPlan.ZERO));

        // overlap 600 > floor(512/2) = 256 -> clamped to 256.
        ChunkPlan clamped = planChunks(1000L, null, 600L);
        check(clamped.chunks() == 3 && clamped.totalTokensWithOverlap() == 1512
                && clamped.overheadTokens() == 512);

        // Negative overlap clamps to 0: 1000 tokens -> ceil(1000/512) = 2 chunks.
        ChunkPlan negative = planChunks(1000L, null, -5L);
        check(negative.chunks() == 2 && negative.totalTokensWithOverlap() == 1000
                && negative.overheadTokens() == 0);

        // chunkSize without overlap: ceil((1000-64)/192) = 5 chunks.
        ChunkPlan partial = planChunks(1000L, 256L, null);
        check(partial.chunks() == 5 && partial.totalTokensWithOverlap() == 1256
                && partial.overheadTokens() == 256);

        // Shorter than the overlap still yields one chunk.
        ChunkPlan tiny = planChunks(50L, null, 64L);
        check(tiny.chunks() == 1 && tiny.totalTokensWithOverlap() == 50 && tiny.overheadTokens() == 0);

        // chunkSize of 1 clamps overlap to 0: ceil(3/1) = 3 chunks.
        ChunkPlan ones = planChunks(3L, 1L, null);
        check(ones.chunks() == 3 && ones.totalTokensWithOverlap() == 3 && ones.overheadTokens() == 0);

        // Pricing: 1,000 tokens on text-embedding-3-small @ 1536 dims.
        EmbeddingPlan priced = planEmbedding(1000L, "text-embedding-3-small", 1536L, null, null).orElseThrow();
        check(priced.chunks() == 3 && priced.totalTokensWithOverlap() == 1128 && priced.vectors() == 3);
        check(Math.abs(priced.cost() - 0.00002256) < 1e-12); // 1128 / 1e6 * $0.02

        // A single-chunk document on voyage-3-lite @ 512 dims.
        EmbeddingPlan single = planEmbedding(512L, "voyage-3-lite", 512L, null, null).orElseThrow();
        check(single.vectors() == 1);
        check(Math.abs(single.cost() - 0.00001024) < 1e-12); // 512 / 1e6 * $0.02

        // Unknown model or unoffered dims -> Optional.empty().
        check(planEmbedding(1000L, "text-embedding-3-small", 999L, null, null).isEmpty());
        check(planEmbedding(1000L, "ghost", 1536L, null, null).isEmpty());

        // Zero tokens price out to a zero-cost plan.
        EmbeddingPlan zero = planEmbedding(0L, "text-embedding-3-small", 1536L, null, null).orElseThrow();
        check(zero.chunks() == 0 && zero.vectors() == 0 && zero.cost() == 0.0);
    }

    private static void check(boolean ok) {
        if (!ok) {
            throw new AssertionError("showcase example failed");
        }
    }
}

/**
 * One embedding model's offered dimensions (ascending, Matryoshka shortening
 * included) and pricing: USD per 1M input tokens.
 */
record EmbeddingModel(String id, String vendor, List<Long> dims, double inputPerM) {
}

/** How a document splits into overlapping chunks. */
record ChunkPlan(long chunks, long totalTokensWithOverlap, long overheadTokens) {
    /** The "nothing to embed" plan the TS source returns for zero/negative
     *  input or a non-positive chunk size. */
    static final ChunkPlan ZERO = new ChunkPlan(0, 0, 0);
}

/**
 * Chunk plan plus pricing for one embedding call. The three chunk fields are
 * flattened in (the TS {@code ...plan} spread) so the record reads like the TS
 * {@code EmbeddingPlan extends ChunkPlan}.
 */
record EmbeddingPlan(long chunks, long totalTokensWithOverlap, long overheadTokens,
                     EmbeddingModel model, long vectors, double cost) {
}

Also available in 12 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →