Skip to content

Cache Breakpoint Planner — Kotlin source

Find what your prompts share — common prefix and suffix blocks — and place prompt-cache breakpoints where they pay, with an estimated cost saving. 100% client-side.

This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Cache Breakpoint Planner — find the blocks a set of prompts share and
// place cache breakpoints where they pay.
//
// Language: Kotlin (JVM 17+, zero dependencies)
// Port of src/lib/cacheBreakpointPlanner.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/cache-breakpoint-planner

import kotlin.math.max
import kotlin.math.min

/** Cached reads bill at ~0.1x — the saving on the cached share is ~90%. */
const val CACHE_READ_DISCOUNT: Double = 0.1

/** One prompt session: an id plus its ordered blocks. */
data class PromptSession(val id: String, val blocks: List<String>)

/** Place the cache breakpoint AFTER this block index (0-based); -1 = terminal. */
data class Breakpoint(
    val afterBlock: Int,
    val label: String,
    val reason: String,
    val cachedTokens: Long,
)

data class PerSessionRow(
    val id: String,
    val totalTokens: Long,
    val uniqueTokens: Long,
    val cachedRatio: Double,
)

/** The full plan: shared blocks, breakpoints, rows, savings. */
data class BreakpointPlan(
    val prefixBlocks: List<String>,
    val prefixTokens: Long,
    val suffixBlocks: List<String>,
    val suffixTokens: Long,
    val breakpoints: List<Breakpoint>,
    val perSession: List<PerSessionRow>,
    /** Estimated cost saving across the sessions vs no caching (0..1). */
    val estimatedSavings: Double,
    val warnings: List<String>,
)

/**
 * The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
 * line costs max(1, round(length / 4)) tokens; empty text is 0.
 */
private fun tok(text: String): Long {
    if (text.isEmpty()) return 0
    return text.split('\n').sumOf { line ->
        if (line.isEmpty()) 0L else max(1L, Math.round(line.length / 4.0))
    }
}

/**
 * Plan cache breakpoints for a set of prompt sessions: find the common
 * leading/trailing blocks across every session and place breakpoints where
 * the cache pays.
 */
fun planBreakpoints(sessions: List<PromptSession>): BreakpointPlan {
    val warnings = mutableListOf<String>()
    // Every Kotlin value carries a list (the type enforces TS's Array.isArray
    // filter); the filter remains as the null-guard stand-in.
    val vs = sessions.filter { it.blocks is List<*> }

    if (vs.isEmpty()) {
        return BreakpointPlan(emptyList(), 0, emptyList(), 0, emptyList(), emptyList(),
            0.0, listOf("No sessions given — paste at least two prompts to compare."))
    }
    if (vs.size == 1) {
        warnings.add("Only one session — a prefix needs at least two prompts to detect.")
    }

    // Common leading blocks by position.
    val shortest = vs.minOf { it.blocks.size }
    var prefixEnd = 0
    while (prefixEnd < shortest &&
        vs.all { it.blocks[prefixEnd] == vs[0].blocks[prefixEnd] }
    ) {
        prefixEnd++
    }

    // Common trailing blocks, matched from each session's own tail, never
    // overlapping the prefix.
    var suffixLen = 0
    while (suffixLen < shortest - prefixEnd &&
        vs.all {
            it.blocks[it.blocks.size - 1 - suffixLen] ==
                vs[0].blocks[vs[0].blocks.size - 1 - suffixLen]
        }
    ) {
        suffixLen++
    }

    val prefixBlocks = vs[0].blocks.take(prefixEnd)
    val suffixBlocks = if (suffixLen > 0) {
        vs[0].blocks.takeLast(suffixLen)
    } else {
        emptyList()
    }
    val prefixTokens = tok(prefixBlocks.joinToString("\n"))
    val suffixTokens = tok(suffixBlocks.joinToString("\n"))

    val breakpoints = mutableListOf<Breakpoint>()
    if (prefixBlocks.isNotEmpty()) {
        breakpoints.add(
            Breakpoint(
                afterBlock = prefixEnd - 1,
                label = "after the shared prefix",
                reason = "${prefixBlocks.size} block(s) identical across every session — " +
                    "cache once, hit on every request.",
                cachedTokens = prefixTokens,
            )
        )
    }
    if (suffixLen > 0) {
        breakpoints.add(
            Breakpoint(
                afterBlock = -1, // terminal: the shared tail sits at the end
                label = "shared tail",
                reason = "$suffixLen trailing block(s) also identical — extend the cache " +
                    "segment or accept the re-read.",
                cachedTokens = suffixTokens,
            )
        )
    }
    if (breakpoints.isEmpty()) {
        warnings.add(
            "No shared leading or trailing blocks — nothing to cache across these sessions."
        )
    }

    val perSession = vs.map { s ->
        val totalTokens = tok(s.blocks.joinToString("\n"))
        val unique = max(totalTokens - prefixTokens - suffixTokens, 0)
        val cachedRatio = if (totalTokens > 0) {
            min((prefixTokens + suffixTokens).toDouble() / totalTokens, 1.0)
        } else 0.0
        PerSessionRow(s.id, totalTokens, unique, cachedRatio)
    }

    val avgTotal = perSession.sumOf { it.totalTokens.toDouble() } / perSession.size
    val cachedShare = if (avgTotal > 0) {
        min((prefixTokens + suffixTokens) / avgTotal, 1.0)
    } else 0.0
    val estimatedSavings = cachedShare * (1 - CACHE_READ_DISCOUNT)

    return BreakpointPlan(
        prefixBlocks, prefixTokens, suffixBlocks, suffixTokens,
        breakpoints, perSession, estimatedSavings, warnings,
    )
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →