Skip to content

Context Window Planner — Kotlin source

Paste your system prompt, docs, and history — see how they fill any model's context window, with overflow warnings and output headroom.

This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Context Window Planner — plan labeled prompt sections against a model's
// context window.
//
// Language: Kotlin (Kotlin 1.9, standard library only)
// Source:   CosmoDev polyglot showcase port of the Context Window Planner
//           tool, ported from src/lib/contextPlanner.ts (the canonical
//           TypeScript implementation).
// Live at:  https://dev.cosmolabs.org/tools/context-window-planner
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; never throws (public API returns plain values).
//   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//   - Self-contained: stdlib only (no kotlinx.serialization — this port ships
//     the same small strict recursive-descent validator the other polyglot
//     siblings use, so the JSON grammar matches JSON.parse exactly rather
//     than approximately).
//
// Port notes: the TS lib delegates to two siblings — `estimateTokens` from
// src/lib/tokenEstimator.ts and `fitsWindow` from src/lib/ai/models.ts (which
// defaults to the bundled pricing snapshot, src/data/ai-models.json). A
// dependency-free port cannot load that file, so the estimator is inlined
// below in the exact form the planner uses it (`estimateTokens(text).tokens`,
// auto content type — the full heuristic lives in the token-estimator port),
// window math is inlined from `fitsWindow` and `models` is an explicit
// parameter, never re-derived.
//
// Faithfulness notes (the places Kotlin's stdlib silently differs from JS):
//   - Length: TS's `String.length` counts UTF-16 code units — and so does
//     Kotlin's `String.length`, so no helper is needed; astral-plane
//     characters (emoji, rare CJK ext-B ideographs) already count as 2 on
//     both sides.
//   - Rounding: `kotlin.math.round` rounds halfway cases away from zero —
//     equal to JS `Math.round` over the non-negative inputs used here;
//     `jsRound()` states the contract explicitly.

import kotlin.math.floor
import kotlin.math.max
import kotlin.math.round

/** One labeled block of the prompt (system / docs / history / ...).
 *  Mirrors the TS `PlanSection` interface. */
data class PlanSection(val label: String, val text: String)

/** Convenience constructor mirroring the TS object literal `{ label, text }`. */
fun sec(label: String, text: String) = PlanSection(label, text)

/** The subset of the TS `AiModel` record the planner reads. Production code
 *  passes the full snapshot entry; only these fields influence the plan. */
data class Model(val id: String, val contextWindow: Long, val maxOutput: Long)

/** Sample table for standalone use (mirrors the shared test fixtures).
 *  Production code passes the model snapshot instead. */
val SAMPLE_MODELS: List<Model> = listOf(
    Model("alpha-mini", 200_000L, 10_000L),
    Model("beta-pro", 1_000_000L, 10_000L),
    Model("gamma-open", 100_000L, 10_000L),
)

/** Result of [planWindow]. Field-for-field twin of the TS `WindowPlan`
 *  interface. */
data class WindowPlan(
    val id: String,               // the model id planned against
    val inputTokens: Long,        // sum of per-section token estimates
    val contextWindow: Long,      // the model's context window
    val free: Long,               // window - input; negative on overflow
    val fits: Boolean,            // raw fit: free >= 0
    val outputReserveOk: Boolean, // room for the output reserve
    val maxOutput: Long,          // the model's output cap (informational)
)

/** Content classification of a single line. The planner only needs each
 *  type's chars-per-token rate (mirrors `CHARS_PER_TOKEN` in
 *  src/lib/tokenEstimator.ts: prose 4, code 3.5, json 3, cjk 1.5). */
private enum class ContentType { PROSE, CODE, JSON, CJK }

private fun ContentType.charsPerToken(): Double = when (this) {
    ContentType.PROSE -> 4.0
    ContentType.CODE -> 3.5
    ContentType.JSON -> 3.0
    ContentType.CJK -> 1.5
}

/** Reports whether `s` contains a CJK ideograph (U+4E00–U+9FFF), kana
 *  (U+3040–U+30FF), or a Hangul syllable (U+AC00–U+D7AF). Mirrors `CJK_RE`
 *  in the TS lib. */
private fun hasCjk(s: String): Boolean = s.any { c ->
    (c in '一'..'鿿') || (c in '぀'..'ヿ') || (c in '가'..'힯')
}

/** Reports whether `c` is one of the code-flavored symbols counted by
 *  `CODE_SYMBOL_RE` (`{}();=<>[]#`). */
private fun isCodeSymbol(c: Char): Boolean =
    c in "{}();=<>[]#"

/** JS `Math.round`: halfway cases round up (`floor(x + 0.5)`). */
private fun jsRound(x: Double): Long = floor(x + 0.5).toLong()

/** Splits `text` on LF or CRLF, mirroring `text.split(/\r?\n/)`: strip the
 *  optional CR that belongs to the newline, then split on LF. A lone CR is
 *  NOT a line break. */
private fun splitLines(text: String): List<String> =
    text.split('\n').map { line -> if (line.endsWith('\r')) line.dropLast(1) else line }

/** Classifies a single line by its shape. Order: json, cjk, code, prose.
 *  Inlined from `detectLineType()` in src/lib/tokenEstimator.ts. */
private fun detectLineType(line: String): ContentType {
    val trimmed = line.trim()
    // JSON-ish: opens like a JSON fragment AND carries a separator.
    val startsJsonish = trimmed.startsWith("{") || trimmed.startsWith("}") ||
        trimmed.startsWith("[") || trimmed.startsWith("\"")
    if (startsJsonish && (":" in line || "," in line)) return ContentType.JSON
    // CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
    if (hasCjk(line)) return ContentType.CJK
    // Code: symbol-dense, or a statement terminator / block opener at EOL.
    val length = line.length // UTF-16 code units, same unit as TS
    val symbols = line.count { isCodeSymbol(it) }
    val density = if (length > 0) symbols.toDouble() / length else 0.0
    if (density > 0.08 || trimmed.endsWith(";") || trimmed.endsWith("{") ||
        trimmed.endsWith("}"))
        return ContentType.CODE
    return ContentType.PROSE
}

/** A strict JSON syntax validator — the exact grammar `JSON.parse` accepts,
 *  walked with a cursor. Same shape as the Rust/C/C++/C#/Java siblings'
 *  validators, so whole-text JSON detection behaves identically across
 *  every port. */
private class JsonParser(private val s: String) {
    private var pos = 0

    /** value := ws* (object | array | string | number | 'true' | 'false' | 'null') ws* */
    fun value(): Boolean {
        skipWs()
        if (pos >= s.length) return false
        return when (s[pos]) {
            '{' -> objekt()
            '[' -> array()
            '"' -> string()
            in '0'..'9', '-' -> number()
            't' -> literal("true")
            'f' -> literal("false")
            'n' -> literal("null")
            else -> false
        }
    }

    fun atEnd(): Boolean {
        skipWs()
        return pos == s.length // reject trailing garbage
    }

    private fun skipWs() {
        while (pos < s.length && s[pos] in " \t\n\r") pos++
    }

    private fun peek(): Char = if (pos < s.length) s[pos] else ''

    private fun eat(b: Char): Boolean {
        if (pos < s.length && s[pos] == b) { pos++; return true }
        return false
    }

    private fun literal(lit: String): Boolean {
        if (s.regionMatches(pos, lit, 0, lit.length)) {
            pos += lit.length
            return true
        }
        return false
    }

    /** object := '{' ws* (string ws* ':' value (ws* ',' ...)*)? ws* '}' */
    private fun objekt(): Boolean {
        if (!eat('{')) return false
        skipWs()
        if (eat('}')) return true
        while (true) {
            if (!string()) return false
            skipWs()
            if (!eat(':')) return false
            if (!value()) return false
            skipWs()
            if (eat(',')) skipWs()
            else return eat('}')
        }
    }

    /** array := '[' ws* (value (ws* ',' ws* value)*)? ws* ']' */
    private fun array(): Boolean {
        if (!eat('[')) return false
        skipWs()
        if (eat(']')) return true
        while (true) {
            if (!value()) return false
            skipWs()
            if (eat(',')) skipWs()
            else return eat(']')
        }
    }

    /** string := '"' (escape | any char >= 0x20)* '"'
     *  escape := '\' ('"' | '/' | '\' | 'b' | 'f' | 'n' | 'r' | 't' | 'u' hex4) */
    private fun string(): Boolean {
        if (!eat('"')) return false
        while (pos < s.length) {
            val c = s[pos]
            when {
                c == '"' -> { pos++; return true }
                c == '\\' -> {
                    pos++
                    if (pos >= s.length) return false
                    when (val esc = s[pos++]) {
                        '"', '/', '\\', 'b', 'f', 'n', 'r', 't' -> {}
                        'u' -> {
                            repeat(4) {
                                val h = peek()
                                val hex = h in '0'..'9' || h in 'a'..'f' || h in 'A'..'F'
                                if (!hex) return false
                                pos++
                            }
                        }
                        else -> return false
                    }
                }
                c < ' ' -> return false // raw control characters not allowed
                else -> pos++
            }
        }
        return false // unterminated string
    }

    /** number := '-'? int frac? exp? — no leading zeros, like JSON.parse. */
    private fun number(): Boolean {
        eat('-')
        when (peek()) {
            '0' -> pos++
            in '1'..'9' -> while (peek() in '0'..'9') pos++
            else -> return false
        }
        if (peek() == '.') {
            pos++
            var digits = 0
            while (peek() in '0'..'9') { pos++; digits++ }
            if (digits == 0) return false
        }
        if (peek() == 'e' || peek() == 'E') {
            pos++
            if (peek() == '+' || peek() == '-') pos++
            var digits = 0
            while (peek() in '0'..'9') { pos++; digits++ }
            if (digits == 0) return false
        }
        return true
    }
}

/** Whole-text JSON gate: a document that parses as JSON is json all the way
 *  down. Mirrors `isValidJson()` (`JSON.parse` in a try/catch);
 *  empty/whitespace text is not. */
internal fun isValidJson(text: String): Boolean {
    if (text.isBlank()) return false
    val p = JsonParser(text)
    return p.value() && p.atEnd()
}

/** Token count of `text` under auto content detection — exactly the slice of
 *  `estimateTokens()` the planner consumes (`.tokens`): per non-empty line,
 *  `max(1, round(length / charsPerToken))`. Framing tokens are the caller's
 *  job. */
internal fun estimateTokens(text: String): Long {
    // AUTO + whole-text JSON: json's 3 chars/token rate applies to every
    // line, not just the reported content type.
    val wholeTextJson = isValidJson(text)
    var tokens = 0L
    for (line in splitLines(text)) {
        if (line.isBlank()) continue
        val type = if (wholeTextJson) ContentType.JSON else detectLineType(line)
        tokens += max(1L, jsRound(line.length / type.charsPerToken()))
    }
    return tokens
}

/** Sum of per-section token estimates (framing tokens are the caller's job).
 *  Mirrors `inputTokenTotal()` in the TS lib. */
fun inputTokenTotal(sections: List<PlanSection>): Long =
    sections.sumOf { estimateTokens(it.text) }

/** Plan one section set against one model's context window. Returns `null`
 *  for an unknown model id (window math is `fitsWindow`'s, never re-derived).
 *  Mirrors `planWindow()` in the TS lib. */
fun planWindow(
    sections: List<PlanSection>,
    modelId: String,
    outputReserve: Long = 0L,
    models: List<Model> = emptyList(),
): WindowPlan? {
    val inputTokens = inputTokenTotal(sections)
    // Fit check inlined from fitsWindow() in src/lib/ai/models.ts.
    val m = models.firstOrNull { it.id == modelId } ?: return null
    val free = m.contextWindow - inputTokens
    return WindowPlan(
        id = modelId,
        inputTokens = inputTokens,
        contextWindow = m.contextWindow,
        free = free,
        fits = free >= 0,
        outputReserveOk = free >= outputReserve,
        maxOutput = m.maxOutput,
    )
}

/** Plan against several models; unknown ids are dropped from the result.
 *  Mirrors `planAll()` in the TS lib. */
fun planAll(
    sections: List<PlanSection>,
    modelIds: List<String>,
    outputReserve: Long = 0L,
    models: List<Model> = emptyList(),
): List<WindowPlan> =
    modelIds.mapNotNull { planWindow(sections, it, outputReserve, models) }

// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------

/** A 1600-char single line of 'a' is pure prose: 1600 / 4 = 400 tokens. */
private fun twoSections(): List<PlanSection> {
    val lineA = "a".repeat(1600)
    return listOf(sec("sys", lineA), sec("docs", lineA))
}

fun main() {
    val two = twoSections()

    // input totals
    check(inputTokenTotal(two) == 800L)
    check(inputTokenTotal(emptyList()) == 0L)
    check(inputTokenTotal(listOf(sec("sys", ""))) == 0L)

    // plans two 400-token sections against beta-pro
    val p = planWindow(two, "beta-pro", 0L, SAMPLE_MODELS)!!
    check(p.id == "beta-pro")
    check(p.inputTokens == 800L)
    check(p.contextWindow == 1_000_000L)
    check(p.free == 999_200L)
    check(p.fits)
    check(p.outputReserveOk)
    check(p.maxOutput == 10_000L)

    // reserve larger than free leaves raw fit true
    val q = planWindow(two, "beta-pro", 1_000_000L, SAMPLE_MODELS)!!
    check(q.fits)
    check(!q.outputReserveOk)

    // reserve exactly equal to free is ok
    val r = planWindow(two, "beta-pro", 999_200L, SAMPLE_MODELS)!!
    check(r.outputReserveOk)

    // smaller window leaves 199,200 free
    val s = planWindow(two, "alpha-mini", 0L, SAMPLE_MODELS)!!
    check(s.contextWindow == 200_000L)
    check(s.free == 199_200L)
    check(s.fits)

    // unknown model id returns null
    check(planWindow(two, "ghost", 0L, SAMPLE_MODELS) == null)

    // no sections: full window free
    val t = planWindow(emptyList(), "beta-pro", 0L, SAMPLE_MODELS)!!
    check(t.inputTokens == 0L)
    check(t.free == 1_000_000L)
    check(t.fits)

    // overflow: fits false, reserve false
    val big = listOf(sec("big", "z".repeat(4_400_000)))
    val u = planWindow(big, "beta-pro", 0L, SAMPLE_MODELS)!!
    check(u.inputTokens == 1_100_000L)
    check(u.free == -100_000L)
    check(!u.fits)
    check(!u.outputReserveOk)

    // plan_all drops unknown ids and keeps order
    val plans = planAll(two, listOf("beta-pro", "alpha-mini", "ghost"), 0L, SAMPLE_MODELS)
    check(plans.size == 2)
    check(plans[0].id == "beta-pro")
    check(plans[1].id == "alpha-mini")
    check(plans[1].free == 199_200L)
    check(planAll(two, emptyList(), 0L, SAMPLE_MODELS).isEmpty())

    println("context-window-planner (Kotlin): all tests passed")
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →