Context Window Planner — Kotlin source
Paste your system prompt, docs, and history — see how they fill any model's context window, with overflow warnings and output headroom.
This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Context Window Planner — plan labeled prompt sections against a model's
// context window.
//
// Language: Kotlin (Kotlin 1.9, standard library only)
// Source: CosmoDev polyglot showcase port of the Context Window Planner
// tool, ported from src/lib/contextPlanner.ts (the canonical
// TypeScript implementation).
// Live at: https://dev.cosmolabs.org/tools/context-window-planner
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; never throws (public API returns plain values).
// - Functionally equivalent to the TS reference: same inputs -> same outputs.
// - Self-contained: stdlib only (no kotlinx.serialization — this port ships
// the same small strict recursive-descent validator the other polyglot
// siblings use, so the JSON grammar matches JSON.parse exactly rather
// than approximately).
//
// Port notes: the TS lib delegates to two siblings — `estimateTokens` from
// src/lib/tokenEstimator.ts and `fitsWindow` from src/lib/ai/models.ts (which
// defaults to the bundled pricing snapshot, src/data/ai-models.json). A
// dependency-free port cannot load that file, so the estimator is inlined
// below in the exact form the planner uses it (`estimateTokens(text).tokens`,
// auto content type — the full heuristic lives in the token-estimator port),
// window math is inlined from `fitsWindow` and `models` is an explicit
// parameter, never re-derived.
//
// Faithfulness notes (the places Kotlin's stdlib silently differs from JS):
// - Length: TS's `String.length` counts UTF-16 code units — and so does
// Kotlin's `String.length`, so no helper is needed; astral-plane
// characters (emoji, rare CJK ext-B ideographs) already count as 2 on
// both sides.
// - Rounding: `kotlin.math.round` rounds halfway cases away from zero —
// equal to JS `Math.round` over the non-negative inputs used here;
// `jsRound()` states the contract explicitly.
import kotlin.math.floor
import kotlin.math.max
import kotlin.math.round
/** One labeled block of the prompt (system / docs / history / ...).
* Mirrors the TS `PlanSection` interface. */
data class PlanSection(val label: String, val text: String)
/** Convenience constructor mirroring the TS object literal `{ label, text }`. */
fun sec(label: String, text: String) = PlanSection(label, text)
/** The subset of the TS `AiModel` record the planner reads. Production code
* passes the full snapshot entry; only these fields influence the plan. */
data class Model(val id: String, val contextWindow: Long, val maxOutput: Long)
/** Sample table for standalone use (mirrors the shared test fixtures).
* Production code passes the model snapshot instead. */
val SAMPLE_MODELS: List<Model> = listOf(
Model("alpha-mini", 200_000L, 10_000L),
Model("beta-pro", 1_000_000L, 10_000L),
Model("gamma-open", 100_000L, 10_000L),
)
/** Result of [planWindow]. Field-for-field twin of the TS `WindowPlan`
* interface. */
data class WindowPlan(
val id: String, // the model id planned against
val inputTokens: Long, // sum of per-section token estimates
val contextWindow: Long, // the model's context window
val free: Long, // window - input; negative on overflow
val fits: Boolean, // raw fit: free >= 0
val outputReserveOk: Boolean, // room for the output reserve
val maxOutput: Long, // the model's output cap (informational)
)
/** Content classification of a single line. The planner only needs each
* type's chars-per-token rate (mirrors `CHARS_PER_TOKEN` in
* src/lib/tokenEstimator.ts: prose 4, code 3.5, json 3, cjk 1.5). */
private enum class ContentType { PROSE, CODE, JSON, CJK }
private fun ContentType.charsPerToken(): Double = when (this) {
ContentType.PROSE -> 4.0
ContentType.CODE -> 3.5
ContentType.JSON -> 3.0
ContentType.CJK -> 1.5
}
/** Reports whether `s` contains a CJK ideograph (U+4E00–U+9FFF), kana
* (U+3040–U+30FF), or a Hangul syllable (U+AC00–U+D7AF). Mirrors `CJK_RE`
* in the TS lib. */
private fun hasCjk(s: String): Boolean = s.any { c ->
(c in '一'..'鿿') || (c in ''..'ヿ') || (c in '가'..'')
}
/** Reports whether `c` is one of the code-flavored symbols counted by
* `CODE_SYMBOL_RE` (`{}();=<>[]#`). */
private fun isCodeSymbol(c: Char): Boolean =
c in "{}();=<>[]#"
/** JS `Math.round`: halfway cases round up (`floor(x + 0.5)`). */
private fun jsRound(x: Double): Long = floor(x + 0.5).toLong()
/** Splits `text` on LF or CRLF, mirroring `text.split(/\r?\n/)`: strip the
* optional CR that belongs to the newline, then split on LF. A lone CR is
* NOT a line break. */
private fun splitLines(text: String): List<String> =
text.split('\n').map { line -> if (line.endsWith('\r')) line.dropLast(1) else line }
/** Classifies a single line by its shape. Order: json, cjk, code, prose.
* Inlined from `detectLineType()` in src/lib/tokenEstimator.ts. */
private fun detectLineType(line: String): ContentType {
val trimmed = line.trim()
// JSON-ish: opens like a JSON fragment AND carries a separator.
val startsJsonish = trimmed.startsWith("{") || trimmed.startsWith("}") ||
trimmed.startsWith("[") || trimmed.startsWith("\"")
if (startsJsonish && (":" in line || "," in line)) return ContentType.JSON
// CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
if (hasCjk(line)) return ContentType.CJK
// Code: symbol-dense, or a statement terminator / block opener at EOL.
val length = line.length // UTF-16 code units, same unit as TS
val symbols = line.count { isCodeSymbol(it) }
val density = if (length > 0) symbols.toDouble() / length else 0.0
if (density > 0.08 || trimmed.endsWith(";") || trimmed.endsWith("{") ||
trimmed.endsWith("}"))
return ContentType.CODE
return ContentType.PROSE
}
/** A strict JSON syntax validator — the exact grammar `JSON.parse` accepts,
* walked with a cursor. Same shape as the Rust/C/C++/C#/Java siblings'
* validators, so whole-text JSON detection behaves identically across
* every port. */
private class JsonParser(private val s: String) {
private var pos = 0
/** value := ws* (object | array | string | number | 'true' | 'false' | 'null') ws* */
fun value(): Boolean {
skipWs()
if (pos >= s.length) return false
return when (s[pos]) {
'{' -> objekt()
'[' -> array()
'"' -> string()
in '0'..'9', '-' -> number()
't' -> literal("true")
'f' -> literal("false")
'n' -> literal("null")
else -> false
}
}
fun atEnd(): Boolean {
skipWs()
return pos == s.length // reject trailing garbage
}
private fun skipWs() {
while (pos < s.length && s[pos] in " \t\n\r") pos++
}
private fun peek(): Char = if (pos < s.length) s[pos] else ' '
private fun eat(b: Char): Boolean {
if (pos < s.length && s[pos] == b) { pos++; return true }
return false
}
private fun literal(lit: String): Boolean {
if (s.regionMatches(pos, lit, 0, lit.length)) {
pos += lit.length
return true
}
return false
}
/** object := '{' ws* (string ws* ':' value (ws* ',' ...)*)? ws* '}' */
private fun objekt(): Boolean {
if (!eat('{')) return false
skipWs()
if (eat('}')) return true
while (true) {
if (!string()) return false
skipWs()
if (!eat(':')) return false
if (!value()) return false
skipWs()
if (eat(',')) skipWs()
else return eat('}')
}
}
/** array := '[' ws* (value (ws* ',' ws* value)*)? ws* ']' */
private fun array(): Boolean {
if (!eat('[')) return false
skipWs()
if (eat(']')) return true
while (true) {
if (!value()) return false
skipWs()
if (eat(',')) skipWs()
else return eat(']')
}
}
/** string := '"' (escape | any char >= 0x20)* '"'
* escape := '\' ('"' | '/' | '\' | 'b' | 'f' | 'n' | 'r' | 't' | 'u' hex4) */
private fun string(): Boolean {
if (!eat('"')) return false
while (pos < s.length) {
val c = s[pos]
when {
c == '"' -> { pos++; return true }
c == '\\' -> {
pos++
if (pos >= s.length) return false
when (val esc = s[pos++]) {
'"', '/', '\\', 'b', 'f', 'n', 'r', 't' -> {}
'u' -> {
repeat(4) {
val h = peek()
val hex = h in '0'..'9' || h in 'a'..'f' || h in 'A'..'F'
if (!hex) return false
pos++
}
}
else -> return false
}
}
c < ' ' -> return false // raw control characters not allowed
else -> pos++
}
}
return false // unterminated string
}
/** number := '-'? int frac? exp? — no leading zeros, like JSON.parse. */
private fun number(): Boolean {
eat('-')
when (peek()) {
'0' -> pos++
in '1'..'9' -> while (peek() in '0'..'9') pos++
else -> return false
}
if (peek() == '.') {
pos++
var digits = 0
while (peek() in '0'..'9') { pos++; digits++ }
if (digits == 0) return false
}
if (peek() == 'e' || peek() == 'E') {
pos++
if (peek() == '+' || peek() == '-') pos++
var digits = 0
while (peek() in '0'..'9') { pos++; digits++ }
if (digits == 0) return false
}
return true
}
}
/** Whole-text JSON gate: a document that parses as JSON is json all the way
* down. Mirrors `isValidJson()` (`JSON.parse` in a try/catch);
* empty/whitespace text is not. */
internal fun isValidJson(text: String): Boolean {
if (text.isBlank()) return false
val p = JsonParser(text)
return p.value() && p.atEnd()
}
/** Token count of `text` under auto content detection — exactly the slice of
* `estimateTokens()` the planner consumes (`.tokens`): per non-empty line,
* `max(1, round(length / charsPerToken))`. Framing tokens are the caller's
* job. */
internal fun estimateTokens(text: String): Long {
// AUTO + whole-text JSON: json's 3 chars/token rate applies to every
// line, not just the reported content type.
val wholeTextJson = isValidJson(text)
var tokens = 0L
for (line in splitLines(text)) {
if (line.isBlank()) continue
val type = if (wholeTextJson) ContentType.JSON else detectLineType(line)
tokens += max(1L, jsRound(line.length / type.charsPerToken()))
}
return tokens
}
/** Sum of per-section token estimates (framing tokens are the caller's job).
* Mirrors `inputTokenTotal()` in the TS lib. */
fun inputTokenTotal(sections: List<PlanSection>): Long =
sections.sumOf { estimateTokens(it.text) }
/** Plan one section set against one model's context window. Returns `null`
* for an unknown model id (window math is `fitsWindow`'s, never re-derived).
* Mirrors `planWindow()` in the TS lib. */
fun planWindow(
sections: List<PlanSection>,
modelId: String,
outputReserve: Long = 0L,
models: List<Model> = emptyList(),
): WindowPlan? {
val inputTokens = inputTokenTotal(sections)
// Fit check inlined from fitsWindow() in src/lib/ai/models.ts.
val m = models.firstOrNull { it.id == modelId } ?: return null
val free = m.contextWindow - inputTokens
return WindowPlan(
id = modelId,
inputTokens = inputTokens,
contextWindow = m.contextWindow,
free = free,
fits = free >= 0,
outputReserveOk = free >= outputReserve,
maxOutput = m.maxOutput,
)
}
/** Plan against several models; unknown ids are dropped from the result.
* Mirrors `planAll()` in the TS lib. */
fun planAll(
sections: List<PlanSection>,
modelIds: List<String>,
outputReserve: Long = 0L,
models: List<Model> = emptyList(),
): List<WindowPlan> =
modelIds.mapNotNull { planWindow(sections, it, outputReserve, models) }
// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
/** A 1600-char single line of 'a' is pure prose: 1600 / 4 = 400 tokens. */
private fun twoSections(): List<PlanSection> {
val lineA = "a".repeat(1600)
return listOf(sec("sys", lineA), sec("docs", lineA))
}
fun main() {
val two = twoSections()
// input totals
check(inputTokenTotal(two) == 800L)
check(inputTokenTotal(emptyList()) == 0L)
check(inputTokenTotal(listOf(sec("sys", ""))) == 0L)
// plans two 400-token sections against beta-pro
val p = planWindow(two, "beta-pro", 0L, SAMPLE_MODELS)!!
check(p.id == "beta-pro")
check(p.inputTokens == 800L)
check(p.contextWindow == 1_000_000L)
check(p.free == 999_200L)
check(p.fits)
check(p.outputReserveOk)
check(p.maxOutput == 10_000L)
// reserve larger than free leaves raw fit true
val q = planWindow(two, "beta-pro", 1_000_000L, SAMPLE_MODELS)!!
check(q.fits)
check(!q.outputReserveOk)
// reserve exactly equal to free is ok
val r = planWindow(two, "beta-pro", 999_200L, SAMPLE_MODELS)!!
check(r.outputReserveOk)
// smaller window leaves 199,200 free
val s = planWindow(two, "alpha-mini", 0L, SAMPLE_MODELS)!!
check(s.contextWindow == 200_000L)
check(s.free == 199_200L)
check(s.fits)
// unknown model id returns null
check(planWindow(two, "ghost", 0L, SAMPLE_MODELS) == null)
// no sections: full window free
val t = planWindow(emptyList(), "beta-pro", 0L, SAMPLE_MODELS)!!
check(t.inputTokens == 0L)
check(t.free == 1_000_000L)
check(t.fits)
// overflow: fits false, reserve false
val big = listOf(sec("big", "z".repeat(4_400_000)))
val u = planWindow(big, "beta-pro", 0L, SAMPLE_MODELS)!!
check(u.inputTokens == 1_100_000L)
check(u.free == -100_000L)
check(!u.fits)
check(!u.outputReserveOk)
// plan_all drops unknown ids and keeps order
val plans = planAll(two, listOf("beta-pro", "alpha-mini", "ghost"), 0L, SAMPLE_MODELS)
check(plans.size == 2)
check(plans[0].id == "beta-pro")
check(plans[1].id == "alpha-mini")
check(plans[1].free == 199_200L)
check(planAll(two, emptyList(), 0L, SAMPLE_MODELS).isEmpty())
println("context-window-planner (Kotlin): all tests passed")
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →