Token Estimator — Kotlin source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.
// token-estimator — Kotlin port: tokenizer-free LLM token estimation.
//
// Display snippet: ports the line classifier and estimator core from the
// TypeScript lib (src/lib/tokenEstimator.ts). Each non-empty line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate; the estimate carries a ±15% band. Kotlin strings
// are UTF-16, so length is the same unit the TS reference counts. The full
// result shape (chars/words/lines/framing) lives in TS/Go, as does the
// whole-text JSON gate (parses as JSON => json throughout).
enum class ContentType { PROSE, CODE, JSON, CJK }
object TokenEstimator {
// Average characters per token by content type (CHARS_PER_TOKEN in TS).
private fun charsPerToken(t: ContentType) = when (t) {
ContentType.JSON -> 3.0
ContentType.CJK -> 1.5
ContentType.CODE -> 3.5
ContentType.PROSE -> 4.0 // the TS fallback type
}
private const val ESTIMATE_TOLERANCE = 0.15
private val CODE_SYMBOLS = "{}();=<>[]#".toSet()
// CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul (U+AC00..U+D7AF).
private fun hasCjk(s: String): Boolean =
s.any { it in '一'..'鿿' || it in ''..'ヿ' || it in '가'..'' }
/** Classify a line by its shape. Order: json, cjk, code, prose. */
fun detectLineType(line: String): ContentType {
val t = line.trim()
val h = t.firstOrNull()
if (h != null && (h == '{' || h == '}' || h == '[' || h == '"') &&
(':' in line || ',' in line)
) return ContentType.JSON
if (hasCjk(line)) return ContentType.CJK
val symbols = line.count { it in CODE_SYMBOLS }
val e = t.lastOrNull() ?: '\0'
if ((line.isNotEmpty() && symbols.toDouble() / line.length > 0.08) ||
e == ';' || e == '{' || e == '}'
) return ContentType.CODE
return ContentType.PROSE
}
/** Sum of per-line estimates plus the ±15% band and per-type breakdown. */
data class Estimate(
val tokens: Double,
val low: Double,
val high: Double,
val dominant: ContentType,
val breakdown: Map<ContentType, Double>,
)
/** Sum per-line estimates for every non-empty line of text. */
fun estimateTokens(text: String): Estimate {
val breakdown = ContentType.entries.associateWith { 0.0 }.toMutableMap()
var tokens = 0.0
for (line in text.replace("\r\n", "\n").split('\n')) {
if (line.isBlank()) continue
val ty = detectLineType(line)
var lt = kotlin.math.round(line.length / charsPerToken(ty))
if (lt < 1.0) lt = 1.0 // max(1, round(len / rate))
tokens += lt
breakdown[ty] = breakdown.getValue(ty) + lt
}
// Dominant type: strictly-greater scan keeps ties on PROSE, as in TS.
var dominant = ContentType.PROSE
for (t in ContentType.entries)
if (breakdown.getValue(t) > breakdown.getValue(dominant)) dominant = t
return Estimate(
tokens,
kotlin.math.round(tokens * (1.0 - ESTIMATE_TOLERANCE)),
kotlin.math.round(tokens * (1.0 + ESTIMATE_TOLERANCE)),
dominant, breakdown,
)
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →