Token Estimator — Swift source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.
// token-estimator — Swift port: tokenizer-free LLM token estimation.
//
// Display snippet: ports the line classifier and estimator core from the
// TypeScript lib (src/lib/tokenEstimator.ts). Each non-empty line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate; the estimate carries a ±15% band. Length is counted
// in UTF-16 code units (utf16.count), the same unit the TS reference
// counts; the full result shape and the whole-text JSON gate live in TS/Go.
import Foundation
enum ContentType: String, CaseIterable {
case prose, code, json, cjk
// Average characters per token by content type (CHARS_PER_TOKEN in TS).
var charsPerToken: Double {
switch self {
case .json: return 3.0
case .cjk: return 1.5
case .code: return 3.5
case .prose: return 4.0 // the TS fallback type
}
}
}
let estimateTolerance = 0.15
let codeSymbols: Set<Character> = ["{", "}", "(", ")", ";", "=", "<", ">", "[", "]", "#"]
// CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul (U+AC00..U+D7AF).
func hasCjk(_ s: String) -> Bool {
s.unicodeScalars.contains {
(0x4E00...0x9FFF).contains($0.value) || (0x3040...0x30FF).contains($0.value) ||
(0xAC00...0xD7AF).contains($0.value)
}
}
// Classify a line by its shape. Order: json, cjk, code, prose.
func detectLineType(_ line: String) -> ContentType {
let t = line.trimmingCharacters(in: .whitespacesAndNewlines)
if let h = t.first, "{}[\"".contains(h),
line.contains(":") || line.contains(",") {
return .json
}
if hasCjk(line) { return .cjk }
let symbols = Double(line.filter { codeSymbols.contains($0) }.count)
let density = symbols / Double(max(line.utf16.count, 1))
if density > 0.08 || (t.last.map { ";{}".contains($0) } ?? false) {
return .code
}
return .prose
}
struct Estimate {
let tokens: Double, low: Double, high: Double, dominant: ContentType
var breakdown: [ContentType: Double]
}
// Sum per-line estimates for every non-empty line of text.
func estimateTokens(_ text: String) -> Estimate {
var breakdown: [ContentType: Double] = [.prose: 0, .code: 0, .json: 0, .cjk: 0]
var tokens = 0.0
let normalized = text.replacingOccurrences(of: "\r\n", with: "\n")
for raw in normalized.split(separator: "\n", omittingEmptySubsequences: false) {
let line = String(raw)
if line.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty { continue }
let ty = detectLineType(line)
var lt = (Double(line.utf16.count) / ty.charsPerToken).rounded()
if lt < 1.0 { lt = 1.0 } // max(1, round(len / rate))
tokens += lt
breakdown[ty, default: 0] += lt
}
// Dominant type: strictly-greater scan keeps ties on .prose, as in TS.
var dominant = ContentType.prose
for t in ContentType.allCases where breakdown[t]! > breakdown[dominant]! { dominant = t }
return Estimate(
tokens: tokens,
low: (tokens * (1.0 - estimateTolerance)).rounded(),
high: (tokens * (1.0 + estimateTolerance)).rounded(),
dominant: dominant,
breakdown: breakdown)
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →