Skip to content

Token Estimator — Swift source

Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.

This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.

// token-estimator — Swift port: tokenizer-free LLM token estimation.
//
// Display snippet: ports the line classifier and estimator core from the
// TypeScript lib (src/lib/tokenEstimator.ts). Each non-empty line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate; the estimate carries a ±15% band. Length is counted
// in UTF-16 code units (utf16.count), the same unit the TS reference
// counts; the full result shape and the whole-text JSON gate live in TS/Go.
import Foundation

enum ContentType: String, CaseIterable {
    case prose, code, json, cjk

    // Average characters per token by content type (CHARS_PER_TOKEN in TS).
    var charsPerToken: Double {
        switch self {
        case .json: return 3.0
        case .cjk: return 1.5
        case .code: return 3.5
        case .prose: return 4.0  // the TS fallback type
        }
    }
}

let estimateTolerance = 0.15
let codeSymbols: Set<Character> = ["{", "}", "(", ")", ";", "=", "<", ">", "[", "]", "#"]

// CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul (U+AC00..U+D7AF).
func hasCjk(_ s: String) -> Bool {
    s.unicodeScalars.contains {
        (0x4E00...0x9FFF).contains($0.value) || (0x3040...0x30FF).contains($0.value) ||
        (0xAC00...0xD7AF).contains($0.value)
    }
}

// Classify a line by its shape. Order: json, cjk, code, prose.
func detectLineType(_ line: String) -> ContentType {
    let t = line.trimmingCharacters(in: .whitespacesAndNewlines)
    if let h = t.first, "{}[\"".contains(h),
       line.contains(":") || line.contains(",") {
        return .json
    }
    if hasCjk(line) { return .cjk }
    let symbols = Double(line.filter { codeSymbols.contains($0) }.count)
    let density = symbols / Double(max(line.utf16.count, 1))
    if density > 0.08 || (t.last.map { ";{}".contains($0) } ?? false) {
        return .code
    }
    return .prose
}

struct Estimate {
    let tokens: Double, low: Double, high: Double, dominant: ContentType
    var breakdown: [ContentType: Double]
}

// Sum per-line estimates for every non-empty line of text.
func estimateTokens(_ text: String) -> Estimate {
    var breakdown: [ContentType: Double] = [.prose: 0, .code: 0, .json: 0, .cjk: 0]
    var tokens = 0.0
    let normalized = text.replacingOccurrences(of: "\r\n", with: "\n")
    for raw in normalized.split(separator: "\n", omittingEmptySubsequences: false) {
        let line = String(raw)
        if line.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty { continue }
        let ty = detectLineType(line)
        var lt = (Double(line.utf16.count) / ty.charsPerToken).rounded()
        if lt < 1.0 { lt = 1.0 }  // max(1, round(len / rate))
        tokens += lt
        breakdown[ty, default: 0] += lt
    }
    // Dominant type: strictly-greater scan keeps ties on .prose, as in TS.
    var dominant = ContentType.prose
    for t in ContentType.allCases where breakdown[t]! > breakdown[dominant]! { dominant = t }
    return Estimate(
        tokens: tokens,
        low: (tokens * (1.0 - estimateTolerance)).rounded(),
        high: (tokens * (1.0 + estimateTolerance)).rounded(),
        dominant: dominant,
        breakdown: breakdown)
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →