Skip to content

Token Estimator — Zig source

Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// token-estimator — Zig port: tokenizer-free LLM token estimation.
//
// Display snippet: ports the line classifier and estimator core from the
// TypeScript lib (src/lib/tokenEstimator.ts). Each non-empty line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate; the estimate carries a ±15% band. Slices count
// bytes here (the TS reference counts UTF-16 code units); the full result
// shape and the whole-text JSON gate live in TS/Go.
const std = @import("std");

pub const ContentType = enum { prose, code, json, cjk };

pub const Estimate = struct {
    tokens: u64, low: u64, high: u64, // tokens plus the ±15% band
    dominant: ContentType,            // line type holding the most token mass
    breakdown: [4]u64,                // tokens per type, index = enum order
};

const chars_per_token = [4]f64{ 4.0, 3.5, 3.0, 1.5 }, tolerance = 0.15; // prose..cjk

fn isCodeSymbol(b: u8) bool {
    return switch (b) {
        '{', '}', '(', ')', ';', '=', '<', '>', '[', ']', '#' => true,
        else => false,
    };
}

// CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul (U+AC00..U+D7AF).
fn hasCjk(s: []const u8) bool {
    const view = std.unicode.Utf8View.init(s) catch return false;
    var it = view.iterator();
    while (it.nextCodepoint()) |cp| {
        if ((cp >= 0x4E00 and cp <= 0x9FFF) or (cp >= 0x3040 and cp <= 0x30FF) or
            (cp >= 0xAC00 and cp <= 0xD7AF)) return true;
    }
    return false;
}

// Classify a line by its shape. Order: json, cjk, code, prose.
pub fn detectLineType(line: []const u8) ContentType {
    const t = std.mem.trim(u8, line, " \t\r");
    const h: u8 = if (t.len > 0) t[0] else 0;
    const e: u8 = if (t.len > 0) t[t.len - 1] else 0;
    const opens = (h == '{' or h == '}' or h == '[' or h == '"');
    const sep = std.mem.indexOfScalar(u8, line, ':') != null or
        std.mem.indexOfScalar(u8, line, ',') != null;
    if (opens and sep) return .json;
    if (hasCjk(line)) return .cjk;
    var symbols: usize = 0;
    for (line) |b| { if (isCodeSymbol(b)) symbols += 1; }
    const density = @as(f64, @floatFromInt(symbols)) / @as(f64, @floatFromInt(line.len));
    if ((line.len > 0 and density > 0.08) or e == ';' or e == '{' or e == '}')
        return .code;
    return .prose;
}

// Sum per-line estimates for every non-empty line of text.
pub fn estimateTokens(text: []const u8) Estimate {
    var est = Estimate{ .tokens = 0, .low = 0, .high = 0, .dominant = .prose,
        .breakdown = [_]u64{0} ** 4 };
    var lines = std.mem.splitScalar(u8, text, '\n');
    while (lines.next()) |raw| {
        const line = std.mem.trimRight(u8, raw, "\r");
        if (std.mem.trim(u8, line, " \t\r").len == 0) continue;
        const ty = detectLineType(line);
        const ratio = @as(f64, @floatFromInt(line.len)) / chars_per_token[@intFromEnum(ty)];
        const lt: u64 = @intFromFloat(@max(1.0, std.math.round(ratio))); // max(1, round(len/rate))
        est.tokens += lt;
        est.breakdown[@intFromEnum(ty)] += lt;
    }
    for ([_]ContentType{ .code, .json, .cjk }) |t| { // ties stay on .prose, as in TS
        if (est.breakdown[@intFromEnum(t)] > est.breakdown[@intFromEnum(est.dominant)])
            est.dominant = t;
    }
    const total: f64 = @floatFromInt(est.tokens);
    est.low = @intFromFloat(std.math.round(total * (1.0 - tolerance)));
    est.high = @intFromFloat(std.math.round(total * (1.0 + tolerance)));
    return est;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →