Skip to content

Token Estimator — Java source

Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.

This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.

// token-estimator — Java port: tokenizer-free LLM token estimation.
//
// Display snippet: ports the line classifier and estimator core from the
// TypeScript lib (src/lib/tokenEstimator.ts). Each non-empty line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate; the estimate carries a ±15% band. Java strings are
// UTF-16, so String.length() is the same unit the TS reference counts. The
// full result shape and the whole-text JSON gate live in TS/Go.
import java.util.EnumMap;
import java.util.Map;

public final class TokenEstimator {

    public enum ContentType { PROSE, CODE, JSON, CJK }

    /** Average chars per token by content type (CHARS_PER_TOKEN in TS). */
    private static double charsPerToken(ContentType t) {
        return switch (t) {
            case JSON -> 3.0; case CJK -> 1.5; case CODE -> 3.5; default -> 4.0;
        }; // default = PROSE, the TS fallback type
    }

    private static final double ESTIMATE_TOLERANCE = 0.15;
    private static final String CODE_SYMBOLS = "{}();=<>[]#";

    // CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul (U+AC00..U+D7AF).
    private static boolean hasCjk(String s) {
        for (int i = 0; i < s.length(); i++) {
            char c = s.charAt(i);
            if ((c >= 0x4E00 && c <= 0x9FFF) || (c >= 0x3040 && c <= 0x30FF) ||
                (c >= 0xAC00 && c <= 0xD7AF)) return true;
        }
        return false;
    }

    /** Classify a line by its shape. Order: json, cjk, code, prose. */
    public static ContentType detectLineType(String line) {
        String t = line.trim();
        char h = t.isEmpty() ? '\0' : t.charAt(0);
        if ((h == '{' || h == '}' || h == '[' || h == '"') &&
            (line.indexOf(':') >= 0 || line.indexOf(',') >= 0)) return ContentType.JSON;
        if (hasCjk(line)) return ContentType.CJK;
        int symbols = 0;
        for (int i = 0; i < line.length(); i++)
            if (CODE_SYMBOLS.indexOf(line.charAt(i)) >= 0) symbols++;
        char e = t.isEmpty() ? '\0' : t.charAt(t.length() - 1);
        if ((line.length() > 0 && (double) symbols / line.length() > 0.08) ||
            e == ';' || e == '{' || e == '}') return ContentType.CODE;
        return ContentType.PROSE;
    }

    /** Sum of per-line estimates plus the ±15% band and per-type breakdown. */
    public record Estimate(double tokens, double low, double high, ContentType dominant,
                           Map<ContentType, Double> breakdown) {}

    /** Sum per-line estimates for every non-empty line of text. */
    public static Estimate estimateTokens(String text) {
        Map<ContentType, Double> breakdown = new EnumMap<>(ContentType.class);
        for (ContentType t : ContentType.values()) breakdown.put(t, 0.0);
        double tokens = 0;
        for (String line : text.replace("\r\n", "\n").split("\n", -1)) {
            if (line.trim().isEmpty()) continue;
            ContentType ty = detectLineType(line);
            double lt = Math.round(line.length() / charsPerToken(ty));
            if (lt < 1.0) lt = 1.0; // max(1, round(len / rate))
            tokens += lt;
            breakdown.merge(ty, lt, Double::sum);
        }
        // Dominant type: strictly-greater scan keeps ties on PROSE, as in TS.
        ContentType dominant = ContentType.PROSE;
        for (ContentType t : ContentType.values())
            if (breakdown.get(t) > breakdown.get(dominant)) dominant = t;
        return new Estimate(tokens,
                Math.round(tokens * (1.0 - ESTIMATE_TOLERANCE)),
                Math.round(tokens * (1.0 + ESTIMATE_TOLERANCE)),
                dominant, breakdown);
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →