Token Estimator — Java source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.
// token-estimator — Java port: tokenizer-free LLM token estimation.
//
// Display snippet: ports the line classifier and estimator core from the
// TypeScript lib (src/lib/tokenEstimator.ts). Each non-empty line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate; the estimate carries a ±15% band. Java strings are
// UTF-16, so String.length() is the same unit the TS reference counts. The
// full result shape and the whole-text JSON gate live in TS/Go.
import java.util.EnumMap;
import java.util.Map;
public final class TokenEstimator {
public enum ContentType { PROSE, CODE, JSON, CJK }
/** Average chars per token by content type (CHARS_PER_TOKEN in TS). */
private static double charsPerToken(ContentType t) {
return switch (t) {
case JSON -> 3.0; case CJK -> 1.5; case CODE -> 3.5; default -> 4.0;
}; // default = PROSE, the TS fallback type
}
private static final double ESTIMATE_TOLERANCE = 0.15;
private static final String CODE_SYMBOLS = "{}();=<>[]#";
// CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul (U+AC00..U+D7AF).
private static boolean hasCjk(String s) {
for (int i = 0; i < s.length(); i++) {
char c = s.charAt(i);
if ((c >= 0x4E00 && c <= 0x9FFF) || (c >= 0x3040 && c <= 0x30FF) ||
(c >= 0xAC00 && c <= 0xD7AF)) return true;
}
return false;
}
/** Classify a line by its shape. Order: json, cjk, code, prose. */
public static ContentType detectLineType(String line) {
String t = line.trim();
char h = t.isEmpty() ? '\0' : t.charAt(0);
if ((h == '{' || h == '}' || h == '[' || h == '"') &&
(line.indexOf(':') >= 0 || line.indexOf(',') >= 0)) return ContentType.JSON;
if (hasCjk(line)) return ContentType.CJK;
int symbols = 0;
for (int i = 0; i < line.length(); i++)
if (CODE_SYMBOLS.indexOf(line.charAt(i)) >= 0) symbols++;
char e = t.isEmpty() ? '\0' : t.charAt(t.length() - 1);
if ((line.length() > 0 && (double) symbols / line.length() > 0.08) ||
e == ';' || e == '{' || e == '}') return ContentType.CODE;
return ContentType.PROSE;
}
/** Sum of per-line estimates plus the ±15% band and per-type breakdown. */
public record Estimate(double tokens, double low, double high, ContentType dominant,
Map<ContentType, Double> breakdown) {}
/** Sum per-line estimates for every non-empty line of text. */
public static Estimate estimateTokens(String text) {
Map<ContentType, Double> breakdown = new EnumMap<>(ContentType.class);
for (ContentType t : ContentType.values()) breakdown.put(t, 0.0);
double tokens = 0;
for (String line : text.replace("\r\n", "\n").split("\n", -1)) {
if (line.trim().isEmpty()) continue;
ContentType ty = detectLineType(line);
double lt = Math.round(line.length() / charsPerToken(ty));
if (lt < 1.0) lt = 1.0; // max(1, round(len / rate))
tokens += lt;
breakdown.merge(ty, lt, Double::sum);
}
// Dominant type: strictly-greater scan keeps ties on PROSE, as in TS.
ContentType dominant = ContentType.PROSE;
for (ContentType t : ContentType.values())
if (breakdown.get(t) > breakdown.get(dominant)) dominant = t;
return new Estimate(tokens,
Math.round(tokens * (1.0 - ESTIMATE_TOLERANCE)),
Math.round(tokens * (1.0 + ESTIMATE_TOLERANCE)),
dominant, breakdown);
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →