Token Estimator — TypeScript source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// LLM token-count estimation heuristics. No tokenizer runs here — each line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate. The result carries a ±15% band (ESTIMATE_TOLERANCE)
// because real BPE tokenizers vary by vocabulary and language mix.
export type ContentType = 'prose' | 'code' | 'json' | 'cjk';
export type AutoType = ContentType | 'auto';
/** Average characters per token, by content type. */
export const CHARS_PER_TOKEN: Record<ContentType, number> = {
prose: 4,
code: 3.5,
json: 3,
cjk: 1.5,
};
/** Reported estimate band on each side of the point estimate. */
export const ESTIMATE_TOLERANCE = 0.15;
/** Chat wrappers (role markers, delimiters) cost roughly this much per message. */
export const CHAT_FRAMING_TOKENS_PER_MESSAGE = 5;
const CJK_RE = /[一-鿿-ヿ가-]/;
const CODE_SYMBOL_RE = /[{}();=<>\[\]#]/g;
const CONTENT_TYPES: ContentType[] = ['prose', 'code', 'json', 'cjk'];
export interface TokenEstimate {
tokens: number; // sum of per-line estimates (excludes framing)
low: number; // round(tokens * (1 - ESTIMATE_TOLERANCE))
high: number; // round(tokens * (1 + ESTIMATE_TOLERANCE))
chars: number; // total characters excluding newlines
words: number; // whitespace-split word count
lines: number; // non-empty line count
contentType: ContentType; // majority of per-line token mass
breakdown: Record<ContentType, number>; // tokens per detected line type (others 0)
framingTokens: number; // (opts.messages ?? 0) * CHAT_FRAMING_TOKENS_PER_MESSAGE
}
export interface EstimateOptions {
contentType?: AutoType;
messages?: number;
}
/** Classify a single line by its shape. Order: json, cjk, code, prose. */
export function detectLineType(line: string): ContentType {
const trimmed = line.trim();
// JSON-ish: opens like a JSON fragment AND carries a separator.
if ((trimmed.startsWith('{') || trimmed.startsWith('}') || trimmed.startsWith('[') || trimmed.startsWith('"')) && (line.includes(':') || line.includes(','))) {
return 'json';
}
// CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
if (CJK_RE.test(line)) return 'cjk';
// Code: symbol-dense, or a statement terminator / block opener at EOL.
const density = (line.match(CODE_SYMBOL_RE) ?? []).length / line.length;
if (density > 0.08 || trimmed.endsWith(';') || trimmed.endsWith('{') || trimmed.endsWith('}')) {
return 'code';
}
return 'prose';
}
function isValidJson(text: string): boolean {
if (!text.trim()) return false;
try {
JSON.parse(text);
return true;
} catch {
return false;
}
}
function emptyBreakdown(): Record<ContentType, number> {
return { prose: 0, code: 0, json: 0, cjk: 0 };
}
/**
* Exact tokenizer: returns the true token count for one encoding of `text`.
* Supplied by callers that have a real BPE tokenizer loaded (e.g. js-tiktoken
* cl100k_base / o200k_base); the estimator itself never loads one.
*/
export type BpeEncoder = (text: string) => number;
/** Named exact-token sources, keyed by encoding. All optional. */
export interface ExactSources {
cl100k?: BpeEncoder;
o200k?: BpeEncoder;
}
/**
* Exact token count, or null when no tokenizer is loaded (`enc` undefined).
* Never falls back to a heuristic — null is the caller's signal to estimate.
*/
export function exactTokens(
text: string,
enc: BpeEncoder | undefined,
): number | null {
return enc === undefined ? null : enc(text);
}
/**
* Heuristic estimate upgraded to an exact one when a tokenizer is available.
* With `enc`: tokens/low/high all carry the exact count (band collapses,
* exact: true); the shape fields (chars, words, lines, contentType,
* breakdown, framingTokens) still come from the heuristic scan. Without
* `enc`: identical to estimateTokens with exact: false.
*/
export function estimateWithExact(
text: string,
type: AutoType,
enc?: BpeEncoder,
): TokenEstimate & { exact: boolean } {
const base = estimateTokens(text, { contentType: type });
const exact = exactTokens(text, enc);
if (exact === null) return { ...base, exact: false };
return { ...base, tokens: exact, low: exact, high: exact, exact: true };
}
export function estimateTokens(text: string, opts?: EstimateOptions): TokenEstimate {
const options = opts ?? {};
const forced = options.contentType && options.contentType !== 'auto' ? options.contentType : null;
// AUTO + whole-text JSON: a document that parses as JSON is json all the way
// down — json's 3 chars/token rate applies to every line, not just the
// reported contentType.
const wholeTextJson = forced === null && isValidJson(text);
const allLines = text.split(/\r?\n/);
const nonEmpty = allLines.filter((line) => line.trim() !== '');
const breakdown = emptyBreakdown();
let tokens = 0;
for (const line of nonEmpty) {
const type = forced ?? (wholeTextJson ? 'json' : detectLineType(line));
const lineTokens = Math.max(1, Math.round(line.length / CHARS_PER_TOKEN[type]));
tokens += lineTokens;
breakdown[type] += lineTokens;
}
// Resolved type = the line type holding the most token mass (ties stay
// 'prose', the first entry of CONTENT_TYPES).
let contentType: ContentType = 'prose';
for (const type of CONTENT_TYPES) {
if (breakdown[type] > breakdown[contentType]) contentType = type;
}
const trimmedText = text.trim();
return {
tokens,
low: Math.round(tokens * (1 - ESTIMATE_TOLERANCE)),
high: Math.round(tokens * (1 + ESTIMATE_TOLERANCE)),
chars: allLines.reduce((sum, line) => sum + line.length, 0),
words: trimmedText === '' ? 0 : trimmedText.split(/\s+/).filter(Boolean).length,
lines: nonEmpty.length,
contentType,
breakdown,
framingTokens: (options.messages ?? 0) * CHAT_FRAMING_TOKENS_PER_MESSAGE,
};
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →