Token Estimator — JavaScript source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
/**
* token-estimator - LLM token-count estimation heuristics.
*
* Language: JavaScript (ES2020+, runs unmodified in Node 16+ and modern browsers)
* Source: CosmoDev polyglot showcase port of the Token Estimator tool, ported from
* src/lib/tokenEstimator.ts (the canonical TypeScript implementation).
* Live at: https://dev.cosmolabs.org/tools/token-estimator
* License: display source - part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; never throws.
* - Functionally equivalent to the TS reference: same inputs -> same outputs.
* - Self-contained: stdlib only (no npm dependencies, no tokenizer).
*
* Heuristic: each line is classified (prose / code / json / cjk) and divided
* by that type's chars-per-token rate; the result carries a +/-15% band
* because real BPE tokenizers vary by vocabulary and language mix. JS strings
* are UTF-16, so `String.length` is the same unit the TS reference counts -
* astral-plane characters (emoji, ...) cost 2 here exactly as they do there.
*/
/** Content classification of a single line. */
// (No `export type` in plain JS - keep the JSDoc union as the contract.)
/**
* @typedef {('prose' | 'code' | 'json' | 'cjk')} ContentType
*/
/**
* Average characters per token, by content type.
* Mirrors CHARS_PER_TOKEN in the TS lib.
*
* @type {Record<ContentType, number>}
*/
export const CHARS_PER_TOKEN = {
prose: 4,
code: 3.5,
json: 3,
cjk: 1.5,
};
/** Reported estimate band on each side of the point estimate (ESTIMATE_TOLERANCE). */
export const ESTIMATE_TOLERANCE = 0.15;
/** Chat wrappers (role markers, delimiters) cost roughly this much per message. */
export const CHAT_FRAMING_TOKENS_PER_MESSAGE = 5;
const CJK_RE = /[一-鿿-ヿ가-]/;
const CODE_SYMBOL_RE = /[{}();=<>\[\]#]/g;
const CONTENT_TYPES = ['prose', 'code', 'json', 'cjk'];
/**
* Result of estimateTokens. Field-for-field twin of the TS TokenEstimate.
*
* @typedef {Object} TokenEstimate
* @property {number} tokens Sum of per-line estimates (excludes framing).
* @property {number} low round(tokens * (1 - ESTIMATE_TOLERANCE))
* @property {number} high round(tokens * (1 + ESTIMATE_TOLERANCE))
* @property {number} chars Total characters excluding newlines.
* @property {number} words Whitespace-split word count.
* @property {number} lines Non-empty line count.
* @property {ContentType} contentType Majority of per-line token mass.
* @property {Record<ContentType, number>} breakdown Tokens per detected line type.
* @property {number} framingTokens messages * CHAT_FRAMING_TOKENS_PER_MESSAGE.
*/
/**
* Options shape. Keys are all optional (EstimateOptions in the TS lib).
*
* @typedef {Object} EstimateOptions
* @property {ContentType | 'auto'} [contentType] Force a type, or 'auto' to detect. Defaults to 'auto'.
* @property {number} [messages] Chat messages the text will be sent as. Defaults to 0.
*/
/**
* Classify a single line by its shape. Order: json, cjk, code, prose.
*
* @param {string} line
* @returns {ContentType}
*/
export function detectLineType(line) {
const trimmed = line.trim();
// JSON-ish: opens like a JSON fragment AND carries a separator.
if ((trimmed.startsWith('{') || trimmed.startsWith('}') || trimmed.startsWith('[') || trimmed.startsWith('"')) && (line.includes(':') || line.includes(','))) {
return 'json';
}
// CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
if (CJK_RE.test(line)) return 'cjk';
// Code: symbol-dense, or a statement terminator / block opener at EOL.
const density = (line.match(CODE_SYMBOL_RE) ?? []).length / line.length;
if (density > 0.08 || trimmed.endsWith(';') || trimmed.endsWith('{') || trimmed.endsWith('}')) {
return 'code';
}
return 'prose';
}
/**
* Whole-text JSON gate: a document that parses as JSON is json all the way
* down. Mirrors isValidJson() (JSON.parse in a try/catch); empty/whitespace
* text is not.
*
* @param {string} text
* @returns {boolean}
*/
function isValidJson(text) {
if (!text.trim()) return false;
try {
JSON.parse(text);
return true;
} catch {
return false;
}
}
/**
* Estimate the LLM token count of text without running a tokenizer.
* Must agree with the TS reference on every shared vector.
*
* @param {string} text
* @param {EstimateOptions} [opts={}]
* @returns {TokenEstimate}
*/
export function estimateTokens(text, opts = {}) {
const forced = opts.contentType && opts.contentType !== 'auto' ? opts.contentType : null;
// AUTO + whole-text JSON: json's 3 chars/token rate applies to every line,
// not just the reported contentType.
const wholeTextJson = forced === null && isValidJson(text);
const allLines = text.split(/\r?\n/);
const nonEmpty = allLines.filter((line) => line.trim() !== '');
const breakdown = { prose: 0, code: 0, json: 0, cjk: 0 };
let tokens = 0;
for (const line of nonEmpty) {
const type = forced ?? (wholeTextJson ? 'json' : detectLineType(line));
const lineTokens = Math.max(1, Math.round(line.length / CHARS_PER_TOKEN[type]));
tokens += lineTokens;
breakdown[type] += lineTokens;
}
// Resolved type = the line type holding the most token mass (ties stay
// 'prose', the first entry of CONTENT_TYPES).
let contentType = 'prose';
for (const type of CONTENT_TYPES) {
if (breakdown[type] > breakdown[contentType]) contentType = type;
}
const trimmedText = text.trim();
return {
tokens,
low: Math.round(tokens * (1 - ESTIMATE_TOLERANCE)),
high: Math.round(tokens * (1 + ESTIMATE_TOLERANCE)),
chars: allLines.reduce((sum, line) => sum + line.length, 0),
words: trimmedText === '' ? 0 : trimmedText.split(/\s+/).filter(Boolean).length,
lines: nonEmpty.length,
contentType,
breakdown,
framingTokens: (opts.messages ?? 0) * CHAT_FRAMING_TOKENS_PER_MESSAGE,
};
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →