Skip to content

Token Estimator — TypeScript source

Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// LLM token-count estimation heuristics. No tokenizer runs here — each line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate. The result carries a ±15% band (ESTIMATE_TOLERANCE)
// because real BPE tokenizers vary by vocabulary and language mix.

export type ContentType = 'prose' | 'code' | 'json' | 'cjk';
export type AutoType = ContentType | 'auto';

/** Average characters per token, by content type. */
export const CHARS_PER_TOKEN: Record<ContentType, number> = {
  prose: 4,
  code: 3.5,
  json: 3,
  cjk: 1.5,
};

/** Reported estimate band on each side of the point estimate. */
export const ESTIMATE_TOLERANCE = 0.15;

/** Chat wrappers (role markers, delimiters) cost roughly this much per message. */
export const CHAT_FRAMING_TOKENS_PER_MESSAGE = 5;

const CJK_RE = /[一-鿿぀-ヿ가-힯]/;
const CODE_SYMBOL_RE = /[{}();=<>\[\]#]/g;
const CONTENT_TYPES: ContentType[] = ['prose', 'code', 'json', 'cjk'];

export interface TokenEstimate {
  tokens: number; // sum of per-line estimates (excludes framing)
  low: number; // round(tokens * (1 - ESTIMATE_TOLERANCE))
  high: number; // round(tokens * (1 + ESTIMATE_TOLERANCE))
  chars: number; // total characters excluding newlines
  words: number; // whitespace-split word count
  lines: number; // non-empty line count
  contentType: ContentType; // majority of per-line token mass
  breakdown: Record<ContentType, number>; // tokens per detected line type (others 0)
  framingTokens: number; // (opts.messages ?? 0) * CHAT_FRAMING_TOKENS_PER_MESSAGE
}

export interface EstimateOptions {
  contentType?: AutoType;
  messages?: number;
}

/** Classify a single line by its shape. Order: json, cjk, code, prose. */
export function detectLineType(line: string): ContentType {
  const trimmed = line.trim();
  // JSON-ish: opens like a JSON fragment AND carries a separator.
  if ((trimmed.startsWith('{') || trimmed.startsWith('}') || trimmed.startsWith('[') || trimmed.startsWith('"')) && (line.includes(':') || line.includes(','))) {
    return 'json';
  }
  // CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
  if (CJK_RE.test(line)) return 'cjk';
  // Code: symbol-dense, or a statement terminator / block opener at EOL.
  const density = (line.match(CODE_SYMBOL_RE) ?? []).length / line.length;
  if (density > 0.08 || trimmed.endsWith(';') || trimmed.endsWith('{') || trimmed.endsWith('}')) {
    return 'code';
  }
  return 'prose';
}

function isValidJson(text: string): boolean {
  if (!text.trim()) return false;
  try {
    JSON.parse(text);
    return true;
  } catch {
    return false;
  }
}

function emptyBreakdown(): Record<ContentType, number> {
  return { prose: 0, code: 0, json: 0, cjk: 0 };
}

/**
 * Exact tokenizer: returns the true token count for one encoding of `text`.
 * Supplied by callers that have a real BPE tokenizer loaded (e.g. js-tiktoken
 * cl100k_base / o200k_base); the estimator itself never loads one.
 */
export type BpeEncoder = (text: string) => number;

/** Named exact-token sources, keyed by encoding. All optional. */
export interface ExactSources {
  cl100k?: BpeEncoder;
  o200k?: BpeEncoder;
}

/**
 * Exact token count, or null when no tokenizer is loaded (`enc` undefined).
 * Never falls back to a heuristic — null is the caller's signal to estimate.
 */
export function exactTokens(
  text: string,
  enc: BpeEncoder | undefined,
): number | null {
  return enc === undefined ? null : enc(text);
}

/**
 * Heuristic estimate upgraded to an exact one when a tokenizer is available.
 * With `enc`: tokens/low/high all carry the exact count (band collapses,
 * exact: true); the shape fields (chars, words, lines, contentType,
 * breakdown, framingTokens) still come from the heuristic scan. Without
 * `enc`: identical to estimateTokens with exact: false.
 */
export function estimateWithExact(
  text: string,
  type: AutoType,
  enc?: BpeEncoder,
): TokenEstimate & { exact: boolean } {
  const base = estimateTokens(text, { contentType: type });
  const exact = exactTokens(text, enc);
  if (exact === null) return { ...base, exact: false };
  return { ...base, tokens: exact, low: exact, high: exact, exact: true };
}

export function estimateTokens(text: string, opts?: EstimateOptions): TokenEstimate {
  const options = opts ?? {};
  const forced = options.contentType && options.contentType !== 'auto' ? options.contentType : null;
  // AUTO + whole-text JSON: a document that parses as JSON is json all the way
  // down — json's 3 chars/token rate applies to every line, not just the
  // reported contentType.
  const wholeTextJson = forced === null && isValidJson(text);

  const allLines = text.split(/\r?\n/);
  const nonEmpty = allLines.filter((line) => line.trim() !== '');

  const breakdown = emptyBreakdown();
  let tokens = 0;
  for (const line of nonEmpty) {
    const type = forced ?? (wholeTextJson ? 'json' : detectLineType(line));
    const lineTokens = Math.max(1, Math.round(line.length / CHARS_PER_TOKEN[type]));
    tokens += lineTokens;
    breakdown[type] += lineTokens;
  }

  // Resolved type = the line type holding the most token mass (ties stay
  // 'prose', the first entry of CONTENT_TYPES).
  let contentType: ContentType = 'prose';
  for (const type of CONTENT_TYPES) {
    if (breakdown[type] > breakdown[contentType]) contentType = type;
  }

  const trimmedText = text.trim();
  return {
    tokens,
    low: Math.round(tokens * (1 - ESTIMATE_TOLERANCE)),
    high: Math.round(tokens * (1 + ESTIMATE_TOLERANCE)),
    chars: allLines.reduce((sum, line) => sum + line.length, 0),
    words: trimmedText === '' ? 0 : trimmedText.split(/\s+/).filter(Boolean).length,
    lines: nonEmpty.length,
    contentType,
    breakdown,
    framingTokens: (options.messages ?? 0) * CHAT_FRAMING_TOKENS_PER_MESSAGE,
  };
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →