Token Estimator — Python source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""token-estimator — LLM token-count estimation heuristics.
Language: Python (3.9+, standard library only)
Source: CosmoDev polyglot showcase port of the Token Estimator tool, ported from
src/lib/tokenEstimator.ts (the canonical TypeScript implementation).
Live at: https://dev.cosmolabs.org/tools/token-estimator
License: display source — part of CosmoDev's polyglot tool pages.
Design goals:
- Pure + deterministic; never raises.
- Functionally equivalent to the TS reference: same inputs -> same outputs.
- Self-contained: stdlib only (no tokenizer, no pip packages).
Heuristic: each line is classified (prose / code / json / cjk) and divided by
that type's chars-per-token rate; the result carries a ±15% band because real
BPE tokenizers vary by vocabulary and language mix.
Faithfulness notes (the two places Python defaults silently differ from JS):
- Rounding: JS ``Math.round`` rounds halfway cases UP, Python's built-in
``round`` rounds halfway cases to the nearest EVEN integer (banker's
rounding — ``round(2.5) == 2``). Every token count goes through
``_js_round`` (``floor(x + 0.5)``) so a 10-char prose line estimates 3
tokens here exactly as it does in TS.
- Length: TS's ``String.length`` counts UTF-16 code units (an astral-plane
character — emoji, rare CJK ext-B ideographs — counts as 2). Python's
``len(str)`` counts code points. Line arithmetic goes through
``_utf16_len`` so multi-byte text estimates identically on both sides.
- JSON validity: ``json.loads`` accepts ``NaN`` / ``Infinity`` literals that
``JSON.parse`` rejects; ``parse_constant=_reject_constant`` restores the
strict grammar.
"""
from __future__ import annotations
import json
import math
import re
from dataclasses import dataclass
from typing import Dict, Optional, Union
__all__ = [
"estimate_tokens",
"detect_line_type",
"CHARS_PER_TOKEN",
"ESTIMATE_TOLERANCE",
"CHAT_FRAMING_TOKENS_PER_MESSAGE",
"TokenEstimate",
"EstimateOptions",
]
ContentType = str # "prose" | "code" | "json" | "cjk" (Literal, kept str so the
# dataclass serializes exactly like the TS object)
#: Average characters per token, by content type. Mirrors CHARS_PER_TOKEN.
CHARS_PER_TOKEN: Dict[str, float] = {
"prose": 4,
"code": 3.5,
"json": 3,
"cjk": 1.5,
}
#: Reported estimate band on each side of the point estimate.
ESTIMATE_TOLERANCE = 0.15
#: Chat wrappers (role markers, delimiters) cost roughly this much per message.
CHAT_FRAMING_TOKENS_PER_MESSAGE = 5
# CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul syllables
# (U+AC00..U+D7AF). Mirrors CJK_RE = /[一-鿿-ヿ가-]/ in the TS lib.
_CJK_RE = re.compile(r"[一-鿿-ヿ가-]")
# Code-flavored symbols, counted over the raw line. Mirrors CODE_SYMBOL_RE.
_CODE_SYMBOL_RE = re.compile(r"[{}();=<>\[\]#]")
# Line split on LF or CRLF. Mirrors text.split(/\r?\n/).
_NEWLINE_RE = re.compile(r"\r?\n")
# Whitespace split for the word count. Mirrors trimmedText.split(/\s+/).
_WHITESPACE_RE = re.compile(r"\s+")
# Scan order used to resolve the majority type; ties keep the earlier entry.
_CONTENT_TYPES = ("prose", "code", "json", "cjk")
@dataclass
class TokenEstimate:
"""Result of ``estimate_tokens``. Field-for-field twin of the TS
``TokenEstimate`` interface (same keys, same meanings)."""
tokens: int # sum of per-line estimates (excludes framing)
low: int # round(tokens * (1 - ESTIMATE_TOLERANCE))
high: int # round(tokens * (1 + ESTIMATE_TOLERANCE))
chars: int # total characters excluding newlines
words: int # whitespace-split word count
lines: int # non-empty line count
contentType: str # majority of per-line token mass
breakdown: Dict[str, int] # tokens per detected line type (others 0)
framingTokens: int # messages * CHAT_FRAMING_TOKENS_PER_MESSAGE
@dataclass
class EstimateOptions:
"""Options mirror the TS ``EstimateOptions`` interface. Defaults match the
TS default (auto detection, no chat framing)."""
contentType: Union[str, None] = "auto"
"""Force a content type ('prose' | 'code' | 'json' | 'cjk'), or 'auto' /
None to detect per line."""
messages: int = 0
"""Chat messages the text will be sent as (adds framing tokens)."""
def _utf16_len(s: str) -> int:
"""Length of ``s`` in UTF-16 code units — the unit TS's ``String.length``
counts. BMP code points are one unit, astral-plane ones two."""
return sum(2 if ord(ch) > 0xFFFF else 1 for ch in s)
def _js_round(x: float) -> int:
"""Round halfway cases up, like JS ``Math.round`` (Python's built-in
``round`` would round 2.5 to 2, not 3)."""
return math.floor(x + 0.5)
def _reject_constant(name: str) -> None:
"""``json.loads`` parse_constant hook: raise on NaN / Infinity so the
validity check matches ``JSON.parse``'s strict grammar."""
raise ValueError(f"non-standard JSON constant: {name}")
def detect_line_type(line: str) -> str:
"""Classify a single line by its shape. Order: json, cjk, code, prose."""
trimmed = line.strip()
# JSON-ish: opens like a JSON fragment AND carries a separator.
if trimmed[:1] in ("{", "}", "[", '"') and (":" in line or "," in line):
return "json"
# CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
if _CJK_RE.search(line):
return "cjk"
# Code: symbol-dense, or a statement terminator / block opener at EOL.
length = _utf16_len(line)
density = len(_CODE_SYMBOL_RE.findall(line)) / length if length else 0.0
if density > 0.08 or trimmed.endswith((";", "{", "}")):
return "code"
return "prose"
def _is_valid_json(text: str) -> bool:
"""Whole-text JSON gate: mirrors isValidJson() (JSON.parse in a
try/except); empty/whitespace text is not."""
if not text.strip():
return False
try:
json.loads(text, parse_constant=_reject_constant)
except ValueError:
return False
return True
def estimate_tokens(text: str, options: Optional[EstimateOptions] = None) -> TokenEstimate:
"""Estimate the LLM token count of ``text`` without running a tokenizer.
Never raises; agrees with the TS reference on every shared vector."""
opts = options if options is not None else EstimateOptions()
forced = opts.contentType if opts.contentType not in (None, "auto") else None
# AUTO + whole-text JSON: a document that parses as JSON is json all the
# way down — json's 3 chars/token rate applies to every line, not just
# the reported contentType.
whole_text_json = forced is None and _is_valid_json(text)
all_lines = _NEWLINE_RE.split(text)
non_empty = [line for line in all_lines if line.strip() != ""]
breakdown: Dict[str, int] = {"prose": 0, "code": 0, "json": 0, "cjk": 0}
tokens = 0
for line in non_empty:
line_type = forced if forced is not None else (
"json" if whole_text_json else detect_line_type(line)
)
line_tokens = max(1, _js_round(_utf16_len(line) / CHARS_PER_TOKEN[line_type]))
tokens += line_tokens
breakdown[line_type] += line_tokens
# Resolved type = the line type holding the most token mass (ties stay
# 'prose', the first entry of _CONTENT_TYPES).
content_type = "prose"
for t in _CONTENT_TYPES:
if breakdown[t] > breakdown[content_type]:
content_type = t
trimmed_text = text.strip()
words = 0 if trimmed_text == "" else len(
[w for w in _WHITESPACE_RE.split(trimmed_text) if w]
)
return TokenEstimate(
tokens=tokens,
low=_js_round(tokens * (1 - ESTIMATE_TOLERANCE)),
high=_js_round(tokens * (1 + ESTIMATE_TOLERANCE)),
chars=sum(_utf16_len(line) for line in all_lines),
words=words,
lines=len(non_empty),
contentType=content_type,
breakdown=breakdown,
framingTokens=opts.messages * CHAT_FRAMING_TOKENS_PER_MESSAGE,
)
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →