Skip to content

Token Estimator — Python source

Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""token-estimator — LLM token-count estimation heuristics.

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the Token Estimator tool, ported from
          src/lib/tokenEstimator.ts (the canonical TypeScript implementation).
Live at:  https://dev.cosmolabs.org/tools/token-estimator
License:  display source — part of CosmoDev's polyglot tool pages.

Design goals:
  - Pure + deterministic; never raises.
  - Functionally equivalent to the TS reference: same inputs -> same outputs.
  - Self-contained: stdlib only (no tokenizer, no pip packages).

Heuristic: each line is classified (prose / code / json / cjk) and divided by
that type's chars-per-token rate; the result carries a ±15% band because real
BPE tokenizers vary by vocabulary and language mix.

Faithfulness notes (the two places Python defaults silently differ from JS):
  - Rounding: JS ``Math.round`` rounds halfway cases UP, Python's built-in
    ``round`` rounds halfway cases to the nearest EVEN integer (banker's
    rounding — ``round(2.5) == 2``). Every token count goes through
    ``_js_round`` (``floor(x + 0.5)``) so a 10-char prose line estimates 3
    tokens here exactly as it does in TS.
  - Length: TS's ``String.length`` counts UTF-16 code units (an astral-plane
    character — emoji, rare CJK ext-B ideographs — counts as 2). Python's
    ``len(str)`` counts code points. Line arithmetic goes through
    ``_utf16_len`` so multi-byte text estimates identically on both sides.
  - JSON validity: ``json.loads`` accepts ``NaN`` / ``Infinity`` literals that
    ``JSON.parse`` rejects; ``parse_constant=_reject_constant`` restores the
    strict grammar.
"""

from __future__ import annotations

import json
import math
import re
from dataclasses import dataclass
from typing import Dict, Optional, Union

__all__ = [
    "estimate_tokens",
    "detect_line_type",
    "CHARS_PER_TOKEN",
    "ESTIMATE_TOLERANCE",
    "CHAT_FRAMING_TOKENS_PER_MESSAGE",
    "TokenEstimate",
    "EstimateOptions",
]

ContentType = str  # "prose" | "code" | "json" | "cjk" (Literal, kept str so the
                   # dataclass serializes exactly like the TS object)

#: Average characters per token, by content type. Mirrors CHARS_PER_TOKEN.
CHARS_PER_TOKEN: Dict[str, float] = {
    "prose": 4,
    "code": 3.5,
    "json": 3,
    "cjk": 1.5,
}

#: Reported estimate band on each side of the point estimate.
ESTIMATE_TOLERANCE = 0.15

#: Chat wrappers (role markers, delimiters) cost roughly this much per message.
CHAT_FRAMING_TOKENS_PER_MESSAGE = 5

# CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul syllables
# (U+AC00..U+D7AF). Mirrors CJK_RE = /[一-鿿぀-ヿ가-힯]/ in the TS lib.
_CJK_RE = re.compile(r"[一-鿿぀-ヿ가-힯]")

# Code-flavored symbols, counted over the raw line. Mirrors CODE_SYMBOL_RE.
_CODE_SYMBOL_RE = re.compile(r"[{}();=<>\[\]#]")

# Line split on LF or CRLF. Mirrors text.split(/\r?\n/).
_NEWLINE_RE = re.compile(r"\r?\n")

# Whitespace split for the word count. Mirrors trimmedText.split(/\s+/).
_WHITESPACE_RE = re.compile(r"\s+")

# Scan order used to resolve the majority type; ties keep the earlier entry.
_CONTENT_TYPES = ("prose", "code", "json", "cjk")


@dataclass
class TokenEstimate:
    """Result of ``estimate_tokens``. Field-for-field twin of the TS
    ``TokenEstimate`` interface (same keys, same meanings)."""

    tokens: int  # sum of per-line estimates (excludes framing)
    low: int  # round(tokens * (1 - ESTIMATE_TOLERANCE))
    high: int  # round(tokens * (1 + ESTIMATE_TOLERANCE))
    chars: int  # total characters excluding newlines
    words: int  # whitespace-split word count
    lines: int  # non-empty line count
    contentType: str  # majority of per-line token mass
    breakdown: Dict[str, int]  # tokens per detected line type (others 0)
    framingTokens: int  # messages * CHAT_FRAMING_TOKENS_PER_MESSAGE


@dataclass
class EstimateOptions:
    """Options mirror the TS ``EstimateOptions`` interface. Defaults match the
    TS default (auto detection, no chat framing)."""

    contentType: Union[str, None] = "auto"
    """Force a content type ('prose' | 'code' | 'json' | 'cjk'), or 'auto' /
    None to detect per line."""

    messages: int = 0
    """Chat messages the text will be sent as (adds framing tokens)."""


def _utf16_len(s: str) -> int:
    """Length of ``s`` in UTF-16 code units — the unit TS's ``String.length``
    counts. BMP code points are one unit, astral-plane ones two."""
    return sum(2 if ord(ch) > 0xFFFF else 1 for ch in s)


def _js_round(x: float) -> int:
    """Round halfway cases up, like JS ``Math.round`` (Python's built-in
    ``round`` would round 2.5 to 2, not 3)."""
    return math.floor(x + 0.5)


def _reject_constant(name: str) -> None:
    """``json.loads`` parse_constant hook: raise on NaN / Infinity so the
    validity check matches ``JSON.parse``'s strict grammar."""
    raise ValueError(f"non-standard JSON constant: {name}")


def detect_line_type(line: str) -> str:
    """Classify a single line by its shape. Order: json, cjk, code, prose."""
    trimmed = line.strip()
    # JSON-ish: opens like a JSON fragment AND carries a separator.
    if trimmed[:1] in ("{", "}", "[", '"') and (":" in line or "," in line):
        return "json"
    # CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
    if _CJK_RE.search(line):
        return "cjk"
    # Code: symbol-dense, or a statement terminator / block opener at EOL.
    length = _utf16_len(line)
    density = len(_CODE_SYMBOL_RE.findall(line)) / length if length else 0.0
    if density > 0.08 or trimmed.endswith((";", "{", "}")):
        return "code"
    return "prose"


def _is_valid_json(text: str) -> bool:
    """Whole-text JSON gate: mirrors isValidJson() (JSON.parse in a
    try/except); empty/whitespace text is not."""
    if not text.strip():
        return False
    try:
        json.loads(text, parse_constant=_reject_constant)
    except ValueError:
        return False
    return True


def estimate_tokens(text: str, options: Optional[EstimateOptions] = None) -> TokenEstimate:
    """Estimate the LLM token count of ``text`` without running a tokenizer.
    Never raises; agrees with the TS reference on every shared vector."""
    opts = options if options is not None else EstimateOptions()
    forced = opts.contentType if opts.contentType not in (None, "auto") else None
    # AUTO + whole-text JSON: a document that parses as JSON is json all the
    # way down — json's 3 chars/token rate applies to every line, not just
    # the reported contentType.
    whole_text_json = forced is None and _is_valid_json(text)

    all_lines = _NEWLINE_RE.split(text)
    non_empty = [line for line in all_lines if line.strip() != ""]

    breakdown: Dict[str, int] = {"prose": 0, "code": 0, "json": 0, "cjk": 0}
    tokens = 0
    for line in non_empty:
        line_type = forced if forced is not None else (
            "json" if whole_text_json else detect_line_type(line)
        )
        line_tokens = max(1, _js_round(_utf16_len(line) / CHARS_PER_TOKEN[line_type]))
        tokens += line_tokens
        breakdown[line_type] += line_tokens

    # Resolved type = the line type holding the most token mass (ties stay
    # 'prose', the first entry of _CONTENT_TYPES).
    content_type = "prose"
    for t in _CONTENT_TYPES:
        if breakdown[t] > breakdown[content_type]:
            content_type = t

    trimmed_text = text.strip()
    words = 0 if trimmed_text == "" else len(
        [w for w in _WHITESPACE_RE.split(trimmed_text) if w]
    )

    return TokenEstimate(
        tokens=tokens,
        low=_js_round(tokens * (1 - ESTIMATE_TOLERANCE)),
        high=_js_round(tokens * (1 + ESTIMATE_TOLERANCE)),
        chars=sum(_utf16_len(line) for line in all_lines),
        words=words,
        lines=len(non_empty),
        contentType=content_type,
        breakdown=breakdown,
        framingTokens=opts.messages * CHAT_FRAMING_TOKENS_PER_MESSAGE,
    )

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →