Token Estimator — PHP source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the PHP implementation — the same logic the interactive tool runs, in a shareable, citable form.
<?php
/**
* token-estimator — LLM token-count estimation heuristics.
*
* Language: PHP (8.1+, standard library only — mbstring for multibyte string
* arithmetic, which is effectively universal in modern PHP)
* Source: CosmoDev polyglot showcase port of the Token Estimator tool, ported from
* src/lib/tokenEstimator.ts (the canonical TypeScript implementation).
* Live at: https://dev.cosmolabs.org/tools/token-estimator
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; never throws.
* - Functionally equivalent to the TS reference: same inputs -> same outputs.
* - Self-contained: stdlib only (no Composer packages, no tokenizer).
*
* Heuristic: each line is classified (prose / code / json / cjk) and divided
* by that type's chars-per-token rate; the result carries a ±15% band
* because real BPE tokenizers vary by vocabulary and language mix.
*
* Faithfulness notes (the places PHP defaults silently differ from JS):
* - Length: TS's String.length counts UTF-16 code units (an astral-plane
* character — emoji, rare CJK ext-B ideographs — counts as 2). PHP
* strings are byte strings, so line arithmetic goes through
* te_utf16_len() (mb_str_split + mb_ord) to count the same unit.
* - Trim: PHP's trim() only strips ASCII whitespace, while JS trim()
* strips Unicode whitespace (NBSP, ideographic space, ...). te_trim()
* strips the ASCII set plus Unicode separator characters (\pZ) so
* whitespace-only-line handling matches the TS reference.
* - Rounding: PHP round() rounds halfway cases away from zero, which for
* the non-negative numbers used here is exactly JS Math.round.
*/
declare(strict_types=1);
/** Content classification of a single line. String values so the returned
* estimate array has the same keys/values as the TS object. */
const CONTENT_TYPE_PROSE = 'prose';
const CONTENT_TYPE_CODE = 'code';
const CONTENT_TYPE_JSON = 'json';
const CONTENT_TYPE_CJK = 'cjk';
const CONTENT_TYPE_AUTO = 'auto';
/**
* Average characters per token, by content type.
* Mirrors CHARS_PER_TOKEN in the TS lib.
*/
const CHARS_PER_TOKEN = [
'prose' => 4.0,
'code' => 3.5,
'json' => 3.0,
'cjk' => 1.5,
];
/** Reported estimate band on each side of the point estimate (ESTIMATE_TOLERANCE). */
const ESTIMATE_TOLERANCE = 0.15;
/** Chat wrappers (role markers, delimiters) cost roughly this much per message. */
const CHAT_FRAMING_TOKENS_PER_MESSAGE = 5;
/** Scan order used to resolve the majority type; ties keep the earlier entry. */
const CONTENT_TYPES = ['prose', 'code', 'json', 'cjk'];
/**
* Length of a string in UTF-16 code units — the unit TS's String.length
* counts. BMP code points are one unit, astral-plane ones two.
*/
function te_utf16_len(string $s): int
{
$n = 0;
foreach (mb_str_split($s, 1, 'UTF-8') as $ch) {
$n += mb_ord($ch, 'UTF-8') > 0xFFFF ? 2 : 1;
}
return $n;
}
/**
* Unicode-aware trim: strips ASCII whitespace plus Unicode separators
* (mirrors JS String.prototype.trim more closely than plain trim()).
*/
function te_trim(string $s): string
{
$trimmed = preg_replace('/^[\s\pZ]+|[\s\pZ]+$/u', '', $s);
return $trimmed ?? '';
}
/**
* Classify a single line by its shape. Order: json, cjk, code, prose.
* Mirrors detectLineType() in the TS lib.
*/
function detect_line_type(string $line): string
{
$trimmed = te_trim($line);
// JSON-ish: opens like a JSON fragment AND carries a separator.
$first = $trimmed !== '' ? mb_substr($trimmed, 0, 1, 'UTF-8') : '';
if (($first === '{' || $first === '}' || $first === '[' || $first === '"')
&& (str_contains($line, ':') || str_contains($line, ','))) {
return CONTENT_TYPE_JSON;
}
// CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul
// syllables (U+AC00..U+D7AF). Mirrors CJK_RE = /[一-鿿-ヿ가-]/.
if (preg_match('/[\x{4E00}-\x{9FFF}\x{3040}-\x{30FF}\x{AC00}-\x{D7AF}]/u', $line) === 1) {
return CONTENT_TYPE_CJK;
}
// Code: symbol-dense, or a statement terminator / block opener at EOL.
// Symbols are ASCII, so they never occur inside a UTF-8 multibyte
// sequence — a byte-oriented match count is exact. Mirrors
// CODE_SYMBOL_RE = /[{}();=<>\[\]#]/g.
$length = te_utf16_len($line);
$symbols = (int) preg_match_all('/[{}();=<>\[\]#]/', $line);
$density = $length > 0 ? $symbols / $length : 0.0;
if ($density > 0.08
|| str_ends_with($trimmed, ';')
|| str_ends_with($trimmed, '{')
|| str_ends_with($trimmed, '}')) {
return CONTENT_TYPE_CODE;
}
return CONTENT_TYPE_PROSE;
}
/**
* Whole-text JSON gate: a document that parses as JSON is json all the way
* down. Mirrors isValidJson() (JSON.parse in a try/catch); empty/whitespace
* text is not. json_decode with JSON_THROW_ON_ERROR gives the same strict
* grammar (no trailing commas, no NaN/Infinity, no comments).
*/
function te_is_valid_json(string $text): bool
{
if (te_trim($text) === '') {
return false;
}
try {
json_decode($text, null, 512, JSON_THROW_ON_ERROR);
return true;
} catch (JsonException) {
return false;
}
}
/**
* Estimate the LLM token count of text without running a tokenizer.
* Mirrors estimateTokens() in the TS lib and must agree with it on every
* shared vector. Never throws.
*
* Options array (all keys optional, mirroring the TS EstimateOptions):
* - contentType: 'prose' | 'code' | 'json' | 'cjk' to force, or 'auto' to
* detect per line (the default).
* - messages: int, chat messages the text will be sent as (adds framing
* tokens). Default 0.
*
* @param array{contentType?: string, messages?: int} $options
* @return array{tokens: int, low: int, high: int, chars: int, words: int,
* lines: int, contentType: string,
* breakdown: array{prose: int, code: int, json: int, cjk: int},
* framingTokens: int}
*/
function estimate_tokens(string $text, array $options = []): array
{
$forced = isset($options['contentType']) && $options['contentType'] !== CONTENT_TYPE_AUTO
? $options['contentType']
: null;
// AUTO + whole-text JSON: json's 3 chars/token rate applies to every
// line, not just the reported contentType.
$whole_text_json = $forced === null && te_is_valid_json($text);
$all_lines = preg_split('/\r?\n/', $text) ?: [];
$non_empty = [];
foreach ($all_lines as $line) {
if (te_trim($line) !== '') {
$non_empty[] = $line;
}
}
$breakdown = ['prose' => 0, 'code' => 0, 'json' => 0, 'cjk' => 0];
$tokens = 0;
$chars = 0;
foreach ($all_lines as $line) {
$chars += te_utf16_len($line);
}
foreach ($non_empty as $line) {
$type = $forced ?? ($whole_text_json ? CONTENT_TYPE_JSON : detect_line_type($line));
$line_tokens = max(1, (int) round(te_utf16_len($line) / CHARS_PER_TOKEN[$type]));
$tokens += $line_tokens;
$breakdown[$type] += $line_tokens;
}
// Resolved type = the line type holding the most token mass (ties stay
// 'prose', the first entry of CONTENT_TYPES).
$content_type = CONTENT_TYPE_PROSE;
foreach (CONTENT_TYPES as $type) {
if ($breakdown[$type] > $breakdown[$content_type]) {
$content_type = $type;
}
}
$trimmed_text = te_trim($text);
$words = 0;
if ($trimmed_text !== '') {
// Mirrors trimmedText.split(/\s+/).filter(Boolean).length: split on
// whitespace runs, drop the leading empty field.
foreach (preg_split('/[\s\pZ]+/u', $trimmed_text) ?: [] as $word) {
if ($word !== '') {
$words++;
}
}
}
return [
'tokens' => $tokens,
'low' => (int) round($tokens * (1 - ESTIMATE_TOLERANCE)),
'high' => (int) round($tokens * (1 + ESTIMATE_TOLERANCE)),
'chars' => $chars,
'words' => $words,
'lines' => count($non_empty),
'contentType' => $content_type,
'breakdown' => $breakdown,
'framingTokens' => ($options['messages'] ?? 0) * CHAT_FRAMING_TOKENS_PER_MESSAGE,
];
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →