Token Estimator — Rust source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! token-estimator — LLM token-count estimation heuristics.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source: CosmoDev polyglot showcase port of the Token Estimator tool, ported from
//! src/lib/tokenEstimator.ts (the canonical TypeScript implementation).
//! Live at: https://dev.cosmolabs.org/tools/token-estimator
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; never panics (public API returns plain values).
//! - Functionally equivalent to the TS reference: same inputs -> same outputs.
//! - Self-contained: std only (no crates.io dependencies — no `serde_json`,
//! no `regex`, no tokenizer).
//!
//! Heuristic: each line is classified (prose / code / json / cjk) and divided
//! by that type's chars-per-token rate; the result carries a ±15% band
//! because real BPE tokenizers vary by vocabulary and language mix.
//!
//! Faithfulness notes (the places Rust's std silently differs from JS):
//! - Length: TS's `String.length` counts UTF-16 code units (an astral-plane
//! character — emoji, rare CJK ext-B ideographs — counts as 2). Rust
//! `str::chars()` counts Unicode scalar values, so line arithmetic goes
//! through `s.encode_utf16().count()` to count the same unit.
//! - JSON: std has no JSON parser and `serde_json` is off-limits, so this
//! port ships a small strict recursive-descent validator (`is_valid_json`)
//! implementing exactly the grammar `JSON.parse` accepts — values,
//! objects, arrays, quoted strings with escapes, JSON numbers, and the
//! three literals — with no trailing commas, comments, NaN/Infinity, or
//! trailing garbage. Whole-text JSON detection therefore behaves
//! identically, not approximately.
//! - Rounding: `f64::round` rounds halfway cases away from zero, which for
//! the non-negative numbers used here is exactly JS `Math.round`.
use std::fmt;
/// Content classification of a single line. Mirrors the TS union
/// `'prose' | 'code' | 'json' | 'cjk'`; `as_str()` keeps the string form the
/// TS object carries.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
pub enum ContentType {
/// Plain natural-language text (the default classification).
#[default]
Prose,
/// Symbol-dense source code.
Code,
/// JSON objects, arrays, and key/value lines.
Json,
/// CJK ideographs, kana, or Hangul.
Cjk,
}
impl ContentType {
/// The TS string literal for this type.
pub fn as_str(self) -> &'static str {
match self {
ContentType::Prose => "prose",
ContentType::Code => "code",
ContentType::Json => "json",
ContentType::Cjk => "cjk",
}
}
/// Average characters per token for this type. Mirrors `CHARS_PER_TOKEN`
/// in the TS lib (prose 4, code 3.5, json 3, cjk 1.5).
pub fn chars_per_token(self) -> f64 {
match self {
ContentType::Prose => 4.0,
ContentType::Code => 3.5,
ContentType::Json => 3.0,
ContentType::Cjk => 1.5,
}
}
}
impl fmt::Display for ContentType {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.write_str(self.as_str())
}
}
/// Reported estimate band on each side of the point estimate.
/// Mirrors `ESTIMATE_TOLERANCE`.
pub const ESTIMATE_TOLERANCE: f64 = 0.15;
/// Chat wrappers (role markers, delimiters) cost roughly this much per
/// message. Mirrors `CHAT_FRAMING_TOKENS_PER_MESSAGE`.
pub const CHAT_FRAMING_TOKENS_PER_MESSAGE: usize = 5;
/// Per-type token mass. Mirrors the TS `Record<ContentType, number>` (same
/// four fields, all starting at zero).
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
pub struct Breakdown {
/// Tokens on lines classified as prose.
pub prose: usize,
/// Tokens on lines classified as code.
pub code: usize,
/// Tokens on lines classified as json.
pub json: usize,
/// Tokens on lines classified as cjk.
pub cjk: usize,
}
impl Breakdown {
fn mass(&self, t: ContentType) -> usize {
match t {
ContentType::Prose => self.prose,
ContentType::Code => self.code,
ContentType::Json => self.json,
ContentType::Cjk => self.cjk,
}
}
fn add(&mut self, t: ContentType, tokens: usize) {
match t {
ContentType::Prose => self.prose += tokens,
ContentType::Code => self.code += tokens,
ContentType::Json => self.json += tokens,
ContentType::Cjk => self.cjk += tokens,
}
}
}
/// Result of `estimate_tokens`. Field-for-field twin of the TS
/// `TokenEstimate` interface.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct TokenEstimate {
/// Sum of per-line estimates (excludes framing).
pub tokens: usize,
/// `round(tokens * (1 - ESTIMATE_TOLERANCE))`.
pub low: usize,
/// `round(tokens * (1 + ESTIMATE_TOLERANCE))`.
pub high: usize,
/// Total characters excluding newlines (UTF-16 code units).
pub chars: usize,
/// Whitespace-split word count.
pub words: usize,
/// Non-empty line count.
pub lines: usize,
/// Majority of per-line token mass.
pub content_type: ContentType,
/// Tokens per detected line type (others stay 0).
pub breakdown: Breakdown,
/// `messages * CHAT_FRAMING_TOKENS_PER_MESSAGE`.
pub framing_tokens: usize,
}
/// Options mirror the TS `EstimateOptions`. The zero value matches the TS
/// default: auto detection (`content_type: None`) with no chat framing.
#[derive(Debug, Clone, Default)]
pub struct Options {
/// Force a content type, or `None` to detect per line (TS `'auto'`).
pub content_type: Option<ContentType>,
/// Chat messages the text will be sent as (adds framing tokens).
pub messages: usize,
}
/// Length of `s` in UTF-16 code units — the unit TS's `String.length`
/// counts. BMP code points are one unit, astral-plane ones two.
fn utf16_len(s: &str) -> usize {
s.encode_utf16().count()
}
/// Reports whether `s` contains a CJK ideograph (U+4E00–U+9FFF), kana
/// (U+3040–U+30FF), or a Hangul syllable (U+AC00–U+D7AF). Mirrors `CJK_RE`
/// in the TS lib.
fn has_cjk(s: &str) -> bool {
s.chars().any(|c| {
('\u{4E00}'..='\u{9FFF}').contains(&c)
|| ('\u{3040}'..='\u{30FF}').contains(&c)
|| ('\u{AC00}'..='\u{D7AF}').contains(&c)
})
}
/// Reports whether `c` is one of the code-flavored symbols counted by
/// `CODE_SYMBOL_RE` (`{}();=<>[]#`).
fn is_code_symbol(c: char) -> bool {
matches!(c, '{' | '}' | '(' | ')' | ';' | '=' | '<' | '>' | '[' | ']' | '#')
}
/// Splits `text` on LF or CRLF, mirroring `text.split(/\r?\n/)`: strip the
/// optional CR that belongs to the newline, then split on LF. A lone CR is
/// NOT a line break.
fn split_lines(text: &str) -> Vec<&str> {
text.split('\n')
.map(|line| line.strip_suffix('\r').unwrap_or(line))
.collect()
}
/// Classifies a single line by its shape. Order: json, cjk, code, prose.
/// Mirrors `detectLineType()` in the TS lib.
pub fn detect_line_type(line: &str) -> ContentType {
let trimmed = line.trim();
// JSON-ish: opens like a JSON fragment AND carries a separator.
let starts_jsonish = trimmed.starts_with('{')
|| trimmed.starts_with('}')
|| trimmed.starts_with('[')
|| trimmed.starts_with('"');
if starts_jsonish && (line.contains(':') || line.contains(',')) {
return ContentType::Json;
}
// CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
if has_cjk(line) {
return ContentType::Cjk;
}
// Code: symbol-dense, or a statement terminator / block opener at EOL.
let length = utf16_len(line);
let symbols = line.chars().filter(|c| is_code_symbol(*c)).count();
let density = if length > 0 { symbols as f64 / length as f64 } else { 0.0 };
if density > 0.08 || trimmed.ends_with(';') || trimmed.ends_with('{') || trimmed.ends_with('}')
{
return ContentType::Code;
}
ContentType::Prose
}
/// A strict JSON syntax validator — the exact grammar `JSON.parse` accepts,
/// walked with a byte cursor. (Multibyte UTF-8 inside strings never contains
/// an ASCII byte, so byte-level scanning is safe.)
struct JsonParser<'a> {
bytes: &'a [u8],
pos: usize,
}
impl<'a> JsonParser<'a> {
fn new(text: &'a str) -> Self {
JsonParser { bytes: text.as_bytes(), pos: 0 }
}
fn skip_ws(&mut self) {
while let Some(&b) = self.bytes.get(self.pos) {
if b == b' ' || b == b'\t' || b == b'\n' || b == b'\r' {
self.pos += 1;
} else {
break;
}
}
}
fn peek(&self) -> Option<u8> {
self.bytes.get(self.pos).copied()
}
fn eat(&mut self, b: u8) -> bool {
if self.peek() == Some(b) {
self.pos += 1;
true
} else {
false
}
}
fn literal(&mut self, lit: &[u8]) -> bool {
if self.bytes[self.pos..].starts_with(lit) {
self.pos += lit.len();
true
} else {
false
}
}
/// value := ws* (object | array | string | number | 'true' | 'false' | 'null') ws*
fn value(&mut self) -> bool {
self.skip_ws();
match self.peek() {
Some(b'{') => self.object(),
Some(b'[') => self.array(),
Some(b'"') => self.string(),
Some(b'-') | Some(b'0'..=b'9') => self.number(),
Some(b't') => self.literal(b"true"),
Some(b'f') => self.literal(b"false"),
Some(b'n') => self.literal(b"null"),
_ => false,
}
}
/// object := '{' ws* (string ws* ':' value (ws* ',' ws* string ws* ':' value)*)? ws* '}'
fn object(&mut self) -> bool {
if !self.eat(b'{') {
return false;
}
self.skip_ws();
if self.eat(b'}') {
return true;
}
loop {
if !self.string() {
return false;
}
self.skip_ws();
if !self.eat(b':') {
return false;
}
if !self.value() {
return false;
}
self.skip_ws();
if self.eat(b',') {
self.skip_ws();
} else {
return self.eat(b'}');
}
}
}
/// array := '[' ws* (value (ws* ',' ws* value)*)? ws* ']'
fn array(&mut self) -> bool {
if !self.eat(b'[') {
return false;
}
self.skip_ws();
if self.eat(b']') {
return true;
}
loop {
if !self.value() {
return false;
}
self.skip_ws();
if self.eat(b',') {
self.skip_ws();
} else {
return self.eat(b']');
}
}
}
/// string := '"' (escape | any byte >= 0x20)* '"'
/// escape := '\' ("\"" | "/" | "\" | 'b' | 'f' | 'n' | 'r' | 't' | 'u' hex hex hex hex)
fn string(&mut self) -> bool {
if !self.eat(b'"') {
return false;
}
while let Some(b) = self.peek() {
match b {
b'"' => {
self.pos += 1;
return true;
}
b'\\' => {
self.pos += 1;
let Some(esc) = self.peek() else { return false };
self.pos += 1;
match esc {
b'"' | b'/' | b'\\' | b'b' | b'f' | b'n' | b'r' | b't' => {}
b'u' => {
for _ in 0..4 {
match self.peek() {
Some(h @ (b'0'..=b'9' | b'a'..=b'f' | b'A'..=b'F')) => {
self.pos += 1;
let _ = h;
}
_ => return false,
}
}
}
_ => return false,
}
}
// Raw control characters are not allowed inside strings.
0x00..=0x1F => return false,
_ => self.pos += 1,
}
}
false // unterminated string
}
/// number := '-'? int frac? exp?
/// int := '0' | [1-9][0-9]* (no leading zeros, like JSON.parse)
/// frac := '.' [0-9]+ ; exp := [eE] [+-]? [0-9]+
fn number(&mut self) -> bool {
self.eat(b'-');
match self.peek() {
Some(b'0') => self.pos += 1,
Some(b'1'..=b'9') => {
while matches!(self.peek(), Some(b'0'..=b'9')) {
self.pos += 1;
}
}
_ => return false,
}
if self.peek() == Some(b'.') {
self.pos += 1;
let mut digits = 0;
while matches!(self.peek(), Some(b'0'..=b'9')) {
self.pos += 1;
digits += 1;
}
if digits == 0 {
return false;
}
}
if matches!(self.peek(), Some(b'e' | b'E')) {
self.pos += 1;
if matches!(self.peek(), Some(b'+' | b'-')) {
self.pos += 1;
}
let mut digits = 0;
while matches!(self.peek(), Some(b'0'..=b'9')) {
self.pos += 1;
digits += 1;
}
if digits == 0 {
return false;
}
}
true
}
}
/// Whole-text JSON gate: a document that parses as JSON is json all the way
/// down. Mirrors `isValidJson()` (`JSON.parse` in a try/catch);
/// empty/whitespace text is not.
fn is_valid_json(text: &str) -> bool {
if text.trim().is_empty() {
return false;
}
let mut p = JsonParser::new(text);
p.value() && {
p.skip_ws();
p.pos == p.bytes.len() // reject trailing garbage
}
}
/// Estimates the LLM token count of `text` without running a tokenizer.
/// Mirrors `estimateTokens()` in the TS lib and must agree with it on every
/// shared vector. `None` options means auto detection with no framing.
pub fn estimate_tokens(text: &str, options: Option<Options>) -> TokenEstimate {
let options = options.unwrap_or_default();
let forced = options.content_type;
// AUTO + whole-text JSON: a document that parses as JSON is json all the
// way down — json's 3 chars/token rate applies to every line, not just
// the reported content type.
let whole_text_json = forced.is_none() && is_valid_json(text);
let all_lines = split_lines(text);
let non_empty: Vec<&str> = all_lines.iter().copied().filter(|l| !l.trim().is_empty()).collect();
let mut breakdown = Breakdown::default();
let mut tokens = 0usize;
for &line in &non_empty {
let t = forced.unwrap_or_else(|| {
if whole_text_json {
ContentType::Json
} else {
detect_line_type(line)
}
});
let line_tokens = ((utf16_len(line) as f64 / t.chars_per_token()) as f64)
.round()
.max(1.0) as usize;
tokens += line_tokens;
breakdown.add(t, line_tokens);
}
// Resolved type = the line type holding the most token mass (ties stay
// prose, the first entry of the scan order).
let scan = [ContentType::Prose, ContentType::Code, ContentType::Json, ContentType::Cjk];
let mut content_type = ContentType::Prose;
for t in scan {
if breakdown.mass(t) > breakdown.mass(content_type) {
content_type = t;
}
}
let trimmed = text.trim();
let words = if trimmed.is_empty() {
0
} else {
// Mirrors trimmedText.split(/\s+/).filter(Boolean).length.
trimmed.split_whitespace().count()
};
TokenEstimate {
tokens,
low: (tokens as f64 * (1.0 - ESTIMATE_TOLERANCE)).round() as usize,
high: (tokens as f64 * (1.0 + ESTIMATE_TOLERANCE)).round() as usize,
chars: all_lines.iter().map(|l| utf16_len(l)).sum(),
words,
lines: non_empty.len(),
content_type,
breakdown,
framing_tokens: options.messages * CHAT_FRAMING_TOKENS_PER_MESSAGE,
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →