Skip to content

Token Estimator — Rust source

Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! token-estimator — LLM token-count estimation heuristics.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source:   CosmoDev polyglot showcase port of the Token Estimator tool, ported from
//!           src/lib/tokenEstimator.ts (the canonical TypeScript implementation).
//! Live at:  https://dev.cosmolabs.org/tools/token-estimator
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; never panics (public API returns plain values).
//!   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//!   - Self-contained: std only (no crates.io dependencies — no `serde_json`,
//!     no `regex`, no tokenizer).
//!
//! Heuristic: each line is classified (prose / code / json / cjk) and divided
//! by that type's chars-per-token rate; the result carries a ±15% band
//! because real BPE tokenizers vary by vocabulary and language mix.
//!
//! Faithfulness notes (the places Rust's std silently differs from JS):
//!   - Length: TS's `String.length` counts UTF-16 code units (an astral-plane
//!     character — emoji, rare CJK ext-B ideographs — counts as 2). Rust
//!     `str::chars()` counts Unicode scalar values, so line arithmetic goes
//!     through `s.encode_utf16().count()` to count the same unit.
//!   - JSON: std has no JSON parser and `serde_json` is off-limits, so this
//!     port ships a small strict recursive-descent validator (`is_valid_json`)
//!     implementing exactly the grammar `JSON.parse` accepts — values,
//!     objects, arrays, quoted strings with escapes, JSON numbers, and the
//!     three literals — with no trailing commas, comments, NaN/Infinity, or
//!     trailing garbage. Whole-text JSON detection therefore behaves
//!     identically, not approximately.
//!   - Rounding: `f64::round` rounds halfway cases away from zero, which for
//!     the non-negative numbers used here is exactly JS `Math.round`.

use std::fmt;

/// Content classification of a single line. Mirrors the TS union
/// `'prose' | 'code' | 'json' | 'cjk'`; `as_str()` keeps the string form the
/// TS object carries.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
pub enum ContentType {
    /// Plain natural-language text (the default classification).
    #[default]
    Prose,
    /// Symbol-dense source code.
    Code,
    /// JSON objects, arrays, and key/value lines.
    Json,
    /// CJK ideographs, kana, or Hangul.
    Cjk,
}

impl ContentType {
    /// The TS string literal for this type.
    pub fn as_str(self) -> &'static str {
        match self {
            ContentType::Prose => "prose",
            ContentType::Code => "code",
            ContentType::Json => "json",
            ContentType::Cjk => "cjk",
        }
    }

    /// Average characters per token for this type. Mirrors `CHARS_PER_TOKEN`
    /// in the TS lib (prose 4, code 3.5, json 3, cjk 1.5).
    pub fn chars_per_token(self) -> f64 {
        match self {
            ContentType::Prose => 4.0,
            ContentType::Code => 3.5,
            ContentType::Json => 3.0,
            ContentType::Cjk => 1.5,
        }
    }
}

impl fmt::Display for ContentType {
    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
        f.write_str(self.as_str())
    }
}

/// Reported estimate band on each side of the point estimate.
/// Mirrors `ESTIMATE_TOLERANCE`.
pub const ESTIMATE_TOLERANCE: f64 = 0.15;

/// Chat wrappers (role markers, delimiters) cost roughly this much per
/// message. Mirrors `CHAT_FRAMING_TOKENS_PER_MESSAGE`.
pub const CHAT_FRAMING_TOKENS_PER_MESSAGE: usize = 5;

/// Per-type token mass. Mirrors the TS `Record<ContentType, number>` (same
/// four fields, all starting at zero).
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
pub struct Breakdown {
    /// Tokens on lines classified as prose.
    pub prose: usize,
    /// Tokens on lines classified as code.
    pub code: usize,
    /// Tokens on lines classified as json.
    pub json: usize,
    /// Tokens on lines classified as cjk.
    pub cjk: usize,
}

impl Breakdown {
    fn mass(&self, t: ContentType) -> usize {
        match t {
            ContentType::Prose => self.prose,
            ContentType::Code => self.code,
            ContentType::Json => self.json,
            ContentType::Cjk => self.cjk,
        }
    }

    fn add(&mut self, t: ContentType, tokens: usize) {
        match t {
            ContentType::Prose => self.prose += tokens,
            ContentType::Code => self.code += tokens,
            ContentType::Json => self.json += tokens,
            ContentType::Cjk => self.cjk += tokens,
        }
    }
}

/// Result of `estimate_tokens`. Field-for-field twin of the TS
/// `TokenEstimate` interface.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct TokenEstimate {
    /// Sum of per-line estimates (excludes framing).
    pub tokens: usize,
    /// `round(tokens * (1 - ESTIMATE_TOLERANCE))`.
    pub low: usize,
    /// `round(tokens * (1 + ESTIMATE_TOLERANCE))`.
    pub high: usize,
    /// Total characters excluding newlines (UTF-16 code units).
    pub chars: usize,
    /// Whitespace-split word count.
    pub words: usize,
    /// Non-empty line count.
    pub lines: usize,
    /// Majority of per-line token mass.
    pub content_type: ContentType,
    /// Tokens per detected line type (others stay 0).
    pub breakdown: Breakdown,
    /// `messages * CHAT_FRAMING_TOKENS_PER_MESSAGE`.
    pub framing_tokens: usize,
}

/// Options mirror the TS `EstimateOptions`. The zero value matches the TS
/// default: auto detection (`content_type: None`) with no chat framing.
#[derive(Debug, Clone, Default)]
pub struct Options {
    /// Force a content type, or `None` to detect per line (TS `'auto'`).
    pub content_type: Option<ContentType>,
    /// Chat messages the text will be sent as (adds framing tokens).
    pub messages: usize,
}

/// Length of `s` in UTF-16 code units — the unit TS's `String.length`
/// counts. BMP code points are one unit, astral-plane ones two.
fn utf16_len(s: &str) -> usize {
    s.encode_utf16().count()
}

/// Reports whether `s` contains a CJK ideograph (U+4E00–U+9FFF), kana
/// (U+3040–U+30FF), or a Hangul syllable (U+AC00–U+D7AF). Mirrors `CJK_RE`
/// in the TS lib.
fn has_cjk(s: &str) -> bool {
    s.chars().any(|c| {
        ('\u{4E00}'..='\u{9FFF}').contains(&c)
            || ('\u{3040}'..='\u{30FF}').contains(&c)
            || ('\u{AC00}'..='\u{D7AF}').contains(&c)
    })
}

/// Reports whether `c` is one of the code-flavored symbols counted by
/// `CODE_SYMBOL_RE` (`{}();=<>[]#`).
fn is_code_symbol(c: char) -> bool {
    matches!(c, '{' | '}' | '(' | ')' | ';' | '=' | '<' | '>' | '[' | ']' | '#')
}

/// Splits `text` on LF or CRLF, mirroring `text.split(/\r?\n/)`: strip the
/// optional CR that belongs to the newline, then split on LF. A lone CR is
/// NOT a line break.
fn split_lines(text: &str) -> Vec<&str> {
    text.split('\n')
        .map(|line| line.strip_suffix('\r').unwrap_or(line))
        .collect()
}

/// Classifies a single line by its shape. Order: json, cjk, code, prose.
/// Mirrors `detectLineType()` in the TS lib.
pub fn detect_line_type(line: &str) -> ContentType {
    let trimmed = line.trim();
    // JSON-ish: opens like a JSON fragment AND carries a separator.
    let starts_jsonish = trimmed.starts_with('{')
        || trimmed.starts_with('}')
        || trimmed.starts_with('[')
        || trimmed.starts_with('"');
    if starts_jsonish && (line.contains(':') || line.contains(',')) {
        return ContentType::Json;
    }
    // CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
    if has_cjk(line) {
        return ContentType::Cjk;
    }
    // Code: symbol-dense, or a statement terminator / block opener at EOL.
    let length = utf16_len(line);
    let symbols = line.chars().filter(|c| is_code_symbol(*c)).count();
    let density = if length > 0 { symbols as f64 / length as f64 } else { 0.0 };
    if density > 0.08 || trimmed.ends_with(';') || trimmed.ends_with('{') || trimmed.ends_with('}')
    {
        return ContentType::Code;
    }
    ContentType::Prose
}

/// A strict JSON syntax validator — the exact grammar `JSON.parse` accepts,
/// walked with a byte cursor. (Multibyte UTF-8 inside strings never contains
/// an ASCII byte, so byte-level scanning is safe.)
struct JsonParser<'a> {
    bytes: &'a [u8],
    pos: usize,
}

impl<'a> JsonParser<'a> {
    fn new(text: &'a str) -> Self {
        JsonParser { bytes: text.as_bytes(), pos: 0 }
    }

    fn skip_ws(&mut self) {
        while let Some(&b) = self.bytes.get(self.pos) {
            if b == b' ' || b == b'\t' || b == b'\n' || b == b'\r' {
                self.pos += 1;
            } else {
                break;
            }
        }
    }

    fn peek(&self) -> Option<u8> {
        self.bytes.get(self.pos).copied()
    }

    fn eat(&mut self, b: u8) -> bool {
        if self.peek() == Some(b) {
            self.pos += 1;
            true
        } else {
            false
        }
    }

    fn literal(&mut self, lit: &[u8]) -> bool {
        if self.bytes[self.pos..].starts_with(lit) {
            self.pos += lit.len();
            true
        } else {
            false
        }
    }

    /// value := ws* (object | array | string | number | 'true' | 'false' | 'null') ws*
    fn value(&mut self) -> bool {
        self.skip_ws();
        match self.peek() {
            Some(b'{') => self.object(),
            Some(b'[') => self.array(),
            Some(b'"') => self.string(),
            Some(b'-') | Some(b'0'..=b'9') => self.number(),
            Some(b't') => self.literal(b"true"),
            Some(b'f') => self.literal(b"false"),
            Some(b'n') => self.literal(b"null"),
            _ => false,
        }
    }

    /// object := '{' ws* (string ws* ':' value (ws* ',' ws* string ws* ':' value)*)? ws* '}'
    fn object(&mut self) -> bool {
        if !self.eat(b'{') {
            return false;
        }
        self.skip_ws();
        if self.eat(b'}') {
            return true;
        }
        loop {
            if !self.string() {
                return false;
            }
            self.skip_ws();
            if !self.eat(b':') {
                return false;
            }
            if !self.value() {
                return false;
            }
            self.skip_ws();
            if self.eat(b',') {
                self.skip_ws();
            } else {
                return self.eat(b'}');
            }
        }
    }

    /// array := '[' ws* (value (ws* ',' ws* value)*)? ws* ']'
    fn array(&mut self) -> bool {
        if !self.eat(b'[') {
            return false;
        }
        self.skip_ws();
        if self.eat(b']') {
            return true;
        }
        loop {
            if !self.value() {
                return false;
            }
            self.skip_ws();
            if self.eat(b',') {
                self.skip_ws();
            } else {
                return self.eat(b']');
            }
        }
    }

    /// string := '"' (escape | any byte >= 0x20)* '"'
    /// escape := '\' ("\"" | "/" | "\" | 'b' | 'f' | 'n' | 'r' | 't' | 'u' hex hex hex hex)
    fn string(&mut self) -> bool {
        if !self.eat(b'"') {
            return false;
        }
        while let Some(b) = self.peek() {
            match b {
                b'"' => {
                    self.pos += 1;
                    return true;
                }
                b'\\' => {
                    self.pos += 1;
                    let Some(esc) = self.peek() else { return false };
                    self.pos += 1;
                    match esc {
                        b'"' | b'/' | b'\\' | b'b' | b'f' | b'n' | b'r' | b't' => {}
                        b'u' => {
                            for _ in 0..4 {
                                match self.peek() {
                                    Some(h @ (b'0'..=b'9' | b'a'..=b'f' | b'A'..=b'F')) => {
                                        self.pos += 1;
                                        let _ = h;
                                    }
                                    _ => return false,
                                }
                            }
                        }
                        _ => return false,
                    }
                }
                // Raw control characters are not allowed inside strings.
                0x00..=0x1F => return false,
                _ => self.pos += 1,
            }
        }
        false // unterminated string
    }

    /// number := '-'? int frac? exp?
    /// int := '0' | [1-9][0-9]*  (no leading zeros, like JSON.parse)
    /// frac := '.' [0-9]+ ; exp := [eE] [+-]? [0-9]+
    fn number(&mut self) -> bool {
        self.eat(b'-');
        match self.peek() {
            Some(b'0') => self.pos += 1,
            Some(b'1'..=b'9') => {
                while matches!(self.peek(), Some(b'0'..=b'9')) {
                    self.pos += 1;
                }
            }
            _ => return false,
        }
        if self.peek() == Some(b'.') {
            self.pos += 1;
            let mut digits = 0;
            while matches!(self.peek(), Some(b'0'..=b'9')) {
                self.pos += 1;
                digits += 1;
            }
            if digits == 0 {
                return false;
            }
        }
        if matches!(self.peek(), Some(b'e' | b'E')) {
            self.pos += 1;
            if matches!(self.peek(), Some(b'+' | b'-')) {
                self.pos += 1;
            }
            let mut digits = 0;
            while matches!(self.peek(), Some(b'0'..=b'9')) {
                self.pos += 1;
                digits += 1;
            }
            if digits == 0 {
                return false;
            }
        }
        true
    }
}

/// Whole-text JSON gate: a document that parses as JSON is json all the way
/// down. Mirrors `isValidJson()` (`JSON.parse` in a try/catch);
/// empty/whitespace text is not.
fn is_valid_json(text: &str) -> bool {
    if text.trim().is_empty() {
        return false;
    }
    let mut p = JsonParser::new(text);
    p.value() && {
        p.skip_ws();
        p.pos == p.bytes.len() // reject trailing garbage
    }
}

/// Estimates the LLM token count of `text` without running a tokenizer.
/// Mirrors `estimateTokens()` in the TS lib and must agree with it on every
/// shared vector. `None` options means auto detection with no framing.
pub fn estimate_tokens(text: &str, options: Option<Options>) -> TokenEstimate {
    let options = options.unwrap_or_default();
    let forced = options.content_type;
    // AUTO + whole-text JSON: a document that parses as JSON is json all the
    // way down — json's 3 chars/token rate applies to every line, not just
    // the reported content type.
    let whole_text_json = forced.is_none() && is_valid_json(text);

    let all_lines = split_lines(text);
    let non_empty: Vec<&str> = all_lines.iter().copied().filter(|l| !l.trim().is_empty()).collect();

    let mut breakdown = Breakdown::default();
    let mut tokens = 0usize;
    for &line in &non_empty {
        let t = forced.unwrap_or_else(|| {
            if whole_text_json {
                ContentType::Json
            } else {
                detect_line_type(line)
            }
        });
        let line_tokens = ((utf16_len(line) as f64 / t.chars_per_token()) as f64)
            .round()
            .max(1.0) as usize;
        tokens += line_tokens;
        breakdown.add(t, line_tokens);
    }

    // Resolved type = the line type holding the most token mass (ties stay
    // prose, the first entry of the scan order).
    let scan = [ContentType::Prose, ContentType::Code, ContentType::Json, ContentType::Cjk];
    let mut content_type = ContentType::Prose;
    for t in scan {
        if breakdown.mass(t) > breakdown.mass(content_type) {
            content_type = t;
        }
    }

    let trimmed = text.trim();
    let words = if trimmed.is_empty() {
        0
    } else {
        // Mirrors trimmedText.split(/\s+/).filter(Boolean).length.
        trimmed.split_whitespace().count()
    };

    TokenEstimate {
        tokens,
        low: (tokens as f64 * (1.0 - ESTIMATE_TOLERANCE)).round() as usize,
        high: (tokens as f64 * (1.0 + ESTIMATE_TOLERANCE)).round() as usize,
        chars: all_lines.iter().map(|l| utf16_len(l)).sum(),
        words,
        lines: non_empty.len(),
        content_type,
        breakdown,
        framing_tokens: options.messages * CHAT_FRAMING_TOKENS_PER_MESSAGE,
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →