Token Estimator — C++ source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// token-estimator — C++ port: tokenizer-free LLM token estimation.
//
// Display snippet: ports the line classifier and estimator core from the
// TypeScript lib (src/lib/tokenEstimator.ts) — each non-empty line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate, with a ±15% band. string_view counts bytes here
// (TS counts UTF-16 units); the full result shape and whole-text JSON
// gate live in TS/Go.
#include <cmath>
#include <string_view>
enum class ContentType { prose, code, json, cjk };
constexpr double kEstimateTolerance = 0.15;
inline double charsPerToken(ContentType t) { // avg chars/token (CHARS_PER_TOKEN in TS)
switch (t) {
case ContentType::json: return 3.0; case ContentType::cjk: return 1.5;
case ContentType::code: return 3.5; default: return 4.0; // prose — TS fallback
}
}
// CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul (U+AC00..U+D7AF).
bool hasCjk(std::string_view s) {
for (size_t i = 0; i < s.size();) {
unsigned char b = (unsigned char)s[i];
size_t n = b < 0x80 ? 1 : (b & 0xE0) == 0xC0 ? 2 : (b & 0xF0) == 0xE0 ? 3 : 4;
if (i + n > s.size()) break; // truncated tail: stop, treat as non-CJK
char32_t r = b & (n == 1 ? 0x7F : n == 2 ? 0x1F : n == 3 ? 0x0F : 0x07);
for (size_t k = 1; k < n; ++k) r = (r << 6) | ((unsigned char)s[i + k] & 0x3F);
if ((r >= 0x4E00 && r <= 0x9FFF) || (r >= 0x3040 && r <= 0x30FF) ||
(r >= 0xAC00 && r <= 0xD7AF)) return true;
i += n;
}
return false;
}
// Classify a line by its shape. Order: json, cjk, code, prose.
ContentType detectLineType(std::string_view line) {
size_t first = line.find_first_not_of(" \t\r");
std::string_view t = first == std::string_view::npos ? std::string_view{} : line.substr(first);
char h = t.empty() ? '\0' : t.front(), e = t.empty() ? '\0' : t.back();
if ((h == '{' || h == '}' || h == '[' || h == '"') &&
(line.find(':') != std::string_view::npos || line.find(',') != std::string_view::npos))
return ContentType::json;
if (hasCjk(line)) return ContentType::cjk;
int symbols = 0;
for (char c : line) if (std::string_view{"{}();=<>[]#"}.find(c) != std::string_view::npos) ++symbols;
bool dense = !line.empty() && double(symbols) / double(line.size()) > 0.08;
if (dense || e == ';' || e == '{' || e == '}') return ContentType::code;
return ContentType::prose;
}
struct Estimate { // tokens ± the band; dominant line type; per-type breakdown
double tokens, low, high; ContentType dominant; // dominant = most token mass
double breakdown[4]; // per type, index = ContentType order
};
// Sum per-line estimates for every non-empty line of text.
Estimate estimateTokens(std::string_view text) {
Estimate est{0, 0, 0, ContentType::prose, {0, 0, 0, 0}};
for (size_t start = 0; start <= text.size();) { // split on \n, tolerating \r
size_t nl = text.find('\n', start);
std::string_view line = text.substr(start, nl == std::string_view::npos ? nl : nl - start);
if (!line.empty() && line.back() == '\r') line.remove_suffix(1);
if (line.find_first_not_of(" \t\r") != std::string_view::npos) {
ContentType ty = detectLineType(line);
double lt = std::round(double(line.size()) / charsPerToken(ty));
if (lt < 1.0) lt = 1.0; // max(1, round(len / rate))
est.tokens += lt;
est.breakdown[size_t(ty)] += lt;
}
if (nl == std::string_view::npos) break;
start = nl + 1;
}
for (int ty = 1; ty < 4; ++ty) // dominant: strictly-greater scan, ties stay on prose
if (est.breakdown[ty] > est.breakdown[size_t(est.dominant)]) est.dominant = ContentType(ty);
est.low = std::round(est.tokens * (1.0 - kEstimateTolerance));
est.high = std::round(est.tokens * (1.0 + kEstimateTolerance));
return est;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →