Token Estimator — C# source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.
// token-estimator — C# port: tokenizer-free LLM token estimation.
//
// Display snippet: ports the line classifier and estimator core from the
// TypeScript lib (src/lib/tokenEstimator.ts). Each non-empty line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate; the estimate carries a ±15% band. C# strings are
// UTF-16, so string.Length is the same unit the TS reference counts; the
// full result shape and the whole-text JSON gate live in TS/Go.
using System;
using System.Collections.Generic;
using System.Linq;
namespace TokenEstimator;
public enum ContentType { Prose, Code, Json, Cjk }
public static class TokenEstimator
{
// Average chars per token by content type (CHARS_PER_TOKEN in TS).
private static double CharsPerToken(ContentType t) => t switch
{
ContentType.Json => 3.0, ContentType.Cjk => 1.5, ContentType.Code => 3.5, _ => 4.0,
}; // _ = Prose, the TS fallback type.
private const double EstimateTolerance = 0.15;
// CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul
// (U+AC00..U+D7AF) sit in the BMP: plain chars suffice.
private static bool HasCjk(string s) =>
s.Any(c => (c >= 0x4E00 && c <= 0x9FFF) || (c >= 0x3040 && c <= 0x30FF) ||
(c >= 0xAC00 && c <= 0xD7AF));
// Classify a line by its shape. Order: json, cjk, code, prose.
public static ContentType DetectLineType(string line)
{
string t = line.Trim();
char h = t.Length > 0 ? t[0] : '\0', e = t.Length > 0 ? t[^1] : '\0';
if ((h == '{' || h == '}' || h == '[' || h == '"') &&
(line.Contains(':') || line.Contains(','))) return ContentType.Json;
if (HasCjk(line)) return ContentType.Cjk;
int symbols = line.Count("{}();=<>[]#".Contains);
if ((line.Length > 0 && (double)symbols / line.Length > 0.08) ||
e == ';' || e == '{' || e == '}') return ContentType.Code;
return ContentType.Prose;
}
public sealed record Estimate(double Tokens, double Low, double High, ContentType Dominant,
IReadOnlyDictionary<ContentType, double> Breakdown);
// Sum per-line estimates for every non-empty line of text.
public static Estimate EstimateTokens(string text)
{
var breakdown = new Dictionary<ContentType, double> { [ContentType.Prose] = 0,
[ContentType.Code] = 0, [ContentType.Json] = 0, [ContentType.Cjk] = 0 };
double tokens = 0;
foreach (string line in text.Replace("\r\n", "\n").Split('\n'))
{
if (line.Trim().Length == 0) continue;
ContentType ty = DetectLineType(line);
double lt = Math.Round(line.Length / CharsPerToken(ty));
if (lt < 1.0) lt = 1.0; // max(1, round(len / rate))
tokens += lt;
breakdown[ty] += lt;
}
// Dominant type: strictly-greater scan keeps ties on Prose, as in TS.
var dominant = ContentType.Prose;
foreach (ContentType ty in new[] { ContentType.Prose, ContentType.Code,
ContentType.Json, ContentType.Cjk })
if (breakdown[ty] > breakdown[dominant]) dominant = ty;
return new Estimate(tokens,
Math.Round(tokens * (1.0 - EstimateTolerance)),
Math.Round(tokens * (1.0 + EstimateTolerance)), dominant, breakdown);
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →