Skip to content

Token Estimator — C# source

Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.

This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.

// token-estimator — C# port: tokenizer-free LLM token estimation.
//
// Display snippet: ports the line classifier and estimator core from the
// TypeScript lib (src/lib/tokenEstimator.ts). Each non-empty line is
// classified (prose / code / json / cjk) and divided by that type's
// chars-per-token rate; the estimate carries a ±15% band. C# strings are
// UTF-16, so string.Length is the same unit the TS reference counts; the
// full result shape and the whole-text JSON gate live in TS/Go.
using System;
using System.Collections.Generic;
using System.Linq;

namespace TokenEstimator;

public enum ContentType { Prose, Code, Json, Cjk }

public static class TokenEstimator
{
    // Average chars per token by content type (CHARS_PER_TOKEN in TS).
    private static double CharsPerToken(ContentType t) => t switch
    {
        ContentType.Json => 3.0, ContentType.Cjk => 1.5, ContentType.Code => 3.5, _ => 4.0,
    }; // _ = Prose, the TS fallback type.

    private const double EstimateTolerance = 0.15;

    // CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul
    // (U+AC00..U+D7AF) sit in the BMP: plain chars suffice.
    private static bool HasCjk(string s) =>
        s.Any(c => (c >= 0x4E00 && c <= 0x9FFF) || (c >= 0x3040 && c <= 0x30FF) ||
                   (c >= 0xAC00 && c <= 0xD7AF));

    // Classify a line by its shape. Order: json, cjk, code, prose.
    public static ContentType DetectLineType(string line)
    {
        string t = line.Trim();
        char h = t.Length > 0 ? t[0] : '\0', e = t.Length > 0 ? t[^1] : '\0';
        if ((h == '{' || h == '}' || h == '[' || h == '"') &&
            (line.Contains(':') || line.Contains(','))) return ContentType.Json;
        if (HasCjk(line)) return ContentType.Cjk;
        int symbols = line.Count("{}();=<>[]#".Contains);
        if ((line.Length > 0 && (double)symbols / line.Length > 0.08) ||
            e == ';' || e == '{' || e == '}') return ContentType.Code;
        return ContentType.Prose;
    }

    public sealed record Estimate(double Tokens, double Low, double High, ContentType Dominant,
        IReadOnlyDictionary<ContentType, double> Breakdown);

    // Sum per-line estimates for every non-empty line of text.
    public static Estimate EstimateTokens(string text)
    {
        var breakdown = new Dictionary<ContentType, double> { [ContentType.Prose] = 0,
            [ContentType.Code] = 0, [ContentType.Json] = 0, [ContentType.Cjk] = 0 };
        double tokens = 0;
        foreach (string line in text.Replace("\r\n", "\n").Split('\n'))
        {
            if (line.Trim().Length == 0) continue;
            ContentType ty = DetectLineType(line);
            double lt = Math.Round(line.Length / CharsPerToken(ty));
            if (lt < 1.0) lt = 1.0; // max(1, round(len / rate))
            tokens += lt;
            breakdown[ty] += lt;
        }
        // Dominant type: strictly-greater scan keeps ties on Prose, as in TS.
        var dominant = ContentType.Prose;
        foreach (ContentType ty in new[] { ContentType.Prose, ContentType.Code,
                                           ContentType.Json, ContentType.Cjk })
            if (breakdown[ty] > breakdown[dominant]) dominant = ty;
        return new Estimate(tokens,
            Math.Round(tokens * (1.0 - EstimateTolerance)),
            Math.Round(tokens * (1.0 + EstimateTolerance)), dominant, breakdown);
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →