Skip to content

Token Estimator — C source

Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/* token-estimator — C port: tokenizer-free LLM token estimation.
 *
 * Display snippet: ports the line classifier and estimator core from the
 * TypeScript lib (src/lib/tokenEstimator.ts). Each non-empty line is
 * classified (prose / code / json / cjk) and divided by that type's
 * chars-per-token rate; the estimate carries a +/-15% band. Lengths are
 * bytes here, lines capped at 1023 (TS counts UTF-16 units, unbounded);
 * the full result shape and the whole-text JSON gate live in TS/Go.
 */
#include <ctype.h>
#include <math.h>
#include <stdint.h>
#include <string.h>

typedef enum { PROSE = 0, CODE, JSON_T, CJK, TYPE_COUNT } content_type;

/* Average chars per token by content type (CHARS_PER_TOKEN in TS). */
static const double CHARS_PER_TOKEN[TYPE_COUNT] = { 4.0, 3.5, 3.0, 1.5 };
static const double ESTIMATE_TOLERANCE = 0.15;

/* Decode one UTF-8 rune from *s, advance the cursor, return its code point. */
static uint32_t next_rune(const char **s) {
    const unsigned char *p = (const unsigned char *)*s;
    uint32_t r = *p++;
    int extra = (r & 0xE0) == 0xC0 ? 1 : (r & 0xF0) == 0xE0 ? 2 : (r & 0xF8) == 0xF0 ? 3 : 0;
    while (extra-- > 0) r = (r << 6) | (*p++ & 0x3F);
    *s = (const char *)p;
    return r;
}

/* CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul (U+AC00..U+D7AF). */
static int has_cjk(const char *s) {
    for (const char *p = s; *p; ) {
        uint32_t r = next_rune(&p);
        if ((r >= 0x4E00 && r <= 0x9FFF) || (r >= 0x3040 && r <= 0x30FF) ||
            (r >= 0xAC00 && r <= 0xD7AF)) return 1;
    }
    return 0;
}

/* Classify a line by its shape. Order: json, cjk, code, prose. */
static content_type detect_line_type(const char *line) {
    const char *t = line;
    while (*t && isspace((unsigned char)*t)) t++;
    if ((*t == '{' || *t == '}' || *t == '[' || *t == '"') &&
        (strchr(line, ':') || strchr(line, ','))) return JSON_T;
    if (has_cjk(line)) return CJK;
    size_t symbols = 0, len = strlen(line), tl = strlen(t);
    for (const char *p = line; *p; p++)
        if (strchr("{}();=<>[]#", *p)) symbols++;
    if ((double)symbols / (double)(len ? len : 1) > 0.08 ||
        (tl > 0 && (t[tl - 1] == ';' || t[tl - 1] == '{' || t[tl - 1] == '}')))
        return CODE;
    return PROSE;
}

/* Sum per-line estimates for every non-empty line; returns tokens, fills band. */
double estimate_tokens(const char *text, double *low, double *high) {
    double tokens = 0;
    char buf[1024];
    const char *line = text;
    while (*line) {
        const char *nl = strchr(line, '\n');
        size_t n = nl ? (size_t)(nl - line) : strlen(line);
        if (n > sizeof buf - 1) n = sizeof buf - 1;
        memcpy(buf, line, n), buf[n] = '\0';
        if (buf[strspn(buf, " \t\r")] != '\0') { /* non-empty line */
            double lt = round((double)n / CHARS_PER_TOKEN[detect_line_type(buf)]);
            tokens += lt < 1.0 ? 1.0 : lt; /* max(1, round(len / rate)) */
        }
        if (!nl) break;
        line = nl + 1;
    }
    if (low) *low = round(tokens * (1.0 - ESTIMATE_TOLERANCE));
    if (high) *high = round(tokens * (1.0 + ESTIMATE_TOLERANCE));
    return tokens;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →