Token Estimator — C source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/* token-estimator — C port: tokenizer-free LLM token estimation.
*
* Display snippet: ports the line classifier and estimator core from the
* TypeScript lib (src/lib/tokenEstimator.ts). Each non-empty line is
* classified (prose / code / json / cjk) and divided by that type's
* chars-per-token rate; the estimate carries a +/-15% band. Lengths are
* bytes here, lines capped at 1023 (TS counts UTF-16 units, unbounded);
* the full result shape and the whole-text JSON gate live in TS/Go.
*/
#include <ctype.h>
#include <math.h>
#include <stdint.h>
#include <string.h>
typedef enum { PROSE = 0, CODE, JSON_T, CJK, TYPE_COUNT } content_type;
/* Average chars per token by content type (CHARS_PER_TOKEN in TS). */
static const double CHARS_PER_TOKEN[TYPE_COUNT] = { 4.0, 3.5, 3.0, 1.5 };
static const double ESTIMATE_TOLERANCE = 0.15;
/* Decode one UTF-8 rune from *s, advance the cursor, return its code point. */
static uint32_t next_rune(const char **s) {
const unsigned char *p = (const unsigned char *)*s;
uint32_t r = *p++;
int extra = (r & 0xE0) == 0xC0 ? 1 : (r & 0xF0) == 0xE0 ? 2 : (r & 0xF8) == 0xF0 ? 3 : 0;
while (extra-- > 0) r = (r << 6) | (*p++ & 0x3F);
*s = (const char *)p;
return r;
}
/* CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul (U+AC00..U+D7AF). */
static int has_cjk(const char *s) {
for (const char *p = s; *p; ) {
uint32_t r = next_rune(&p);
if ((r >= 0x4E00 && r <= 0x9FFF) || (r >= 0x3040 && r <= 0x30FF) ||
(r >= 0xAC00 && r <= 0xD7AF)) return 1;
}
return 0;
}
/* Classify a line by its shape. Order: json, cjk, code, prose. */
static content_type detect_line_type(const char *line) {
const char *t = line;
while (*t && isspace((unsigned char)*t)) t++;
if ((*t == '{' || *t == '}' || *t == '[' || *t == '"') &&
(strchr(line, ':') || strchr(line, ','))) return JSON_T;
if (has_cjk(line)) return CJK;
size_t symbols = 0, len = strlen(line), tl = strlen(t);
for (const char *p = line; *p; p++)
if (strchr("{}();=<>[]#", *p)) symbols++;
if ((double)symbols / (double)(len ? len : 1) > 0.08 ||
(tl > 0 && (t[tl - 1] == ';' || t[tl - 1] == '{' || t[tl - 1] == '}')))
return CODE;
return PROSE;
}
/* Sum per-line estimates for every non-empty line; returns tokens, fills band. */
double estimate_tokens(const char *text, double *low, double *high) {
double tokens = 0;
char buf[1024];
const char *line = text;
while (*line) {
const char *nl = strchr(line, '\n');
size_t n = nl ? (size_t)(nl - line) : strlen(line);
if (n > sizeof buf - 1) n = sizeof buf - 1;
memcpy(buf, line, n), buf[n] = '\0';
if (buf[strspn(buf, " \t\r")] != '\0') { /* non-empty line */
double lt = round((double)n / CHARS_PER_TOKEN[detect_line_type(buf)]);
tokens += lt < 1.0 ? 1.0 : lt; /* max(1, round(len / rate)) */
}
if (!nl) break;
line = nl + 1;
}
if (low) *low = round(tokens * (1.0 - ESTIMATE_TOLERANCE));
if (high) *high = round(tokens * (1.0 + ESTIMATE_TOLERANCE));
return tokens;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →