Token Estimator — Go source
Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.
This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Package tokenestimator is the Go twin of CosmoDev's
// src/lib/tokenEstimator.ts (dual source: the web lib is TypeScript, the CLI
// lib is Go — kept in lock-step). Pure + deterministic, never panics. The
// table-driven tests in token-estimator_test.go share vectors with
// src/lib/tokenEstimator.test.ts so the two implementations are held to the
// same contract.
//
// No tokenizer runs here: each line is classified (prose / code / json / cjk)
// and divided by that type's chars-per-token rate. The result carries a
// ±15% band (EstimateTolerance) because real BPE tokenizers vary by vocabulary
// and language mix. Line arithmetic counts UTF-16 code units — the unit TS's
// String.length counts — so multi-byte text estimates identically on both
// sides.
package tokenestimator
import (
"encoding/json"
"math"
"regexp"
"strings"
)
// ContentType is the classification of a line of text. Mirrors the TS union
// 'prose' | 'code' | 'json' | 'cjk' (string literals, so JSON output matches).
type ContentType string
const (
// TypeProse is the zero value, matching the TS fallback type.
TypeProse ContentType = "prose"
TypeCode ContentType = "code"
TypeJSON ContentType = "json"
TypeCJK ContentType = "cjk"
)
// AutoType is ContentType plus the auto-detect sentinel, mirroring the TS
// union AutoType = ContentType | 'auto'. It is an alias, so a ContentType
// value (or Auto) assigns directly. The zero value "" is treated as auto, so
// an unset EstimateOptions mirrors TS's omitted contentType.
type AutoType = ContentType
// Auto lets EstimateTokens detect each line's type (and apply whole-text
// JSON detection). "" behaves the same way.
const Auto AutoType = "auto"
// CharsPerToken holds the average characters per token, by content type.
// Mirrors CHARS_PER_TOKEN in the TS lib.
var CharsPerToken = map[ContentType]float64{
TypeProse: 4,
TypeCode: 3.5,
TypeJSON: 3,
TypeCJK: 1.5,
}
// EstimateTolerance is the reported estimate band on each side of the point
// estimate. Mirrors ESTIMATE_TOLERANCE.
const EstimateTolerance = 0.15
// ChatFramingTokensPerMessage is the rough cost of chat wrappers (role
// markers, delimiters) per message. Mirrors CHAT_FRAMING_TOKENS_PER_MESSAGE.
const ChatFramingTokensPerMessage = 5
// contentTypes fixes the scan order used to resolve the majority type:
// prose, code, json, cjk — ties keep the earlier entry (prose).
var contentTypes = []ContentType{TypeProse, TypeCode, TypeJSON, TypeCJK}
// newlines splits on LF or CRLF, mirroring text.split(/\r?\n/) in the TS lib.
var newlines = regexp.MustCompile(`\r?\n`)
// Breakdown is the per-type token mass. Mirrors the TS
// Record<ContentType, number> (JSON keys prose/code/json/cjk).
type Breakdown struct {
Prose int `json:"prose"`
Code int `json:"code"`
JSON int `json:"json"`
CJK int `json:"cjk"`
}
func (b Breakdown) mass(t ContentType) int {
switch t {
case TypeProse:
return b.Prose
case TypeCode:
return b.Code
case TypeJSON:
return b.JSON
case TypeCJK:
return b.CJK
}
return 0
}
// TokenEstimate is the result of EstimateTokens. Field-for-field twin of the
// TS TokenEstimate interface (camelCase JSON tags, aimodels convention).
type TokenEstimate struct {
Tokens int `json:"tokens"` // sum of per-line estimates (excludes framing)
Low int `json:"low"` // round(tokens * (1 - EstimateTolerance))
High int `json:"high"` // round(tokens * (1 + EstimateTolerance))
Chars int `json:"chars"` // total characters excluding newlines
Words int `json:"words"` // whitespace-split word count
Lines int `json:"lines"` // non-empty line count
ContentType ContentType `json:"contentType"` // majority of per-line token mass
Breakdown Breakdown `json:"breakdown"` // tokens per detected line type
FramingTokens int `json:"framingTokens"` // Messages * ChatFramingTokensPerMessage
}
// EstimateOptions configures EstimateTokens. The zero value (and a nil
// pointer) matches the TS default: auto detection, no chat framing.
type EstimateOptions struct {
ContentType AutoType // "" or "auto" → detect; a ContentType value forces it
Messages int // chat messages the text will be sent as
}
// utf16Len reports the length of s in UTF-16 code units — the unit TS's
// String.length counts. BMP runes are one unit, astral runes two.
func utf16Len(s string) int {
n := 0
for _, r := range s {
if r > 0xFFFF {
n += 2
} else {
n++
}
}
return n
}
// hasCJK reports whether s contains a CJK ideograph (U+4E00–U+9FFF), kana
// (U+3040–U+30FF), or Hangul (U+AC00–U+D7AF). Mirrors CJK_RE in the TS lib.
func hasCJK(s string) bool {
for _, r := range s {
if (r >= 0x4E00 && r <= 0x9FFF) ||
(r >= 0x3040 && r <= 0x30FF) ||
(r >= 0xAC00 && r <= 0xD7AF) {
return true
}
}
return false
}
// DetectLineType classifies a single line by its shape. Order: json, cjk,
// code, prose. It is the twin of detectLineType() in src/lib/tokenEstimator.ts.
func DetectLineType(line string) ContentType {
trimmed := strings.TrimSpace(line)
// JSON-ish: opens like a JSON fragment AND carries a separator.
if (strings.HasPrefix(trimmed, "{") || strings.HasPrefix(trimmed, "}") ||
strings.HasPrefix(trimmed, "[") || strings.HasPrefix(trimmed, "\"")) &&
(strings.Contains(line, ":") || strings.Contains(line, ",")) {
return TypeJSON
}
// CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
if hasCJK(line) {
return TypeCJK
}
// Code: symbol-dense, or a statement terminator / block opener at EOL.
length := utf16Len(line)
symbols := 0
for _, r := range line {
if strings.ContainsRune("{}();=<>[]#", r) {
symbols++
}
}
density := 0.0
if length > 0 {
density = float64(symbols) / float64(length)
}
if density > 0.08 || strings.HasSuffix(trimmed, ";") ||
strings.HasSuffix(trimmed, "{") || strings.HasSuffix(trimmed, "}") {
return TypeCode
}
return TypeProse
}
// isValidJSON reports whether the whole text parses as JSON. Mirrors
// isValidJson() (JSON.parse in a try/catch); empty/whitespace text is not.
func isValidJSON(text string) bool {
if strings.TrimSpace(text) == "" {
return false
}
var v any
return json.Unmarshal([]byte(text), &v) == nil
}
// EstimateTokens estimates the LLM token count of text without running a
// tokenizer. It is the twin of estimateTokens() in src/lib/tokenEstimator.ts
// and must agree with it on every shared vector. A nil opts means auto
// detection with no chat framing.
func EstimateTokens(text string, opts *EstimateOptions) TokenEstimate {
var options EstimateOptions
if opts != nil {
options = *opts
}
// Forced type: a known ContentType value; "", "auto", or an unrecognized
// string falls back to auto (the TS lib cannot express an invalid union
// member — treat one as "not forced" rather than divide by a missing rate).
var forced ContentType
hasForced := false
if c := ContentType(options.ContentType); c != "" && options.ContentType != Auto {
if _, ok := CharsPerToken[c]; ok {
forced = c
hasForced = true
}
}
// AUTO + whole-text JSON: a document that parses as JSON is json all the
// way down — json's 3 chars/token rate applies to every line, not just
// the reported contentType.
wholeTextJSON := !hasForced && isValidJSON(text)
allLines := newlines.Split(text, -1)
var breakdown Breakdown
tokens := 0
chars := 0
lines := 0
for _, line := range allLines {
chars += utf16Len(line)
if strings.TrimSpace(line) == "" {
continue
}
lines++
t := DetectLineType(line)
if hasForced {
t = forced
} else if wholeTextJSON {
t = TypeJSON
}
lineTokens := int(math.Max(1, math.Round(float64(utf16Len(line))/CharsPerToken[t])))
tokens += lineTokens
switch t {
case TypeProse:
breakdown.Prose += lineTokens
case TypeCode:
breakdown.Code += lineTokens
case TypeJSON:
breakdown.JSON += lineTokens
case TypeCJK:
breakdown.CJK += lineTokens
}
}
// Resolved type = the line type holding the most token mass (ties stay
// prose, the first entry of contentTypes).
contentType := TypeProse
for _, t := range contentTypes {
if breakdown.mass(t) > breakdown.mass(contentType) {
contentType = t
}
}
trimmedText := strings.TrimSpace(text)
words := 0
if trimmedText != "" {
words = len(strings.Fields(trimmedText))
}
return TokenEstimate{
Tokens: tokens,
Low: int(math.Round(float64(tokens) * (1 - EstimateTolerance))),
High: int(math.Round(float64(tokens) * (1 + EstimateTolerance))),
Chars: chars,
Words: words,
Lines: lines,
ContentType: contentType,
Breakdown: breakdown,
FramingTokens: options.Messages * ChatFramingTokensPerMessage,
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →