Skip to content

Token Estimator — Go source

Estimate LLM token counts for any text or code - per-content-type heuristics (prose, code, JSON, CJK) with a ±15% range, plus chat-framing overhead. Runs entirely in your browser.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package tokenestimator is the Go twin of CosmoDev's
// src/lib/tokenEstimator.ts (dual source: the web lib is TypeScript, the CLI
// lib is Go — kept in lock-step). Pure + deterministic, never panics. The
// table-driven tests in token-estimator_test.go share vectors with
// src/lib/tokenEstimator.test.ts so the two implementations are held to the
// same contract.
//
// No tokenizer runs here: each line is classified (prose / code / json / cjk)
// and divided by that type's chars-per-token rate. The result carries a
// ±15% band (EstimateTolerance) because real BPE tokenizers vary by vocabulary
// and language mix. Line arithmetic counts UTF-16 code units — the unit TS's
// String.length counts — so multi-byte text estimates identically on both
// sides.
package tokenestimator

import (
	"encoding/json"
	"math"
	"regexp"
	"strings"
)

// ContentType is the classification of a line of text. Mirrors the TS union
// 'prose' | 'code' | 'json' | 'cjk' (string literals, so JSON output matches).
type ContentType string

const (
	// TypeProse is the zero value, matching the TS fallback type.
	TypeProse ContentType = "prose"
	TypeCode  ContentType = "code"
	TypeJSON  ContentType = "json"
	TypeCJK   ContentType = "cjk"
)

// AutoType is ContentType plus the auto-detect sentinel, mirroring the TS
// union AutoType = ContentType | 'auto'. It is an alias, so a ContentType
// value (or Auto) assigns directly. The zero value "" is treated as auto, so
// an unset EstimateOptions mirrors TS's omitted contentType.
type AutoType = ContentType

// Auto lets EstimateTokens detect each line's type (and apply whole-text
// JSON detection). "" behaves the same way.
const Auto AutoType = "auto"

// CharsPerToken holds the average characters per token, by content type.
// Mirrors CHARS_PER_TOKEN in the TS lib.
var CharsPerToken = map[ContentType]float64{
	TypeProse: 4,
	TypeCode:  3.5,
	TypeJSON:  3,
	TypeCJK:   1.5,
}

// EstimateTolerance is the reported estimate band on each side of the point
// estimate. Mirrors ESTIMATE_TOLERANCE.
const EstimateTolerance = 0.15

// ChatFramingTokensPerMessage is the rough cost of chat wrappers (role
// markers, delimiters) per message. Mirrors CHAT_FRAMING_TOKENS_PER_MESSAGE.
const ChatFramingTokensPerMessage = 5

// contentTypes fixes the scan order used to resolve the majority type:
// prose, code, json, cjk — ties keep the earlier entry (prose).
var contentTypes = []ContentType{TypeProse, TypeCode, TypeJSON, TypeCJK}

// newlines splits on LF or CRLF, mirroring text.split(/\r?\n/) in the TS lib.
var newlines = regexp.MustCompile(`\r?\n`)

// Breakdown is the per-type token mass. Mirrors the TS
// Record<ContentType, number> (JSON keys prose/code/json/cjk).
type Breakdown struct {
	Prose int `json:"prose"`
	Code  int `json:"code"`
	JSON  int `json:"json"`
	CJK   int `json:"cjk"`
}

func (b Breakdown) mass(t ContentType) int {
	switch t {
	case TypeProse:
		return b.Prose
	case TypeCode:
		return b.Code
	case TypeJSON:
		return b.JSON
	case TypeCJK:
		return b.CJK
	}
	return 0
}

// TokenEstimate is the result of EstimateTokens. Field-for-field twin of the
// TS TokenEstimate interface (camelCase JSON tags, aimodels convention).
type TokenEstimate struct {
	Tokens        int         `json:"tokens"`        // sum of per-line estimates (excludes framing)
	Low           int         `json:"low"`           // round(tokens * (1 - EstimateTolerance))
	High          int         `json:"high"`          // round(tokens * (1 + EstimateTolerance))
	Chars         int         `json:"chars"`         // total characters excluding newlines
	Words         int         `json:"words"`         // whitespace-split word count
	Lines         int         `json:"lines"`         // non-empty line count
	ContentType   ContentType `json:"contentType"`   // majority of per-line token mass
	Breakdown     Breakdown   `json:"breakdown"`     // tokens per detected line type
	FramingTokens int         `json:"framingTokens"` // Messages * ChatFramingTokensPerMessage
}

// EstimateOptions configures EstimateTokens. The zero value (and a nil
// pointer) matches the TS default: auto detection, no chat framing.
type EstimateOptions struct {
	ContentType AutoType // "" or "auto" → detect; a ContentType value forces it
	Messages    int      // chat messages the text will be sent as
}

// utf16Len reports the length of s in UTF-16 code units — the unit TS's
// String.length counts. BMP runes are one unit, astral runes two.
func utf16Len(s string) int {
	n := 0
	for _, r := range s {
		if r > 0xFFFF {
			n += 2
		} else {
			n++
		}
	}
	return n
}

// hasCJK reports whether s contains a CJK ideograph (U+4E00–U+9FFF), kana
// (U+3040–U+30FF), or Hangul (U+AC00–U+D7AF). Mirrors CJK_RE in the TS lib.
func hasCJK(s string) bool {
	for _, r := range s {
		if (r >= 0x4E00 && r <= 0x9FFF) ||
			(r >= 0x3040 && r <= 0x30FF) ||
			(r >= 0xAC00 && r <= 0xD7AF) {
			return true
		}
	}
	return false
}

// DetectLineType classifies a single line by its shape. Order: json, cjk,
// code, prose. It is the twin of detectLineType() in src/lib/tokenEstimator.ts.
func DetectLineType(line string) ContentType {
	trimmed := strings.TrimSpace(line)
	// JSON-ish: opens like a JSON fragment AND carries a separator.
	if (strings.HasPrefix(trimmed, "{") || strings.HasPrefix(trimmed, "}") ||
		strings.HasPrefix(trimmed, "[") || strings.HasPrefix(trimmed, "\"")) &&
		(strings.Contains(line, ":") || strings.Contains(line, ",")) {
		return TypeJSON
	}
	// CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
	if hasCJK(line) {
		return TypeCJK
	}
	// Code: symbol-dense, or a statement terminator / block opener at EOL.
	length := utf16Len(line)
	symbols := 0
	for _, r := range line {
		if strings.ContainsRune("{}();=<>[]#", r) {
			symbols++
		}
	}
	density := 0.0
	if length > 0 {
		density = float64(symbols) / float64(length)
	}
	if density > 0.08 || strings.HasSuffix(trimmed, ";") ||
		strings.HasSuffix(trimmed, "{") || strings.HasSuffix(trimmed, "}") {
		return TypeCode
	}
	return TypeProse
}

// isValidJSON reports whether the whole text parses as JSON. Mirrors
// isValidJson() (JSON.parse in a try/catch); empty/whitespace text is not.
func isValidJSON(text string) bool {
	if strings.TrimSpace(text) == "" {
		return false
	}
	var v any
	return json.Unmarshal([]byte(text), &v) == nil
}

// EstimateTokens estimates the LLM token count of text without running a
// tokenizer. It is the twin of estimateTokens() in src/lib/tokenEstimator.ts
// and must agree with it on every shared vector. A nil opts means auto
// detection with no chat framing.
func EstimateTokens(text string, opts *EstimateOptions) TokenEstimate {
	var options EstimateOptions
	if opts != nil {
		options = *opts
	}
	// Forced type: a known ContentType value; "", "auto", or an unrecognized
	// string falls back to auto (the TS lib cannot express an invalid union
	// member — treat one as "not forced" rather than divide by a missing rate).
	var forced ContentType
	hasForced := false
	if c := ContentType(options.ContentType); c != "" && options.ContentType != Auto {
		if _, ok := CharsPerToken[c]; ok {
			forced = c
			hasForced = true
		}
	}
	// AUTO + whole-text JSON: a document that parses as JSON is json all the
	// way down — json's 3 chars/token rate applies to every line, not just
	// the reported contentType.
	wholeTextJSON := !hasForced && isValidJSON(text)

	allLines := newlines.Split(text, -1)

	var breakdown Breakdown
	tokens := 0
	chars := 0
	lines := 0
	for _, line := range allLines {
		chars += utf16Len(line)
		if strings.TrimSpace(line) == "" {
			continue
		}
		lines++
		t := DetectLineType(line)
		if hasForced {
			t = forced
		} else if wholeTextJSON {
			t = TypeJSON
		}
		lineTokens := int(math.Max(1, math.Round(float64(utf16Len(line))/CharsPerToken[t])))
		tokens += lineTokens
		switch t {
		case TypeProse:
			breakdown.Prose += lineTokens
		case TypeCode:
			breakdown.Code += lineTokens
		case TypeJSON:
			breakdown.JSON += lineTokens
		case TypeCJK:
			breakdown.CJK += lineTokens
		}
	}

	// Resolved type = the line type holding the most token mass (ties stay
	// prose, the first entry of contentTypes).
	contentType := TypeProse
	for _, t := range contentTypes {
		if breakdown.mass(t) > breakdown.mass(contentType) {
			contentType = t
		}
	}

	trimmedText := strings.TrimSpace(text)
	words := 0
	if trimmedText != "" {
		words = len(strings.Fields(trimmedText))
	}

	return TokenEstimate{
		Tokens:        tokens,
		Low:           int(math.Round(float64(tokens) * (1 - EstimateTolerance))),
		High:          int(math.Round(float64(tokens) * (1 + EstimateTolerance))),
		Chars:         chars,
		Words:         words,
		Lines:         lines,
		ContentType:   contentType,
		Breakdown:     breakdown,
		FramingTokens: options.Messages * ChatFramingTokensPerMessage,
	}
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →