Skip to content

HTML Entity Encoder/Decoder — Go source

Encode text to HTML entities and decode entities back to text (named + numeric). UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package htmlentityencoder is the Go twin of CosmoDev's
// src/lib/htmlEntities.ts (dual source: the web lib is TypeScript, the CLI lib
// is Go — kept in lock-step). Pure + deterministic, never panics. The
// table-driven tests in html-entity-encoder_test.go share vectors with
// src/lib/htmlEntities.test.ts so the two implementations are held to the same
// contract.
//
// Behavior mirrors the TS lib exactly: encode escapes the five
// HTML-significant characters (& replaced first so emitted entities are never
// re-escaped), optionally converting every non-ASCII code point to a decimal
// numeric reference under Ascii mode; decode resolves named, decimal, and hex
// references, passing unknown / malformed / out-of-range references through
// untouched (never returning an error).
package htmlentityencoder

import (
	"regexp"
	"strconv"
	"strings"
)

// EncodeOptions configures EncodeHtml. The zero value (EncodeOptions{})
// matches the TS default (encodeHtml(text) with no options): non-ASCII passes
// through untouched.
type EncodeOptions struct {
	// Ascii, when true, additionally encodes every non-ASCII code point as a
	// decimal numeric reference (e.g. © -> ©), producing ASCII-safe output.
	// Mirrors the TS lib's ascii option (default false).
	Ascii bool
}

// EncodeHtml escapes the five HTML-significant characters. It is the Go twin
// of encodeHtml() in src/lib/htmlEntities.ts and must agree with it on every
// shared vector. '&' is replaced first so the entities it emits (e.g. "<")
// are never re-escaped by the subsequent replacements — exactly like the TS
// lib's sequential replace chain. When opts.Ascii is true, every non-ASCII
// code point is additionally converted to a decimal numeric reference (one
// reference per code point, so astral characters like emoji encode as a single
// reference).
func EncodeHtml(text string, opts EncodeOptions) string {
	s := text
	s = strings.ReplaceAll(s, "&", "&")
	s = strings.ReplaceAll(s, "<", "&lt;")
	s = strings.ReplaceAll(s, ">", "&gt;")
	s = strings.ReplaceAll(s, "'", "&apos;")
	s = strings.ReplaceAll(s, "\"", "&quot;")

	if !opts.Ascii {
		return s
	}
	// \P{ASCII} in the TS matches each non-ASCII code point as one unit;
	// ranging over the Go string decodes UTF-8 into runes, so astral chars
	// (e.g. 🚀) produce a single decimal reference — matching codePointAt().
	var b strings.Builder
	b.Grow(len(s))
	for _, r := range s {
		if r <= 0x7F {
			b.WriteRune(r)
			continue
		}
		b.WriteString("&#")
		b.WriteString(strconv.FormatInt(int64(r), 10))
		b.WriteByte(';')
	}
	return b.String()
}

// entityRe matches a well-formed reference: "&#xHH;", "&#NN;", or "&name;".
// It is the Go RE2 equivalent of the TS lib's ENTITY regex
//   /&(#[xX][0-9a-fA-F]+|#[0-9]+|[A-Za-z][A-Za-z0-9]*);/g
var entityRe = regexp.MustCompile(`&(#[xX][0-9a-fA-F]+|#[0-9]+|[A-Za-z][A-Za-z0-9]*);`)

// decodeReference resolves a single matched reference to its character. The
// argument is the full match including the leading '&' and trailing ';'.
// Unknown, malformed, out-of-range, or lone-surrogate references are returned
// untouched — mirroring decodeReference() in the TS lib (which never throws).
func decodeReference(match string) string {
	// Strip the leading '&' and trailing ';' to recover the body. match[0] is
	// '&' and the final byte is ';' by construction of entityRe.
	body := match[1 : len(match)-1]
	if body[0] == '#' {
		// Numeric reference. body is at least "#0" (length >= 2).
		hex := body[1] == 'x' || body[1] == 'X'
		var digits string
		var base int
		if hex {
			digits = body[2:]
			base = 16
		} else {
			digits = body[1:]
			base = 10
		}
		// ParseInt returns a range error for inputs that overflow int64; the
		// TS lib's parseInt would instead yield a finite but out-of-range
		// Number. Either way the code point is invalid and we keep the original.
		num, err := strconv.ParseInt(digits, base, 64)
		if err != nil {
			return match
		}
		// Lone surrogates are not valid standalone code points — keep original.
		if num >= 0xD800 && num <= 0xDFFF {
			return match
		}
		// String.fromCodePoint throws a RangeError for code points > U+10FFFF;
		// the TS lib falls back to the original text. Do the same. (num is
		// non-negative here because the regex has no sign, but the lower bound
		// is checked defensively.)
		if num < 0 || num > 0x10FFFF {
			return match
		}
		return string(rune(num))
	}
	if v, ok := htmlEntityTable[body]; ok {
		return v
	}
	return match
}

// DecodeHtml resolves named ("&amp;", "&copy;", ...), decimal ("&#169;"), and
// hex ("&#xA9;") references. It is the Go twin of decodeHtml() in
// src/lib/htmlEntities.ts and must agree with it on every shared vector.
// Unknown / malformed / out-of-range references pass through untouched.
func DecodeHtml(text string) string {
	return entityRe.ReplaceAllStringFunc(text, decodeReference)
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →