HTML Entity Encoder/Decoder — Go source
Encode text to HTML entities and decode entities back to text (named + numeric). UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.
This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Package htmlentityencoder is the Go twin of CosmoDev's
// src/lib/htmlEntities.ts (dual source: the web lib is TypeScript, the CLI lib
// is Go — kept in lock-step). Pure + deterministic, never panics. The
// table-driven tests in html-entity-encoder_test.go share vectors with
// src/lib/htmlEntities.test.ts so the two implementations are held to the same
// contract.
//
// Behavior mirrors the TS lib exactly: encode escapes the five
// HTML-significant characters (& replaced first so emitted entities are never
// re-escaped), optionally converting every non-ASCII code point to a decimal
// numeric reference under Ascii mode; decode resolves named, decimal, and hex
// references, passing unknown / malformed / out-of-range references through
// untouched (never returning an error).
package htmlentityencoder
import (
"regexp"
"strconv"
"strings"
)
// EncodeOptions configures EncodeHtml. The zero value (EncodeOptions{})
// matches the TS default (encodeHtml(text) with no options): non-ASCII passes
// through untouched.
type EncodeOptions struct {
// Ascii, when true, additionally encodes every non-ASCII code point as a
// decimal numeric reference (e.g. © -> ©), producing ASCII-safe output.
// Mirrors the TS lib's ascii option (default false).
Ascii bool
}
// EncodeHtml escapes the five HTML-significant characters. It is the Go twin
// of encodeHtml() in src/lib/htmlEntities.ts and must agree with it on every
// shared vector. '&' is replaced first so the entities it emits (e.g. "<")
// are never re-escaped by the subsequent replacements — exactly like the TS
// lib's sequential replace chain. When opts.Ascii is true, every non-ASCII
// code point is additionally converted to a decimal numeric reference (one
// reference per code point, so astral characters like emoji encode as a single
// reference).
func EncodeHtml(text string, opts EncodeOptions) string {
s := text
s = strings.ReplaceAll(s, "&", "&")
s = strings.ReplaceAll(s, "<", "<")
s = strings.ReplaceAll(s, ">", ">")
s = strings.ReplaceAll(s, "'", "'")
s = strings.ReplaceAll(s, "\"", """)
if !opts.Ascii {
return s
}
// \P{ASCII} in the TS matches each non-ASCII code point as one unit;
// ranging over the Go string decodes UTF-8 into runes, so astral chars
// (e.g. 🚀) produce a single decimal reference — matching codePointAt().
var b strings.Builder
b.Grow(len(s))
for _, r := range s {
if r <= 0x7F {
b.WriteRune(r)
continue
}
b.WriteString("&#")
b.WriteString(strconv.FormatInt(int64(r), 10))
b.WriteByte(';')
}
return b.String()
}
// entityRe matches a well-formed reference: "&#xHH;", "&#NN;", or "&name;".
// It is the Go RE2 equivalent of the TS lib's ENTITY regex
// /&(#[xX][0-9a-fA-F]+|#[0-9]+|[A-Za-z][A-Za-z0-9]*);/g
var entityRe = regexp.MustCompile(`&(#[xX][0-9a-fA-F]+|#[0-9]+|[A-Za-z][A-Za-z0-9]*);`)
// decodeReference resolves a single matched reference to its character. The
// argument is the full match including the leading '&' and trailing ';'.
// Unknown, malformed, out-of-range, or lone-surrogate references are returned
// untouched — mirroring decodeReference() in the TS lib (which never throws).
func decodeReference(match string) string {
// Strip the leading '&' and trailing ';' to recover the body. match[0] is
// '&' and the final byte is ';' by construction of entityRe.
body := match[1 : len(match)-1]
if body[0] == '#' {
// Numeric reference. body is at least "#0" (length >= 2).
hex := body[1] == 'x' || body[1] == 'X'
var digits string
var base int
if hex {
digits = body[2:]
base = 16
} else {
digits = body[1:]
base = 10
}
// ParseInt returns a range error for inputs that overflow int64; the
// TS lib's parseInt would instead yield a finite but out-of-range
// Number. Either way the code point is invalid and we keep the original.
num, err := strconv.ParseInt(digits, base, 64)
if err != nil {
return match
}
// Lone surrogates are not valid standalone code points — keep original.
if num >= 0xD800 && num <= 0xDFFF {
return match
}
// String.fromCodePoint throws a RangeError for code points > U+10FFFF;
// the TS lib falls back to the original text. Do the same. (num is
// non-negative here because the regex has no sign, but the lower bound
// is checked defensively.)
if num < 0 || num > 0x10FFFF {
return match
}
return string(rune(num))
}
if v, ok := htmlEntityTable[body]; ok {
return v
}
return match
}
// DecodeHtml resolves named ("&", "©", ...), decimal ("©"), and
// hex ("©") references. It is the Go twin of decodeHtml() in
// src/lib/htmlEntities.ts and must agree with it on every shared vector.
// Unknown / malformed / out-of-range references pass through untouched.
func DecodeHtml(text string) string {
return entityRe.ReplaceAllStringFunc(text, decodeReference)
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →