Slugify — Go source
Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.
This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Package slugify is the Go twin of CosmoDev's src/lib/slugify.ts (dual source:
// the web lib is TypeScript, the CLI lib is Go — kept in lock-step). Pure +
// deterministic, never panics. The table-driven tests in slugify_test.go share
// vectors with src/lib/slugify.test.ts so the two implementations are held to
// the same contract.
//
// Pipeline mirrors the TS lib exactly: transliterate non-decomposing ligatures
// → NFKD decompose → strip combining diacritics → collapse non-alphanumeric
// runs → split into words → apply casing → (strip stopwords) → join with the
// separator → (truncate to max length at a word boundary).
package slugify
import (
"regexp"
"strings"
"golang.org/x/text/unicode/norm"
)
// Case is the letter-casing mode of the resulting slug.
type Case int
const (
// CaseLower lowercases every word. It is the zero value, matching the TS
// default (case: 'lower').
CaseLower Case = iota
CasePreserve
CaseUpper
)
// Options configures Slugify. The zero value (Options{}) matches the TS default
// (slugify(text) with no options): hyphen separator, lower case, unlimited.
//
// Separator is a *string so the zero value means "default hyphen" while a
// non-nil empty string still means "concatenate" — exactly like the TS lib's
// distinction between omitted (→ '-') and '' (→ concatenate).
type Options struct {
Separator *string // nil → "-" (default); non-nil, including "", used verbatim
MaxLength int // <=0 means unlimited
Case Case // default CaseLower (zero value)
StripStopwords bool
}
// Letters/ligatures that NFKD does NOT decompose into an ASCII base + combining
// mark. Same entries as TRANSLIT in src/lib/slugify.ts.
var translit = map[rune]string{
// Germanic
'ß': "ss",
// Latin ligatures
'æ': "ae", 'Æ': "ae",
'œ': "oe", 'Œ': "oe",
'ff': "ff", 'fi': "fi", 'fl': "fl", 'ffi': "ffi", 'ffl': "ffl", 'ſt': "st", 'st': "st",
// Nordic / insular
'ð': "d", 'Ð': "d",
'þ': "th", 'Þ': "th",
'ø': "o", 'Ø': "o",
// Eastern European / strokes
'ł': "l", 'Ł': "l",
'đ': "d", 'Đ': "d",
'ħ': "h", 'Ħ': "h",
}
// Common English stopwords. Compared case-insensitively. Mirrors STOPWORDS.
var stopwords = map[string]bool{
"the": true, "a": true, "an": true, "and": true, "or": true, "but": true,
"of": true, "to": true, "in": true, "on": true, "at": true, "for": true,
"with": true, "by": true, "from": true,
}
var (
nonAlnum = regexp.MustCompile(`[^a-zA-Z0-9]+`)
newlines = regexp.MustCompile(`\r?\n`)
)
// tokenize breaks text into clean ASCII words (transliterated, diacritics
// stripped, cased). It mirrors the tokenize() helper in the TS lib.
func tokenize(text string, opts Options) []string {
// 1. transliterate letters that don't decompose on their own.
var b strings.Builder
for _, r := range text {
if r < 0x80 {
b.WriteRune(r)
continue
}
if rep, ok := translit[r]; ok {
b.WriteString(rep)
} else {
b.WriteRune(r)
}
}
// 2. decompose accented characters into base + combining marks (NFKD).
// 3. drop combining diacritical marks (U+0300–U+036F).
var stripped strings.Builder
for _, r := range norm.NFKD.String(b.String()) {
if r >= 0x0300 && r <= 0x036F {
continue
}
stripped.WriteRune(r)
}
// 4. collapse every run of non-alphanumeric characters to a single space.
ascii := strings.TrimSpace(nonAlnum.ReplaceAllString(stripped.String(), " "))
var words []string
if ascii != "" {
words = strings.Split(ascii, " ")
}
switch opts.Case {
case CaseUpper:
for i, w := range words {
words[i] = strings.ToUpper(w)
}
case CaseLower:
for i, w := range words {
words[i] = strings.ToLower(w)
}
// CasePreserve: leave the original casing untouched.
}
if opts.StripStopwords {
filtered := words[:0]
for _, w := range words {
if !stopwords[strings.ToLower(w)] {
filtered = append(filtered, w)
}
}
words = filtered
}
return words
}
// truncateAtWord truncates slug to max chars at the last whole-word boundary.
func truncateAtWord(slug, separator string, max int) string {
if len(slug) <= max {
return slug
}
if separator == "" {
return slug[:max] // nothing to break on — hard cut (slug is ASCII here)
}
cut := slug[:max]
last := strings.LastIndex(cut, separator)
if last > 0 {
return cut[:last]
}
return cut // no separator found — hard cut
}
// Slugify converts arbitrary text into a URL-safe slug. It is the Go twin of
// slugify() in src/lib/slugify.ts and must agree with it on every shared vector.
func Slugify(text string, opts Options) string {
sep := "-"
if opts.Separator != nil {
sep = *opts.Separator
}
slug := strings.Join(tokenize(text, opts), sep)
if opts.MaxLength > 0 {
return truncateAtWord(slug, sep, opts.MaxLength)
}
return slug
}
// SlugifyLines slugifies each line independently (batch mode). It returns
// exactly one slug per input line, matching slugifyLines() in the TS lib.
func SlugifyLines(text string, opts Options) []string {
lines := newlines.Split(text, -1)
out := make([]string, len(lines))
for i, line := range lines {
out[i] = Slugify(line, opts)
}
return out
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →