Slugify — Python source
Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""slugify — URL-safe slug generator with locale-aware Unicode transliteration.
Language: Python (3.9+, standard library only)
Source: CosmoDev polyglot showcase port of the Slugify tool, ported from
src/lib/slugify.ts (the canonical TypeScript implementation).
License: display source — part of CosmoDev's polyglot tool pages.
Design goals:
- Pure + deterministic; never raises.
- Functionally equivalent to the TS reference: same inputs -> same outputs.
- Self-contained: stdlib only (no pip packages — no `unicodedata` magic
beyond what's needed, no third-party slugify libs).
Pipeline: transliterate ligatures -> NFKD decompose -> strip combining
diacritics -> collapse non-alphanumeric runs -> split into words -> apply
casing -> (strip stopwords) -> join with the separator -> (truncate at a
word boundary). Anything that can't be transliterated to ASCII (emoji, CJK,
...) collapses to a word separator.
Unicode note: unlike the dependency-free ports in this showcase, Python's
stdlib ships ``unicodedata.normalize``, so we can use the REAL NFKD path the
TS source uses — no manual block stripping needed. ``unicodedata`` is part of
the Python language, not an external dependency.
"""
from __future__ import annotations
import re
import unicodedata
from dataclasses import dataclass
from typing import List, Literal, Optional
__all__ = ["slugify", "slugify_lines", "SlugifyOptions"]
CaseMode = Literal["lower", "preserve", "upper"]
@dataclass
class SlugifyOptions:
"""Options mirror the TS ``SlugifyOptions`` interface.
Every field defaults to the TS default, so callers can construct the
object incrementally with keyword args (a dataclass gives us that for free
and keeps the option set self-documenting).
"""
separator: str = "-"
"""Character(s) joining words. Empty string concatenates."""
max_length: int = 0
"""Max slug length; truncated at the last word boundary at or under the
limit. ``0`` or negative = unlimited."""
case: CaseMode = "lower"
"""Letter casing of the result."""
strip_stopwords: bool = False
"""Strip common English stopwords (the, a, an, of, ...)."""
# Letters / ligatures that NFKD does NOT decompose into an ASCII base + a
# combining mark. Mapping them up front turns "Straße" -> "strasse",
# "Æsir" -> "aesir", "Søren" -> "soren". Accented Latin letters (á é ñ ü ...)
# need no entry here — NFKD splits them into base + diacritic and we strip the
# diacritic below.
TRANSLIT: dict[str, str] = {
# Germanic
"ß": "ss",
# Latin ligatures
"æ": "ae", "Æ": "ae",
"œ": "oe", "Œ": "oe",
"ff": "ff", "fi": "fi", "fl": "fl", "ffi": "ffi", "ffl": "ffl", "ſt": "st", "st": "st",
# Nordic / insular
"ð": "d", "Ð": "d",
"þ": "th", "Þ": "th",
"ø": "o", "Ø": "o",
# Eastern European / strokes
"ł": "l", "Ł": "l",
"đ": "d", "Đ": "d",
"ħ": "h", "Ħ": "h",
}
# Common English stopwords, lowercased. Compared case-insensitively so
# ``preserve`` / ``upper`` modes still drop them. A frozenset is the idiomatic
# immutable membership set.
STOPWORDS: frozenset[str] = frozenset({
"the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at",
"for", "with", "by", "from",
})
# Pre-compiled patterns mirror the TS regex literals. Compiling once at import
# is the idiomatic Python way to avoid per-call recompilation.
_NONASCII_RE = re.compile(r"[^\x00-\x7F]")
_DIACRITIC_RE = re.compile(r"[̀-ͯ]") # combining diacritical marks
_NONALNUM_RE = re.compile(r"[^a-zA-Z0-9]+")
def _translit_replacer(match: re.Match[str]) -> str:
"""``re.sub`` callback: look the matched non-ASCII char up in TRANSLIT,
leaving it untouched if it isn't a known ligature."""
return TRANSLIT.get(match.group(0), match.group(0))
def tokenize(text: str, options: SlugifyOptions) -> List[str]:
"""Break text into a list of clean ASCII words:
transliterated, diacritics stripped, and cased per ``options``.
Pipeline:
1. transliterate letters that don't decompose on their own,
2. NFKD-decompose accented characters into base + combining marks,
3. drop combining diacritical marks (U+0300..U+036F),
4. collapse every run of non-alphanumeric characters to a single space.
"""
ascii_text = _NONALNUM_RE.sub(
" ",
_DIACRITIC_RE.sub(
"",
unicodedata.normalize(
"NFKD",
_NONASCII_RE.sub(_translit_replacer, text),
),
),
).strip()
words: List[str] = ascii_text.split(" ") if ascii_text else []
if options.case == "upper":
words = [w.upper() for w in words]
elif options.case == "lower":
words = [w.lower() for w in words]
# case == "preserve" -> leave the original casing untouched
if options.strip_stopwords:
words = [w for w in words if w.lower() not in STOPWORDS]
return words
def _truncate_at_word(slug: str, separator: str, max_length: int) -> str:
"""Truncate ``slug`` to ``max_length`` chars at the last whole-word boundary."""
if len(slug) <= max_length:
return slug
if separator == "":
return slug[:max_length] # nothing to break on — hard cut
cut = slug[:max_length]
last = cut.rfind(separator)
return cut[:last] if last > 0 else cut # no separator found -> hard cut
def slugify(text: str, options: Optional[SlugifyOptions] = None) -> str:
"""Convert arbitrary text into a URL-safe slug. Never raises; an empty or
all-symbol input simply yields an empty string."""
if options is None:
options = SlugifyOptions()
# str.join handles the empty-separator case naturally ("".join -> concat),
# so we don't need a branch — same behavior as TS's Array.prototype.join.
slug = options.separator.join(tokenize(text, options))
if options.max_length and options.max_length > 0:
return _truncate_at_word(slug, options.separator, options.max_length)
return slug
def slugify_lines(text: str, options: Optional[SlugifyOptions] = None) -> List[str]:
"""Slugify each line independently (batch mode). Returns exactly one slug
per input line, matching the TS ``/\\r?\\n/`` split."""
if options is None:
options = SlugifyOptions()
return [slugify(line, options) for line in re.split(r"\r?\n", text)]
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →