Slugify — TypeScript source
Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// URL-safe slug generator with locale-aware Unicode transliteration.
// Pure + deterministic, never throws - the unit-test surface for the Slugify tool.
export type SlugCase = 'lower' | 'preserve' | 'upper';
export interface SlugifyOptions {
/** Character(s) joining words. Defaults to '-'. Empty string concatenates. */
separator?: string;
/** Max slug length; truncated at the last word boundary at or under the limit. 0 / negative = unlimited. */
maxLength?: number;
/** Letter casing of the result. Defaults to 'lower'. */
case?: SlugCase;
/** Strip common English stopwords (the, a, an, of, …). Defaults to false. */
stripStopwords?: boolean;
}
// Letters/ligatures that NFKD does NOT decompose into an ASCII base + combining
// mark. Map them to ASCII up front so "Straße" -> "strasse", "Æsir" -> "aesir",
// "Søren" -> "soren". Accented Latin letters (á é ñ ü …) need no entry - NFKD
// splits them into base + diacritic and we strip the diacritic below.
const TRANSLIT: Record<string, string> = {
// Germanic
ß: 'ss',
// Latin ligatures
æ: 'ae', Æ: 'ae',
œ: 'oe', Œ: 'oe',
ff: 'ff', fi: 'fi', fl: 'fl', ffi: 'ffi', ffl: 'ffl', ſt: 'st', st: 'st',
// Nordic / insular
ð: 'd', Ð: 'd',
þ: 'th', Þ: 'th',
ø: 'o', Ø: 'o',
// Eastern European / strokes
ł: 'l', Ł: 'l',
đ: 'd', Đ: 'd',
ħ: 'h', Ħ: 'h',
};
// Common English stopwords, lowercased. Compared case-insensitively so preserve
// / upper modes still drop them. Stripped only when `stripStopwords` is on.
const STOPWORDS = new Set([
'the', 'a', 'an', 'and', 'or', 'but', 'of', 'to', 'in', 'on', 'at', 'for',
'with', 'by', 'from',
]);
/** Break text into a list of clean ASCII words (transliterated, diacritics stripped, cased). */
function tokenize(text: string, options: SlugifyOptions): string[] {
const mode: SlugCase = options.case ?? 'lower';
const ascii = text
// 1. transliterate letters that don't decompose on their own
.replace(/[^\x00-\x7F]/g, (ch) => TRANSLIT[ch] ?? ch)
// 2. decompose accented characters into base + combining marks
.normalize('NFKD')
// 3. drop combining diacritical marks (U+0300-U+036F)
.replace(/[̀-ͯ]/g, '')
// 4. collapse every run of non-alphanumeric characters to a single space
.replace(/[^a-zA-Z0-9]+/g, ' ')
.trim();
let words = ascii ? ascii.split(' ') : [];
if (mode === 'upper') words = words.map((w) => w.toUpperCase());
else if (mode === 'lower') words = words.map((w) => w.toLowerCase());
// mode === 'preserve' -> leave the original casing untouched
if (options.stripStopwords) {
words = words.filter((w) => !STOPWORDS.has(w.toLowerCase()));
}
return words;
}
/** Truncate `slug` to `max` chars at the last whole-word boundary. */
function truncateAtWord(slug: string, separator: string, max: number): string {
if (slug.length <= max) return slug;
if (separator === '') return slug.slice(0, max); // nothing to break on - hard cut
const cut = slug.slice(0, max);
const last = cut.lastIndexOf(separator);
return last > 0 ? cut.slice(0, last) : cut; // no separator found -> hard cut
}
/**
* Convert arbitrary text into a URL-safe slug.
*
* Pipeline: transliterate ligatures -> NFKD decompose -> strip combining
* diacritics -> collapse non-alphanumeric runs -> split into words -> apply
* casing -> (strip stopwords) -> join with the separator -> (truncate to max
* length at a word boundary). Anything that can't be transliterated to ASCII
* (emoji, CJK, etc.) collapses to a word separator. Never throws.
*/
export function slugify(text: string, options: SlugifyOptions = {}): string {
const separator = options.separator ?? '-';
const slug = tokenize(text, options).join(separator);
if (options.maxLength && options.maxLength > 0) {
return truncateAtWord(slug, separator, options.maxLength);
}
return slug;
}
/** Slugify each line independently (batch mode). Returns exactly one slug per input line. */
export function slugifyLines(text: string, options: SlugifyOptions = {}): string[] {
return text.split(/\r?\n/).map((line) => slugify(line, options));
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →