Skip to content

Slugify — TypeScript source

Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// URL-safe slug generator with locale-aware Unicode transliteration.
// Pure + deterministic, never throws - the unit-test surface for the Slugify tool.

export type SlugCase = 'lower' | 'preserve' | 'upper';

export interface SlugifyOptions {
  /** Character(s) joining words. Defaults to '-'. Empty string concatenates. */
  separator?: string;
  /** Max slug length; truncated at the last word boundary at or under the limit. 0 / negative = unlimited. */
  maxLength?: number;
  /** Letter casing of the result. Defaults to 'lower'. */
  case?: SlugCase;
  /** Strip common English stopwords (the, a, an, of, …). Defaults to false. */
  stripStopwords?: boolean;
}

// Letters/ligatures that NFKD does NOT decompose into an ASCII base + combining
// mark. Map them to ASCII up front so "Straße" -> "strasse", "Æsir" -> "aesir",
// "Søren" -> "soren". Accented Latin letters (á é ñ ü …) need no entry - NFKD
// splits them into base + diacritic and we strip the diacritic below.
const TRANSLIT: Record<string, string> = {
  // Germanic
  ß: 'ss',
  // Latin ligatures
  æ: 'ae', Æ: 'ae',
  œ: 'oe', Œ: 'oe',
  ff: 'ff', fi: 'fi', fl: 'fl', ffi: 'ffi', ffl: 'ffl', ſt: 'st', st: 'st',
  // Nordic / insular
  ð: 'd', Ð: 'd',
  þ: 'th', Þ: 'th',
  ø: 'o', Ø: 'o',
  // Eastern European / strokes
  ł: 'l', Ł: 'l',
  đ: 'd', Đ: 'd',
  ħ: 'h', Ħ: 'h',
};

// Common English stopwords, lowercased. Compared case-insensitively so preserve
// / upper modes still drop them. Stripped only when `stripStopwords` is on.
const STOPWORDS = new Set([
  'the', 'a', 'an', 'and', 'or', 'but', 'of', 'to', 'in', 'on', 'at', 'for',
  'with', 'by', 'from',
]);

/** Break text into a list of clean ASCII words (transliterated, diacritics stripped, cased). */
function tokenize(text: string, options: SlugifyOptions): string[] {
  const mode: SlugCase = options.case ?? 'lower';
  const ascii = text
    // 1. transliterate letters that don't decompose on their own
    .replace(/[^\x00-\x7F]/g, (ch) => TRANSLIT[ch] ?? ch)
    // 2. decompose accented characters into base + combining marks
    .normalize('NFKD')
    // 3. drop combining diacritical marks (U+0300-U+036F)
    .replace(/[̀-ͯ]/g, '')
    // 4. collapse every run of non-alphanumeric characters to a single space
    .replace(/[^a-zA-Z0-9]+/g, ' ')
    .trim();

  let words = ascii ? ascii.split(' ') : [];
  if (mode === 'upper') words = words.map((w) => w.toUpperCase());
  else if (mode === 'lower') words = words.map((w) => w.toLowerCase());
  // mode === 'preserve' -> leave the original casing untouched

  if (options.stripStopwords) {
    words = words.filter((w) => !STOPWORDS.has(w.toLowerCase()));
  }
  return words;
}

/** Truncate `slug` to `max` chars at the last whole-word boundary. */
function truncateAtWord(slug: string, separator: string, max: number): string {
  if (slug.length <= max) return slug;
  if (separator === '') return slug.slice(0, max); // nothing to break on - hard cut
  const cut = slug.slice(0, max);
  const last = cut.lastIndexOf(separator);
  return last > 0 ? cut.slice(0, last) : cut; // no separator found -> hard cut
}

/**
 * Convert arbitrary text into a URL-safe slug.
 *
 * Pipeline: transliterate ligatures -> NFKD decompose -> strip combining
 * diacritics -> collapse non-alphanumeric runs -> split into words -> apply
 * casing -> (strip stopwords) -> join with the separator -> (truncate to max
 * length at a word boundary). Anything that can't be transliterated to ASCII
 * (emoji, CJK, etc.) collapses to a word separator. Never throws.
 */
export function slugify(text: string, options: SlugifyOptions = {}): string {
  const separator = options.separator ?? '-';
  const slug = tokenize(text, options).join(separator);
  if (options.maxLength && options.maxLength > 0) {
    return truncateAtWord(slug, separator, options.maxLength);
  }
  return slug;
}

/** Slugify each line independently (batch mode). Returns exactly one slug per input line. */
export function slugifyLines(text: string, options: SlugifyOptions = {}): string[] {
  return text.split(/\r?\n/).map((line) => slugify(line, options));
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →