Skip to content

Slugify — JavaScript source

Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.

This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

/**
 * slugify - URL-safe slug generator with locale-aware Unicode transliteration.
 *
 * Language:   JavaScript (ES2020+, runs unmodified in Node 16+ and modern browsers)
 * Source:     CosmoDev polyglot showcase port of the Slugify tool, ported from
 *             src/lib/slugify.ts (the canonical TypeScript implementation).
 * License:    display source - part of CosmoDev's polyglot tool pages.
 *
 * Design goals:
 *   - Pure + deterministic; never throws.
 *   - Functionally equivalent to the TS reference: same inputs -> same outputs.
 *   - Self-contained: stdlib only (no npm dependencies).
 *
 * Pipeline: transliterate ligatures -> NFKD decompose -> strip combining
 * diacritics -> collapse non-alphanumeric runs -> split into words -> apply
 * casing -> (strip stopwords) -> join with the separator -> (truncate at a
 * word boundary). Anything that can't be transliterated to ASCII (emoji, CJK,
 * ...) collapses to a word separator.
 */

'use strict';

/** Letter casing for the produced slug. */
// (No `export type` in plain JS - keep the JSDoc union as the contract.)
/**
 * @typedef {('lower' | 'preserve' | 'upper')} SlugCase
 */

/**
 * Options shape. Keys are all optional.
 * @typedef {Object} SlugifyOptions
 * @property {string} [separator]      Character(s) joining words. Defaults to '-'. Empty string concatenates.
 * @property {number} [maxLength]      Max slug length; truncated at the last word boundary at or under the limit. 0 / negative = unlimited.
 * @property {SlugCase} [case]         Letter casing of the result. Defaults to 'lower'.
 * @property {boolean} [stripStopwords] Strip common English stopwords (the, a, an, of, ...). Defaults to false.
 */

/**
 * Letters / ligatures that NFKD does NOT decompose into an ASCII base + a
 * combining mark. We map them to ASCII up front so "Straße" -> "strasse",
 * "Æsir" -> "aesir", "Søren" -> "soren". Accented Latin letters (á é ñ ü ...)
 * need no entry here - NFKD splits them into base + diacritic and we strip the
 * diacritic below.
 */
const TRANSLIT = {
  // Germanic
  ß: 'ss',
  // Latin ligatures
  æ: 'ae', Æ: 'ae',
  œ: 'oe', Œ: 'oe',
  ff: 'ff', fi: 'fi', fl: 'fl', ffi: 'ffi', ffl: 'ffl', ſt: 'st', st: 'st',
  // Nordic / insular
  ð: 'd', Ð: 'd',
  þ: 'th', Þ: 'th',
  ø: 'o', Ø: 'o',
  // Eastern European / strokes
  ł: 'l', Ł: 'l',
  đ: 'd', Đ: 'd',
  ħ: 'h', Ħ: 'h',
};

/**
 * Common English stopwords, lowercased. Compared case-insensitively so
 * `preserve` / `upper` modes still drop them. Stripped only when
 * `stripStopwords` is on.
 */
const STOPWORDS = new Set([
  'the', 'a', 'an', 'and', 'or', 'but', 'of', 'to', 'in', 'on', 'at', 'for',
  'with', 'by', 'from',
]);

/**
 * Break text into a list of clean ASCII words:
 * transliterated, diacritics stripped, and cased per options.
 *
 * @param {string} text
 * @param {SlugifyOptions} options
 * @returns {string[]}
 */
function tokenize(text, options) {
  const mode = options.case ?? 'lower';

  // 1. transliterate the letters that NFKD won't decompose on their own
  // 2. NFKD-decompose accented characters into base + combining marks
  // 3. drop combining diacritical marks (U+0300 .. U+036F)
  // 4. collapse every run of non-alphanumeric characters to a single space
  const ascii = text
    .replace(/[^\x00-\x7F]/g, (ch) => TRANSLIT[ch] ?? ch)
    .normalize('NFKD')
    .replace(/[̀-ͯ]/g, '')
    .replace(/[^a-zA-Z0-9]+/g, ' ')
    .trim();

  let words = ascii ? ascii.split(' ') : [];
  if (mode === 'upper') {
    words = words.map((w) => w.toUpperCase());
  } else if (mode === 'lower') {
    words = words.map((w) => w.toLowerCase());
  }
  // mode === 'preserve' -> leave the original casing untouched

  if (options.stripStopwords) {
    words = words.filter((w) => !STOPWORDS.has(w.toLowerCase()));
  }
  return words;
}

/**
 * Truncate `slug` to `max` chars at the last whole-word boundary.
 *
 * @param {string} slug
 * @param {string} separator
 * @param {number} max
 * @returns {string}
 */
function truncateAtWord(slug, separator, max) {
  if (slug.length <= max) return slug;
  if (separator === '') return slug.slice(0, max); // nothing to break on - hard cut
  const cut = slug.slice(0, max);
  const last = cut.lastIndexOf(separator);
  return last > 0 ? cut.slice(0, last) : cut; // no separator found -> hard cut
}

/**
 * Convert arbitrary text into a URL-safe slug.
 *
 * @param {string} text
 * @param {SlugifyOptions} [options={}]
 * @returns {string}
 */
function slugify(text, options = {}) {
  const separator = options.separator ?? '-';
  const slug = tokenize(text, options).join(separator);
  if (options.maxLength && options.maxLength > 0) {
    return truncateAtWord(slug, separator, options.maxLength);
  }
  return slug;
}

/**
 * Slugify each line independently (batch mode). Returns exactly one slug per
 * input line.
 *
 * @param {string} text
 * @param {SlugifyOptions} [options={}]
 * @returns {string[]}
 */
function slugifyLines(text, options = {}) {
  return text.split(/\r?\n/).map((line) => slugify(line, options));
}

// CommonJS export so the file is consumable from Node without a build step,
// while staying dependency-free and framework-agnostic.
module.exports = { slugify, slugifyLines, tokenize, truncateAtWord, TRANSLIT, STOPWORDS };

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →