Slugify — JavaScript source
Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.
This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
/**
* slugify - URL-safe slug generator with locale-aware Unicode transliteration.
*
* Language: JavaScript (ES2020+, runs unmodified in Node 16+ and modern browsers)
* Source: CosmoDev polyglot showcase port of the Slugify tool, ported from
* src/lib/slugify.ts (the canonical TypeScript implementation).
* License: display source - part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; never throws.
* - Functionally equivalent to the TS reference: same inputs -> same outputs.
* - Self-contained: stdlib only (no npm dependencies).
*
* Pipeline: transliterate ligatures -> NFKD decompose -> strip combining
* diacritics -> collapse non-alphanumeric runs -> split into words -> apply
* casing -> (strip stopwords) -> join with the separator -> (truncate at a
* word boundary). Anything that can't be transliterated to ASCII (emoji, CJK,
* ...) collapses to a word separator.
*/
'use strict';
/** Letter casing for the produced slug. */
// (No `export type` in plain JS - keep the JSDoc union as the contract.)
/**
* @typedef {('lower' | 'preserve' | 'upper')} SlugCase
*/
/**
* Options shape. Keys are all optional.
* @typedef {Object} SlugifyOptions
* @property {string} [separator] Character(s) joining words. Defaults to '-'. Empty string concatenates.
* @property {number} [maxLength] Max slug length; truncated at the last word boundary at or under the limit. 0 / negative = unlimited.
* @property {SlugCase} [case] Letter casing of the result. Defaults to 'lower'.
* @property {boolean} [stripStopwords] Strip common English stopwords (the, a, an, of, ...). Defaults to false.
*/
/**
* Letters / ligatures that NFKD does NOT decompose into an ASCII base + a
* combining mark. We map them to ASCII up front so "Straße" -> "strasse",
* "Æsir" -> "aesir", "Søren" -> "soren". Accented Latin letters (á é ñ ü ...)
* need no entry here - NFKD splits them into base + diacritic and we strip the
* diacritic below.
*/
const TRANSLIT = {
// Germanic
ß: 'ss',
// Latin ligatures
æ: 'ae', Æ: 'ae',
œ: 'oe', Œ: 'oe',
ff: 'ff', fi: 'fi', fl: 'fl', ffi: 'ffi', ffl: 'ffl', ſt: 'st', st: 'st',
// Nordic / insular
ð: 'd', Ð: 'd',
þ: 'th', Þ: 'th',
ø: 'o', Ø: 'o',
// Eastern European / strokes
ł: 'l', Ł: 'l',
đ: 'd', Đ: 'd',
ħ: 'h', Ħ: 'h',
};
/**
* Common English stopwords, lowercased. Compared case-insensitively so
* `preserve` / `upper` modes still drop them. Stripped only when
* `stripStopwords` is on.
*/
const STOPWORDS = new Set([
'the', 'a', 'an', 'and', 'or', 'but', 'of', 'to', 'in', 'on', 'at', 'for',
'with', 'by', 'from',
]);
/**
* Break text into a list of clean ASCII words:
* transliterated, diacritics stripped, and cased per options.
*
* @param {string} text
* @param {SlugifyOptions} options
* @returns {string[]}
*/
function tokenize(text, options) {
const mode = options.case ?? 'lower';
// 1. transliterate the letters that NFKD won't decompose on their own
// 2. NFKD-decompose accented characters into base + combining marks
// 3. drop combining diacritical marks (U+0300 .. U+036F)
// 4. collapse every run of non-alphanumeric characters to a single space
const ascii = text
.replace(/[^\x00-\x7F]/g, (ch) => TRANSLIT[ch] ?? ch)
.normalize('NFKD')
.replace(/[̀-ͯ]/g, '')
.replace(/[^a-zA-Z0-9]+/g, ' ')
.trim();
let words = ascii ? ascii.split(' ') : [];
if (mode === 'upper') {
words = words.map((w) => w.toUpperCase());
} else if (mode === 'lower') {
words = words.map((w) => w.toLowerCase());
}
// mode === 'preserve' -> leave the original casing untouched
if (options.stripStopwords) {
words = words.filter((w) => !STOPWORDS.has(w.toLowerCase()));
}
return words;
}
/**
* Truncate `slug` to `max` chars at the last whole-word boundary.
*
* @param {string} slug
* @param {string} separator
* @param {number} max
* @returns {string}
*/
function truncateAtWord(slug, separator, max) {
if (slug.length <= max) return slug;
if (separator === '') return slug.slice(0, max); // nothing to break on - hard cut
const cut = slug.slice(0, max);
const last = cut.lastIndexOf(separator);
return last > 0 ? cut.slice(0, last) : cut; // no separator found -> hard cut
}
/**
* Convert arbitrary text into a URL-safe slug.
*
* @param {string} text
* @param {SlugifyOptions} [options={}]
* @returns {string}
*/
function slugify(text, options = {}) {
const separator = options.separator ?? '-';
const slug = tokenize(text, options).join(separator);
if (options.maxLength && options.maxLength > 0) {
return truncateAtWord(slug, separator, options.maxLength);
}
return slug;
}
/**
* Slugify each line independently (batch mode). Returns exactly one slug per
* input line.
*
* @param {string} text
* @param {SlugifyOptions} [options={}]
* @returns {string[]}
*/
function slugifyLines(text, options = {}) {
return text.split(/\r?\n/).map((line) => slugify(line, options));
}
// CommonJS export so the file is consumable from Node without a build step,
// while staying dependency-free and framework-agnostic.
module.exports = { slugify, slugifyLines, tokenize, truncateAtWord, TRANSLIT, STOPWORDS };
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →