Skip to content

Slugify — Rust source

Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! slugify — URL-safe slug generator with locale-aware Unicode transliteration.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source:   CosmoDev polyglot showcase port of the Slugify tool, ported from
//!           src/lib/slugify.ts (the canonical TypeScript implementation).
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; never panics (public API returns Strings, no Result).
//!   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//!   - Self-contained: std only (no crates.io dependencies — no `unicode-normalization`).
//!
//! Pipeline: transliterate ligatures -> NFKD decompose -> strip combining
//! diacritics -> collapse non-alphanumeric runs -> split into words -> apply
//! casing -> (strip stopwords) -> join with the separator -> (truncate at a
//! word boundary). Anything that can't be transliterated to ASCII (emoji, CJK,
//! ...) collapses to a word separator.
//!
//! Unicode note: Rust's std has no NFKD normalizer (the `unicode-normalization`
//! crate provides one, but the brief forbids external deps). We reproduce the
//! SAME observable effect the TS code relies on:
//!   (a) map the fixed set of non-decomposing ligatures (TRANSLIT) to ASCII,
//!   (b) strip every code point in the combining-diacritic block U+0300..U+036F.
//! In the TS source NFKD exists solely to split a precomposed accented letter
//! into base + U+03xx combining mark so step (b) can drop the mark. For the
//! Latin range — the script this tool targets — that decomposition always lands
//! in U+0300..U+036F, so removing the range reproduces the TS output for every
//! input the tool is designed for. This is a deliberate, documented trade-off
//! to keep the port dependency-free.

/// Letter casing for the produced slug.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
pub enum Case {
    /// Lowercase everything (the default).
    #[default]
    Lower,
    /// Leave original letter casing untouched.
    Preserve,
    /// Uppercase everything.
    Upper,
}

/// Options mirror the TS `SlugifyOptions`. Every field has a sensible default,
/// so callers can construct it incrementally with `..Default::default()`.
#[derive(Debug, Clone, Default)]
pub struct Options {
    /// Joining chars. `None` -> default `-`. `Some("")` concatenates words.
    pub separator: Option<String>,
    /// Max slug length; truncated at the last word boundary at or under the
    /// limit. `None` or `Some(0)` / negative = unlimited.
    pub max_length: Option<usize>,
    /// Letter casing of the result. Defaults to [`Case::Lower`].
    pub case: Case,
    /// Strip common English stopwords (the, a, an, of, ...). Defaults to false.
    pub strip_stopwords: bool,
}

/// One entry of the non-decomposing transliteration table: a source rune and
/// its ASCII replacement (which may be multiple chars, e.g. ß -> "ss").
type Ligature = (char, &'static str);

/// Letters / ligatures that NFKD does NOT decompose into an ASCII base +
/// combining mark. Mapping them up front turns "Straße" -> "strasse",
/// "Æsir" -> "aesir", "Søren" -> "soren". Accented Latin letters (á é ñ ü ...)
/// need no entry — their diacritic lands in U+0300..U+036F and is stripped.
const TRANSLIT: &[Ligature] = &[
    // Germanic
    ('ß', "ss"),
    // Latin ligatures
    ('æ', "ae"), ('Æ', "ae"),
    ('œ', "oe"), ('Œ', "oe"),
    ('ff', "ff"), ('fi', "fi"), ('fl', "fl"), ('ffi', "ffi"), ('ffl', "ffl"), ('ſt', "st"), ('st', "st"),
    // Nordic / insular
    ('ð', "d"), ('Ð', "d"),
    ('þ', "th"), ('Þ', "th"),
    ('ø', "o"), ('Ø', "o"),
    // Eastern European / strokes
    ('ł', "l"), ('Ł', "l"),
    ('đ', "d"), ('Đ', "d"),
    ('ħ', "h"), ('Ħ', "h"),
];

/// Common English stopwords, lowercased. Compared case-insensitively so
/// `Preserve` / `Upper` modes still drop them. A `const` slice scanned linearly
/// — small enough that hashing would cost more than it saves, and avoids a
/// once-cell initialization.
const STOPWORDS: &[&str] = &[
    "the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at", "for",
    "with", "by", "from",
];

/// Look up `c` in the fixed ligature table. Returns `Some(replacement)` when
/// found (the replacement may be multi-char). Table is tiny, so a linear scan
/// is the right tool.
fn transliterate_once(c: char) -> Option<&'static str> {
    TRANSLIT.iter().find(|(from, _)| *from == c).map(|(_, to)| *to)
}

/// Reports whether `c` is a combining diacritical mark in U+0300..U+036F —
/// the block the TS source strips after NFKD.
fn is_combining_mark(c: char) -> bool {
    (c as u32) >= 0x0300 && (c as u32) <= 0x036F
}

/// Mirrors TS's `[a-zA-Z0-9]` class: the only characters that survive into
/// words. We deliberately restrict to the ASCII range so non-ASCII letters
/// (Greek, Cyrillic, ...) and emoji act as word separators — exactly as TS's
/// `[^a-zA-Z0-9]+` rule dictates.
fn is_ascii_alnum(c: char) -> bool {
    c.is_ascii_alphanumeric()
}

/// Break text into clean ASCII words: ligatures transliterated, diacritics
/// stripped, and cased per `opts`.
///
/// Single pass over chars: transliterate non-ASCII ligatures, drop combining
/// marks, push ASCII alphanumerics into the current word, and treat every
/// other char (whitespace, punctuation, symbols, untransliterated non-ASCII)
/// as a word boundary.
fn tokenize(text: &str, opts: &Options) -> Vec<String> {
    let mut words: Vec<String> = Vec::new();
    let mut current = String::new();

    for c in text.chars() {
        if is_combining_mark(c) {
            continue; // strip diacritic
        }
        if let Some(rep) = transliterate_once(c) {
            current.push_str(rep);
            continue;
        }
        if is_ascii_alnum(c) {
            current.push(c);
            continue;
        }
        // Separator char — flush the in-progress word if any.
        if !current.is_empty() {
            words.push(std::mem::take(&mut current));
        }
    }
    if !current.is_empty() {
        words.push(current);
    }

    // Apply casing. `Preserve` is a no-op.
    match opts.case {
        Case::Upper => {
            for w in &mut words {
                *w = w.to_uppercase().collect();
            }
        }
        Case::Lower => {
            for w in &mut words {
                *w = w.to_lowercase().collect();
            }
        }
        Case::Preserve => {}
    }

    // Optionally drop English stopwords (case-insensitive match). `retain`
    // is the idiomatic in-place filter on Vec.
    if opts.strip_stopwords {
        words.retain(|w| !STOPWORDS.contains(&w.to_lowercase().as_str()));
    }

    words
}

/// Resolve the effective separator: `None` and `Some("")` both mean "no
/// separator" is impossible in TS — `None` maps to the default `-`, while
/// `Some("")` genuinely concatenates. We honor that distinction here.
fn effective_separator(opts: &Options) -> &str {
    match &opts.separator {
        None => "-",
        Some(s) => s.as_str(),
    }
}

/// Truncate `slug` to `max` chars at the last whole-word boundary.
///
/// Operates on `char` counts (not bytes) to match TS's `String.prototype.slice`
/// semantics. The slug is ASCII-only by construction, so byte and char lengths
/// coincide — but we use the char view defensively for clarity.
fn truncate_at_word(slug: &str, separator: &str, max: usize) -> String {
    let char_count = slug.chars().count();
    if char_count <= max {
        return slug.to_string();
    }
    let cut: String = slug.chars().take(max).collect();
    if separator.is_empty() {
        return cut; // nothing to break on — hard cut
    }
    match cut.rfind(separator) {
        // Match TS: only treat as a word boundary if it's not at index 0.
        Some(idx) if idx > 0 => cut[..idx].to_string(),
        _ => cut, // no usable separator found -> hard cut
    }
}

/// Convert arbitrary text into a URL-safe slug. Never panics; an empty or
/// all-symbol input simply yields an empty `String`.
pub fn slugify(text: &str, opts: &Options) -> String {
    let separator = effective_separator(opts);
    let slug = tokenize(text, opts).join(separator);
    if let Some(max) = opts.max_length {
        if max > 0 {
            return truncate_at_word(&slug, separator, max);
        }
    }
    slug
}

/// Convenience wrapper using default options — the common "give me a normal
/// kebab slug" case.
pub fn slugify_default(text: &str) -> String {
    slugify(text, &Options::default())
}

/// Slugify each line independently (batch mode). Returns exactly one slug per
/// input line, matching the TS `/\r?\n/` split.
pub fn slugify_lines(text: &str, opts: &Options) -> Vec<String> {
    // Split on `\n` after stripping a preceding `\r` reproduces `\r?\n`
    // exactly (handles `\n`, `\r\n`, and a lone `\r` becomes an empty line +
    // the remainder, identical to the regex).
    text.split('\n')
        .map(|line| {
            let line = line.strip_suffix('\r').unwrap_or(line);
            slugify(line, opts)
        })
        .collect()
}

// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn basic_kebab() {
        assert_eq!(slugify_default("Hello, World!"), "hello-world");
    }

    #[test]
    fn transliteration_and_diacritics() {
        assert_eq!(slugify_default("Straße & Æsir: Søren"), "strasse-aesir-soren");
        assert_eq!(slugify_default("Café Niño"), "cafe-nino");
    }

    #[test]
    fn collapse_untransliterable() {
        // Emoji and CJK collapse to separators, never crash.
        assert_eq!(slugify_default("Hello 🌍世界"), "hello");
    }

    #[test]
    fn max_length_at_word_boundary() {
        let opts = Options { max_length: Some(10), ..Default::default() };
        assert_eq!(slugify("a b c d e f", &opts), "a-b-c-d");
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →