Slugify — Rust source
Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! slugify — URL-safe slug generator with locale-aware Unicode transliteration.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source: CosmoDev polyglot showcase port of the Slugify tool, ported from
//! src/lib/slugify.ts (the canonical TypeScript implementation).
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; never panics (public API returns Strings, no Result).
//! - Functionally equivalent to the TS reference: same inputs -> same outputs.
//! - Self-contained: std only (no crates.io dependencies — no `unicode-normalization`).
//!
//! Pipeline: transliterate ligatures -> NFKD decompose -> strip combining
//! diacritics -> collapse non-alphanumeric runs -> split into words -> apply
//! casing -> (strip stopwords) -> join with the separator -> (truncate at a
//! word boundary). Anything that can't be transliterated to ASCII (emoji, CJK,
//! ...) collapses to a word separator.
//!
//! Unicode note: Rust's std has no NFKD normalizer (the `unicode-normalization`
//! crate provides one, but the brief forbids external deps). We reproduce the
//! SAME observable effect the TS code relies on:
//! (a) map the fixed set of non-decomposing ligatures (TRANSLIT) to ASCII,
//! (b) strip every code point in the combining-diacritic block U+0300..U+036F.
//! In the TS source NFKD exists solely to split a precomposed accented letter
//! into base + U+03xx combining mark so step (b) can drop the mark. For the
//! Latin range — the script this tool targets — that decomposition always lands
//! in U+0300..U+036F, so removing the range reproduces the TS output for every
//! input the tool is designed for. This is a deliberate, documented trade-off
//! to keep the port dependency-free.
/// Letter casing for the produced slug.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
pub enum Case {
/// Lowercase everything (the default).
#[default]
Lower,
/// Leave original letter casing untouched.
Preserve,
/// Uppercase everything.
Upper,
}
/// Options mirror the TS `SlugifyOptions`. Every field has a sensible default,
/// so callers can construct it incrementally with `..Default::default()`.
#[derive(Debug, Clone, Default)]
pub struct Options {
/// Joining chars. `None` -> default `-`. `Some("")` concatenates words.
pub separator: Option<String>,
/// Max slug length; truncated at the last word boundary at or under the
/// limit. `None` or `Some(0)` / negative = unlimited.
pub max_length: Option<usize>,
/// Letter casing of the result. Defaults to [`Case::Lower`].
pub case: Case,
/// Strip common English stopwords (the, a, an, of, ...). Defaults to false.
pub strip_stopwords: bool,
}
/// One entry of the non-decomposing transliteration table: a source rune and
/// its ASCII replacement (which may be multiple chars, e.g. ß -> "ss").
type Ligature = (char, &'static str);
/// Letters / ligatures that NFKD does NOT decompose into an ASCII base +
/// combining mark. Mapping them up front turns "Straße" -> "strasse",
/// "Æsir" -> "aesir", "Søren" -> "soren". Accented Latin letters (á é ñ ü ...)
/// need no entry — their diacritic lands in U+0300..U+036F and is stripped.
const TRANSLIT: &[Ligature] = &[
// Germanic
('ß', "ss"),
// Latin ligatures
('æ', "ae"), ('Æ', "ae"),
('œ', "oe"), ('Œ', "oe"),
('ff', "ff"), ('fi', "fi"), ('fl', "fl"), ('ffi', "ffi"), ('ffl', "ffl"), ('ſt', "st"), ('st', "st"),
// Nordic / insular
('ð', "d"), ('Ð', "d"),
('þ', "th"), ('Þ', "th"),
('ø', "o"), ('Ø', "o"),
// Eastern European / strokes
('ł', "l"), ('Ł', "l"),
('đ', "d"), ('Đ', "d"),
('ħ', "h"), ('Ħ', "h"),
];
/// Common English stopwords, lowercased. Compared case-insensitively so
/// `Preserve` / `Upper` modes still drop them. A `const` slice scanned linearly
/// — small enough that hashing would cost more than it saves, and avoids a
/// once-cell initialization.
const STOPWORDS: &[&str] = &[
"the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at", "for",
"with", "by", "from",
];
/// Look up `c` in the fixed ligature table. Returns `Some(replacement)` when
/// found (the replacement may be multi-char). Table is tiny, so a linear scan
/// is the right tool.
fn transliterate_once(c: char) -> Option<&'static str> {
TRANSLIT.iter().find(|(from, _)| *from == c).map(|(_, to)| *to)
}
/// Reports whether `c` is a combining diacritical mark in U+0300..U+036F —
/// the block the TS source strips after NFKD.
fn is_combining_mark(c: char) -> bool {
(c as u32) >= 0x0300 && (c as u32) <= 0x036F
}
/// Mirrors TS's `[a-zA-Z0-9]` class: the only characters that survive into
/// words. We deliberately restrict to the ASCII range so non-ASCII letters
/// (Greek, Cyrillic, ...) and emoji act as word separators — exactly as TS's
/// `[^a-zA-Z0-9]+` rule dictates.
fn is_ascii_alnum(c: char) -> bool {
c.is_ascii_alphanumeric()
}
/// Break text into clean ASCII words: ligatures transliterated, diacritics
/// stripped, and cased per `opts`.
///
/// Single pass over chars: transliterate non-ASCII ligatures, drop combining
/// marks, push ASCII alphanumerics into the current word, and treat every
/// other char (whitespace, punctuation, symbols, untransliterated non-ASCII)
/// as a word boundary.
fn tokenize(text: &str, opts: &Options) -> Vec<String> {
let mut words: Vec<String> = Vec::new();
let mut current = String::new();
for c in text.chars() {
if is_combining_mark(c) {
continue; // strip diacritic
}
if let Some(rep) = transliterate_once(c) {
current.push_str(rep);
continue;
}
if is_ascii_alnum(c) {
current.push(c);
continue;
}
// Separator char — flush the in-progress word if any.
if !current.is_empty() {
words.push(std::mem::take(&mut current));
}
}
if !current.is_empty() {
words.push(current);
}
// Apply casing. `Preserve` is a no-op.
match opts.case {
Case::Upper => {
for w in &mut words {
*w = w.to_uppercase().collect();
}
}
Case::Lower => {
for w in &mut words {
*w = w.to_lowercase().collect();
}
}
Case::Preserve => {}
}
// Optionally drop English stopwords (case-insensitive match). `retain`
// is the idiomatic in-place filter on Vec.
if opts.strip_stopwords {
words.retain(|w| !STOPWORDS.contains(&w.to_lowercase().as_str()));
}
words
}
/// Resolve the effective separator: `None` and `Some("")` both mean "no
/// separator" is impossible in TS — `None` maps to the default `-`, while
/// `Some("")` genuinely concatenates. We honor that distinction here.
fn effective_separator(opts: &Options) -> &str {
match &opts.separator {
None => "-",
Some(s) => s.as_str(),
}
}
/// Truncate `slug` to `max` chars at the last whole-word boundary.
///
/// Operates on `char` counts (not bytes) to match TS's `String.prototype.slice`
/// semantics. The slug is ASCII-only by construction, so byte and char lengths
/// coincide — but we use the char view defensively for clarity.
fn truncate_at_word(slug: &str, separator: &str, max: usize) -> String {
let char_count = slug.chars().count();
if char_count <= max {
return slug.to_string();
}
let cut: String = slug.chars().take(max).collect();
if separator.is_empty() {
return cut; // nothing to break on — hard cut
}
match cut.rfind(separator) {
// Match TS: only treat as a word boundary if it's not at index 0.
Some(idx) if idx > 0 => cut[..idx].to_string(),
_ => cut, // no usable separator found -> hard cut
}
}
/// Convert arbitrary text into a URL-safe slug. Never panics; an empty or
/// all-symbol input simply yields an empty `String`.
pub fn slugify(text: &str, opts: &Options) -> String {
let separator = effective_separator(opts);
let slug = tokenize(text, opts).join(separator);
if let Some(max) = opts.max_length {
if max > 0 {
return truncate_at_word(&slug, separator, max);
}
}
slug
}
/// Convenience wrapper using default options — the common "give me a normal
/// kebab slug" case.
pub fn slugify_default(text: &str) -> String {
slugify(text, &Options::default())
}
/// Slugify each line independently (batch mode). Returns exactly one slug per
/// input line, matching the TS `/\r?\n/` split.
pub fn slugify_lines(text: &str, opts: &Options) -> Vec<String> {
// Split on `\n` after stripping a preceding `\r` reproduces `\r?\n`
// exactly (handles `\n`, `\r\n`, and a lone `\r` becomes an empty line +
// the remainder, identical to the regex).
text.split('\n')
.map(|line| {
let line = line.strip_suffix('\r').unwrap_or(line);
slugify(line, opts)
})
.collect()
}
// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn basic_kebab() {
assert_eq!(slugify_default("Hello, World!"), "hello-world");
}
#[test]
fn transliteration_and_diacritics() {
assert_eq!(slugify_default("Straße & Æsir: Søren"), "strasse-aesir-soren");
assert_eq!(slugify_default("Café Niño"), "cafe-nino");
}
#[test]
fn collapse_untransliterable() {
// Emoji and CJK collapse to separators, never crash.
assert_eq!(slugify_default("Hello 🌍世界"), "hello");
}
#[test]
fn max_length_at_word_boundary() {
let opts = Options { max_length: Some(10), ..Default::default() };
assert_eq!(slugify("a b c d e f", &opts), "a-b-c-d");
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →