Skip to content

Slugify — Swift source

Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.

This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.

// slugify — Swift port: URL-safe slugs with locale-aware Unicode transliteration.
import Foundation

/// Letter casing for the produced slug.
enum SlugCase: String { case lower, preserve, upper }

/// Options mirror the TS SlugifyOptions; every field defaults.
struct SlugifyOptions {
    var separator: String = "-"
    var maxLength: Int = 0               // <= 0 = unlimited
    var casing: SlugCase = .lower
    var stripStopwords = false
}

/// Pure slug logic, ported from the canonical TS lib. Deterministic and total:
/// any input yields a slug, never a trap.
private let translit: [Character: String] = [
    "ß": "ss",
    "æ": "ae", "Æ": "ae", "œ": "oe", "Œ": "oe",
    "ff": "ff", "fi": "fi", "fl": "fl", "ffi": "ffi", "ffl": "ffl", "ſt": "st", "st": "st",
    "ð": "d", "Ð": "d", "þ": "th", "Þ": "th", "ø": "o", "Ø": "o",
    "ł": "l", "Ł": "l", "đ": "d", "Đ": "d", "ħ": "h", "Ħ": "h",
]

private let stopwords: Set<String> = [
    "the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at",
    "for", "with", "by", "from",
]

/// Break text into clean ASCII words (transliterated, diacritics stripped,
/// cased). Mirrors the TS tokenize().
private func tokenize(_ text: String, _ opts: SlugifyOptions) -> [String] {
    // 1. transliterate the letters NFKD will not decompose on their own
    var t = ""
    for ch in text { t += translit[ch] ?? String(ch) }
    // 2. NFKD-decompose, 3. drop combining diacritics (U+0300...U+036F)
    var ascii = ""
    for scalar in t.decomposedStringWithCompatibilityMapping.unicodeScalars
    where !(0x0300...0x036F).contains(scalar.value) {
        ascii.unicodeScalars.append(scalar)
    }
    // 4. collapse every run of non-alphanumeric characters (split omits empties)
    var words = ascii
        .split { !($0.isASCII && ($0.isLetter || $0.isNumber)) }
        .map(String.init)
    words = opts.casing == .upper ? words.map { $0.uppercased() }
        : opts.casing == .lower ? words.map { $0.lowercased() }
        : words
    if opts.stripStopwords {
        words = words.filter { !stopwords.contains($0.lowercased()) }
    }
    return words
}

/// Truncate to max chars at the last whole-word boundary (hard cut when the
/// separator is empty or absent from the head).
private func truncateAtWord(_ slug: String, _ separator: String, _ max: Int) -> String {
    if slug.count <= max { return slug }
    let cut = String(slug.prefix(max))
    guard !separator.isEmpty,
          let range = cut.range(of: separator, options: .backwards) else { return cut }
    let last = cut.distance(from: cut.startIndex, to: range.lowerBound)
    return last > 0 ? String(cut.prefix(last)) : cut
}

/// Convert arbitrary text into a URL-safe slug.
func slugify(_ text: String, _ opts: SlugifyOptions = SlugifyOptions()) -> String {
    let slug = tokenize(text, opts).joined(separator: opts.separator)
    return opts.maxLength > 0 ? truncateAtWord(slug, opts.separator, opts.maxLength) : slug
}

/// Slugify each line independently (batch mode), matching the TS /\r?\n/ split.
/// Swift folds "\r\n" into ONE Character (a grapheme cluster), so the split
/// tests both spellings; a lone "\r" stays inside its line, like the regex.
func slugifyLines(_ text: String, _ opts: SlugifyOptions = SlugifyOptions()) -> [String] {
    var lines: [String] = []
    var cur = ""
    for ch in text {
        if ch == "\n" || ch == "\r\n" {
            lines.append(String(cur))
            cur = ""
        } else {
            cur.append(ch)
        }
    }
    lines.append(String(cur)) // JS split keeps the tail line ("" for trailing \n)
    return lines.map { slugify($0, opts) }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →