Skip to content

Slugify — Kotlin source

Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.

This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.

// slugify — Kotlin port: URL-safe slugs with locale-aware Unicode transliteration.
import java.text.Normalizer

/** Letter casing for the produced slug. */
enum class SlugCase { LOWER, PRESERVE, UPPER }

/** Options mirror the TS SlugifyOptions; every field defaults. */
data class SlugifyOptions(
    val separator: String = "-",
    val maxLength: Int = 0,               // <= 0 = unlimited
    val case: SlugCase = SlugCase.LOWER,
    val stripStopwords: Boolean = false,
)

// Letters/ligatures NFKD does not decompose into an ASCII base + combining
// mark. Accented Latin (á é ñ …) needs no entry: NFKD splits it and the
// U+0300..U+036F strip below drops the diacritic.
private val TRANSLIT = mapOf(
    'ß' to "ss",
    'æ' to "ae", 'Æ' to "ae", 'œ' to "oe", 'Œ' to "oe",
    'ff' to "ff", 'fi' to "fi", 'fl' to "fl", 'ffi' to "ffi", 'ffl' to "ffl", 'ſt' to "st", 'st' to "st",
    'ð' to "d", 'Ð' to "d", 'þ' to "th", 'Þ' to "th", 'ø' to "o", 'Ø' to "o",
    'ł' to "l", 'Ł' to "l", 'đ' to "d", 'Đ' to "d", 'ħ' to "h", 'Ħ' to "h",
)

private val STOPWORDS = setOf(
    "the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at",
    "for", "with", "by", "from",
)

/** Break text into clean ASCII words (transliterated, diacritics stripped,
 * cased). Mirrors the TS tokenize(). */
private fun tokenize(text: String, opts: SlugifyOptions): List<String> {
    val translit = buildString(text.length) { text.forEach { append(TRANSLIT[it] ?: it) } }
    val ascii = Normalizer.normalize(translit, Normalizer.Form.NFKD)
        .replace(Regex("[̀-ͯ]"), "")     // drop combining diacritics
        .replace(Regex("[^a-zA-Z0-9]+"), " ")      // collapse runs to one space
        .trim()
    var words = if (ascii.isEmpty()) emptyList() else ascii.split(" ")
    words = when (opts.case) {
        SlugCase.UPPER -> words.map { it.uppercase() }
        SlugCase.LOWER -> words.map { it.lowercase() }
        SlugCase.PRESERVE -> words
    }
    if (opts.stripStopwords) words = words.filterNot { it.lowercase() in STOPWORDS }
    return words
}

// Truncate to max chars at the last whole-word boundary (hard cut when the
// separator is empty or absent from the head).
private fun truncateAtWord(slug: String, separator: String, max: Int): String {
    if (slug.length <= max) return slug
    val cut = slug.take(max)
    if (separator.isEmpty()) return cut
    val last = cut.lastIndexOf(separator)
    return if (last > 0) cut.take(last) else cut
}

/** Convert arbitrary text into a URL-safe slug. */
fun slugify(text: String, opts: SlugifyOptions = SlugifyOptions()): String {
    val slug = tokenize(text, opts).joinToString(opts.separator)
    return if (opts.maxLength > 0) truncateAtWord(slug, opts.separator, opts.maxLength) else slug
}

/** Slugify each line independently (batch mode). */
fun slugifyLines(text: String, opts: SlugifyOptions = SlugifyOptions()): List<String> =
    text.split(Regex("\r?\n")).map { slugify(it, opts) }

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →