Skip to content

Slugify — Java source

Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.

This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.

// slugify — Java port: URL-safe slugs with locale-aware Unicode transliteration.
import java.text.Normalizer;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.List;
import java.util.Locale;
import java.util.Map;
import java.util.Set;
import java.util.stream.Collectors;

/** Pure slug logic, ported from the canonical TS lib. Deterministic and
 * total: any input yields a slug, never an exception. */
final class Slugify {

    /** Letter casing for the produced slug. */
    public enum Case { LOWER, PRESERVE, UPPER }

    /** Options mirror the TS SlugifyOptions; fluent setters default everything. */
    public static final class Options {
        String separator = "-";
        int maxLength;                       // <= 0 = unlimited
        Case casing = Case.LOWER;
        boolean stripStopwords;

        public Options separator(String s) { separator = s; return this; }
        public Options maxLength(int n) { maxLength = n; return this; }
        public Options casing(Case c) { casing = c; return this; }
        public Options stripStopwords(boolean b) { stripStopwords = b; return this; }
    }

    public static Options options() { return new Options(); }

    // Letters/ligatures NFKD does not decompose into an ASCII base + combining
    // mark. Accented Latin (á é ñ …) needs no entry: NFKD splits it and the
    // U+0300..U+036F strip below drops the diacritic.
    private static final Map<String, String> TRANSLIT = Map.ofEntries(
        Map.entry("ß", "ss"),
        Map.entry("æ", "ae"), Map.entry("Æ", "ae"),
        Map.entry("œ", "oe"), Map.entry("Œ", "oe"),
        Map.entry("ff", "ff"), Map.entry("fi", "fi"), Map.entry("fl", "fl"),
        Map.entry("ffi", "ffi"), Map.entry("ffl", "ffl"),
        Map.entry("ſt", "st"), Map.entry("st", "st"),
        Map.entry("ð", "d"), Map.entry("Ð", "d"),
        Map.entry("þ", "th"), Map.entry("Þ", "th"),
        Map.entry("ø", "o"), Map.entry("Ø", "o"),
        Map.entry("ł", "l"), Map.entry("Ł", "l"),
        Map.entry("đ", "d"), Map.entry("Đ", "d"),
        Map.entry("ħ", "h"), Map.entry("Ħ", "h"));

    private static final Set<String> STOPWORDS = Set.of(
        "the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at",
        "for", "with", "by", "from");

    /** Break text into clean ASCII words (transliterated, diacritics stripped,
     * cased). Mirrors the TS tokenize(). */
    private static List<String> tokenize(String text, Options o) {
        StringBuilder t = new StringBuilder(text.length());
        for (int i = 0; i < text.length(); i++) {
            String ch = String.valueOf(text.charAt(i));
            t.append(TRANSLIT.getOrDefault(ch, ch));
        }
        String ascii = Normalizer.normalize(t, Normalizer.Form.NFKD)
                .replaceAll("[̀-ͯ]", "")    // drop combining diacritics
                .replaceAll("[^a-zA-Z0-9]+", " ")     // collapse runs to one space
                .trim();
        List<String> words = ascii.isEmpty()
                ? new ArrayList<>()
                : new ArrayList<>(Arrays.asList(ascii.split(" ")));
        for (int i = 0; i < words.size(); i++) {
            String w = words.get(i);
            words.set(i, o.casing == Case.UPPER ? w.toUpperCase(Locale.ROOT)
                    : o.casing == Case.LOWER ? w.toLowerCase(Locale.ROOT) : w);
        }
        if (o.stripStopwords) {
            words.removeIf(w -> STOPWORDS.contains(w.toLowerCase(Locale.ROOT)));
        }
        return words;
    }

    // Truncate to max chars at the last whole-word boundary (hard cut when the
    // separator is empty or absent from the head).
    private static String truncateAtWord(String slug, String separator, int max) {
        if (slug.length() <= max) return slug;
        String cut = slug.substring(0, max);
        if (separator.isEmpty()) return cut;
        int last = cut.lastIndexOf(separator);
        return last > 0 ? cut.substring(0, last) : cut;
    }

    /** Convert arbitrary text into a URL-safe slug. */
    public static String slugify(String text, Options o) {
        String slug = String.join(o.separator, tokenize(text, o));
        return o.maxLength > 0 ? truncateAtWord(slug, o.separator, o.maxLength) : slug;
    }

    public static String slugify(String text) { return slugify(text, new Options()); }

    /** Slugify each line independently (batch mode). */
    public static List<String> slugifyLines(String text, Options o) {
        return Arrays.stream(text.split("\r?\n", -1))
                .map(line -> slugify(line, o))
                .collect(Collectors.toList());
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →