Skip to content

Slugify — Zig source

Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// slugify — Zig port: URL-safe slugs with locale-aware Unicode transliteration.
const std = @import("std");
const Allocator = std.mem.Allocator;

/// Letter casing for the produced slug.
pub const Case = enum { lower, preserve, upper };

/// Options mirror the TS SlugifyOptions; every field defaults.
pub const Options = struct {
    separator: []const u8 = "-",
    max_length: usize = 0, // 0 = unlimited
    case: Case = .lower,
    strip_stopwords: bool = false,
};

// Non-decomposing letters keyed by code point (0x00DF = ß, 0x00E6 = æ, ...).
// Accented Latin needs no entry — its combining mark is stripped below.
// Zig's stdlib has no NFKD normalizer, so (like the Rust port) we reproduce
// the TS pipeline's observable effect: transliterate these letters, drop the
// U+0300..U+036F combining block, keep only ASCII alphanumerics as word
// characters. Precomposed NFC accented input is not decomposed.
const Ligature = struct { cp: u21, to: []const u8 };
const TRANSLIT = [_]Ligature{
    .{ .cp = 0x00DF, .to = "ss" },
    .{ .cp = 0x00E6, .to = "ae" }, .{ .cp = 0x00C6, .to = "ae" },
    .{ .cp = 0x0153, .to = "oe" }, .{ .cp = 0x0152, .to = "oe" },
    .{ .cp = 0xFB00, .to = "ff" }, .{ .cp = 0xFB01, .to = "fi" },
    .{ .cp = 0xFB02, .to = "fl" }, .{ .cp = 0xFB03, .to = "ffi" },
    .{ .cp = 0xFB04, .to = "ffl" }, .{ .cp = 0xFB05, .to = "st" },
    .{ .cp = 0xFB06, .to = "st" },
    .{ .cp = 0x00F0, .to = "d" }, .{ .cp = 0x00D0, .to = "d" },
    .{ .cp = 0x00FE, .to = "th" }, .{ .cp = 0x00DE, .to = "th" },
    .{ .cp = 0x00F8, .to = "o" }, .{ .cp = 0x00D8, .to = "o" },
    .{ .cp = 0x0142, .to = "l" }, .{ .cp = 0x0141, .to = "l" },
    .{ .cp = 0x0111, .to = "d" }, .{ .cp = 0x0110, .to = "d" },
    .{ .cp = 0x0127, .to = "h" }, .{ .cp = 0x0126, .to = "h" },
};

const STOPWORDS = [_][]const u8{
    "the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at",
    "for", "with", "by", "from",
};

fn isAsciiAlnum(c: u8) bool {
    return (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or (c >= '0' and c <= '9');
}

// Decode one UTF-8 rune at s[i] (advancing i). Invalid or over-3-byte
// sequences yield 0xFFFD — untransliterable, so it just acts as a word
// separator and the port stays total like the TS original.
fn nextCp(s: []const u8, i: *usize) u21 {
    const b = s[i.*];
    const len = std.unicode.utf8ByteSequenceLength(b) catch {
        i.* += 1;
        return 0xFFFD;
    };
    if (i.* + len > s.len) {
        i.* += 1;
        return 0xFFFD;
    }
    const cp = std.unicode.utf8Decode(s[i.* .. i.* + len]) catch {
        i.* += 1;
        return 0xFFFD;
    };
    i.* += len;
    return cp;
}

// ASCII-only, case-insensitive stopword test.
fn isStopword(w: []const u8) bool {
    for (STOPWORDS) |sw| {
        if (w.len != sw.len) continue;
        var eq = true;
        for (w, sw) |a, b| {
            eq = eq and std.ascii.toLower(a) == std.ascii.toLower(b);
        }
        if (eq) return true;
    }
    return false;
}

// Case-map the buffer in place (words are ASCII by construction), then hand
// ownership of a copy to the list.
fn pushWord(a: Allocator, words: *std.ArrayList([]const u8), cur: *std.ArrayList(u8), mode: Case) Allocator.Error!void {
    for (cur.items) |*c| {
        switch (mode) {
            .lower => c.* = std.ascii.toLower(c.*),
            .upper => c.* = std.ascii.toUpper(c.*),
            .preserve => {},
        }
    }
    const owned = try a.dupe(u8, cur.items);
    words.append(owned) catch |e| {
        a.free(owned);
        return e;
    };
    cur.clearRetainingCapacity();
}

/// Break text into clean ASCII words (transliterated, diacritics stripped,
/// cased per options). Mirrors the TS tokenize(). Caller owns the result;
/// release with freeWords().
pub fn tokenize(a: Allocator, text: []const u8, o: Options) Allocator.Error![][]const u8 {
    var words = std.ArrayList([]const u8).init(a);
    errdefer {
        for (words.items) |w| a.free(w);
        words.deinit();
    }
    var cur = std.ArrayList(u8).init(a);
    defer cur.deinit();

    var i: usize = 0;
    while (i < text.len) {
        const cp = nextCp(text, &i);
        var rep: ?[]const u8 = null;
        for (TRANSLIT) |e| {
            if (e.cp == cp) {
                rep = e.to;
                break;
            }
        }
        if (rep) |r| {
            try cur.appendSlice(r); // transliterated ligature
        } else if (cp < 0x80 and isAsciiAlnum(@intCast(cp))) {
            try cur.append(@intCast(cp)); // word character
        } else if (cur.items.len > 0) {
            // separator rune (combining marks, emoji, CJK, ...) — flush the word
            try pushWord(a, &words, &cur, o.case);
        }
    }
    if (cur.items.len > 0) try pushWord(a, &words, &cur, o.case);

    if (o.strip_stopwords) {
        var w: usize = 0;
        for (words.items) |word| {
            if (isStopword(word)) {
                a.free(word);
                continue;
            }
            words.items[w] = word;
            w += 1;
        }
        words.shrinkRetainingCapacity(w);
    }
    return words.toOwnedSlice();
}

/// Release a list returned by tokenize / slugifyLines.
pub fn freeWords(a: Allocator, ws: []const []const u8) void {
    for (ws) |w| a.free(@constCast(w));
    a.free(ws);
}

// Truncate to max chars at the last whole-word boundary (hard cut when the
// separator is empty or absent from the head).
fn truncateAtWord(a: Allocator, slug: []const u8, sep: []const u8, max: usize) Allocator.Error![]u8 {
    if (slug.len <= max) return a.dupe(u8, slug);
    const cut = slug[0..max];
    if (sep.len == 0) return a.dupe(u8, cut);
    const idx = std.mem.lastIndexOf(u8, cut, sep) orelse return a.dupe(u8, cut);
    if (idx == 0) return a.dupe(u8, cut); // TS: index 0 is not a boundary
    return a.dupe(u8, cut[0..idx]);
}

/// Convert arbitrary text into a URL-safe slug. Caller owns the result.
pub fn slugify(a: Allocator, text: []const u8, o: Options) Allocator.Error![]u8 {
    const words = try tokenize(a, text, o);
    defer freeWords(a, words);

    var out = std.ArrayList(u8).init(a);
    errdefer out.deinit();
    for (words, 0..) |w, k| {
        if (k > 0) try out.appendSlice(o.separator);
        try out.appendSlice(w);
    }
    const slug = try out.toOwnedSlice();
    if (o.max_length > 0) {
        const t = try truncateAtWord(a, slug, o.separator, o.max_length);
        a.free(slug);
        return t;
    }
    return slug;
}

/// Slugify each line independently (batch mode), matching the TS /\r?\n/ split.
/// Caller owns the result; release with freeWords().
pub fn slugifyLines(a: Allocator, text: []const u8, o: Options) Allocator.Error![][]const u8 {
    var out = std.ArrayList([]const u8).init(a);
    errdefer {
        for (out.items) |l| a.free(@constCast(l));
        out.deinit();
    }
    var i: usize = 0;
    while (i < text.len) {
        const nl = std.mem.indexOfScalarPos(u8, text, i, '\n');
        const end = nl orelse text.len;
        var n = end - i;
        if (n > 0 and text[i + n - 1] == '\r') n -= 1; // strip the CR of CRLF
        const line = try slugify(a, text[i .. i + n], o);
        out.append(line) catch |e| {
            a.free(line);
            return e;
        };
        i = if (nl) |p| p + 1 else end;
    }
    return out.toOwnedSlice();
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →