Slugify — Zig source
Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
// slugify — Zig port: URL-safe slugs with locale-aware Unicode transliteration.
const std = @import("std");
const Allocator = std.mem.Allocator;
/// Letter casing for the produced slug.
pub const Case = enum { lower, preserve, upper };
/// Options mirror the TS SlugifyOptions; every field defaults.
pub const Options = struct {
separator: []const u8 = "-",
max_length: usize = 0, // 0 = unlimited
case: Case = .lower,
strip_stopwords: bool = false,
};
// Non-decomposing letters keyed by code point (0x00DF = ß, 0x00E6 = æ, ...).
// Accented Latin needs no entry — its combining mark is stripped below.
// Zig's stdlib has no NFKD normalizer, so (like the Rust port) we reproduce
// the TS pipeline's observable effect: transliterate these letters, drop the
// U+0300..U+036F combining block, keep only ASCII alphanumerics as word
// characters. Precomposed NFC accented input is not decomposed.
const Ligature = struct { cp: u21, to: []const u8 };
const TRANSLIT = [_]Ligature{
.{ .cp = 0x00DF, .to = "ss" },
.{ .cp = 0x00E6, .to = "ae" }, .{ .cp = 0x00C6, .to = "ae" },
.{ .cp = 0x0153, .to = "oe" }, .{ .cp = 0x0152, .to = "oe" },
.{ .cp = 0xFB00, .to = "ff" }, .{ .cp = 0xFB01, .to = "fi" },
.{ .cp = 0xFB02, .to = "fl" }, .{ .cp = 0xFB03, .to = "ffi" },
.{ .cp = 0xFB04, .to = "ffl" }, .{ .cp = 0xFB05, .to = "st" },
.{ .cp = 0xFB06, .to = "st" },
.{ .cp = 0x00F0, .to = "d" }, .{ .cp = 0x00D0, .to = "d" },
.{ .cp = 0x00FE, .to = "th" }, .{ .cp = 0x00DE, .to = "th" },
.{ .cp = 0x00F8, .to = "o" }, .{ .cp = 0x00D8, .to = "o" },
.{ .cp = 0x0142, .to = "l" }, .{ .cp = 0x0141, .to = "l" },
.{ .cp = 0x0111, .to = "d" }, .{ .cp = 0x0110, .to = "d" },
.{ .cp = 0x0127, .to = "h" }, .{ .cp = 0x0126, .to = "h" },
};
const STOPWORDS = [_][]const u8{
"the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at",
"for", "with", "by", "from",
};
fn isAsciiAlnum(c: u8) bool {
return (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or (c >= '0' and c <= '9');
}
// Decode one UTF-8 rune at s[i] (advancing i). Invalid or over-3-byte
// sequences yield 0xFFFD — untransliterable, so it just acts as a word
// separator and the port stays total like the TS original.
fn nextCp(s: []const u8, i: *usize) u21 {
const b = s[i.*];
const len = std.unicode.utf8ByteSequenceLength(b) catch {
i.* += 1;
return 0xFFFD;
};
if (i.* + len > s.len) {
i.* += 1;
return 0xFFFD;
}
const cp = std.unicode.utf8Decode(s[i.* .. i.* + len]) catch {
i.* += 1;
return 0xFFFD;
};
i.* += len;
return cp;
}
// ASCII-only, case-insensitive stopword test.
fn isStopword(w: []const u8) bool {
for (STOPWORDS) |sw| {
if (w.len != sw.len) continue;
var eq = true;
for (w, sw) |a, b| {
eq = eq and std.ascii.toLower(a) == std.ascii.toLower(b);
}
if (eq) return true;
}
return false;
}
// Case-map the buffer in place (words are ASCII by construction), then hand
// ownership of a copy to the list.
fn pushWord(a: Allocator, words: *std.ArrayList([]const u8), cur: *std.ArrayList(u8), mode: Case) Allocator.Error!void {
for (cur.items) |*c| {
switch (mode) {
.lower => c.* = std.ascii.toLower(c.*),
.upper => c.* = std.ascii.toUpper(c.*),
.preserve => {},
}
}
const owned = try a.dupe(u8, cur.items);
words.append(owned) catch |e| {
a.free(owned);
return e;
};
cur.clearRetainingCapacity();
}
/// Break text into clean ASCII words (transliterated, diacritics stripped,
/// cased per options). Mirrors the TS tokenize(). Caller owns the result;
/// release with freeWords().
pub fn tokenize(a: Allocator, text: []const u8, o: Options) Allocator.Error![][]const u8 {
var words = std.ArrayList([]const u8).init(a);
errdefer {
for (words.items) |w| a.free(w);
words.deinit();
}
var cur = std.ArrayList(u8).init(a);
defer cur.deinit();
var i: usize = 0;
while (i < text.len) {
const cp = nextCp(text, &i);
var rep: ?[]const u8 = null;
for (TRANSLIT) |e| {
if (e.cp == cp) {
rep = e.to;
break;
}
}
if (rep) |r| {
try cur.appendSlice(r); // transliterated ligature
} else if (cp < 0x80 and isAsciiAlnum(@intCast(cp))) {
try cur.append(@intCast(cp)); // word character
} else if (cur.items.len > 0) {
// separator rune (combining marks, emoji, CJK, ...) — flush the word
try pushWord(a, &words, &cur, o.case);
}
}
if (cur.items.len > 0) try pushWord(a, &words, &cur, o.case);
if (o.strip_stopwords) {
var w: usize = 0;
for (words.items) |word| {
if (isStopword(word)) {
a.free(word);
continue;
}
words.items[w] = word;
w += 1;
}
words.shrinkRetainingCapacity(w);
}
return words.toOwnedSlice();
}
/// Release a list returned by tokenize / slugifyLines.
pub fn freeWords(a: Allocator, ws: []const []const u8) void {
for (ws) |w| a.free(@constCast(w));
a.free(ws);
}
// Truncate to max chars at the last whole-word boundary (hard cut when the
// separator is empty or absent from the head).
fn truncateAtWord(a: Allocator, slug: []const u8, sep: []const u8, max: usize) Allocator.Error![]u8 {
if (slug.len <= max) return a.dupe(u8, slug);
const cut = slug[0..max];
if (sep.len == 0) return a.dupe(u8, cut);
const idx = std.mem.lastIndexOf(u8, cut, sep) orelse return a.dupe(u8, cut);
if (idx == 0) return a.dupe(u8, cut); // TS: index 0 is not a boundary
return a.dupe(u8, cut[0..idx]);
}
/// Convert arbitrary text into a URL-safe slug. Caller owns the result.
pub fn slugify(a: Allocator, text: []const u8, o: Options) Allocator.Error![]u8 {
const words = try tokenize(a, text, o);
defer freeWords(a, words);
var out = std.ArrayList(u8).init(a);
errdefer out.deinit();
for (words, 0..) |w, k| {
if (k > 0) try out.appendSlice(o.separator);
try out.appendSlice(w);
}
const slug = try out.toOwnedSlice();
if (o.max_length > 0) {
const t = try truncateAtWord(a, slug, o.separator, o.max_length);
a.free(slug);
return t;
}
return slug;
}
/// Slugify each line independently (batch mode), matching the TS /\r?\n/ split.
/// Caller owns the result; release with freeWords().
pub fn slugifyLines(a: Allocator, text: []const u8, o: Options) Allocator.Error![][]const u8 {
var out = std.ArrayList([]const u8).init(a);
errdefer {
for (out.items) |l| a.free(@constCast(l));
out.deinit();
}
var i: usize = 0;
while (i < text.len) {
const nl = std.mem.indexOfScalarPos(u8, text, i, '\n');
const end = nl orelse text.len;
var n = end - i;
if (n > 0 and text[i + n - 1] == '\r') n -= 1; // strip the CR of CRLF
const line = try slugify(a, text[i .. i + n], o);
out.append(line) catch |e| {
a.free(line);
return e;
};
i = if (nl) |p| p + 1 else end;
}
return out.toOwnedSlice();
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →