Slugify — C source
Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/* slugify — C port: URL-safe slugs with locale-aware Unicode transliteration. */
#include <stdbool.h>
#include <stddef.h>
#include <stdlib.h>
#include <string.h>
/* Letter casing for the produced slug. */
typedef enum { SLUG_LOWER, SLUG_PRESERVE, SLUG_UPPER } slug_case;
typedef struct {
const char *separator; /* NULL -> "-" */
int max_length; /* <= 0 = unlimited */
slug_case case_mode;
bool strip_stopwords;
} slugify_options;
/* Non-decomposing letters keyed by Unicode code point (U+00DF = ß, U+00E6 = æ,
* ...). Accented Latin needs no entry — its combining mark is stripped below.
* C has no NFKD normalizer, so (like the Rust port) we reproduce the TS
* pipeline's observable effect: transliterate these letters, drop the
* U+0300..U+036F combining block, keep only ASCII alphanumerics as word
* characters. Precomposed NFC accented input is not decomposed. */
static const struct { unsigned cp; const char *to; } TRANSLIT[] = {
{ 0x00DF, "ss" },
{ 0x00E6, "ae" }, { 0x00C6, "ae" }, { 0x0153, "oe" }, { 0x0152, "oe" },
{ 0xFB00, "ff" }, { 0xFB01, "fi" }, { 0xFB02, "fl" }, { 0xFB03, "ffi" },
{ 0xFB04, "ffl" }, { 0xFB05, "st" }, { 0xFB06, "st" },
{ 0x00F0, "d" }, { 0x00D0, "d" }, { 0x00FE, "th" }, { 0x00DE, "th" },
{ 0x00F8, "o" }, { 0x00D8, "o" }, { 0x0142, "l" }, { 0x0141, "l" },
{ 0x0111, "d" }, { 0x0110, "d" }, { 0x0127, "h" }, { 0x0126, "h" },
};
static const char *const STOPWORDS[] = {
"the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at",
"for", "with", "by", "from",
};
static bool is_ascii_alnum(unsigned char c)
{
return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9');
}
static char lower_ascii(unsigned char c) { return (c >= 'A' && c <= 'Z') ? (char)(c + 32) : (char)c; }
/* Decode one UTF-8 rune at s[*i] (advancing *i). Invalid or over-3-byte
* sequences yield 0xFFFD — an untransliterable code point that simply acts
* as a word separator, so the port stays total like the TS original. */
static unsigned next_cp(const unsigned char *s, size_t n, size_t *i)
{
unsigned c = s[*i];
if (c < 0x80) { (*i)++; return c; }
if ((c & 0xE0) == 0xC0 && *i + 1 < n && (s[*i + 1] & 0xC0) == 0x80) {
unsigned cp = ((c & 0x1F) << 6) | (s[*i + 1] & 0x3F);
*i += 2;
return cp;
}
if ((c & 0xF0) == 0xE0 && *i + 2 < n && (s[*i + 1] & 0xC0) == 0x80
&& (s[*i + 2] & 0xC0) == 0x80) {
unsigned cp = ((c & 0x0F) << 12) | ((s[*i + 1] & 0x3F) << 6) | (s[*i + 2] & 0x3F);
*i += 3;
return cp;
}
(*i)++;
return 0xFFFD;
}
static bool is_stopword(const char *w) /* ASCII-only word, case-insensitive scan */
{
for (size_t k = 0; k < sizeof STOPWORDS / sizeof STOPWORDS[0]; k++) {
const char *a = STOPWORDS[k], *b = w;
while (*a && *b && lower_ascii((unsigned char)*a) == lower_ascii((unsigned char)*b)) { a++; b++; }
if (*a == '\0' && *b == '\0')
return true;
}
return false;
}
static char *push_word(char ***words, size_t *len, size_t *cap, char *owned)
{
if (*len == *cap) {
size_t grown = *cap * 2;
char **p = realloc(*words, grown * sizeof *p);
if (!p) { free(owned); return NULL; }
*words = p;
*cap = grown;
}
(*words)[(*len)++] = owned;
return owned;
}
/* Break text into clean ASCII words: transliterated, diacritics stripped,
* cased per options. Returns a NULL-terminated array of malloc'd words
* (NULL on OOM); release with slugify_words_free(). */
static char **slugify_words(const char *text, const slugify_options *o)
{
size_t cap = 8, len = 0, i = 0, n = strlen(text), cl = 0;
char **words = malloc(cap * sizeof *words);
char *cur = malloc(n + 1);
if (!words || !cur)
goto oom;
while (i < n) {
unsigned cp = next_cp((const unsigned char *)text, n, &i);
const char *rep = NULL;
for (size_t k = 0; k < sizeof TRANSLIT / sizeof TRANSLIT[0]; k++)
if (TRANSLIT[k].cp == cp) { rep = TRANSLIT[k].to; break; }
if (rep != NULL) { /* transliterated ligature */
while (*rep) cur[cl++] = *rep++;
} else if (cp < 0x80 && is_ascii_alnum((unsigned char)cp)) {
cur[cl++] = (char)cp; /* word character */
} else if (cl > 0) { /* separator rune (incl. marks, emoji, CJK) */
cur[cl] = '\0';
if (!push_word(&words, &len, &cap, cur))
goto oom;
cur = malloc(n + 1);
cl = 0;
if (!cur)
goto oom;
}
}
if (cl > 0) {
cur[cl] = '\0';
if (!push_word(&words, &len, &cap, cur))
goto oom;
} else {
free(cur);
}
words[len] = NULL;
for (size_t k = 0; k < len; k++) { /* casing: words are ASCII by construction */
if (o->case_mode == SLUG_LOWER)
for (char *p = words[k]; *p; p++) *p = lower_ascii((unsigned char)*p);
else if (o->case_mode == SLUG_UPPER)
for (char *p = words[k]; *p; p++)
if (*p >= 'a' && *p <= 'z') *p = (char)(*p - 32);
}
if (o->strip_stopwords) {
size_t w = 0;
for (size_t k = 0; k < len; k++) {
if (is_stopword(words[k])) { free(words[k]); continue; }
words[w++] = words[k];
}
words[w] = NULL;
}
return words;
oom:
free(cur);
for (size_t k = 0; k < len; k++) free(words[k]);
free(words);
return NULL;
}
static void slugify_words_free(char **words)
{
if (!words) return;
for (size_t k = 0; words[k]; k++) free(words[k]);
free(words);
}
/* Truncate to max chars at the last whole-word boundary (hard cut when the
* separator is empty or absent from the head). Caller frees. */
static char *truncate_at_word(const char *slug, const char *sep, int max)
{
size_t n = strlen(slug);
if ((int)n <= max) { char *dup = malloc(n + 1); return dup ? memcpy(dup, slug, n + 1) : NULL; }
char *cut = malloc((size_t)max + 1);
if (!cut) return NULL;
memcpy(cut, slug, (size_t)max);
cut[max] = '\0';
if (*sep == '\0') return cut;
const char *found = NULL; /* last separator occurrence inside the cut */
for (const char *p = cut; (p = strstr(p, sep)) != NULL; p++) found = p;
if (found == NULL || found == cut) return cut; /* TS: index 0 is not a boundary */
size_t keep = (size_t)(found - cut);
char *out = malloc(keep + 1);
if (!out) { free(cut); return NULL; }
memcpy(out, cut, keep);
out[keep] = '\0';
free(cut);
return out;
}
/* Convert arbitrary text into a URL-safe slug. Caller frees; NULL on OOM. */
char *slugify(const char *text, const slugify_options *opts)
{
static const slugify_options def = { NULL, 0, SLUG_LOWER, false };
const slugify_options *o = opts ? opts : &def;
const char *sep = o->separator ? o->separator : "-";
char **words = slugify_words(text, o);
if (!words) return NULL;
size_t sepl = strlen(sep), total = 1;
for (size_t k = 0; words[k]; k++) total += strlen(words[k]) + sepl;
char *slug = malloc(total);
if (slug) {
char *p = slug;
for (size_t k = 0; words[k]; k++) {
if (k > 0) { memcpy(p, sep, sepl); p += sepl; }
size_t wl = strlen(words[k]);
memcpy(p, words[k], wl);
p += wl;
}
*p = '\0';
}
slugify_words_free(words);
if (!slug) return NULL;
if (o->max_length > 0) {
char *out = truncate_at_word(slug, sep, o->max_length);
free(slug);
return out;
}
return slug;
}
/* Slugify each line independently (batch mode), matching the TS /\r?\n/ split.
* Returns a NULL-terminated array of malloc'd lines; free each element and
* the array itself (NULL on OOM). */
char **slugify_lines(const char *text, const slugify_options *opts)
{
size_t cap = 8, len = 0;
char **out = malloc(cap * sizeof *out);
if (!out) return NULL;
const char *p = text;
if (*p == '\0') { /* JS ''.split(...) yields exactly one empty line */
char *s = slugify("", opts);
if (!s) { free(out); return NULL; }
out[len++] = s;
}
while (*p) {
const char *nl = strchr(p, '\n');
const char *end = nl ? nl : p + strlen(p);
size_t n = (size_t)(end - p);
if (n > 0 && p[n - 1] == '\r') n--; /* strip the CR of CRLF */
char *line = malloc(n + 1);
if (!line) goto fail;
memcpy(line, p, n);
line[n] = '\0';
char *s = slugify(line, opts);
free(line);
if (!s) goto fail;
if (len == cap) {
char **grown = realloc(out, (cap *= 2) * sizeof *grown);
if (!grown) { free(s); goto fail; }
out = grown;
}
out[len++] = s;
p = nl ? nl + 1 : end;
}
if (len == cap) { /* room for the terminator */
char **grown = realloc(out, (cap + 1) * sizeof *grown);
if (!grown) goto fail;
out = grown;
}
out[len] = NULL;
return out;
fail:
for (size_t k = 0; k < len; k++) free(out[k]);
free(out);
return NULL;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →