Skip to content

Slugify — C source

Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/* slugify — C port: URL-safe slugs with locale-aware Unicode transliteration. */
#include <stdbool.h>
#include <stddef.h>
#include <stdlib.h>
#include <string.h>

/* Letter casing for the produced slug. */
typedef enum { SLUG_LOWER, SLUG_PRESERVE, SLUG_UPPER } slug_case;

typedef struct {
    const char *separator; /* NULL -> "-" */
    int max_length;        /* <= 0 = unlimited */
    slug_case case_mode;
    bool strip_stopwords;
} slugify_options;

/* Non-decomposing letters keyed by Unicode code point (U+00DF = ß, U+00E6 = æ,
 * ...). Accented Latin needs no entry — its combining mark is stripped below.
 * C has no NFKD normalizer, so (like the Rust port) we reproduce the TS
 * pipeline's observable effect: transliterate these letters, drop the
 * U+0300..U+036F combining block, keep only ASCII alphanumerics as word
 * characters. Precomposed NFC accented input is not decomposed. */
static const struct { unsigned cp; const char *to; } TRANSLIT[] = {
    { 0x00DF, "ss" },
    { 0x00E6, "ae" }, { 0x00C6, "ae" }, { 0x0153, "oe" }, { 0x0152, "oe" },
    { 0xFB00, "ff" }, { 0xFB01, "fi" }, { 0xFB02, "fl" }, { 0xFB03, "ffi" },
    { 0xFB04, "ffl" }, { 0xFB05, "st" }, { 0xFB06, "st" },
    { 0x00F0, "d" }, { 0x00D0, "d" }, { 0x00FE, "th" }, { 0x00DE, "th" },
    { 0x00F8, "o" }, { 0x00D8, "o" }, { 0x0142, "l" }, { 0x0141, "l" },
    { 0x0111, "d" }, { 0x0110, "d" }, { 0x0127, "h" }, { 0x0126, "h" },
};

static const char *const STOPWORDS[] = {
    "the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at",
    "for", "with", "by", "from",
};

static bool is_ascii_alnum(unsigned char c)
{
    return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9');
}

static char lower_ascii(unsigned char c) { return (c >= 'A' && c <= 'Z') ? (char)(c + 32) : (char)c; }

/* Decode one UTF-8 rune at s[*i] (advancing *i). Invalid or over-3-byte
 * sequences yield 0xFFFD — an untransliterable code point that simply acts
 * as a word separator, so the port stays total like the TS original. */
static unsigned next_cp(const unsigned char *s, size_t n, size_t *i)
{
    unsigned c = s[*i];
    if (c < 0x80) { (*i)++; return c; }
    if ((c & 0xE0) == 0xC0 && *i + 1 < n && (s[*i + 1] & 0xC0) == 0x80) {
        unsigned cp = ((c & 0x1F) << 6) | (s[*i + 1] & 0x3F);
        *i += 2;
        return cp;
    }
    if ((c & 0xF0) == 0xE0 && *i + 2 < n && (s[*i + 1] & 0xC0) == 0x80
        && (s[*i + 2] & 0xC0) == 0x80) {
        unsigned cp = ((c & 0x0F) << 12) | ((s[*i + 1] & 0x3F) << 6) | (s[*i + 2] & 0x3F);
        *i += 3;
        return cp;
    }
    (*i)++;
    return 0xFFFD;
}

static bool is_stopword(const char *w) /* ASCII-only word, case-insensitive scan */
{
    for (size_t k = 0; k < sizeof STOPWORDS / sizeof STOPWORDS[0]; k++) {
        const char *a = STOPWORDS[k], *b = w;
        while (*a && *b && lower_ascii((unsigned char)*a) == lower_ascii((unsigned char)*b)) { a++; b++; }
        if (*a == '\0' && *b == '\0')
            return true;
    }
    return false;
}

static char *push_word(char ***words, size_t *len, size_t *cap, char *owned)
{
    if (*len == *cap) {
        size_t grown = *cap * 2;
        char **p = realloc(*words, grown * sizeof *p);
        if (!p) { free(owned); return NULL; }
        *words = p;
        *cap = grown;
    }
    (*words)[(*len)++] = owned;
    return owned;
}

/* Break text into clean ASCII words: transliterated, diacritics stripped,
 * cased per options. Returns a NULL-terminated array of malloc'd words
 * (NULL on OOM); release with slugify_words_free(). */
static char **slugify_words(const char *text, const slugify_options *o)
{
    size_t cap = 8, len = 0, i = 0, n = strlen(text), cl = 0;
    char **words = malloc(cap * sizeof *words);
    char *cur = malloc(n + 1);
    if (!words || !cur)
        goto oom;

    while (i < n) {
        unsigned cp = next_cp((const unsigned char *)text, n, &i);
        const char *rep = NULL;
        for (size_t k = 0; k < sizeof TRANSLIT / sizeof TRANSLIT[0]; k++)
            if (TRANSLIT[k].cp == cp) { rep = TRANSLIT[k].to; break; }
        if (rep != NULL) {                /* transliterated ligature */
            while (*rep) cur[cl++] = *rep++;
        } else if (cp < 0x80 && is_ascii_alnum((unsigned char)cp)) {
            cur[cl++] = (char)cp;         /* word character */
        } else if (cl > 0) {              /* separator rune (incl. marks, emoji, CJK) */
            cur[cl] = '\0';
            if (!push_word(&words, &len, &cap, cur))
                goto oom;
            cur = malloc(n + 1);
            cl = 0;
            if (!cur)
                goto oom;
        }
    }
    if (cl > 0) {
        cur[cl] = '\0';
        if (!push_word(&words, &len, &cap, cur))
            goto oom;
    } else {
        free(cur);
    }
    words[len] = NULL;

    for (size_t k = 0; k < len; k++) {    /* casing: words are ASCII by construction */
        if (o->case_mode == SLUG_LOWER)
            for (char *p = words[k]; *p; p++) *p = lower_ascii((unsigned char)*p);
        else if (o->case_mode == SLUG_UPPER)
            for (char *p = words[k]; *p; p++)
                if (*p >= 'a' && *p <= 'z') *p = (char)(*p - 32);
    }
    if (o->strip_stopwords) {
        size_t w = 0;
        for (size_t k = 0; k < len; k++) {
            if (is_stopword(words[k])) { free(words[k]); continue; }
            words[w++] = words[k];
        }
        words[w] = NULL;
    }
    return words;

oom:
    free(cur);
    for (size_t k = 0; k < len; k++) free(words[k]);
    free(words);
    return NULL;
}

static void slugify_words_free(char **words)
{
    if (!words) return;
    for (size_t k = 0; words[k]; k++) free(words[k]);
    free(words);
}

/* Truncate to max chars at the last whole-word boundary (hard cut when the
 * separator is empty or absent from the head). Caller frees. */
static char *truncate_at_word(const char *slug, const char *sep, int max)
{
    size_t n = strlen(slug);
    if ((int)n <= max) { char *dup = malloc(n + 1); return dup ? memcpy(dup, slug, n + 1) : NULL; }
    char *cut = malloc((size_t)max + 1);
    if (!cut) return NULL;
    memcpy(cut, slug, (size_t)max);
    cut[max] = '\0';
    if (*sep == '\0') return cut;
    const char *found = NULL; /* last separator occurrence inside the cut */
    for (const char *p = cut; (p = strstr(p, sep)) != NULL; p++) found = p;
    if (found == NULL || found == cut) return cut; /* TS: index 0 is not a boundary */
    size_t keep = (size_t)(found - cut);
    char *out = malloc(keep + 1);
    if (!out) { free(cut); return NULL; }
    memcpy(out, cut, keep);
    out[keep] = '\0';
    free(cut);
    return out;
}

/* Convert arbitrary text into a URL-safe slug. Caller frees; NULL on OOM. */
char *slugify(const char *text, const slugify_options *opts)
{
    static const slugify_options def = { NULL, 0, SLUG_LOWER, false };
    const slugify_options *o = opts ? opts : &def;
    const char *sep = o->separator ? o->separator : "-";

    char **words = slugify_words(text, o);
    if (!words) return NULL;
    size_t sepl = strlen(sep), total = 1;
    for (size_t k = 0; words[k]; k++) total += strlen(words[k]) + sepl;
    char *slug = malloc(total);
    if (slug) {
        char *p = slug;
        for (size_t k = 0; words[k]; k++) {
            if (k > 0) { memcpy(p, sep, sepl); p += sepl; }
            size_t wl = strlen(words[k]);
            memcpy(p, words[k], wl);
            p += wl;
        }
        *p = '\0';
    }
    slugify_words_free(words);
    if (!slug) return NULL;
    if (o->max_length > 0) {
        char *out = truncate_at_word(slug, sep, o->max_length);
        free(slug);
        return out;
    }
    return slug;
}

/* Slugify each line independently (batch mode), matching the TS /\r?\n/ split.
 * Returns a NULL-terminated array of malloc'd lines; free each element and
 * the array itself (NULL on OOM). */
char **slugify_lines(const char *text, const slugify_options *opts)
{
    size_t cap = 8, len = 0;
    char **out = malloc(cap * sizeof *out);
    if (!out) return NULL;
    const char *p = text;
    if (*p == '\0') { /* JS ''.split(...) yields exactly one empty line */
        char *s = slugify("", opts);
        if (!s) { free(out); return NULL; }
        out[len++] = s;
    }
    while (*p) {
        const char *nl = strchr(p, '\n');
        const char *end = nl ? nl : p + strlen(p);
        size_t n = (size_t)(end - p);
        if (n > 0 && p[n - 1] == '\r') n--; /* strip the CR of CRLF */
        char *line = malloc(n + 1);
        if (!line) goto fail;
        memcpy(line, p, n);
        line[n] = '\0';
        char *s = slugify(line, opts);
        free(line);
        if (!s) goto fail;
        if (len == cap) {
            char **grown = realloc(out, (cap *= 2) * sizeof *grown);
            if (!grown) { free(s); goto fail; }
            out = grown;
        }
        out[len++] = s;
        p = nl ? nl + 1 : end;
    }
    if (len == cap) { /* room for the terminator */
        char **grown = realloc(out, (cap + 1) * sizeof *grown);
        if (!grown) goto fail;
        out = grown;
    }
    out[len] = NULL;
    return out;

fail:
    for (size_t k = 0; k < len; k++) free(out[k]);
    free(out);
    return NULL;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →