Skip to content

Slugify — C++ source

Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// slugify — C++ port: URL-safe slugs with locale-aware Unicode transliteration.
#include <algorithm>
#include <cstddef>
#include <cstdint>
#include <string>
#include <string_view>
#include <unordered_map>
#include <vector>

/// Letter casing for the produced slug.
enum class case_mode { lower, preserve, upper };

/// Options mirror the TS SlugifyOptions; every field defaults.
struct options {
    std::string separator = "-";
    int max_length = 0;               // <= 0 = unlimited
    case_mode casing = case_mode::lower;
    bool strip_stopwords = false;
};

// Non-decomposing letters keyed by code point (U'ß', U'æ', ...). Accented
// Latin needs no entry — its combining mark is stripped below. C++ has no
// NFKD normalizer in the stdlib, so (like the Rust port) we reproduce the TS
// pipeline's observable effect: transliterate these letters, drop the
// U+0300..U+036F combining block, keep only ASCII alphanumerics as word
// characters. Precomposed NFC accented input is not decomposed.
inline const std::unordered_map<char32_t, std::string_view>& translit_table() {
    static const std::unordered_map<char32_t, std::string_view> t = {
        {U'ß', "ss"},
        {U'æ', "ae"}, {U'Æ', "ae"}, {U'œ', "oe"}, {U'Œ', "oe"},
        {U'ff', "ff"}, {U'fi', "fi"}, {U'fl', "fl"}, {U'ffi', "ffi"},
        {U'ffl', "ffl"}, {U'ſt', "st"}, {U'st', "st"},
        {U'ð', "d"}, {U'Ð', "d"}, {U'þ', "th"}, {U'Þ', "th"},
        {U'ø', "o"}, {U'Ø', "o"}, {U'ł', "l"}, {U'Ł', "l"},
        {U'đ', "d"}, {U'Đ', "d"}, {U'ħ', "h"}, {U'Ħ', "h"},
    };
    return t;
}

inline bool is_stopword(std::string_view w) { // ASCII-only, case-insensitive
    static constexpr std::string_view list[] = {
        "the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at",
        "for", "with", "by", "from",
    };
    for (auto sw : list) {
        if (w.size() != sw.size()) continue;
        bool eq = true;
        for (std::size_t i = 0; i < w.size(); i++) {
            char a = w[i], b = sw[i];
            a += (a >= 'A' && a <= 'Z') ? 32 : 0;
            b += (b >= 'A' && b <= 'Z') ? 32 : 0;
            eq = eq && a == b;
        }
        if (eq) return true;
    }
    return false;
}

inline bool is_ascii_alnum(char c) {
    return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9');
}

// Decode one UTF-8 rune at s[i] (advancing i). Invalid or over-3-byte
// sequences yield 0xFFFD — untransliterable, so it just acts as a word
// separator and the port stays total like the TS original.
inline char32_t next_cp(std::string_view s, std::size_t& i) {
    unsigned char c = static_cast<unsigned char>(s[i]);
    if (c < 0x80) { ++i; return c; }
    auto cont = [&](std::size_t k) {
        return k < s.size() && (static_cast<unsigned char>(s[k]) & 0xC0) == 0x80;
    };
    if ((c & 0xE0) == 0xC0 && cont(i + 1)) {
        char32_t cp = (static_cast<char32_t>(c & 0x1F) << 6)
                    | (static_cast<unsigned char>(s[i + 1]) & 0x3F);
        i += 2;
        return cp;
    }
    if ((c & 0xF0) == 0xE0 && cont(i + 1) && cont(i + 2)) {
        char32_t cp = (static_cast<char32_t>(c & 0x0F) << 12)
                    | (static_cast<char32_t>(static_cast<unsigned char>(s[i + 1]) & 0x3F) << 6)
                    | (static_cast<unsigned char>(s[i + 2]) & 0x3F);
        i += 3;
        return cp;
    }
    ++i;
    return 0xFFFD;
}

/// Break text into clean ASCII words (transliterated, diacritics stripped,
/// cased per options). Mirrors the TS tokenize().
inline std::vector<std::string> tokenize(std::string_view text, const options& o) {
    const auto& translit = translit_table();
    std::vector<std::string> words;
    std::string cur;
    for (std::size_t i = 0; i < text.size();) {
        char32_t cp = next_cp(text, i);
        if (auto it = translit.find(cp); it != translit.end()) {
            cur += it->second;           // transliterated ligature
        } else if (cp < 0x80 && is_ascii_alnum(static_cast<char>(cp))) {
            cur += static_cast<char>(cp); // word character
        } else if (!cur.empty()) {       // separator (combining marks, emoji, CJK, ...)
            words.push_back(std::move(cur));
            cur.clear();
        }
    }
    if (!cur.empty()) words.push_back(std::move(cur));

    for (auto& w : words) {
        switch (o.casing) {
        case case_mode::lower:
            for (char& c : w) if (c >= 'A' && c <= 'Z') c = static_cast<char>(c + 32);
            break;
        case case_mode::upper:
            for (char& c : w) if (c >= 'a' && c <= 'z') c = static_cast<char>(c - 32);
            break;
        case case_mode::preserve: break;
        }
    }
    if (o.strip_stopwords)
        words.erase(std::remove_if(words.begin(), words.end(),
                                   [](const std::string& w) { return is_stopword(w); }),
                    words.end());
    return words;
}

// Truncate to max chars at the last whole-word boundary (hard cut when the
// separator is empty or absent from the head).
inline std::string truncate_at_word(std::string slug, std::string_view separator, int max) {
    if (static_cast<int>(slug.size()) <= max) return slug;
    std::string cut = slug.substr(0, static_cast<std::size_t>(max));
    if (separator.empty()) return cut;
    auto pos = cut.rfind(separator);
    return pos > 0 ? cut.substr(0, pos) : cut; // TS: index 0 is not a boundary
}

/// Convert arbitrary text into a URL-safe slug.
inline std::string slugify(std::string_view text, const options& o = {}) {
    const std::string& sep = o.separator;
    std::string slug;
    auto words = tokenize(text, o);
    for (std::size_t k = 0; k < words.size(); k++) {
        if (k > 0) slug += sep;
        slug += words[k];
    }
    return o.max_length > 0 ? truncate_at_word(std::move(slug), sep, o.max_length) : slug;
}

/// Slugify each line independently (batch mode), matching the TS /\r?\n/ split.
inline std::vector<std::string> slugify_lines(std::string_view text, const options& o = {}) {
    std::vector<std::string> out;
    std::size_t i = 0;
    while (i < text.size()) {
        auto nl = text.find('\n', i);
        auto end = nl == std::string_view::npos ? text.size() : nl;
        auto n = end - i;
        if (n > 0 && text[i + n - 1] == '\r') n--; // strip the CR of CRLF
        out.push_back(slugify(text.substr(i, n), o));
        i = nl == std::string_view::npos ? end : nl + 1;
    }
    return out;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →