Slugify — C++ source
Generate clean, URL-safe slugs from any text with locale-aware Unicode transliteration. Accents, emoji, and punctuation are handled automatically - runs entirely in your browser.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// slugify — C++ port: URL-safe slugs with locale-aware Unicode transliteration.
#include <algorithm>
#include <cstddef>
#include <cstdint>
#include <string>
#include <string_view>
#include <unordered_map>
#include <vector>
/// Letter casing for the produced slug.
enum class case_mode { lower, preserve, upper };
/// Options mirror the TS SlugifyOptions; every field defaults.
struct options {
std::string separator = "-";
int max_length = 0; // <= 0 = unlimited
case_mode casing = case_mode::lower;
bool strip_stopwords = false;
};
// Non-decomposing letters keyed by code point (U'ß', U'æ', ...). Accented
// Latin needs no entry — its combining mark is stripped below. C++ has no
// NFKD normalizer in the stdlib, so (like the Rust port) we reproduce the TS
// pipeline's observable effect: transliterate these letters, drop the
// U+0300..U+036F combining block, keep only ASCII alphanumerics as word
// characters. Precomposed NFC accented input is not decomposed.
inline const std::unordered_map<char32_t, std::string_view>& translit_table() {
static const std::unordered_map<char32_t, std::string_view> t = {
{U'ß', "ss"},
{U'æ', "ae"}, {U'Æ', "ae"}, {U'œ', "oe"}, {U'Œ', "oe"},
{U'ff', "ff"}, {U'fi', "fi"}, {U'fl', "fl"}, {U'ffi', "ffi"},
{U'ffl', "ffl"}, {U'ſt', "st"}, {U'st', "st"},
{U'ð', "d"}, {U'Ð', "d"}, {U'þ', "th"}, {U'Þ', "th"},
{U'ø', "o"}, {U'Ø', "o"}, {U'ł', "l"}, {U'Ł', "l"},
{U'đ', "d"}, {U'Đ', "d"}, {U'ħ', "h"}, {U'Ħ', "h"},
};
return t;
}
inline bool is_stopword(std::string_view w) { // ASCII-only, case-insensitive
static constexpr std::string_view list[] = {
"the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at",
"for", "with", "by", "from",
};
for (auto sw : list) {
if (w.size() != sw.size()) continue;
bool eq = true;
for (std::size_t i = 0; i < w.size(); i++) {
char a = w[i], b = sw[i];
a += (a >= 'A' && a <= 'Z') ? 32 : 0;
b += (b >= 'A' && b <= 'Z') ? 32 : 0;
eq = eq && a == b;
}
if (eq) return true;
}
return false;
}
inline bool is_ascii_alnum(char c) {
return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9');
}
// Decode one UTF-8 rune at s[i] (advancing i). Invalid or over-3-byte
// sequences yield 0xFFFD — untransliterable, so it just acts as a word
// separator and the port stays total like the TS original.
inline char32_t next_cp(std::string_view s, std::size_t& i) {
unsigned char c = static_cast<unsigned char>(s[i]);
if (c < 0x80) { ++i; return c; }
auto cont = [&](std::size_t k) {
return k < s.size() && (static_cast<unsigned char>(s[k]) & 0xC0) == 0x80;
};
if ((c & 0xE0) == 0xC0 && cont(i + 1)) {
char32_t cp = (static_cast<char32_t>(c & 0x1F) << 6)
| (static_cast<unsigned char>(s[i + 1]) & 0x3F);
i += 2;
return cp;
}
if ((c & 0xF0) == 0xE0 && cont(i + 1) && cont(i + 2)) {
char32_t cp = (static_cast<char32_t>(c & 0x0F) << 12)
| (static_cast<char32_t>(static_cast<unsigned char>(s[i + 1]) & 0x3F) << 6)
| (static_cast<unsigned char>(s[i + 2]) & 0x3F);
i += 3;
return cp;
}
++i;
return 0xFFFD;
}
/// Break text into clean ASCII words (transliterated, diacritics stripped,
/// cased per options). Mirrors the TS tokenize().
inline std::vector<std::string> tokenize(std::string_view text, const options& o) {
const auto& translit = translit_table();
std::vector<std::string> words;
std::string cur;
for (std::size_t i = 0; i < text.size();) {
char32_t cp = next_cp(text, i);
if (auto it = translit.find(cp); it != translit.end()) {
cur += it->second; // transliterated ligature
} else if (cp < 0x80 && is_ascii_alnum(static_cast<char>(cp))) {
cur += static_cast<char>(cp); // word character
} else if (!cur.empty()) { // separator (combining marks, emoji, CJK, ...)
words.push_back(std::move(cur));
cur.clear();
}
}
if (!cur.empty()) words.push_back(std::move(cur));
for (auto& w : words) {
switch (o.casing) {
case case_mode::lower:
for (char& c : w) if (c >= 'A' && c <= 'Z') c = static_cast<char>(c + 32);
break;
case case_mode::upper:
for (char& c : w) if (c >= 'a' && c <= 'z') c = static_cast<char>(c - 32);
break;
case case_mode::preserve: break;
}
}
if (o.strip_stopwords)
words.erase(std::remove_if(words.begin(), words.end(),
[](const std::string& w) { return is_stopword(w); }),
words.end());
return words;
}
// Truncate to max chars at the last whole-word boundary (hard cut when the
// separator is empty or absent from the head).
inline std::string truncate_at_word(std::string slug, std::string_view separator, int max) {
if (static_cast<int>(slug.size()) <= max) return slug;
std::string cut = slug.substr(0, static_cast<std::size_t>(max));
if (separator.empty()) return cut;
auto pos = cut.rfind(separator);
return pos > 0 ? cut.substr(0, pos) : cut; // TS: index 0 is not a boundary
}
/// Convert arbitrary text into a URL-safe slug.
inline std::string slugify(std::string_view text, const options& o = {}) {
const std::string& sep = o.separator;
std::string slug;
auto words = tokenize(text, o);
for (std::size_t k = 0; k < words.size(); k++) {
if (k > 0) slug += sep;
slug += words[k];
}
return o.max_length > 0 ? truncate_at_word(std::move(slug), sep, o.max_length) : slug;
}
/// Slugify each line independently (batch mode), matching the TS /\r?\n/ split.
inline std::vector<std::string> slugify_lines(std::string_view text, const options& o = {}) {
std::vector<std::string> out;
std::size_t i = 0;
while (i < text.size()) {
auto nl = text.find('\n', i);
auto end = nl == std::string_view::npos ? text.size() : nl;
auto n = end - i;
if (n > 0 && text[i + n - 1] == '\r') n--; // strip the CR of CRLF
out.push_back(slugify(text.substr(i, n), o));
i = nl == std::string_view::npos ? end : nl + 1;
}
return out;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →