Sort Lines & Remove Duplicates — C++ source
Alphabetize, reverse, shuffle, dedupe, or length-sort lines of text. Supports case-insensitive and natural sorting (file2 before file10).
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// sort-lines — multi-mode line sorter. Language: C++ (20). Port of src/lib/sortLines.ts — same contract as this dir's go.go (the live Go twin): split on "\n", apply the mode (asc/desc/length-asc/length-desc/reverse/shuffle/unique), join back. std::stable_sort keeps ties in input order (the TS sort is stable); shuffle uses mulberry32 with uint32 wraparound, so a seed reproduces the TS order.
#include <algorithm>
#include <cctype>
#include <cstdint>
#include <iostream>
#include <set>
#include <string>
#include <vector>
enum class Mode { Asc, Desc, LengthAsc, LengthDesc, Reverse, Shuffle, Unique };
struct Options {
bool case_sensitive = true; // TS defaults: case sensitive, no trim, no natural, seed 1
bool trim = false, natural = false;
int seed = 1;
};
struct Result {
std::vector<std::string> lines;
std::string text;
int removed_duplicates = 0;
};
// mulberry32 — deterministic PRNG (not cryptographic); a seed reproduces the same shuffle.
struct Mulberry32 {
uint32_t a;
explicit Mulberry32(uint32_t seed) : a(seed) {}
double next() {
a += 0x6d2b79f5u;
uint32_t t = (a ^ (a >> 15)) * (1u | a);
t = (t + (t ^ (t >> 7)) * (61u | t)) ^ t;
return double(t ^ (t >> 14)) / 4294967296.0;
}
};
static bool is_digit_ch(char c) { return c >= '0' && c <= '9'; }
static std::string lowercase(const std::string &s) {
std::string out = s;
std::transform(out.begin(), out.end(), out.begin(), [](unsigned char c) { return char(std::tolower(c)); });
return out;
}
// Natural order: split into ASCII digit / non-digit runs, compare chunk-wise so numbers
// order by value — "file2" sorts before "file10".
static int natural_compare(const std::string &a, const std::string &b, bool cs) {
auto chunks = [](const std::string &s) {
std::vector<std::string> out;
size_t i = 0;
while (i < s.size()) {
bool d = is_digit_ch(s[i]);
size_t j = i + 1;
while (j < s.size() && is_digit_ch(s[j]) == d) j++;
out.push_back(s.substr(i, j - i));
i = j;
}
if (out.empty()) out.push_back(""); // "" -> [""] like the TS ?? fallback
return out;
};
auto noz = [](const std::string &s) { // numeric value: strip leading zeros
size_t k = 0;
while (k < s.size() && s[k] == '0') k++;
return s.substr(k);
};
std::vector<std::string> aa = chunks(cs ? a : lowercase(a)), bb = chunks(cs ? b : lowercase(b));
for (size_t i = 0; i < std::min(aa.size(), bb.size()); i++) {
const std::string &x = aa[i], &y = bb[i];
bool dn = !x.empty() && is_digit_ch(x[0]), dm = !y.empty() && is_digit_ch(y[0]);
if (dn != dm) return x < y ? -1 : 1; // digit run vs text run: raw compare
if (dn) {
std::string vx = noz(x), vy = noz(y);
if (vx.size() != vy.size()) return vx.size() < vy.size() ? -1 : 1;
if (vx != vy) return vx < vy ? -1 : 1;
} else if (x != y) {
return x < y ? -1 : 1;
}
}
return int(aa.size()) - int(bb.size());
}
static Result sort_lines(const std::string &input, Mode mode, const Options &o = Options{}) {
std::vector<std::string> lines; // split("\n") semantics: trailing empty line kept
std::string cur;
for (char c : input) {
if (c == '\n') { lines.push_back(cur); cur.clear(); } else cur += c;
}
lines.push_back(cur);
if (o.trim) {
for (std::string &l : lines) {
l.erase(0, l.find_first_not_of(" \t\r"));
l.erase(l.find_last_not_of(" \t\r") + 1);
}
}
auto norm = [&](const std::string &s) { return o.case_sensitive ? s : lowercase(s); };
int removed = 0;
switch (mode) {
case Mode::Unique: { // keep each normalized line's first occurrence; count the rest
std::set<std::string> seen;
std::vector<std::string> out;
for (const std::string &l : lines) {
if (seen.count(norm(l))) removed++;
else { seen.insert(norm(l)); out.push_back(l); }
}
lines = out;
break;
}
case Mode::Shuffle: { // Fisher-Yates with the seeded PRNG -> reproducible order
Mulberry32 rng(uint32_t(o.seed));
for (size_t i = lines.size(); i-- > 1; ) {
size_t j = size_t(rng.next() * double(i + 1));
std::swap(lines[i], lines[j]);
}
break;
}
case Mode::Reverse:
std::reverse(lines.begin(), lines.end());
break;
case Mode::LengthAsc:
case Mode::LengthDesc: { // stable by length (ties keep input order), then reverse for desc
std::stable_sort(lines.begin(), lines.end(), [](const std::string &x, const std::string &y) { return x.size() < y.size(); });
if (mode == Mode::LengthDesc) std::reverse(lines.begin(), lines.end());
break;
}
default: { // Asc / Desc — stable, ties keep input order in both directions
int dir = mode == Mode::Asc ? 1 : -1;
std::stable_sort(lines.begin(), lines.end(), [&](const std::string &x, const std::string &y) {
int c = o.natural ? natural_compare(x, y, o.case_sensitive)
: (norm(x) < norm(y) ? -1 : norm(x) > norm(y) ? 1 : 0);
return c * dir < 0;
});
break;
}
}
Result r;
r.lines = lines;
r.removed_duplicates = removed;
for (size_t i = 0; i < lines.size(); i++) { if (i) r.text += '\n'; r.text += lines[i]; }
return r;
}
int main() {
std::string text = "pear\napple\nBanana\napple\nfig10\nfig2";
auto show = [](const char *label, const Result &r) {
std::cout << label << r.text;
if (r.removed_duplicates) std::cout << " (removed " << r.removed_duplicates << ")";
std::cout << "\n";
};
show("asc: ", sort_lines(text, Mode::Asc));
show("ci-asc: ", sort_lines(text, Mode::Asc, Options{.case_sensitive = false}));
show("uniq: ", sort_lines(text, Mode::Unique, Options{.case_sensitive = false}));
show("shuf-7: ", sort_lines(text, Mode::Shuffle, Options{.seed = 7}));
show("nat-ci: ", sort_lines(text, Mode::Asc, Options{.case_sensitive = false, .natural = true}));
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →