Text Extractor — C++ source
Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
// out of arbitrary text (logs, headers, config).
//
// Language: C++ (C++17, standard library only)
// Source: CosmoDev polyglot showcase port of the Extract tool, ported from
// src/lib/extract.ts (the canonical TypeScript implementation) and
// held in lock-step with its Go twin cli/extract/extract.go.
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; the public API never throws.
// - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
// - The six per-kind patterns are reproduced VERBATIM from the Go twin so the
// "lock-step contract" between the two implementations is auditable at a
// glance.
//
// Engine note: std::regex defaults to the ECMAScript dialect, which supports
// every construct the six patterns use — \b word boundaries, (?:...) groups,
// counted repetition, alternation. std::regex is a backtracking engine while
// Go's regexp is RE2, but both implement leftmost-first (Perl-order)
// alternation and none of the patterns hinge on a backtracking-only or
// RE2-only subtlety, so matches agree on every input. Raw string literals
// keep the pattern text character-for-character identical to the Go twin.
#include <algorithm>
#include <array>
#include <regex>
#include <string>
#include <unordered_set>
#include <vector>
namespace extract {
/// One of the six canonical extraction kinds. Mirrors the TS `ExtractType`
/// union ('url' | 'email' | 'ipv4' | 'ipv6' | 'hash' | 'domain') and Go's
/// `Type`.
enum class Kind { Url, Email, Ipv4, Ipv6, Hash, Domain };
/// Canonical spelling used in the TS/Go string-literal contract.
inline const char* kind_str(Kind k) {
switch (k) {
case Kind::Url: return "url";
case Kind::Email: return "email";
case Kind::Ipv4: return "ipv4";
case Kind::Ipv6: return "ipv6";
case Kind::Hash: return "hash";
case Kind::Domain: return "domain";
}
return ""; // unreachable
}
/// The canonical kinds in display order — Go's `ExtractTypes` / TS's
/// `EXTRACT_TYPES`.
inline constexpr std::array<Kind, 6> ALL_KINDS = {
Kind::Url, Kind::Email, Kind::Ipv4, Kind::Ipv6, Kind::Hash, Kind::Domain,
};
// The per-kind patterns, compiled once per program (`inline` variables have
// exactly one instance across translation units). They mirror the `RE` record
// in src/lib/extract.ts and the `re*` vars in the Go twin VERBATIM: same
// anchors (\b), character classes, and counted repetition.
inline const std::regex RE_URL{R"(https?://[^\s]+)"};
inline const std::regex RE_EMAIL{
R"([a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,})"};
inline const std::regex RE_IPV4{R"(\b(?:\d{1,3}\.){3}\d{1,3}\b)"};
/// Intentionally permissive — a hex/colon run — post-filtered by `is_ipv6`.
inline const std::regex RE_IPV6{R"([0-9a-fA-F:]+)"};
/// md5 (32) / sha1 (40) / sha256 (64) / sha512 (128). `\b` keeps each length
/// honest, so a 64-char run does not also match as a leading 32-char hash.
inline const std::regex RE_HASH{
R"(\b[a-fA-F0-9]{32}\b|\b[a-fA-F0-9]{40}\b|\b[a-fA-F0-9]{64}\b|\b[a-fA-F0-9]{128}\b)"};
inline const std::regex RE_DOMAIN{
R"(\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b)"};
/// Validates a single 1–4 hex-digit IPv6 hextet (regex_match pins both ends,
/// so the ^/$ anchors in the sibling ports are implied).
inline const std::regex RE_IPV6_GROUP{"^[0-9a-fA-F]{1,4}$"};
/// The C++ twin of the TS `Record<ExtractType, string[]>`. Every field is
/// always present (an empty vector, never "absent"): unselected kinds and
/// empty matches are empty, matching the TS lib's "always all six keys"
/// contract. Value-initialisation gives the all-empty shape for free.
struct ExtractResult {
std::vector<std::string> url;
std::vector<std::string> email;
std::vector<std::string> ipv4;
std::vector<std::string> ipv6;
std::vector<std::string> hash;
std::vector<std::string> domain;
};
/// Deduplicate values, preserving first-occurrence order. The C++ twin of the
/// `uniq()` helper in src/lib/extract.ts. `insert` returns false on a repeat,
/// so order is kept without a separate `contains` scan.
std::vector<std::string> uniq(const std::vector<std::string>& values) {
std::unordered_set<std::string> seen;
std::vector<std::string> out;
out.reserve(values.size());
for (const auto& v : values) {
if (seen.insert(v).second) out.push_back(v);
}
return out;
}
/// Every non-overlapping match of `re` in `text`, left to right — the
/// std::regex equivalent of Go's `regexp.FindAllString` and JS's
/// `String.prototype.match` with the global flag.
std::vector<std::string> all_matches(const std::regex& re, const std::string& text) {
std::vector<std::string> out;
for (auto it = std::sregex_iterator(text.begin(), text.end(), re);
it != std::sregex_iterator(); ++it) {
out.push_back(it->str());
}
return out;
}
/// Reports whether a hex/colon run is a plausible IPv6 address: it must
/// contain a colon AND either hold a compressed zero-run (`::`) or be exactly
/// eight groups of 1–4 hex digits. Mirrors `isIpv6()` in src/lib/extract.ts.
bool is_ipv6(const std::string& run) {
if (run.find(':') == std::string::npos) return false;
if (run.find("::") != std::string::npos) return true;
// Manual split on ':' so empty segments survive at both ends, exactly like
// split(':') in the TS/Go/Python siblings: a trailing ':' must yield a
// trailing empty group and fail the count, not be silently dropped.
std::vector<std::string> groups;
std::size_t start = 0;
for (std::size_t i = 0; i <= run.size(); ++i) {
if (i == run.size() || run[i] == ':') {
groups.push_back(run.substr(start, i - start));
start = i + 1;
}
}
if (groups.size() != 8) return false;
return std::all_of(groups.begin(), groups.end(),
[](const std::string& g) { return std::regex_match(g, RE_IPV6_GROUP); });
}
/// The domain part (after the last `@`) of a matched email. Mirrors
/// `domainOf()` in src/lib/extract.ts.
std::string domain_of(const std::string& email) {
std::size_t idx = email.rfind('@');
return idx == std::string::npos ? email : email.substr(idx + 1);
}
/// Extract pulls every occurrence of the given `types` (default: all six) from
/// `input`. It returns an `ExtractResult` with one field per kind — always all
/// six, populated only for the selected types. Matches are deduped per kind,
/// preserving first-occurrence order. An email also contributes its domain to
/// the `domain` list when both Email and Domain are selected. It is the C++
/// twin of `extract()` in src/lib/extract.ts and must agree with it on every
/// shared vector.
ExtractResult extract(const std::string& input, const std::vector<Kind>& types = {}) {
const std::vector<Kind> selected =
types.empty() ? std::vector<Kind>(ALL_KINDS.begin(), ALL_KINDS.end()) : types;
auto want = [&selected](Kind k) {
return std::find(selected.begin(), selected.end(), k) != selected.end();
};
ExtractResult out;
if (want(Kind::Url)) out.url = uniq(all_matches(RE_URL, input));
if (want(Kind::Email)) out.email = uniq(all_matches(RE_EMAIL, input));
if (want(Kind::Ipv4)) out.ipv4 = uniq(all_matches(RE_IPV4, input));
if (want(Kind::Ipv6)) {
std::vector<std::string> runs = all_matches(RE_IPV6, input);
runs.erase(std::remove_if(runs.begin(), runs.end(),
[](const std::string& r) { return !is_ipv6(r); }),
runs.end());
out.ipv6 = uniq(runs);
}
if (want(Kind::Hash)) out.hash = uniq(all_matches(RE_HASH, input));
if (want(Kind::Domain)) {
std::vector<std::string> combined = all_matches(RE_DOMAIN, input);
// Cross-rule: an email also yields its domain in the domain list.
if (want(Kind::Email)) {
for (const auto& e : all_matches(RE_EMAIL, input)) {
combined.push_back(domain_of(e));
}
}
out.domain = uniq(combined);
}
return out;
}
} // namespace extract
// ---------- showcase tests (the canonical suite lives in src/lib) ----------
// Build with -DEXTRACT_SELFTEST for a runnable five-vector check — the C++
// analogue of the Rust sibling's #[cfg(test)] module.
#ifdef EXTRACT_SELFTEST
#include <cassert>
#include <iostream>
int main() {
using extract::Kind;
using extract::extract;
// Well-known digests of the empty string (real hash values), shared with
// src/lib/extract.test.ts so the showcase uses identical vectors.
const std::string md5_empty = "d41d8cd98f00b204e9800998ecf8427e"; // 32
const std::string sha256_empty =
"e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"; // 64
// Extracts and dedupes URLs, keeping first-occurrence order.
auto r = extract("a https://x.com b https://y.com c https://x.com");
assert((r.url == std::vector<std::string>{"https://x.com", "https://y.com"}));
// Plus-tags and multi-part-TLD domains are matched; an email also
// contributes its domain when email AND domain are both selected.
r = extract("reach a.b+tag@mail.example.co.uk please");
assert((r.email == std::vector<std::string>{"a.b+tag@mail.example.co.uk"}));
assert((r.domain == std::vector<std::string>{"mail.example.co.uk"}));
// `::1` is a compressed zero-run; `12:30:45` has no `::` and only three
// groups, so it is rejected as a clock, not an address.
r = extract("loopback ::1 and time 12:30:45 now");
assert((r.ipv6 == std::vector<std::string>{"::1"}));
// Hashes by length: md5 (32) and sha256 (64).
r = extract("m " + md5_empty + " s " + sha256_empty);
assert((r.hash == std::vector<std::string>{md5_empty, sha256_empty}));
// Type selection: unselected kinds stay empty.
r = extract("https://x.com and a@b.com", {Kind::Url});
assert((r.url == std::vector<std::string>{"https://x.com"}));
assert(r.email.empty());
std::cout << "extract: all showcase tests passed\n";
return 0;
}
#endif
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →