Text Extractor — C source
Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/*
* extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
* out of arbitrary text (logs, headers, config).
*
* Language: C (C11, standard library only)
* Source: CosmoDev polyglot showcase port of the Extract tool, ported from
* src/lib/extract.ts (the canonical TypeScript implementation) and
* held in lock-step with its Go twin cli/extract/extract.go.
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; no global state, no locale dependence.
* - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
* - Self-contained: C11 stdlib only.
*
* Dependency note: C's standard library ships no regex engine, and POSIX
* regexec(3) (ERE) cannot express the non-capturing groups and \b anchors the
* six patterns rely on. Each pattern is therefore a hand-rolled scanner that
* reproduces the regex semantics explicitly — leftmost match, greedy with
* backtracking, word boundaries — so every rule the TS lib encodes in a
* pattern string appears here as visible, auditable code.
*/
#include <assert.h>
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
/* ---------- character classes (locale-independent regex-class mirrors) ---------- */
static bool is_digit_(char c) { return c >= '0' && c <= '9'; }
static bool is_hex_(char c) {
return is_digit_(c) || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F');
}
static bool is_alnum_(char c) {
return is_digit_(c) || (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z');
}
/* \w in the TS/Go/JS regex dialect: [A-Za-z0-9_]. */
static bool is_word_(char c) { return is_alnum_(c) || c == '_'; }
static bool is_letter_(char c) {
return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z');
}
static bool is_space_(char c) {
return c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '\v' || c == '\f';
}
/* Email local-part class [a-zA-Z0-9._%+-] (note: no '_'). */
static bool is_email_local_(char c) {
return is_alnum_(c) || c == '.' || c == '%' || c == '+' || c == '-';
}
/* Email domain class [a-zA-Z0-9.-]. */
static bool is_email_domain_(char c) { return is_alnum_(c) || c == '.' || c == '-'; }
/* Domain label class [a-zA-Z0-9-]. */
static bool is_label_char_(char c) { return is_alnum_(c) || c == '-'; }
/* \b at `pos`: exactly one of s[pos-1] / s[pos] is a word char (string edges
* count as non-word). Mirrors the \b assertion in src/lib/extract.ts. */
static bool word_boundary_at_(const char *s, size_t n, size_t pos) {
bool before = pos > 0 && is_word_(s[pos - 1]);
bool after = pos < n && is_word_(s[pos]);
return before != after;
}
/* ---------- kinds ---------- */
/* One of the six canonical extraction kinds. Mirrors the TS `ExtractType`
* union and Go's `Type`. */
typedef enum {
EXTRACT_URL,
EXTRACT_EMAIL,
EXTRACT_IPV4,
EXTRACT_IPV6,
EXTRACT_HASH,
EXTRACT_DOMAIN
} extract_kind;
/* Canonical spelling used in the TS/Go string-literal contract. */
static const char *extract_kind_name(extract_kind k) {
switch (k) {
case EXTRACT_URL: return "url";
case EXTRACT_EMAIL: return "email";
case EXTRACT_IPV4: return "ipv4";
case EXTRACT_IPV6: return "ipv6";
case EXTRACT_HASH: return "hash";
case EXTRACT_DOMAIN: return "domain";
}
return "";
}
/* The canonical kinds in display order — Go's `ExtractTypes` / TS's
* `EXTRACT_TYPES`. */
static const extract_kind EXTRACT_ALL_KINDS[6] = {
EXTRACT_URL, EXTRACT_EMAIL, EXTRACT_IPV4, EXTRACT_IPV6, EXTRACT_HASH, EXTRACT_DOMAIN
};
/* ---------- growable string list (the char* twin of TS string[]) ---------- */
typedef struct {
char **items;
size_t len;
size_t cap;
} str_list;
static void str_list_push(str_list *l, const char *s, size_t n) {
if (l->len == l->cap) {
size_t cap = l->cap ? l->cap * 2 : 8;
char **items = realloc(l->items, cap * sizeof *items);
if (!items) { perror("extract: out of memory"); exit(1); }
l->items = items;
l->cap = cap;
}
char *copy = malloc(n + 1);
if (!copy) { perror("extract: out of memory"); exit(1); }
memcpy(copy, s, n);
copy[n] = '\0';
l->items[l->len++] = copy;
}
/* Deduplicate in place, preserving first-occurrence order. The C twin of the
* `uniq()` helper in src/lib/extract.ts. */
static void str_list_uniq(str_list *l) {
size_t w = 0;
for (size_t i = 0; i < l->len; i++) {
bool dup = false;
for (size_t j = 0; j < w; j++) {
if (strcmp(l->items[j], l->items[i]) == 0) { dup = true; break; }
}
if (!dup) l->items[w++] = l->items[i];
else free(l->items[i]);
}
l->len = w;
}
static void str_list_free(str_list *l) {
for (size_t i = 0; i < l->len; i++) free(l->items[i]);
free(l->items);
l->items = NULL;
l->len = l->cap = 0;
}
/* ---------- result ---------- */
/* The C twin of the TS `Record<ExtractType, string[]>`. Every field is always
* present (possibly empty): unselected kinds and empty matches are empty
* lists, matching the TS lib's "always all six keys" contract. */
typedef struct {
str_list url, email, ipv4, ipv6, hash, domain;
} extract_result;
static void extract_result_free(extract_result *r) {
str_list_free(&r->url);
str_list_free(&r->email);
str_list_free(&r->ipv4);
str_list_free(&r->ipv6);
str_list_free(&r->hash);
str_list_free(&r->domain);
}
static bool want_kind(const extract_kind *types, size_t types_len, extract_kind k) {
for (size_t i = 0; i < types_len; i++) {
if (types[i] == k) return true;
}
return false;
}
/* ---------- scanners (one per TS pattern) ---------- */
/* https?://[^\s]+ — leftmost "http", optional "s", "://", then a non-empty
* run of non-whitespace (greedy, unanchored end). */
static void scan_urls(const char *s, size_t n, str_list *out) {
size_t i = 0;
for (;;) {
const char *h = memchr(s + i, 'h', n - i);
if (!h) return;
size_t p = (size_t)(h - s); /* candidate "http" start */
if (n - p < 4 || strncmp(s + p, "http", 4) != 0) { i = p + 1; continue; }
size_t q = p + 4;
if (q < n && s[q] == 's') q++; /* optional "s" (greedy, tried first) */
if (q + 3 > n || s[q] != ':' || s[q + 1] != '/' || s[q + 2] != '/') {
i = p + 1;
continue;
}
size_t run = q + 3;
while (run < n && !is_space_(s[run])) run++;
if (run == q + 3) { i = p + 1; continue; } /* [^\s]+ needs >= 1 char */
str_list_push(out, s + p, run - p);
i = run; /* non-overlapping, like findall */
}
}
/* [a-zA-Z0-9.-]+\.[a-zA-Z]{2,} starting at `pos`. Returns the match end, or
* SIZE_MAX. The greedy class run is backtracked from the right to the last
* '.' followed by >= 2 letters — exactly the engine's backtrack order. */
static size_t match_email_domain_(const char *s, size_t n, size_t pos) {
size_t de = pos;
while (de < n && is_email_domain_(s[de])) de++;
/* p scans right-to-left over the run; p > pos keeps >= 1 char for the
* `[a-zA-Z0-9.-]+` part before the '.'. */
for (size_t p = de - 1; p > pos; p--) {
if (s[p] != '.') continue;
size_t l = p + 1;
while (l < n && is_letter_(s[l])) l++;
if (l - (p + 1) >= 2) return l;
}
return SIZE_MAX;
}
/* [a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,} — scan '@' left to right.
* The local part is the maximal run before it: '@' is not in the class, so a
* backtracked (shorter) local part always ends on a class char and can never
* reach the '@'. The match starts at the leftmost available position, capped
* by the non-overlap cursor from a previous match. */
static void scan_emails(const char *s, size_t n, str_list *out) {
size_t i = 0;
for (;;) {
const char *at = memchr(s + i, '@', n - i);
if (!at) return;
size_t a = (size_t)(at - s);
size_t ls = a;
while (ls > i && is_email_local_(s[ls - 1])) ls--;
if (ls < a) { /* non-empty local part */
size_t end = match_email_domain_(s, n, a + 1);
if (end != SIZE_MAX) {
str_list_push(out, s + ls, end - ls);
i = end;
continue;
}
}
i = a + 1;
}
}
/* (\d{1,3}\.){3}\d{1,3}\b starting at `pos` (the leading \b is checked by the
* caller). Returns the match end, or SIZE_MAX. Mirrors greedy backtracking:
* each group prefers 3 digits and backs off to 2, then 1; the final group
* must land on a word boundary. */
static size_t match_ipv4_body_(const char *s, size_t n, size_t pos, int groups_left) {
size_t run = 0;
while (run < 3 && pos + run < n && is_digit_(s[pos + run])) run++;
if (groups_left == 0) {
for (size_t len = run; len >= 1; len--) {
if (word_boundary_at_(s, n, pos + len)) return pos + len;
}
return SIZE_MAX;
}
for (size_t len = run; len >= 1; len--) {
if (pos + len < n && s[pos + len] == '.') {
size_t end = match_ipv4_body_(s, n, pos + len + 1, groups_left - 1);
if (end != SIZE_MAX) return end;
}
}
return SIZE_MAX;
}
/* \b(?:\d{1,3}\.){3}\d{1,3}\b — attempted at every digit where a leading word
* boundary holds (a \b pattern start needs both). */
static void scan_ipv4(const char *s, size_t n, str_list *out) {
size_t i = 0;
while (i < n) {
if (is_digit_(s[i]) && word_boundary_at_(s, n, i)) {
size_t end = match_ipv4_body_(s, n, i, 3);
if (end != SIZE_MAX) {
str_list_push(out, s + i, end - i);
i = end;
continue;
}
}
i++;
}
}
/* A hex/colon run is a plausible IPv6: it has a colon AND either contains
* "::" (a compressed zero-run) or is exactly eight groups of 1-4 hex digits.
* Mirrors `isIpv6()` in src/lib/extract.ts. */
static bool is_ipv6_(const char *run, size_t len) {
bool has_colon = false, has_double = false;
for (size_t i = 0; i < len; i++) {
if (run[i] == ':') {
has_colon = true;
if (i + 1 < len && run[i + 1] == ':') has_double = true;
}
}
if (!has_colon) return false;
if (has_double) return true;
size_t groups = 1, glen = 0;
for (size_t i = 0; i <= len; i++) {
if (i == len || run[i] == ':') {
if (glen < 1 || glen > 4) return false;
if (i < len) { groups++; glen = 0; }
} else {
glen++;
}
}
return groups == 8;
}
/* [0-9a-fA-F:]+ — intentionally permissive; callers post-filter with
* is_ipv6_, exactly as the TS lib does. */
static void scan_ipv6(const char *s, size_t n, str_list *out) {
size_t i = 0;
while (i < n) {
if (is_hex_(s[i]) || s[i] == ':') {
size_t e = i;
while (e < n && (is_hex_(s[e]) || s[e] == ':')) e++;
if (is_ipv6_(s + i, e - i)) str_list_push(out, s + i, e - i);
i = e;
} else {
i++;
}
}
}
/* \b[a-fA-F0-9]{32}\b | ...{40}... | ...{64}... | ...{128}... — a maximal hex
* run qualifies iff its length is exactly 32/40/64/128 and word boundaries
* hold at both ends (a boundary can never hold mid-run, so sub-runs of a
* longer run never match). */
static void scan_hashes(const char *s, size_t n, str_list *out) {
size_t i = 0;
while (i < n) {
if (is_hex_(s[i])) {
size_t e = i;
while (e < n && is_hex_(s[e])) e++;
size_t len = e - i;
if ((len == 32 || len == 40 || len == 64 || len == 128)
&& word_boundary_at_(s, n, i) && word_boundary_at_(s, n, e)) {
str_list_push(out, s + i, len);
}
i = e;
} else {
i++;
}
}
}
/* \b[label](\.TLD)+\b at `pos` (leading \b checked by the caller). Returns
* the match end, or SIZE_MAX.
*
* Label = alnum + up to 61 label chars ending alnum (<= 63 chars). Greedy
* label end = the LAST alnum of the class run (capped at 63); a shorter label
* always leaves a class char — never '.' — next, so only the greedy end can
* precede the TLD tail: no label backtracking is needed.
*
* Tail = (\.letters{2,})+ greedy. On \b failure at the greedy end (the next
* char is a digit or '_'), the engine backtracks by dropping the LAST
* iteration — whose end always sits on a '.', a word boundary — so a >=2
* iteration tail always yields a match one iteration shorter. */
static size_t match_domain_at_(const char *s, size_t n, size_t pos) {
size_t run = pos, cap = pos + 63;
while (run < n && run < cap && is_label_char_(s[run])) run++;
size_t le = run;
while (le > pos && !is_alnum_(s[le - 1])) le--; /* drop trailing '-' */
if (le >= n || s[le] != '.') return SIZE_MAX;
size_t iterations = 0, last_end = 0, prev_end = 0;
for (size_t p = le;;) {
if (p < n && s[p] == '.') {
size_t l = p + 1;
while (l < n && is_letter_(s[l])) l++;
if (l - (p + 1) >= 2) {
prev_end = last_end;
last_end = l;
iterations++;
p = l;
continue;
}
}
break;
}
if (iterations == 0) return SIZE_MAX;
if (word_boundary_at_(s, n, last_end)) return last_end;
if (iterations >= 2) return prev_end; /* ends on '.', a boundary */
return SIZE_MAX;
}
/* \b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b. */
static void scan_domains(const char *s, size_t n, str_list *out) {
size_t i = 0;
while (i < n) {
if (is_alnum_(s[i]) && word_boundary_at_(s, n, i)) {
size_t end = match_domain_at_(s, n, i);
if (end != SIZE_MAX) {
str_list_push(out, s + i, end - i);
i = end;
continue;
}
}
i++;
}
}
/* The domain part (after the last '@') of a matched email. Mirrors
* `domainOf()` in src/lib/extract.ts (lastIndexOf('@')). */
static size_t domain_of_(const char *email, size_t len) {
size_t at = len;
for (size_t i = len; i > 0; i--) {
if (email[i - 1] == '@') { at = i - 1; break; }
}
return at + 1; /* slice start; == len+1 only if no '@', impossible here */
}
/* ---------- extract ---------- */
/* Extract pulls every occurrence of the given `types` (default: all six)
* kinds from `input` (NULL is treated as empty text). Returns an
* extract_result with one field per kind — always all six, populated only for
* the selected types. Matches are deduped per kind, preserving first-occurrence
* order. An email also contributes its domain to the `domain` list when both
* email and domain are selected. Free the result with extract_result_free. */
static extract_result extract(const char *input, const extract_kind *types, size_t types_len) {
const char *s = input ? input : "";
size_t n = strlen(s);
const extract_kind *sel = types;
size_t sel_len = types_len;
extract_kind all[6];
if (sel_len == 0) { /* TS defaults the param AND re-defaults an empty list */
memcpy(all, EXTRACT_ALL_KINDS, sizeof all);
sel = all;
sel_len = 6;
}
extract_result r;
memset(&r, 0, sizeof r);
if (want_kind(sel, sel_len, EXTRACT_URL)) {
scan_urls(s, n, &r.url);
str_list_uniq(&r.url);
}
if (want_kind(sel, sel_len, EXTRACT_EMAIL)) {
scan_emails(s, n, &r.email);
str_list_uniq(&r.email);
}
if (want_kind(sel, sel_len, EXTRACT_IPV4)) {
scan_ipv4(s, n, &r.ipv4);
str_list_uniq(&r.ipv4);
}
if (want_kind(sel, sel_len, EXTRACT_IPV6)) {
scan_ipv6(s, n, &r.ipv6);
str_list_uniq(&r.ipv6);
}
if (want_kind(sel, sel_len, EXTRACT_HASH)) {
scan_hashes(s, n, &r.hash);
str_list_uniq(&r.hash);
}
if (want_kind(sel, sel_len, EXTRACT_DOMAIN)) {
scan_domains(s, n, &r.domain);
/* Cross-rule: an email also yields its domain in the domain list. */
if (want_kind(sel, sel_len, EXTRACT_EMAIL)) {
str_list emails;
memset(&emails, 0, sizeof emails);
scan_emails(s, n, &emails);
for (size_t i = 0; i < emails.len; i++) {
size_t at = domain_of_(emails.items[i], strlen(emails.items[i]));
if (at < strlen(emails.items[i])) {
const char *e = emails.items[i];
str_list_push(&r.domain, e + at, strlen(e) - at);
}
}
str_list_free(&emails);
}
str_list_uniq(&r.domain);
}
return r;
}
/* ---------- showcase tests (the canonical suite lives in src/lib) ---------- */
/* Well-known digests of the empty string (real hash values), shared with
* src/lib/extract.test.ts so the showcase uses identical vectors. */
static const char MD5_EMPTY[] = "d41d8cd98f00b204e9800998ecf8427e"; /* 32 */
static const char SHA256_EMPTY[] =
"e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"; /* 64 */
static bool str_list_eq(const str_list *l, const char *const *want, size_t want_len) {
if (l->len != want_len) return false;
for (size_t i = 0; i < want_len; i++) {
if (strcmp(l->items[i], want[i]) != 0) return false;
}
return true;
}
int main(void) {
extract_result r;
/* URLs: deduped, first-occurrence order preserved. */
r = extract("a https://x.com b https://y.com c https://x.com", NULL, 0);
const char *want_urls[] = {"https://x.com", "https://y.com"};
assert(str_list_eq(&r.url, want_urls, 2));
extract_result_free(&r);
/* Emails contribute their domain when email AND domain are selected. */
r = extract("reach a.b+tag@mail.example.co.uk please", NULL, 0);
const char *want_emails[] = {"a.b+tag@mail.example.co.uk"};
const char *want_domains[] = {"mail.example.co.uk"};
assert(str_list_eq(&r.email, want_emails, 1));
assert(str_list_eq(&r.domain, want_domains, 1));
extract_result_free(&r);
/* IPv6: compressed zero-runs pass; clock times (no '::', 3 groups) don't. */
r = extract("loopback ::1 and time 12:30:45 now", NULL, 0);
const char *want_ipv6[] = {"::1"};
assert(str_list_eq(&r.ipv6, want_ipv6, 1));
extract_result_free(&r);
/* Hashes by length: md5 (32) and sha256 (64). */
{
char text[256];
snprintf(text, sizeof text, "m %s s %s", MD5_EMPTY, SHA256_EMPTY);
r = extract(text, NULL, 0);
const char *want_hash[] = {MD5_EMPTY, SHA256_EMPTY};
assert(str_list_eq(&r.hash, want_hash, 2));
extract_result_free(&r);
}
/* Type selection: unselected kinds stay empty. */
{
extract_kind only_url[1] = {EXTRACT_URL};
r = extract("https://x.com and a@b.com", only_url, 1);
const char *want_sel[] = {"https://x.com"};
assert(str_list_eq(&r.url, want_sel, 1));
assert(r.email.len == 0);
extract_result_free(&r);
}
printf("extract: all showcase tests passed\n");
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →