Skip to content

Text Extractor — C source

Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
 * out of arbitrary text (logs, headers, config).
 *
 * Language: C (C11, standard library only)
 * Source:   CosmoDev polyglot showcase port of the Extract tool, ported from
 *           src/lib/extract.ts (the canonical TypeScript implementation) and
 *           held in lock-step with its Go twin cli/extract/extract.go.
 * License:  display source — part of CosmoDev's polyglot tool pages.
 *
 * Design goals:
 *   - Pure + deterministic; no global state, no locale dependence.
 *   - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
 *   - Self-contained: C11 stdlib only.
 *
 * Dependency note: C's standard library ships no regex engine, and POSIX
 * regexec(3) (ERE) cannot express the non-capturing groups and \b anchors the
 * six patterns rely on. Each pattern is therefore a hand-rolled scanner that
 * reproduces the regex semantics explicitly — leftmost match, greedy with
 * backtracking, word boundaries — so every rule the TS lib encodes in a
 * pattern string appears here as visible, auditable code.
 */

#include <assert.h>
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

/* ---------- character classes (locale-independent regex-class mirrors) ---------- */

static bool is_digit_(char c) { return c >= '0' && c <= '9'; }

static bool is_hex_(char c) {
    return is_digit_(c) || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F');
}

static bool is_alnum_(char c) {
    return is_digit_(c) || (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z');
}

/* \w in the TS/Go/JS regex dialect: [A-Za-z0-9_]. */
static bool is_word_(char c) { return is_alnum_(c) || c == '_'; }

static bool is_letter_(char c) {
    return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z');
}

static bool is_space_(char c) {
    return c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '\v' || c == '\f';
}

/* Email local-part class [a-zA-Z0-9._%+-] (note: no '_'). */
static bool is_email_local_(char c) {
    return is_alnum_(c) || c == '.' || c == '%' || c == '+' || c == '-';
}

/* Email domain class [a-zA-Z0-9.-]. */
static bool is_email_domain_(char c) { return is_alnum_(c) || c == '.' || c == '-'; }

/* Domain label class [a-zA-Z0-9-]. */
static bool is_label_char_(char c) { return is_alnum_(c) || c == '-'; }

/* \b at `pos`: exactly one of s[pos-1] / s[pos] is a word char (string edges
 * count as non-word). Mirrors the \b assertion in src/lib/extract.ts. */
static bool word_boundary_at_(const char *s, size_t n, size_t pos) {
    bool before = pos > 0 && is_word_(s[pos - 1]);
    bool after = pos < n && is_word_(s[pos]);
    return before != after;
}

/* ---------- kinds ---------- */

/* One of the six canonical extraction kinds. Mirrors the TS `ExtractType`
 * union and Go's `Type`. */
typedef enum {
    EXTRACT_URL,
    EXTRACT_EMAIL,
    EXTRACT_IPV4,
    EXTRACT_IPV6,
    EXTRACT_HASH,
    EXTRACT_DOMAIN
} extract_kind;

/* Canonical spelling used in the TS/Go string-literal contract. */
static const char *extract_kind_name(extract_kind k) {
    switch (k) {
        case EXTRACT_URL:    return "url";
        case EXTRACT_EMAIL:  return "email";
        case EXTRACT_IPV4:   return "ipv4";
        case EXTRACT_IPV6:   return "ipv6";
        case EXTRACT_HASH:   return "hash";
        case EXTRACT_DOMAIN: return "domain";
    }
    return "";
}

/* The canonical kinds in display order — Go's `ExtractTypes` / TS's
 * `EXTRACT_TYPES`. */
static const extract_kind EXTRACT_ALL_KINDS[6] = {
    EXTRACT_URL, EXTRACT_EMAIL, EXTRACT_IPV4, EXTRACT_IPV6, EXTRACT_HASH, EXTRACT_DOMAIN
};

/* ---------- growable string list (the char* twin of TS string[]) ---------- */

typedef struct {
    char **items;
    size_t len;
    size_t cap;
} str_list;

static void str_list_push(str_list *l, const char *s, size_t n) {
    if (l->len == l->cap) {
        size_t cap = l->cap ? l->cap * 2 : 8;
        char **items = realloc(l->items, cap * sizeof *items);
        if (!items) { perror("extract: out of memory"); exit(1); }
        l->items = items;
        l->cap = cap;
    }
    char *copy = malloc(n + 1);
    if (!copy) { perror("extract: out of memory"); exit(1); }
    memcpy(copy, s, n);
    copy[n] = '\0';
    l->items[l->len++] = copy;
}

/* Deduplicate in place, preserving first-occurrence order. The C twin of the
 * `uniq()` helper in src/lib/extract.ts. */
static void str_list_uniq(str_list *l) {
    size_t w = 0;
    for (size_t i = 0; i < l->len; i++) {
        bool dup = false;
        for (size_t j = 0; j < w; j++) {
            if (strcmp(l->items[j], l->items[i]) == 0) { dup = true; break; }
        }
        if (!dup) l->items[w++] = l->items[i];
        else free(l->items[i]);
    }
    l->len = w;
}

static void str_list_free(str_list *l) {
    for (size_t i = 0; i < l->len; i++) free(l->items[i]);
    free(l->items);
    l->items = NULL;
    l->len = l->cap = 0;
}

/* ---------- result ---------- */

/* The C twin of the TS `Record<ExtractType, string[]>`. Every field is always
 * present (possibly empty): unselected kinds and empty matches are empty
 * lists, matching the TS lib's "always all six keys" contract. */
typedef struct {
    str_list url, email, ipv4, ipv6, hash, domain;
} extract_result;

static void extract_result_free(extract_result *r) {
    str_list_free(&r->url);
    str_list_free(&r->email);
    str_list_free(&r->ipv4);
    str_list_free(&r->ipv6);
    str_list_free(&r->hash);
    str_list_free(&r->domain);
}

static bool want_kind(const extract_kind *types, size_t types_len, extract_kind k) {
    for (size_t i = 0; i < types_len; i++) {
        if (types[i] == k) return true;
    }
    return false;
}

/* ---------- scanners (one per TS pattern) ---------- */

/* https?://[^\s]+ — leftmost "http", optional "s", "://", then a non-empty
 * run of non-whitespace (greedy, unanchored end). */
static void scan_urls(const char *s, size_t n, str_list *out) {
    size_t i = 0;
    for (;;) {
        const char *h = memchr(s + i, 'h', n - i);
        if (!h) return;
        size_t p = (size_t)(h - s); /* candidate "http" start */
        if (n - p < 4 || strncmp(s + p, "http", 4) != 0) { i = p + 1; continue; }
        size_t q = p + 4;
        if (q < n && s[q] == 's') q++; /* optional "s" (greedy, tried first) */
        if (q + 3 > n || s[q] != ':' || s[q + 1] != '/' || s[q + 2] != '/') {
            i = p + 1;
            continue;
        }
        size_t run = q + 3;
        while (run < n && !is_space_(s[run])) run++;
        if (run == q + 3) { i = p + 1; continue; } /* [^\s]+ needs >= 1 char */
        str_list_push(out, s + p, run - p);
        i = run; /* non-overlapping, like findall */
    }
}

/* [a-zA-Z0-9.-]+\.[a-zA-Z]{2,} starting at `pos`. Returns the match end, or
 * SIZE_MAX. The greedy class run is backtracked from the right to the last
 * '.' followed by >= 2 letters — exactly the engine's backtrack order. */
static size_t match_email_domain_(const char *s, size_t n, size_t pos) {
    size_t de = pos;
    while (de < n && is_email_domain_(s[de])) de++;
    /* p scans right-to-left over the run; p > pos keeps >= 1 char for the
     * `[a-zA-Z0-9.-]+` part before the '.'. */
    for (size_t p = de - 1; p > pos; p--) {
        if (s[p] != '.') continue;
        size_t l = p + 1;
        while (l < n && is_letter_(s[l])) l++;
        if (l - (p + 1) >= 2) return l;
    }
    return SIZE_MAX;
}

/* [a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,} — scan '@' left to right.
 * The local part is the maximal run before it: '@' is not in the class, so a
 * backtracked (shorter) local part always ends on a class char and can never
 * reach the '@'. The match starts at the leftmost available position, capped
 * by the non-overlap cursor from a previous match. */
static void scan_emails(const char *s, size_t n, str_list *out) {
    size_t i = 0;
    for (;;) {
        const char *at = memchr(s + i, '@', n - i);
        if (!at) return;
        size_t a = (size_t)(at - s);
        size_t ls = a;
        while (ls > i && is_email_local_(s[ls - 1])) ls--;
        if (ls < a) { /* non-empty local part */
            size_t end = match_email_domain_(s, n, a + 1);
            if (end != SIZE_MAX) {
                str_list_push(out, s + ls, end - ls);
                i = end;
                continue;
            }
        }
        i = a + 1;
    }
}

/* (\d{1,3}\.){3}\d{1,3}\b starting at `pos` (the leading \b is checked by the
 * caller). Returns the match end, or SIZE_MAX. Mirrors greedy backtracking:
 * each group prefers 3 digits and backs off to 2, then 1; the final group
 * must land on a word boundary. */
static size_t match_ipv4_body_(const char *s, size_t n, size_t pos, int groups_left) {
    size_t run = 0;
    while (run < 3 && pos + run < n && is_digit_(s[pos + run])) run++;
    if (groups_left == 0) {
        for (size_t len = run; len >= 1; len--) {
            if (word_boundary_at_(s, n, pos + len)) return pos + len;
        }
        return SIZE_MAX;
    }
    for (size_t len = run; len >= 1; len--) {
        if (pos + len < n && s[pos + len] == '.') {
            size_t end = match_ipv4_body_(s, n, pos + len + 1, groups_left - 1);
            if (end != SIZE_MAX) return end;
        }
    }
    return SIZE_MAX;
}

/* \b(?:\d{1,3}\.){3}\d{1,3}\b — attempted at every digit where a leading word
 * boundary holds (a \b pattern start needs both). */
static void scan_ipv4(const char *s, size_t n, str_list *out) {
    size_t i = 0;
    while (i < n) {
        if (is_digit_(s[i]) && word_boundary_at_(s, n, i)) {
            size_t end = match_ipv4_body_(s, n, i, 3);
            if (end != SIZE_MAX) {
                str_list_push(out, s + i, end - i);
                i = end;
                continue;
            }
        }
        i++;
    }
}

/* A hex/colon run is a plausible IPv6: it has a colon AND either contains
 * "::" (a compressed zero-run) or is exactly eight groups of 1-4 hex digits.
 * Mirrors `isIpv6()` in src/lib/extract.ts. */
static bool is_ipv6_(const char *run, size_t len) {
    bool has_colon = false, has_double = false;
    for (size_t i = 0; i < len; i++) {
        if (run[i] == ':') {
            has_colon = true;
            if (i + 1 < len && run[i + 1] == ':') has_double = true;
        }
    }
    if (!has_colon) return false;
    if (has_double) return true;
    size_t groups = 1, glen = 0;
    for (size_t i = 0; i <= len; i++) {
        if (i == len || run[i] == ':') {
            if (glen < 1 || glen > 4) return false;
            if (i < len) { groups++; glen = 0; }
        } else {
            glen++;
        }
    }
    return groups == 8;
}

/* [0-9a-fA-F:]+ — intentionally permissive; callers post-filter with
 * is_ipv6_, exactly as the TS lib does. */
static void scan_ipv6(const char *s, size_t n, str_list *out) {
    size_t i = 0;
    while (i < n) {
        if (is_hex_(s[i]) || s[i] == ':') {
            size_t e = i;
            while (e < n && (is_hex_(s[e]) || s[e] == ':')) e++;
            if (is_ipv6_(s + i, e - i)) str_list_push(out, s + i, e - i);
            i = e;
        } else {
            i++;
        }
    }
}

/* \b[a-fA-F0-9]{32}\b | ...{40}... | ...{64}... | ...{128}... — a maximal hex
 * run qualifies iff its length is exactly 32/40/64/128 and word boundaries
 * hold at both ends (a boundary can never hold mid-run, so sub-runs of a
 * longer run never match). */
static void scan_hashes(const char *s, size_t n, str_list *out) {
    size_t i = 0;
    while (i < n) {
        if (is_hex_(s[i])) {
            size_t e = i;
            while (e < n && is_hex_(s[e])) e++;
            size_t len = e - i;
            if ((len == 32 || len == 40 || len == 64 || len == 128)
                && word_boundary_at_(s, n, i) && word_boundary_at_(s, n, e)) {
                str_list_push(out, s + i, len);
            }
            i = e;
        } else {
            i++;
        }
    }
}

/* \b[label](\.TLD)+\b at `pos` (leading \b checked by the caller). Returns
 * the match end, or SIZE_MAX.
 *
 * Label = alnum + up to 61 label chars ending alnum (<= 63 chars). Greedy
 * label end = the LAST alnum of the class run (capped at 63); a shorter label
 * always leaves a class char — never '.' — next, so only the greedy end can
 * precede the TLD tail: no label backtracking is needed.
 *
 * Tail = (\.letters{2,})+ greedy. On \b failure at the greedy end (the next
 * char is a digit or '_'), the engine backtracks by dropping the LAST
 * iteration — whose end always sits on a '.', a word boundary — so a >=2
 * iteration tail always yields a match one iteration shorter. */
static size_t match_domain_at_(const char *s, size_t n, size_t pos) {
    size_t run = pos, cap = pos + 63;
    while (run < n && run < cap && is_label_char_(s[run])) run++;
    size_t le = run;
    while (le > pos && !is_alnum_(s[le - 1])) le--; /* drop trailing '-' */
    if (le >= n || s[le] != '.') return SIZE_MAX;
    size_t iterations = 0, last_end = 0, prev_end = 0;
    for (size_t p = le;;) {
        if (p < n && s[p] == '.') {
            size_t l = p + 1;
            while (l < n && is_letter_(s[l])) l++;
            if (l - (p + 1) >= 2) {
                prev_end = last_end;
                last_end = l;
                iterations++;
                p = l;
                continue;
            }
        }
        break;
    }
    if (iterations == 0) return SIZE_MAX;
    if (word_boundary_at_(s, n, last_end)) return last_end;
    if (iterations >= 2) return prev_end; /* ends on '.', a boundary */
    return SIZE_MAX;
}

/* \b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b. */
static void scan_domains(const char *s, size_t n, str_list *out) {
    size_t i = 0;
    while (i < n) {
        if (is_alnum_(s[i]) && word_boundary_at_(s, n, i)) {
            size_t end = match_domain_at_(s, n, i);
            if (end != SIZE_MAX) {
                str_list_push(out, s + i, end - i);
                i = end;
                continue;
            }
        }
        i++;
    }
}

/* The domain part (after the last '@') of a matched email. Mirrors
 * `domainOf()` in src/lib/extract.ts (lastIndexOf('@')). */
static size_t domain_of_(const char *email, size_t len) {
    size_t at = len;
    for (size_t i = len; i > 0; i--) {
        if (email[i - 1] == '@') { at = i - 1; break; }
    }
    return at + 1; /* slice start; == len+1 only if no '@', impossible here */
}

/* ---------- extract ---------- */

/* Extract pulls every occurrence of the given `types` (default: all six)
 * kinds from `input` (NULL is treated as empty text). Returns an
 * extract_result with one field per kind — always all six, populated only for
 * the selected types. Matches are deduped per kind, preserving first-occurrence
 * order. An email also contributes its domain to the `domain` list when both
 * email and domain are selected. Free the result with extract_result_free. */
static extract_result extract(const char *input, const extract_kind *types, size_t types_len) {
    const char *s = input ? input : "";
    size_t n = strlen(s);
    const extract_kind *sel = types;
    size_t sel_len = types_len;
    extract_kind all[6];
    if (sel_len == 0) { /* TS defaults the param AND re-defaults an empty list */
        memcpy(all, EXTRACT_ALL_KINDS, sizeof all);
        sel = all;
        sel_len = 6;
    }

    extract_result r;
    memset(&r, 0, sizeof r);

    if (want_kind(sel, sel_len, EXTRACT_URL)) {
        scan_urls(s, n, &r.url);
        str_list_uniq(&r.url);
    }
    if (want_kind(sel, sel_len, EXTRACT_EMAIL)) {
        scan_emails(s, n, &r.email);
        str_list_uniq(&r.email);
    }
    if (want_kind(sel, sel_len, EXTRACT_IPV4)) {
        scan_ipv4(s, n, &r.ipv4);
        str_list_uniq(&r.ipv4);
    }
    if (want_kind(sel, sel_len, EXTRACT_IPV6)) {
        scan_ipv6(s, n, &r.ipv6);
        str_list_uniq(&r.ipv6);
    }
    if (want_kind(sel, sel_len, EXTRACT_HASH)) {
        scan_hashes(s, n, &r.hash);
        str_list_uniq(&r.hash);
    }
    if (want_kind(sel, sel_len, EXTRACT_DOMAIN)) {
        scan_domains(s, n, &r.domain);
        /* Cross-rule: an email also yields its domain in the domain list. */
        if (want_kind(sel, sel_len, EXTRACT_EMAIL)) {
            str_list emails;
            memset(&emails, 0, sizeof emails);
            scan_emails(s, n, &emails);
            for (size_t i = 0; i < emails.len; i++) {
                size_t at = domain_of_(emails.items[i], strlen(emails.items[i]));
                if (at < strlen(emails.items[i])) {
                    const char *e = emails.items[i];
                    str_list_push(&r.domain, e + at, strlen(e) - at);
                }
            }
            str_list_free(&emails);
        }
        str_list_uniq(&r.domain);
    }
    return r;
}

/* ---------- showcase tests (the canonical suite lives in src/lib) ---------- */

/* Well-known digests of the empty string (real hash values), shared with
 * src/lib/extract.test.ts so the showcase uses identical vectors. */
static const char MD5_EMPTY[] = "d41d8cd98f00b204e9800998ecf8427e";    /* 32 */
static const char SHA256_EMPTY[] =
    "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"; /* 64 */

static bool str_list_eq(const str_list *l, const char *const *want, size_t want_len) {
    if (l->len != want_len) return false;
    for (size_t i = 0; i < want_len; i++) {
        if (strcmp(l->items[i], want[i]) != 0) return false;
    }
    return true;
}

int main(void) {
    extract_result r;

    /* URLs: deduped, first-occurrence order preserved. */
    r = extract("a https://x.com b https://y.com c https://x.com", NULL, 0);
    const char *want_urls[] = {"https://x.com", "https://y.com"};
    assert(str_list_eq(&r.url, want_urls, 2));
    extract_result_free(&r);

    /* Emails contribute their domain when email AND domain are selected. */
    r = extract("reach a.b+tag@mail.example.co.uk please", NULL, 0);
    const char *want_emails[] = {"a.b+tag@mail.example.co.uk"};
    const char *want_domains[] = {"mail.example.co.uk"};
    assert(str_list_eq(&r.email, want_emails, 1));
    assert(str_list_eq(&r.domain, want_domains, 1));
    extract_result_free(&r);

    /* IPv6: compressed zero-runs pass; clock times (no '::', 3 groups) don't. */
    r = extract("loopback ::1 and time 12:30:45 now", NULL, 0);
    const char *want_ipv6[] = {"::1"};
    assert(str_list_eq(&r.ipv6, want_ipv6, 1));
    extract_result_free(&r);

    /* Hashes by length: md5 (32) and sha256 (64). */
    {
        char text[256];
        snprintf(text, sizeof text, "m %s s %s", MD5_EMPTY, SHA256_EMPTY);
        r = extract(text, NULL, 0);
        const char *want_hash[] = {MD5_EMPTY, SHA256_EMPTY};
        assert(str_list_eq(&r.hash, want_hash, 2));
        extract_result_free(&r);
    }

    /* Type selection: unselected kinds stay empty. */
    {
        extract_kind only_url[1] = {EXTRACT_URL};
        r = extract("https://x.com and a@b.com", only_url, 1);
        const char *want_sel[] = {"https://x.com"};
        assert(str_list_eq(&r.url, want_sel, 1));
        assert(r.email.len == 0);
        extract_result_free(&r);
    }

    printf("extract: all showcase tests passed\n");
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →