Skip to content

Text Extractor — Zig source

Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
//! out of arbitrary text (logs, headers, config).
//!
//! Language: Zig 0.13 (standard library only)
//! Source:   CosmoDev polyglot showcase port of the Extract tool, ported from
//!           src/lib/extract.ts (the canonical TypeScript implementation) and
//!           held in lock-step with its Go twin cli/extract/extract.go.
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; allocation failure is the only error path.
//!   - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
//!
//! Dependency note: unlike Go / Python / PHP / JS (and Rust's `regex` crate),
//! Zig ships no regex engine in std. The six per-kind patterns are small and
//! regular enough to hand-roll, so this port re-implements each pattern's
//! exact semantics — greedy runs, `\b` word boundaries, counted repetition,
//! the alternation of hash lengths — as explicit left-to-right byte scanners.
//! Each matcher documents the pattern it mirrors from the `RE` record in
//! src/lib/extract.ts, so equivalence stays auditable without an engine.
//! The scanners work byte-wise over UTF-8: every pattern class is ASCII, and
//! `\b`/`\s` are ASCII-defined exactly as in Go's regexp (a multibyte
//! character is simply a run of non-word, non-space bytes).

const std = @import("std");

/// One of the six canonical extraction kinds. Mirrors the TS `ExtractType`
/// union ('url' | 'email' | 'ipv4' | 'ipv6' | 'hash' | 'domain') and Go's
/// `Type`.
pub const Kind = enum { url, email, ipv4, ipv6, hash, domain };

/// The canonical kinds in display order — TS's `EXTRACT_TYPES`.
pub const all_kinds = [6]Kind{ .url, .email, .ipv4, .ipv6, .hash, .domain };

/// One field per kind — always all six, empty when unselected. The inner
/// slices are allocated by `extract` (free them with `deinitResult`); the
/// matched bytes themselves point into the caller's `input`, zero-copy.
pub const Result = struct {
    url: []const []const u8 = &.{},
    email: []const []const u8 = &.{},
    ipv4: []const []const u8 = &.{},
    ipv6: []const []const u8 = &.{},
    hash: []const []const u8 = &.{},
    domain: []const []const u8 = &.{},
};

// ---------- byte classes (ASCII, exactly the pattern classes) ----------

fn isAlphaByte(c: u8) bool {
    return (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z');
}

fn isDigitByte(c: u8) bool {
    return c >= '0' and c <= '9';
}

fn isAlnumByte(c: u8) bool {
    return isAlphaByte(c) or isDigitByte(c);
}

fn isHexByte(c: u8) bool {
    return isDigitByte(c) or (c >= 'a' and c <= 'f') or (c >= 'A' and c <= 'F');
}

/// `\w` as Go/JS define it: [A-Za-z0-9_] — the class behind `\b`.
fn isWordByte(c: u8) bool {
    return isAlnumByte(c) or c == '_';
}

/// `\s` as Go defines it (ASCII whitespace); the `[^\\s]` of the URL tail is
/// its complement, so multibyte UTF-8 bytes read as non-space, same as Go.
fn isSpaceByte(c: u8) bool {
    return c == ' ' or c == '\t' or c == '\n' or c == '\r' or c == 0x0b or c == 0x0c;
}

/// `\b` at index `i` of `text`: exactly one side of the index is a word byte.
fn wordBoundary(text: []const u8, i: usize) bool {
    const before = i > 0 and isWordByte(text[i - 1]);
    const after = i < text.len and isWordByte(text[i]);
    return before != after;
}

/// Length of the maximal digit run starting at `i` (greedy `\d+`).
fn digitRunLen(text: []const u8, i: usize) usize {
    var p = i;
    while (p < text.len and isDigitByte(text[p])) p += 1;
    return p - i;
}

// ---------- per-kind matchers: length of the match at `i`, or null ----------
// Each returns the match length for a match anchored at `i`; the driver
// below walks the text and advances past each hit — exactly how a global
// regex scans (leftmost, non-overlapping).

/// /https?:\/\/[^\s]+/ — `http`, an optional `s`, `://`, then a maximal
/// non-whitespace run of at least one byte.
fn matchUrlAt(text: []const u8, i: usize) ?usize {
    const rest = text[i..];
    if (!std.mem.startsWith(u8, rest, "http")) return null;
    var p: usize = 4;
    if (p < rest.len and rest[p] == 's') p += 1;
    if (!std.mem.startsWith(u8, rest[p..], "://")) return null;
    p += 3;
    const body = p;
    while (p < rest.len and !isSpaceByte(rest[p])) p += 1;
    if (p == body) return null; // `+` needs at least one byte
    return p;
}

/// The local-part class [a-zA-Z0-9._%+-].
fn isLocalByte(c: u8) bool {
    return isAlnumByte(c) or c == '.' or c == '_' or c == '%' or c == '+' or c == '-';
}

/// The email-domain class [a-zA-Z0-9.-].
fn isDomainRunByte(c: u8) bool {
    return isAlnumByte(c) or c == '.' or c == '-';
}

/// /[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}/ — a greedy local run
/// ('@' is not in the class, so no backtracking there), then a domain run
/// whose qualifying dot is the LAST one followed by two or more letters:
/// that is where greedy backtracking of `[a-zA-Z0-9.-]+` lands.
fn matchEmailAt(text: []const u8, i: usize) ?usize {
    var p = i;
    while (p < text.len and isLocalByte(text[p])) p += 1;
    if (p == i) return null;
    if (p >= text.len or text[p] != '@') return null;
    p += 1;
    const run_start = p;
    while (p < text.len and isDomainRunByte(text[p])) p += 1;
    // The `+` gives back from the longest run, so try dots right to left.
    var d = p;
    while (d > run_start) {
        d -= 1;
        if (text[d] != '.') continue;
        var e = d + 1;
        while (e < p and isAlphaByte(text[e])) e += 1;
        if (e - (d + 1) >= 2) return e - i;
    }
    return null;
}

/// /\b(?:\d{1,3}\.){3}\d{1,3}\b/ — three dotted 1-3 digit groups, a final
/// 1-3 digit group, word boundaries at both ends. A 4+ digit run can never
/// satisfy `\d{1,3}` followed by `.` (or the trailing `\b`), exactly as the
/// regex rejects it.
fn matchIpv4At(text: []const u8, i: usize) ?usize {
    if (!wordBoundary(text, i)) return null;
    var p = i;
    var group: usize = 0;
    while (group < 3) : (group += 1) {
        const run = digitRunLen(text, p);
        if (run < 1 or run > 3) return null;
        p += run;
        if (p >= text.len or text[p] != '.') return null;
        p += 1;
    }
    const tail = digitRunLen(text, p);
    if (tail < 1 or tail > 3) return null;
    p += tail;
    if (!wordBoundary(text, p)) return null;
    return p - i;
}

/// /[0-9a-fA-F:]+/ — a maximal hex/colon run, post-filtered by `isIpv6`
/// (exactly as the TS lib does, so bare hex words, times, and MACs drop out).
fn matchIpv6RunAt(text: []const u8, i: usize) ?usize {
    var p = i;
    while (p < text.len and (isHexByte(text[p]) or text[p] == ':')) p += 1;
    return if (p == i) null else p - i;
}

/// /\b[a-fA-F0-9]{32}\b|{40}\b|{64}\b|{128}\b/ — a maximal hex run at a
/// word boundary whose length is exactly one of the four digest sizes
/// (md5/sha1/sha256/sha512). The trailing `\b` of every alternation arm
/// makes any other length fail, so set membership IS the alternation.
fn matchHashAt(text: []const u8, i: usize) ?usize {
    if (!wordBoundary(text, i)) return null;
    if (!isHexByte(text[i])) return null;
    var p = i;
    while (p < text.len and isHexByte(text[p])) p += 1;
    const len = p - i;
    if (len != 32 and len != 40 and len != 64 and len != 128) return null;
    if (!wordBoundary(text, p)) return null;
    return len;
}

/// /\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b/
/// — a 1-63 byte label (alnum edges, hyphens inside), then one or more
/// `.letters` tails of at least two letters, ending on a word boundary.
fn matchDomainAt(text: []const u8, i: usize) ?usize {
    if (!wordBoundary(text, i)) return null;
    if (!isAlnumByte(text[i])) return null;

    // Label: the maximal [a-zA-Z0-9-] run. The `.` that starts the tail can
    // only sit at the run's end (the class has no dot), so backtracking the
    // label shorter never helps — it must be maximal, at most 63 bytes
    // (1 + 61 + 1), and end alnum (a trailing hyphen has nowhere to go).
    var p = i + 1;
    while (p < text.len and (isAlnumByte(text[p]) or text[p] == '-')) p += 1;
    if (p - i > 63) return null;
    if (!isAlnumByte(text[p - 1])) return null;

    // Tail groups, greedy: `.letters` runs of >= 2 letters, as many as fit.
    // The trailing `\b` drops a final group that runs into a word byte (a
    // digit or underscore) — that is the `+` backtracking; after dropping,
    // the next byte is '.', a non-word byte, so the boundary then holds.
    var match_end: ?usize = null;
    var e = p;
    while (e < text.len and text[e] == '.') {
        const tail_start = e + 1;
        var t = tail_start;
        while (t < text.len and isAlphaByte(text[t])) t += 1;
        if (t - tail_start < 2) break; // `{2,}` unsatisfiable — `+` stops
        e = t;
        if (t >= text.len or !isWordByte(text[t])) match_end = e;
        if (t >= text.len or text[t] != '.') break;
    }
    return if (match_end) |end| end - i else null;
}

/// A hex/colon run is a plausible IPv6: it has a colon AND either contains
/// `::` (a compressed zero-run) or is exactly eight groups of 1-4 hex
/// digits. Mirrors `isIpv6()` in src/lib/extract.ts.
pub fn isIpv6(run: []const u8) bool {
    if (std.mem.indexOfScalar(u8, run, ':') == null) return false;
    if (std.mem.indexOf(u8, run, "::") != null) return true;
    var groups: usize = 0;
    var it = std.mem.splitScalar(u8, run, ':');
    while (it.next()) |g| {
        if (g.len < 1 or g.len > 4) return false;
        for (g) |c| {
            if (!isHexByte(c)) return false;
        }
        groups += 1;
    }
    return groups == 8;
}

/// Domain part (after the last `@`) of a matched email. Mirrors `domainOf()`
/// in src/lib/extract.ts (lastIndexOf('@') + slice).
fn domainOf(email: []const u8) []const u8 {
    if (std.mem.lastIndexOfScalar(u8, email, '@')) |at| {
        return email[at + 1 ..];
    }
    return email;
}

fn want(selected: []const Kind, k: Kind) bool {
    for (selected) |s| {
        if (s == k) return true;
    }
    return false;
}

/// An order-preserving dedup collector — the Go twin of the `uniq()` helper
/// in src/lib/extract.ts. Keys are the match slices themselves
/// (content-hashed by StringHashMap), so equal-text matches dedupe.
const Collector = struct {
    list: std.ArrayList([]const u8),
    seen: std.StringHashMap(void),

    fn init(allocator: std.mem.Allocator) Collector {
        return .{
            .list = std.ArrayList([]const u8).init(allocator),
            .seen = std.StringHashMap(void).init(allocator),
        };
    }

    fn deinit(self: *Collector) void {
        self.list.deinit();
        self.seen.deinit();
    }

    fn push(self: *Collector, s: []const u8) !void {
        if (self.seen.contains(s)) return;
        try self.seen.put(s, {});
        try self.list.append(s);
    }
};

/// Walk `text` left to right; at each index try `matchFn`; on a match emit
/// it and resume after it, otherwise step one byte. That is exactly how a
/// global regex scans (leftmost, non-overlapping).
fn scanAll(
    comptime matchFn: fn ([]const u8, usize) ?usize,
    out: *Collector,
    text: []const u8,
) !void {
    var i: usize = 0;
    while (i < text.len) {
        if (matchFn(text, i)) |len| {
            try out.push(text[i .. i + len]);
            i += len;
        } else {
            i += 1;
        }
    }
}

/// Extract every occurrence of the given `types` (default: all six) from
/// `input`. Returns a `Result` with one field per kind — always all six,
/// populated only for the selected types (unselected kinds stay empty).
/// Matches are deduped per kind, preserving first-occurrence order. An email
/// also contributes its domain to the `domain` list when both `.email` and
/// `.domain` are selected.
pub fn extract(allocator: std.mem.Allocator, input: []const u8, types: []const Kind) !Result {
    const selected = if (types.len == 0) &all_kinds else types;

    var out = Result{};

    if (want(selected, .url)) {
        var c = Collector.init(allocator);
        defer c.deinit();
        try scanAll(matchUrlAt, &c, input);
        out.url = try c.list.toOwnedSlice();
    }
    if (want(selected, .email)) {
        var c = Collector.init(allocator);
        defer c.deinit();
        try scanAll(matchEmailAt, &c, input);
        out.email = try c.list.toOwnedSlice();
    }
    if (want(selected, .ipv4)) {
        var c = Collector.init(allocator);
        defer c.deinit();
        try scanAll(matchIpv4At, &c, input);
        out.ipv4 = try c.list.toOwnedSlice();
    }
    if (want(selected, .ipv6)) {
        var c = Collector.init(allocator);
        defer c.deinit();
        // The permissive hex/colon run, post-filtered — as the TS lib does.
        var i: usize = 0;
        while (i < input.len) {
            if (matchIpv6RunAt(input, i)) |len| {
                if (isIpv6(input[i .. i + len])) try c.push(input[i .. i + len]);
                i += len;
            } else {
                i += 1;
            }
        }
        out.ipv6 = try c.list.toOwnedSlice();
    }
    if (want(selected, .hash)) {
        var c = Collector.init(allocator);
        defer c.deinit();
        try scanAll(matchHashAt, &c, input);
        out.hash = try c.list.toOwnedSlice();
    }
    if (want(selected, .domain)) {
        var c = Collector.init(allocator);
        defer c.deinit();
        try scanAll(matchDomainAt, &c, input);
        // Cross-rule: an email also yields its domain in the domain list.
        if (want(selected, .email)) {
            var em = Collector.init(allocator);
            defer em.deinit();
            try scanAll(matchEmailAt, &em, input);
            for (em.list.items) |e| {
                try c.push(domainOf(e));
            }
        }
        out.domain = try c.list.toOwnedSlice();
    }

    return out;
}

/// Free every slice in `result` (the slice arrays; the matched bytes live in
/// the caller's `input`). Freeing the empty default is a no-op.
pub fn deinitResult(allocator: std.mem.Allocator, result: *const Result) void {
    allocator.free(result.url);
    allocator.free(result.email);
    allocator.free(result.ipv4);
    allocator.free(result.ipv6);
    allocator.free(result.hash);
    allocator.free(result.domain);
}

// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------

const testing = std.testing;

// Well-known digests of the empty string (real hash values), shared with
// src/lib/extract.test.ts so the showcase uses identical vectors.
const md5_empty = "d41d8cd98f00b204e9800998ecf8427e"; // 32
const sha256_empty = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"; // 64

test "extracts and dedupes urls" {
    var r = try extract(
        testing.allocator,
        "a https://x.com b https://y.com c https://x.com",
        &.{},
    );
    defer deinitResult(testing.allocator, &r);
    try testing.expectEqual(@as(usize, 2), r.url.len);
    try testing.expectEqualStrings("https://x.com", r.url[0]);
    try testing.expectEqualStrings("https://y.com", r.url[1]);
}

test "emails and their domains" {
    // Plus-tags and multi-part-TLD domains are matched; when email AND
    // domain are both selected, an email also contributes its domain.
    var r = try extract(
        testing.allocator,
        "reach a.b+tag@mail.example.co.uk please",
        &.{},
    );
    defer deinitResult(testing.allocator, &r);
    try testing.expectEqual(@as(usize, 1), r.email.len);
    try testing.expectEqualStrings("a.b+tag@mail.example.co.uk", r.email[0]);
    try testing.expectEqual(@as(usize, 1), r.domain.len);
    try testing.expectEqualStrings("mail.example.co.uk", r.domain[0]);
}

test "ipv6 keeps compressed rejects times" {
    // `::1` is a compressed zero-run; `12:30:45` has no `::` and only 3
    // groups, so it is rejected as a clock, not an address.
    var r = try extract(
        testing.allocator,
        "loopback ::1 and time 12:30:45 now",
        &.{},
    );
    defer deinitResult(testing.allocator, &r);
    try testing.expectEqual(@as(usize, 1), r.ipv6.len);
    try testing.expectEqualStrings("::1", r.ipv6[0]);
}

test "hashes by length" {
    const text = "m " ++ md5_empty ++ " s " ++ sha256_empty;
    var r = try extract(testing.allocator, text, &.{});
    defer deinitResult(testing.allocator, &r);
    try testing.expectEqual(@as(usize, 2), r.hash.len);
    try testing.expectEqualStrings(md5_empty, r.hash[0]);
    try testing.expectEqualStrings(sha256_empty, r.hash[1]);
}

test "type selection returns only selected" {
    var r = try extract(testing.allocator, "https://x.com and a@b.com", &.{.url});
    defer deinitResult(testing.allocator, &r);
    try testing.expectEqual(@as(usize, 1), r.url.len);
    try testing.expectEqualStrings("https://x.com", r.url[0]);
    try testing.expectEqual(@as(usize, 0), r.email.len); // email not selected
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →