Text Extractor — Zig source
Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
//! out of arbitrary text (logs, headers, config).
//!
//! Language: Zig 0.13 (standard library only)
//! Source: CosmoDev polyglot showcase port of the Extract tool, ported from
//! src/lib/extract.ts (the canonical TypeScript implementation) and
//! held in lock-step with its Go twin cli/extract/extract.go.
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; allocation failure is the only error path.
//! - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
//!
//! Dependency note: unlike Go / Python / PHP / JS (and Rust's `regex` crate),
//! Zig ships no regex engine in std. The six per-kind patterns are small and
//! regular enough to hand-roll, so this port re-implements each pattern's
//! exact semantics — greedy runs, `\b` word boundaries, counted repetition,
//! the alternation of hash lengths — as explicit left-to-right byte scanners.
//! Each matcher documents the pattern it mirrors from the `RE` record in
//! src/lib/extract.ts, so equivalence stays auditable without an engine.
//! The scanners work byte-wise over UTF-8: every pattern class is ASCII, and
//! `\b`/`\s` are ASCII-defined exactly as in Go's regexp (a multibyte
//! character is simply a run of non-word, non-space bytes).
const std = @import("std");
/// One of the six canonical extraction kinds. Mirrors the TS `ExtractType`
/// union ('url' | 'email' | 'ipv4' | 'ipv6' | 'hash' | 'domain') and Go's
/// `Type`.
pub const Kind = enum { url, email, ipv4, ipv6, hash, domain };
/// The canonical kinds in display order — TS's `EXTRACT_TYPES`.
pub const all_kinds = [6]Kind{ .url, .email, .ipv4, .ipv6, .hash, .domain };
/// One field per kind — always all six, empty when unselected. The inner
/// slices are allocated by `extract` (free them with `deinitResult`); the
/// matched bytes themselves point into the caller's `input`, zero-copy.
pub const Result = struct {
url: []const []const u8 = &.{},
email: []const []const u8 = &.{},
ipv4: []const []const u8 = &.{},
ipv6: []const []const u8 = &.{},
hash: []const []const u8 = &.{},
domain: []const []const u8 = &.{},
};
// ---------- byte classes (ASCII, exactly the pattern classes) ----------
fn isAlphaByte(c: u8) bool {
return (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z');
}
fn isDigitByte(c: u8) bool {
return c >= '0' and c <= '9';
}
fn isAlnumByte(c: u8) bool {
return isAlphaByte(c) or isDigitByte(c);
}
fn isHexByte(c: u8) bool {
return isDigitByte(c) or (c >= 'a' and c <= 'f') or (c >= 'A' and c <= 'F');
}
/// `\w` as Go/JS define it: [A-Za-z0-9_] — the class behind `\b`.
fn isWordByte(c: u8) bool {
return isAlnumByte(c) or c == '_';
}
/// `\s` as Go defines it (ASCII whitespace); the `[^\\s]` of the URL tail is
/// its complement, so multibyte UTF-8 bytes read as non-space, same as Go.
fn isSpaceByte(c: u8) bool {
return c == ' ' or c == '\t' or c == '\n' or c == '\r' or c == 0x0b or c == 0x0c;
}
/// `\b` at index `i` of `text`: exactly one side of the index is a word byte.
fn wordBoundary(text: []const u8, i: usize) bool {
const before = i > 0 and isWordByte(text[i - 1]);
const after = i < text.len and isWordByte(text[i]);
return before != after;
}
/// Length of the maximal digit run starting at `i` (greedy `\d+`).
fn digitRunLen(text: []const u8, i: usize) usize {
var p = i;
while (p < text.len and isDigitByte(text[p])) p += 1;
return p - i;
}
// ---------- per-kind matchers: length of the match at `i`, or null ----------
// Each returns the match length for a match anchored at `i`; the driver
// below walks the text and advances past each hit — exactly how a global
// regex scans (leftmost, non-overlapping).
/// /https?:\/\/[^\s]+/ — `http`, an optional `s`, `://`, then a maximal
/// non-whitespace run of at least one byte.
fn matchUrlAt(text: []const u8, i: usize) ?usize {
const rest = text[i..];
if (!std.mem.startsWith(u8, rest, "http")) return null;
var p: usize = 4;
if (p < rest.len and rest[p] == 's') p += 1;
if (!std.mem.startsWith(u8, rest[p..], "://")) return null;
p += 3;
const body = p;
while (p < rest.len and !isSpaceByte(rest[p])) p += 1;
if (p == body) return null; // `+` needs at least one byte
return p;
}
/// The local-part class [a-zA-Z0-9._%+-].
fn isLocalByte(c: u8) bool {
return isAlnumByte(c) or c == '.' or c == '_' or c == '%' or c == '+' or c == '-';
}
/// The email-domain class [a-zA-Z0-9.-].
fn isDomainRunByte(c: u8) bool {
return isAlnumByte(c) or c == '.' or c == '-';
}
/// /[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}/ — a greedy local run
/// ('@' is not in the class, so no backtracking there), then a domain run
/// whose qualifying dot is the LAST one followed by two or more letters:
/// that is where greedy backtracking of `[a-zA-Z0-9.-]+` lands.
fn matchEmailAt(text: []const u8, i: usize) ?usize {
var p = i;
while (p < text.len and isLocalByte(text[p])) p += 1;
if (p == i) return null;
if (p >= text.len or text[p] != '@') return null;
p += 1;
const run_start = p;
while (p < text.len and isDomainRunByte(text[p])) p += 1;
// The `+` gives back from the longest run, so try dots right to left.
var d = p;
while (d > run_start) {
d -= 1;
if (text[d] != '.') continue;
var e = d + 1;
while (e < p and isAlphaByte(text[e])) e += 1;
if (e - (d + 1) >= 2) return e - i;
}
return null;
}
/// /\b(?:\d{1,3}\.){3}\d{1,3}\b/ — three dotted 1-3 digit groups, a final
/// 1-3 digit group, word boundaries at both ends. A 4+ digit run can never
/// satisfy `\d{1,3}` followed by `.` (or the trailing `\b`), exactly as the
/// regex rejects it.
fn matchIpv4At(text: []const u8, i: usize) ?usize {
if (!wordBoundary(text, i)) return null;
var p = i;
var group: usize = 0;
while (group < 3) : (group += 1) {
const run = digitRunLen(text, p);
if (run < 1 or run > 3) return null;
p += run;
if (p >= text.len or text[p] != '.') return null;
p += 1;
}
const tail = digitRunLen(text, p);
if (tail < 1 or tail > 3) return null;
p += tail;
if (!wordBoundary(text, p)) return null;
return p - i;
}
/// /[0-9a-fA-F:]+/ — a maximal hex/colon run, post-filtered by `isIpv6`
/// (exactly as the TS lib does, so bare hex words, times, and MACs drop out).
fn matchIpv6RunAt(text: []const u8, i: usize) ?usize {
var p = i;
while (p < text.len and (isHexByte(text[p]) or text[p] == ':')) p += 1;
return if (p == i) null else p - i;
}
/// /\b[a-fA-F0-9]{32}\b|{40}\b|{64}\b|{128}\b/ — a maximal hex run at a
/// word boundary whose length is exactly one of the four digest sizes
/// (md5/sha1/sha256/sha512). The trailing `\b` of every alternation arm
/// makes any other length fail, so set membership IS the alternation.
fn matchHashAt(text: []const u8, i: usize) ?usize {
if (!wordBoundary(text, i)) return null;
if (!isHexByte(text[i])) return null;
var p = i;
while (p < text.len and isHexByte(text[p])) p += 1;
const len = p - i;
if (len != 32 and len != 40 and len != 64 and len != 128) return null;
if (!wordBoundary(text, p)) return null;
return len;
}
/// /\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b/
/// — a 1-63 byte label (alnum edges, hyphens inside), then one or more
/// `.letters` tails of at least two letters, ending on a word boundary.
fn matchDomainAt(text: []const u8, i: usize) ?usize {
if (!wordBoundary(text, i)) return null;
if (!isAlnumByte(text[i])) return null;
// Label: the maximal [a-zA-Z0-9-] run. The `.` that starts the tail can
// only sit at the run's end (the class has no dot), so backtracking the
// label shorter never helps — it must be maximal, at most 63 bytes
// (1 + 61 + 1), and end alnum (a trailing hyphen has nowhere to go).
var p = i + 1;
while (p < text.len and (isAlnumByte(text[p]) or text[p] == '-')) p += 1;
if (p - i > 63) return null;
if (!isAlnumByte(text[p - 1])) return null;
// Tail groups, greedy: `.letters` runs of >= 2 letters, as many as fit.
// The trailing `\b` drops a final group that runs into a word byte (a
// digit or underscore) — that is the `+` backtracking; after dropping,
// the next byte is '.', a non-word byte, so the boundary then holds.
var match_end: ?usize = null;
var e = p;
while (e < text.len and text[e] == '.') {
const tail_start = e + 1;
var t = tail_start;
while (t < text.len and isAlphaByte(text[t])) t += 1;
if (t - tail_start < 2) break; // `{2,}` unsatisfiable — `+` stops
e = t;
if (t >= text.len or !isWordByte(text[t])) match_end = e;
if (t >= text.len or text[t] != '.') break;
}
return if (match_end) |end| end - i else null;
}
/// A hex/colon run is a plausible IPv6: it has a colon AND either contains
/// `::` (a compressed zero-run) or is exactly eight groups of 1-4 hex
/// digits. Mirrors `isIpv6()` in src/lib/extract.ts.
pub fn isIpv6(run: []const u8) bool {
if (std.mem.indexOfScalar(u8, run, ':') == null) return false;
if (std.mem.indexOf(u8, run, "::") != null) return true;
var groups: usize = 0;
var it = std.mem.splitScalar(u8, run, ':');
while (it.next()) |g| {
if (g.len < 1 or g.len > 4) return false;
for (g) |c| {
if (!isHexByte(c)) return false;
}
groups += 1;
}
return groups == 8;
}
/// Domain part (after the last `@`) of a matched email. Mirrors `domainOf()`
/// in src/lib/extract.ts (lastIndexOf('@') + slice).
fn domainOf(email: []const u8) []const u8 {
if (std.mem.lastIndexOfScalar(u8, email, '@')) |at| {
return email[at + 1 ..];
}
return email;
}
fn want(selected: []const Kind, k: Kind) bool {
for (selected) |s| {
if (s == k) return true;
}
return false;
}
/// An order-preserving dedup collector — the Go twin of the `uniq()` helper
/// in src/lib/extract.ts. Keys are the match slices themselves
/// (content-hashed by StringHashMap), so equal-text matches dedupe.
const Collector = struct {
list: std.ArrayList([]const u8),
seen: std.StringHashMap(void),
fn init(allocator: std.mem.Allocator) Collector {
return .{
.list = std.ArrayList([]const u8).init(allocator),
.seen = std.StringHashMap(void).init(allocator),
};
}
fn deinit(self: *Collector) void {
self.list.deinit();
self.seen.deinit();
}
fn push(self: *Collector, s: []const u8) !void {
if (self.seen.contains(s)) return;
try self.seen.put(s, {});
try self.list.append(s);
}
};
/// Walk `text` left to right; at each index try `matchFn`; on a match emit
/// it and resume after it, otherwise step one byte. That is exactly how a
/// global regex scans (leftmost, non-overlapping).
fn scanAll(
comptime matchFn: fn ([]const u8, usize) ?usize,
out: *Collector,
text: []const u8,
) !void {
var i: usize = 0;
while (i < text.len) {
if (matchFn(text, i)) |len| {
try out.push(text[i .. i + len]);
i += len;
} else {
i += 1;
}
}
}
/// Extract every occurrence of the given `types` (default: all six) from
/// `input`. Returns a `Result` with one field per kind — always all six,
/// populated only for the selected types (unselected kinds stay empty).
/// Matches are deduped per kind, preserving first-occurrence order. An email
/// also contributes its domain to the `domain` list when both `.email` and
/// `.domain` are selected.
pub fn extract(allocator: std.mem.Allocator, input: []const u8, types: []const Kind) !Result {
const selected = if (types.len == 0) &all_kinds else types;
var out = Result{};
if (want(selected, .url)) {
var c = Collector.init(allocator);
defer c.deinit();
try scanAll(matchUrlAt, &c, input);
out.url = try c.list.toOwnedSlice();
}
if (want(selected, .email)) {
var c = Collector.init(allocator);
defer c.deinit();
try scanAll(matchEmailAt, &c, input);
out.email = try c.list.toOwnedSlice();
}
if (want(selected, .ipv4)) {
var c = Collector.init(allocator);
defer c.deinit();
try scanAll(matchIpv4At, &c, input);
out.ipv4 = try c.list.toOwnedSlice();
}
if (want(selected, .ipv6)) {
var c = Collector.init(allocator);
defer c.deinit();
// The permissive hex/colon run, post-filtered — as the TS lib does.
var i: usize = 0;
while (i < input.len) {
if (matchIpv6RunAt(input, i)) |len| {
if (isIpv6(input[i .. i + len])) try c.push(input[i .. i + len]);
i += len;
} else {
i += 1;
}
}
out.ipv6 = try c.list.toOwnedSlice();
}
if (want(selected, .hash)) {
var c = Collector.init(allocator);
defer c.deinit();
try scanAll(matchHashAt, &c, input);
out.hash = try c.list.toOwnedSlice();
}
if (want(selected, .domain)) {
var c = Collector.init(allocator);
defer c.deinit();
try scanAll(matchDomainAt, &c, input);
// Cross-rule: an email also yields its domain in the domain list.
if (want(selected, .email)) {
var em = Collector.init(allocator);
defer em.deinit();
try scanAll(matchEmailAt, &em, input);
for (em.list.items) |e| {
try c.push(domainOf(e));
}
}
out.domain = try c.list.toOwnedSlice();
}
return out;
}
/// Free every slice in `result` (the slice arrays; the matched bytes live in
/// the caller's `input`). Freeing the empty default is a no-op.
pub fn deinitResult(allocator: std.mem.Allocator, result: *const Result) void {
allocator.free(result.url);
allocator.free(result.email);
allocator.free(result.ipv4);
allocator.free(result.ipv6);
allocator.free(result.hash);
allocator.free(result.domain);
}
// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
const testing = std.testing;
// Well-known digests of the empty string (real hash values), shared with
// src/lib/extract.test.ts so the showcase uses identical vectors.
const md5_empty = "d41d8cd98f00b204e9800998ecf8427e"; // 32
const sha256_empty = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"; // 64
test "extracts and dedupes urls" {
var r = try extract(
testing.allocator,
"a https://x.com b https://y.com c https://x.com",
&.{},
);
defer deinitResult(testing.allocator, &r);
try testing.expectEqual(@as(usize, 2), r.url.len);
try testing.expectEqualStrings("https://x.com", r.url[0]);
try testing.expectEqualStrings("https://y.com", r.url[1]);
}
test "emails and their domains" {
// Plus-tags and multi-part-TLD domains are matched; when email AND
// domain are both selected, an email also contributes its domain.
var r = try extract(
testing.allocator,
"reach a.b+tag@mail.example.co.uk please",
&.{},
);
defer deinitResult(testing.allocator, &r);
try testing.expectEqual(@as(usize, 1), r.email.len);
try testing.expectEqualStrings("a.b+tag@mail.example.co.uk", r.email[0]);
try testing.expectEqual(@as(usize, 1), r.domain.len);
try testing.expectEqualStrings("mail.example.co.uk", r.domain[0]);
}
test "ipv6 keeps compressed rejects times" {
// `::1` is a compressed zero-run; `12:30:45` has no `::` and only 3
// groups, so it is rejected as a clock, not an address.
var r = try extract(
testing.allocator,
"loopback ::1 and time 12:30:45 now",
&.{},
);
defer deinitResult(testing.allocator, &r);
try testing.expectEqual(@as(usize, 1), r.ipv6.len);
try testing.expectEqualStrings("::1", r.ipv6[0]);
}
test "hashes by length" {
const text = "m " ++ md5_empty ++ " s " ++ sha256_empty;
var r = try extract(testing.allocator, text, &.{});
defer deinitResult(testing.allocator, &r);
try testing.expectEqual(@as(usize, 2), r.hash.len);
try testing.expectEqualStrings(md5_empty, r.hash[0]);
try testing.expectEqualStrings(sha256_empty, r.hash[1]);
}
test "type selection returns only selected" {
var r = try extract(testing.allocator, "https://x.com and a@b.com", &.{.url});
defer deinitResult(testing.allocator, &r);
try testing.expectEqual(@as(usize, 1), r.url.len);
try testing.expectEqualStrings("https://x.com", r.url[0]);
try testing.expectEqual(@as(usize, 0), r.email.len); // email not selected
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →