Skip to content

PII Redactor — Zig source

Paste text and automatically detect and mask personal data — emails, phone numbers, IP addresses, SSNs, credit card numbers, and dates.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! pii-redactor — PII detection & redaction (email, phone, IPs, SSN, cards, dates).
//!
//! Language: Zig 0.14 (standard library only)
//! Ported from: src/lib/pii-redactor.ts (the canonical TypeScript implementation).
//! display source — part of CosmoDev's polyglot tool pages.
//!
//! Deterministic detection for seven personal-data types. The TS reference
//! drives each detector with a regex; Zig's standard library has no regex
//! engine, so this port re-expresses the same patterns as hand-rolled
//! cursor scanners with identical candidate spans. Every candidate still
//! passes the same structural validator (octet ranges, Luhn checksum,
//! month/day bounds, E.164 digit count) to keep false positives low.
//! Overlapping candidates resolve by type priority - unambiguous types
//! (email, Luhn-passing card numbers, SSNs, IPs, dates) claim their span
//! before the fuzzy phone pattern. Never fails.
//!
//! The fuzzy phone pattern is a greedy forward match (no backtracking),
//! which covers the practical shapes `+C (A) G1…G4`.

const std = @import("std");

/// The seven PII types the detector knows.
pub const PiiType = enum {
    email,
    phone,
    ipv4,
    ipv6,
    ssn,
    credit_card,
    date,
};

/// One detected personal-data item: where it is (indices into the input).
pub const PiiMatch = struct {
    type: PiiType,
    start: usize, // index of the first byte in the input
    end: usize, // index one past the last byte

    /// The matched substring, verbatim.
    pub fn original(self: PiiMatch, text: []const u8) []const u8 {
        return text[self.start..self.end];
    }
};

/// All PII types, in display order.
pub const pii_types = [_]PiiType{
    .email, .phone, .ipv4, .ipv6, .ssn, .credit_card, .date,
};

/// Overlap resolution: when two candidates cover the same span, the more
/// specific type wins. Phone is deliberately last - a date, SSN, IP, or card
/// number can all masquerade as one.
fn priority(t: PiiType) u8 {
    return switch (t) {
        .email => 0,
        .credit_card => 1,
        .ssn => 2,
        .ipv6 => 3,
        .ipv4 => 4,
        .date => 5,
        .phone => 6,
    };
}

// --- Character classes ----------------------------------------------------

fn isDigit(c: u8) bool {
    return c >= '0' and c <= '9';
}

fn isWordChar(c: u8) bool {
    return isDigit(c) or (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or c == '_';
}

fn isHexDigit(c: u8) bool {
    return isDigit(c) or (c >= 'a' and c <= 'f') or (c >= 'A' and c <= 'F');
}

fn isLocalChar(c: u8) bool { // [A-Za-z0-9._%+-]
    return isWordChar(c) or c == '.' or c == '%' or c == '+' or c == '-';
}

fn isDomainChar(c: u8) bool { // [A-Za-z0-9.-]
    return isWordChar(c) or c == '.' or c == '-';
}

fn isSeparator(c: u8) bool { // [ .-]
    return c == ' ' or c == '.' or c == '-';
}

// --- Validators (identical logic to the TS reference) ----------------------

/// Luhn checksum. `digits` must be a non-empty string of 0-9; any other
/// character makes it invalid.
pub fn isValidLuhn(digits: []const u8) bool {
    if (digits.len == 0) return false;
    var sum: u32 = 0;
    var double = false;
    var i: usize = digits.len;
    while (i > 0) {
        i -= 1;
        const c = digits[i];
        if (!isDigit(c)) return false;
        var d: u32 = c - '0';
        if (double) {
            d *= 2;
            if (d > 9) d -= 9;
        }
        sum += d;
        double = !double;
    }
    return sum % 10 == 0;
}

/// Octets 0-255 each; the scanner already bounds the shape to a dotted quad.
fn isValidIpv4(candidate: []const u8) bool {
    var it = std.mem.splitScalar(u8, candidate, '.');
    while (it.next()) |octet| {
        const n = std.fmt.parseInt(u32, octet, 10) catch return false;
        if (n > 255) return false;
    }
    return true;
}

fn isHexGroup1to4(g: []const u8) bool {
    if (g.len == 0 or g.len > 4) return false;
    for (g) |c| if (!isHexDigit(c)) return false;
    return true;
}

/// Full 8-group form, or a compressed `::` form expanding to exactly 8.
fn isValidIpv6(candidate: []const u8) bool {
    // Lone ":" / "::" (URL scheme separators like https://) carry no hex digits.
    var has_hex = false;
    for (candidate) |c| {
        if (isHexDigit(c)) {
            has_hex = true;
            break;
        }
    }
    if (!has_hex) return false;

    var groups = std.mem.splitScalar(u8, candidate, ':');
    var has_empty_group = false;
    while (groups.next()) |g| {
        if (g.len == 0) has_empty_group = true;
    }

    if (has_empty_group) {
        // Compressed: at most one "::", its sides together hold < 8 groups.
        const dbl = std.mem.indexOf(u8, candidate, "::") orelse return false;
        if (std.mem.indexOfPos(u8, candidate, dbl + 2, "::") != null) return false;
        const left = candidate[0..dbl];
        const right = candidate[dbl + 2 ..];
        var left_count: usize = 0;
        if (left.len > 0) {
            var lg = std.mem.splitScalar(u8, left, ':');
            while (lg.next()) |g| {
                if (!isHexGroup1to4(g)) return false;
                left_count += 1;
            }
        }
        var right_count: usize = 0;
        if (right.len > 0) {
            var rg = std.mem.splitScalar(u8, right, ':');
            while (rg.next()) |g| {
                if (!isHexGroup1to4(g)) return false;
                right_count += 1;
            }
        }
        if (left_count + right_count > 7) return false;
        return true;
    }

    var count: usize = 0;
    var g2 = std.mem.splitScalar(u8, candidate, ':');
    while (g2.next()) |g| {
        if (!isHexGroup1to4(g)) return false;
        count += 1;
    }
    return count == 8;
}

/// ISO calendar plausibility: month 01-12, day 01-31.
fn isValidDate(candidate: []const u8) bool {
    if (candidate.len != 10 or candidate[4] != '-' or candidate[7] != '-') return false;
    for (candidate, 0..) |c, i| {
        if (i == 4 or i == 7) continue;
        if (!isDigit(c)) return false;
    }
    const month = (candidate[5] - '0') * 10 + (candidate[6] - '0');
    const day = (candidate[8] - '0') * 10 + (candidate[9] - '0');
    return month >= 1 and month <= 12 and day >= 1 and day <= 31;
}

fn countDigits(candidate: []const u8) usize {
    var n: usize = 0;
    for (candidate) |c| {
        if (isDigit(c)) n += 1;
    }
    return n;
}

fn looksLikeDottedQuad(candidate: []const u8) bool {
    var dots: usize = 0;
    var group_digits: usize = 0;
    for (candidate) |c| {
        if (c == '.') {
            dots += 1;
            if (group_digits == 0 or group_digits > 3) return false;
            group_digits = 0;
        } else if (isDigit(c)) {
            group_digits += 1;
        } else {
            return false;
        }
    }
    return dots == 3 and group_digits >= 1 and group_digits <= 3;
}

/// E.164 digit budget (7-15) and structural guards for the fuzzy phone shape.
fn isValidPhone(candidate: []const u8) bool {
    const digits = countDigits(candidate);
    if (digits < 7 or digits > 15) return false;
    // A dotted quad is IP-shaped: if it were a valid IP it was already claimed
    // by the ipv4 detector; an invalid one (999.x) is likelier a version string.
    if (looksLikeDottedQuad(candidate)) return false;
    // YYYY-MM-DD shaped (even an impossible date) is never a phone number.
    if (candidate.len == 13 and isValidDateShape(candidate)) return false;
    return true;
}

fn isValidDateShape(candidate: []const u8) bool {
    // "^\\d{4}-\\d{2}-\\d{2}$" shape check (plausibility is checked elsewhere).
    if (candidate.len != 10) return false;
    for (candidate, 0..) |c, i| {
        if (i == 4 or i == 7) {
            if (c != '-') return false;
        } else if (!isDigit(c)) return false;
    }
    return true;
}

/// 13-19 digits with optional space/dash grouping, plus a Luhn checksum.
fn isValidCard(candidate: []const u8) bool {
    var buf: [24]u8 = undefined;
    var n: usize = 0;
    for (candidate) |c| {
        if (isDigit(c)) {
            buf[n] = c;
            n += 1;
        }
    }
    return n >= 13 and n <= 19 and isValidLuhn(buf[0..n]);
}

// --- Scanners (cursor re-expressions of the TS regexes) ---------------------

const CandidateList = std.ArrayList(PiiMatch);

fn push(list: *CandidateList, t: PiiType, start: usize, end: usize) void {
    list.append(.{ .type = t, .start = start, .end = end }) catch {};
}

/// [A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}
fn scanEmail(text: []const u8, out: *CandidateList) void {
    var i: usize = 0;
    while (i < text.len) : (i += 1) {
        if (text[i] != '@') continue;
        // Extend left over the local part (must be non-empty and start-bounded).
        var s = i;
        while (s > 0 and isLocalChar(text[s - 1])) s -= 1;
        if (s == i) continue; // empty local part
        // Extend right over the domain, then backtrack to the last "."
        // followed by 2+ letters running to the end of the domain run
        // (the greedy `[A-Za-z0-9.-]+\.[A-Za-z]{2,}` tail).
        var e = i + 1;
        while (e < text.len and isDomainChar(text[e])) e += 1;
        var last_dot: ?usize = null;
        var d = e;
        while (d > i + 1) {
            d -= 1;
            if (text[d] == '.') {
                var m = d + 1;
                while (m < e and std.ascii.isAlphabetic(text[m])) m += 1;
                if (m == e and m - d - 1 >= 2) {
                    last_dot = d;
                    break;
                }
            }
        }
        if (last_dot == null) continue;
        push(out, .email, s, e);
        i = e - 1; // resume after the whole match
    }
}

/// \d(?:[ -]?\d){11,} — a maximal run of 12+ digits with single space/dash
/// separators; isValidCard then enforces 13-19 digits + Luhn on the run.
fn scanCreditCard(text: []const u8, out: *CandidateList) void {
    var i: usize = 0;
    while (i < text.len) {
        if (!isDigit(text[i])) {
            i += 1;
            continue;
        }
        var e = i + 1;
        var digits: usize = 1;
        while (e < text.len) {
            if (isDigit(text[e])) {
                digits += 1;
                e += 1;
            } else if ((text[e] == ' ' or text[e] == '-') and e + 1 < text.len and
                isDigit(text[e + 1]))
            {
                digits += 1;
                e += 2;
            } else break;
        }
        if (digits >= 12) {
            if (isValidCard(text[i..e])) push(out, .credit_card, i, e);
            i = e;
        } else {
            i += 1;
        }
    }
}

/// \b\d{3}-\d{2}-\d{4}\b
fn scanSsn(text: []const u8, out: *CandidateList) void {
    var i: usize = 0;
    while (i + 11 <= text.len) : (i += 1) {
        const c = text[i .. i + 11];
        if (!(isDigit(c[0]) and isDigit(c[1]) and isDigit(c[2]) and c[3] == '-' and
            isDigit(c[4]) and isDigit(c[5]) and c[6] == '-' and
            isDigit(c[7]) and isDigit(c[8]) and isDigit(c[9]) and isDigit(c[10]))) continue;
        const before_ok = i == 0 or !isWordChar(text[i - 1]);
        const after = i + 11;
        const after_ok = after >= text.len or !isWordChar(text[after]);
        if (before_ok and after_ok) {
            push(out, .ssn, i, after);
            i = after - 1;
        }
    }
}

/// Hex groups joined by colons; isValidIpv6 rejects prose like "10:30:45".
/// Scans maximal [0-9a-fA-F:] runs containing at least one colon.
fn scanIpv6(text: []const u8, out: *CandidateList) void {
    var i: usize = 0;
    while (i < text.len) {
        if (!(isHexDigit(text[i]) or text[i] == ':')) {
            i += 1;
            continue;
        }
        var e = i;
        var colons: usize = 0;
        while (e < text.len and (isHexDigit(text[e]) or text[e] == ':')) {
            if (text[e] == ':') colons += 1;
            e += 1;
        }
        if (colons >= 1 and isValidIpv6(text[i..e])) push(out, .ipv6, i, e);
        i = e;
    }
}

/// (?<![\w.])(?:\d{1,3}\.){3}\d{1,3}(?!\.?\d)(?!\w) — dotted quad with guards
/// that keep it out of versions ("v1.2.3.4") and longer quintets ("1.2.3.4.5").
fn scanIpv4(text: []const u8, out: *CandidateList) void {
    var i: usize = 0;
    while (i < text.len) : (i += 1) {
        if (i > 0 and (isWordChar(text[i - 1]) or text[i - 1] == '.')) continue;
        // Parse four 1-3 digit octets separated by dots.
        var e = i;
        var ok = true;
        var octet: usize = 0;
        while (octet < 4) : (octet += 1) {
            if (octet > 0) {
                if (e < text.len and text[e] == '.') {
                    e += 1;
                } else {
                    ok = false;
                    break;
                }
            }
            var n: usize = 0;
            while (e < text.len and isDigit(text[e]) and n < 3) : (n += 1) e += 1;
            if (n == 0 or (n == 3 and e < text.len and isDigit(text[e]))) {
                ok = false; // empty octet, or a 4+ digit run the regex can't consume
                break;
            }
        }
        if (!ok) continue;
        // Lookahead: no ".digit" continuation, no word char.
        if (e < text.len) {
            if (text[e] == '.' and e + 1 < text.len and isDigit(text[e + 1])) continue;
            if (isWordChar(text[e])) continue;
        }
        if (isValidIpv4(text[i..e])) {
            push(out, .ipv4, i, e);
            i = e - 1;
        }
    }
}

/// (?<!\d)\d{4}-\d{2}-\d{2}(?!\d)
fn scanDate(text: []const u8, out: *CandidateList) void {
    var i: usize = 0;
    while (i + 10 <= text.len) : (i += 1) {
        if (i > 0 and isDigit(text[i - 1])) continue;
        const c = text[i .. i + 10];
        if (!isValidDateShape(c)) continue;
        if (i + 10 < text.len and isDigit(text[i + 10])) continue;
        if (isValidDate(c)) {
            push(out, .date, i, i + 10);
            i += 9;
        }
    }
}

fn takeDigits(text: []const u8, i: *usize, max: usize) usize {
    var n: usize = 0;
    while (i.* < text.len and isDigit(text[i.*]) and n < max) : (n += 1) i.* += 1;
    return n;
}

/// (?<![\d(])(?:\+\d{1,3}[ .-]?)?(?:\(\d{1,4}\)|\d{1,4})(?:[ .-]?\d{2,4}){1,4}(?!\d)
/// Greedy forward match (no backtracking) - covers the practical phone shapes.
fn scanPhone(text: []const u8, out: *CandidateList) void {
    var i: usize = 0;
    while (i < text.len) : (i += 1) {
        if (i > 0 and (isDigit(text[i - 1]) or text[i - 1] == '(')) continue;
        const start = i;
        var j = i;
        // Optional +country prefix.
        if (j < text.len and text[j] == '+') {
            j += 1;
            const n = takeDigits(text, &j, 3);
            if (n == 0) continue;
            if (j < text.len and isSeparator(text[j]) and text[j] != '.') j += 1;
        }
        // Area: (1-4 digits) or 1-4 digits.
        if (j < text.len and text[j] == '(') {
            j += 1;
            const n = takeDigits(text, &j, 4);
            if (n == 0 or j >= text.len or text[j] != ')') continue;
            j += 1;
        } else {
            const n = takeDigits(text, &j, 4);
            if (n == 0) continue;
        }
        // 1-4 groups of [ .-]?\d{2,4}.
        var groups: usize = 0;
        while (groups < 4) : (groups += 1) {
            var k = j;
            if (k < text.len and isSeparator(text[k])) k += 1;
            const n = takeDigits(text, &k, 4);
            if (n < 2) break;
            j = k;
        }
        if (groups == 0) continue;
        if (j < text.len and isDigit(text[j])) continue; // trailing digit guard
        if (isValidPhone(text[start..j])) {
            push(out, .phone, start, j);
            i = j - 1;
        }
    }
}

// --- Public API --------------------------------------------------------------

fn typeEnabled(active: ?*const std.EnumSet(PiiType), t: PiiType) bool {
    if (active) |set| return set.contains(t);
    return true;
}

/// Detect personal data in `text`. Pass `types` to scan for a subset (the
/// per-type toggles); pass null to scan for everything. Returns matches in
/// document order, non-overlapping, with exact `start`/`end` indices.
/// Caller owns the returned list.
pub fn detectPii(
    allocator: std.mem.Allocator,
    text: []const u8,
    types: ?[]const PiiType,
) std.mem.Allocator.Error!std.ArrayList(PiiMatch) {
    var active_set: ?std.EnumSet(PiiType) = null;
    if (types) |list| {
        var s = std.EnumSet(PiiType).initEmpty();
        for (list) |t| s.insert(t);
        active_set = s;
    }

    var candidates = std.ArrayList(PiiMatch).init(allocator);
    errdefer candidates.deinit();

    if (typeEnabled(if (active_set) |*s| s else null, .email)) scanEmail(text, &candidates);
    if (typeEnabled(if (active_set) |*s| s else null, .credit_card)) scanCreditCard(text, &candidates);
    if (typeEnabled(if (active_set) |*s| s else null, .ssn)) scanSsn(text, &candidates);
    if (typeEnabled(if (active_set) |*s| s else null, .ipv6)) scanIpv6(text, &candidates);
    if (typeEnabled(if (active_set) |*s| s else null, .ipv4)) scanIpv4(text, &candidates);
    if (typeEnabled(if (active_set) |*s| s else null, .date)) scanDate(text, &candidates);
    if (typeEnabled(if (active_set) |*s| s else null, .phone)) scanPhone(text, &candidates);

    // Highest-priority (lowest number) candidates claim their span first.
    std.mem.sort(PiiMatch, candidates.items, {}, struct {
        fn lt(_: void, a: PiiMatch, b: PiiMatch) bool {
            const pa = priority(a.type);
            const pb = priority(b.type);
            if (pa != pb) return pa < pb;
            return a.start < b.start;
        }
    }.lt);

    var kept = std.ArrayList(PiiMatch).init(allocator);
    errdefer kept.deinit();
    for (candidates.items) |c| {
        var overlaps = false;
        for (kept.items) |k| {
            if (c.start < k.end and k.start < c.end) {
                overlaps = true;
                break;
            }
        }
        if (!overlaps) try kept.append(c);
    }
    candidates.deinit();

    std.mem.sort(PiiMatch, kept.items, {}, struct {
        fn lt(_: void, a: PiiMatch, b: PiiMatch) bool {
            return a.start < b.start;
        }
    }.lt);
    return kept;
}

pub const RedactOptions = struct {
    mask: []const u8 = "[REDACTED]",
    types: ?[]const PiiType = null,
};

/// Redact personal data from `text`, replacing every detected span with
/// `options.mask` (default `[REDACTED]`). Accepts the same `types` subset as
/// detectPii. Caller owns the returned string.
pub fn redactPii(
    allocator: std.mem.Allocator,
    text: []const u8,
    options: RedactOptions,
) std.mem.Allocator.Error![]u8 {
    const matches = try detectPii(allocator, text, options.types);
    defer matches.deinit();

    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    var pos: usize = 0;
    for (matches.items) |m| {
        try out.appendSlice(text[pos..m.start]);
        try out.appendSlice(options.mask);
        pos = m.end;
    }
    try out.appendSlice(text[pos..]);
    return out.toOwnedSlice();
}

Also available in 8 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →