PII Redactor — Zig source
Paste text and automatically detect and mask personal data — emails, phone numbers, IP addresses, SSNs, credit card numbers, and dates.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! pii-redactor — PII detection & redaction (email, phone, IPs, SSN, cards, dates).
//!
//! Language: Zig 0.14 (standard library only)
//! Ported from: src/lib/pii-redactor.ts (the canonical TypeScript implementation).
//! display source — part of CosmoDev's polyglot tool pages.
//!
//! Deterministic detection for seven personal-data types. The TS reference
//! drives each detector with a regex; Zig's standard library has no regex
//! engine, so this port re-expresses the same patterns as hand-rolled
//! cursor scanners with identical candidate spans. Every candidate still
//! passes the same structural validator (octet ranges, Luhn checksum,
//! month/day bounds, E.164 digit count) to keep false positives low.
//! Overlapping candidates resolve by type priority - unambiguous types
//! (email, Luhn-passing card numbers, SSNs, IPs, dates) claim their span
//! before the fuzzy phone pattern. Never fails.
//!
//! The fuzzy phone pattern is a greedy forward match (no backtracking),
//! which covers the practical shapes `+C (A) G1…G4`.
const std = @import("std");
/// The seven PII types the detector knows.
pub const PiiType = enum {
email,
phone,
ipv4,
ipv6,
ssn,
credit_card,
date,
};
/// One detected personal-data item: where it is (indices into the input).
pub const PiiMatch = struct {
type: PiiType,
start: usize, // index of the first byte in the input
end: usize, // index one past the last byte
/// The matched substring, verbatim.
pub fn original(self: PiiMatch, text: []const u8) []const u8 {
return text[self.start..self.end];
}
};
/// All PII types, in display order.
pub const pii_types = [_]PiiType{
.email, .phone, .ipv4, .ipv6, .ssn, .credit_card, .date,
};
/// Overlap resolution: when two candidates cover the same span, the more
/// specific type wins. Phone is deliberately last - a date, SSN, IP, or card
/// number can all masquerade as one.
fn priority(t: PiiType) u8 {
return switch (t) {
.email => 0,
.credit_card => 1,
.ssn => 2,
.ipv6 => 3,
.ipv4 => 4,
.date => 5,
.phone => 6,
};
}
// --- Character classes ----------------------------------------------------
fn isDigit(c: u8) bool {
return c >= '0' and c <= '9';
}
fn isWordChar(c: u8) bool {
return isDigit(c) or (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or c == '_';
}
fn isHexDigit(c: u8) bool {
return isDigit(c) or (c >= 'a' and c <= 'f') or (c >= 'A' and c <= 'F');
}
fn isLocalChar(c: u8) bool { // [A-Za-z0-9._%+-]
return isWordChar(c) or c == '.' or c == '%' or c == '+' or c == '-';
}
fn isDomainChar(c: u8) bool { // [A-Za-z0-9.-]
return isWordChar(c) or c == '.' or c == '-';
}
fn isSeparator(c: u8) bool { // [ .-]
return c == ' ' or c == '.' or c == '-';
}
// --- Validators (identical logic to the TS reference) ----------------------
/// Luhn checksum. `digits` must be a non-empty string of 0-9; any other
/// character makes it invalid.
pub fn isValidLuhn(digits: []const u8) bool {
if (digits.len == 0) return false;
var sum: u32 = 0;
var double = false;
var i: usize = digits.len;
while (i > 0) {
i -= 1;
const c = digits[i];
if (!isDigit(c)) return false;
var d: u32 = c - '0';
if (double) {
d *= 2;
if (d > 9) d -= 9;
}
sum += d;
double = !double;
}
return sum % 10 == 0;
}
/// Octets 0-255 each; the scanner already bounds the shape to a dotted quad.
fn isValidIpv4(candidate: []const u8) bool {
var it = std.mem.splitScalar(u8, candidate, '.');
while (it.next()) |octet| {
const n = std.fmt.parseInt(u32, octet, 10) catch return false;
if (n > 255) return false;
}
return true;
}
fn isHexGroup1to4(g: []const u8) bool {
if (g.len == 0 or g.len > 4) return false;
for (g) |c| if (!isHexDigit(c)) return false;
return true;
}
/// Full 8-group form, or a compressed `::` form expanding to exactly 8.
fn isValidIpv6(candidate: []const u8) bool {
// Lone ":" / "::" (URL scheme separators like https://) carry no hex digits.
var has_hex = false;
for (candidate) |c| {
if (isHexDigit(c)) {
has_hex = true;
break;
}
}
if (!has_hex) return false;
var groups = std.mem.splitScalar(u8, candidate, ':');
var has_empty_group = false;
while (groups.next()) |g| {
if (g.len == 0) has_empty_group = true;
}
if (has_empty_group) {
// Compressed: at most one "::", its sides together hold < 8 groups.
const dbl = std.mem.indexOf(u8, candidate, "::") orelse return false;
if (std.mem.indexOfPos(u8, candidate, dbl + 2, "::") != null) return false;
const left = candidate[0..dbl];
const right = candidate[dbl + 2 ..];
var left_count: usize = 0;
if (left.len > 0) {
var lg = std.mem.splitScalar(u8, left, ':');
while (lg.next()) |g| {
if (!isHexGroup1to4(g)) return false;
left_count += 1;
}
}
var right_count: usize = 0;
if (right.len > 0) {
var rg = std.mem.splitScalar(u8, right, ':');
while (rg.next()) |g| {
if (!isHexGroup1to4(g)) return false;
right_count += 1;
}
}
if (left_count + right_count > 7) return false;
return true;
}
var count: usize = 0;
var g2 = std.mem.splitScalar(u8, candidate, ':');
while (g2.next()) |g| {
if (!isHexGroup1to4(g)) return false;
count += 1;
}
return count == 8;
}
/// ISO calendar plausibility: month 01-12, day 01-31.
fn isValidDate(candidate: []const u8) bool {
if (candidate.len != 10 or candidate[4] != '-' or candidate[7] != '-') return false;
for (candidate, 0..) |c, i| {
if (i == 4 or i == 7) continue;
if (!isDigit(c)) return false;
}
const month = (candidate[5] - '0') * 10 + (candidate[6] - '0');
const day = (candidate[8] - '0') * 10 + (candidate[9] - '0');
return month >= 1 and month <= 12 and day >= 1 and day <= 31;
}
fn countDigits(candidate: []const u8) usize {
var n: usize = 0;
for (candidate) |c| {
if (isDigit(c)) n += 1;
}
return n;
}
fn looksLikeDottedQuad(candidate: []const u8) bool {
var dots: usize = 0;
var group_digits: usize = 0;
for (candidate) |c| {
if (c == '.') {
dots += 1;
if (group_digits == 0 or group_digits > 3) return false;
group_digits = 0;
} else if (isDigit(c)) {
group_digits += 1;
} else {
return false;
}
}
return dots == 3 and group_digits >= 1 and group_digits <= 3;
}
/// E.164 digit budget (7-15) and structural guards for the fuzzy phone shape.
fn isValidPhone(candidate: []const u8) bool {
const digits = countDigits(candidate);
if (digits < 7 or digits > 15) return false;
// A dotted quad is IP-shaped: if it were a valid IP it was already claimed
// by the ipv4 detector; an invalid one (999.x) is likelier a version string.
if (looksLikeDottedQuad(candidate)) return false;
// YYYY-MM-DD shaped (even an impossible date) is never a phone number.
if (candidate.len == 13 and isValidDateShape(candidate)) return false;
return true;
}
fn isValidDateShape(candidate: []const u8) bool {
// "^\\d{4}-\\d{2}-\\d{2}$" shape check (plausibility is checked elsewhere).
if (candidate.len != 10) return false;
for (candidate, 0..) |c, i| {
if (i == 4 or i == 7) {
if (c != '-') return false;
} else if (!isDigit(c)) return false;
}
return true;
}
/// 13-19 digits with optional space/dash grouping, plus a Luhn checksum.
fn isValidCard(candidate: []const u8) bool {
var buf: [24]u8 = undefined;
var n: usize = 0;
for (candidate) |c| {
if (isDigit(c)) {
buf[n] = c;
n += 1;
}
}
return n >= 13 and n <= 19 and isValidLuhn(buf[0..n]);
}
// --- Scanners (cursor re-expressions of the TS regexes) ---------------------
const CandidateList = std.ArrayList(PiiMatch);
fn push(list: *CandidateList, t: PiiType, start: usize, end: usize) void {
list.append(.{ .type = t, .start = start, .end = end }) catch {};
}
/// [A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}
fn scanEmail(text: []const u8, out: *CandidateList) void {
var i: usize = 0;
while (i < text.len) : (i += 1) {
if (text[i] != '@') continue;
// Extend left over the local part (must be non-empty and start-bounded).
var s = i;
while (s > 0 and isLocalChar(text[s - 1])) s -= 1;
if (s == i) continue; // empty local part
// Extend right over the domain, then backtrack to the last "."
// followed by 2+ letters running to the end of the domain run
// (the greedy `[A-Za-z0-9.-]+\.[A-Za-z]{2,}` tail).
var e = i + 1;
while (e < text.len and isDomainChar(text[e])) e += 1;
var last_dot: ?usize = null;
var d = e;
while (d > i + 1) {
d -= 1;
if (text[d] == '.') {
var m = d + 1;
while (m < e and std.ascii.isAlphabetic(text[m])) m += 1;
if (m == e and m - d - 1 >= 2) {
last_dot = d;
break;
}
}
}
if (last_dot == null) continue;
push(out, .email, s, e);
i = e - 1; // resume after the whole match
}
}
/// \d(?:[ -]?\d){11,} — a maximal run of 12+ digits with single space/dash
/// separators; isValidCard then enforces 13-19 digits + Luhn on the run.
fn scanCreditCard(text: []const u8, out: *CandidateList) void {
var i: usize = 0;
while (i < text.len) {
if (!isDigit(text[i])) {
i += 1;
continue;
}
var e = i + 1;
var digits: usize = 1;
while (e < text.len) {
if (isDigit(text[e])) {
digits += 1;
e += 1;
} else if ((text[e] == ' ' or text[e] == '-') and e + 1 < text.len and
isDigit(text[e + 1]))
{
digits += 1;
e += 2;
} else break;
}
if (digits >= 12) {
if (isValidCard(text[i..e])) push(out, .credit_card, i, e);
i = e;
} else {
i += 1;
}
}
}
/// \b\d{3}-\d{2}-\d{4}\b
fn scanSsn(text: []const u8, out: *CandidateList) void {
var i: usize = 0;
while (i + 11 <= text.len) : (i += 1) {
const c = text[i .. i + 11];
if (!(isDigit(c[0]) and isDigit(c[1]) and isDigit(c[2]) and c[3] == '-' and
isDigit(c[4]) and isDigit(c[5]) and c[6] == '-' and
isDigit(c[7]) and isDigit(c[8]) and isDigit(c[9]) and isDigit(c[10]))) continue;
const before_ok = i == 0 or !isWordChar(text[i - 1]);
const after = i + 11;
const after_ok = after >= text.len or !isWordChar(text[after]);
if (before_ok and after_ok) {
push(out, .ssn, i, after);
i = after - 1;
}
}
}
/// Hex groups joined by colons; isValidIpv6 rejects prose like "10:30:45".
/// Scans maximal [0-9a-fA-F:] runs containing at least one colon.
fn scanIpv6(text: []const u8, out: *CandidateList) void {
var i: usize = 0;
while (i < text.len) {
if (!(isHexDigit(text[i]) or text[i] == ':')) {
i += 1;
continue;
}
var e = i;
var colons: usize = 0;
while (e < text.len and (isHexDigit(text[e]) or text[e] == ':')) {
if (text[e] == ':') colons += 1;
e += 1;
}
if (colons >= 1 and isValidIpv6(text[i..e])) push(out, .ipv6, i, e);
i = e;
}
}
/// (?<![\w.])(?:\d{1,3}\.){3}\d{1,3}(?!\.?\d)(?!\w) — dotted quad with guards
/// that keep it out of versions ("v1.2.3.4") and longer quintets ("1.2.3.4.5").
fn scanIpv4(text: []const u8, out: *CandidateList) void {
var i: usize = 0;
while (i < text.len) : (i += 1) {
if (i > 0 and (isWordChar(text[i - 1]) or text[i - 1] == '.')) continue;
// Parse four 1-3 digit octets separated by dots.
var e = i;
var ok = true;
var octet: usize = 0;
while (octet < 4) : (octet += 1) {
if (octet > 0) {
if (e < text.len and text[e] == '.') {
e += 1;
} else {
ok = false;
break;
}
}
var n: usize = 0;
while (e < text.len and isDigit(text[e]) and n < 3) : (n += 1) e += 1;
if (n == 0 or (n == 3 and e < text.len and isDigit(text[e]))) {
ok = false; // empty octet, or a 4+ digit run the regex can't consume
break;
}
}
if (!ok) continue;
// Lookahead: no ".digit" continuation, no word char.
if (e < text.len) {
if (text[e] == '.' and e + 1 < text.len and isDigit(text[e + 1])) continue;
if (isWordChar(text[e])) continue;
}
if (isValidIpv4(text[i..e])) {
push(out, .ipv4, i, e);
i = e - 1;
}
}
}
/// (?<!\d)\d{4}-\d{2}-\d{2}(?!\d)
fn scanDate(text: []const u8, out: *CandidateList) void {
var i: usize = 0;
while (i + 10 <= text.len) : (i += 1) {
if (i > 0 and isDigit(text[i - 1])) continue;
const c = text[i .. i + 10];
if (!isValidDateShape(c)) continue;
if (i + 10 < text.len and isDigit(text[i + 10])) continue;
if (isValidDate(c)) {
push(out, .date, i, i + 10);
i += 9;
}
}
}
fn takeDigits(text: []const u8, i: *usize, max: usize) usize {
var n: usize = 0;
while (i.* < text.len and isDigit(text[i.*]) and n < max) : (n += 1) i.* += 1;
return n;
}
/// (?<![\d(])(?:\+\d{1,3}[ .-]?)?(?:\(\d{1,4}\)|\d{1,4})(?:[ .-]?\d{2,4}){1,4}(?!\d)
/// Greedy forward match (no backtracking) - covers the practical phone shapes.
fn scanPhone(text: []const u8, out: *CandidateList) void {
var i: usize = 0;
while (i < text.len) : (i += 1) {
if (i > 0 and (isDigit(text[i - 1]) or text[i - 1] == '(')) continue;
const start = i;
var j = i;
// Optional +country prefix.
if (j < text.len and text[j] == '+') {
j += 1;
const n = takeDigits(text, &j, 3);
if (n == 0) continue;
if (j < text.len and isSeparator(text[j]) and text[j] != '.') j += 1;
}
// Area: (1-4 digits) or 1-4 digits.
if (j < text.len and text[j] == '(') {
j += 1;
const n = takeDigits(text, &j, 4);
if (n == 0 or j >= text.len or text[j] != ')') continue;
j += 1;
} else {
const n = takeDigits(text, &j, 4);
if (n == 0) continue;
}
// 1-4 groups of [ .-]?\d{2,4}.
var groups: usize = 0;
while (groups < 4) : (groups += 1) {
var k = j;
if (k < text.len and isSeparator(text[k])) k += 1;
const n = takeDigits(text, &k, 4);
if (n < 2) break;
j = k;
}
if (groups == 0) continue;
if (j < text.len and isDigit(text[j])) continue; // trailing digit guard
if (isValidPhone(text[start..j])) {
push(out, .phone, start, j);
i = j - 1;
}
}
}
// --- Public API --------------------------------------------------------------
fn typeEnabled(active: ?*const std.EnumSet(PiiType), t: PiiType) bool {
if (active) |set| return set.contains(t);
return true;
}
/// Detect personal data in `text`. Pass `types` to scan for a subset (the
/// per-type toggles); pass null to scan for everything. Returns matches in
/// document order, non-overlapping, with exact `start`/`end` indices.
/// Caller owns the returned list.
pub fn detectPii(
allocator: std.mem.Allocator,
text: []const u8,
types: ?[]const PiiType,
) std.mem.Allocator.Error!std.ArrayList(PiiMatch) {
var active_set: ?std.EnumSet(PiiType) = null;
if (types) |list| {
var s = std.EnumSet(PiiType).initEmpty();
for (list) |t| s.insert(t);
active_set = s;
}
var candidates = std.ArrayList(PiiMatch).init(allocator);
errdefer candidates.deinit();
if (typeEnabled(if (active_set) |*s| s else null, .email)) scanEmail(text, &candidates);
if (typeEnabled(if (active_set) |*s| s else null, .credit_card)) scanCreditCard(text, &candidates);
if (typeEnabled(if (active_set) |*s| s else null, .ssn)) scanSsn(text, &candidates);
if (typeEnabled(if (active_set) |*s| s else null, .ipv6)) scanIpv6(text, &candidates);
if (typeEnabled(if (active_set) |*s| s else null, .ipv4)) scanIpv4(text, &candidates);
if (typeEnabled(if (active_set) |*s| s else null, .date)) scanDate(text, &candidates);
if (typeEnabled(if (active_set) |*s| s else null, .phone)) scanPhone(text, &candidates);
// Highest-priority (lowest number) candidates claim their span first.
std.mem.sort(PiiMatch, candidates.items, {}, struct {
fn lt(_: void, a: PiiMatch, b: PiiMatch) bool {
const pa = priority(a.type);
const pb = priority(b.type);
if (pa != pb) return pa < pb;
return a.start < b.start;
}
}.lt);
var kept = std.ArrayList(PiiMatch).init(allocator);
errdefer kept.deinit();
for (candidates.items) |c| {
var overlaps = false;
for (kept.items) |k| {
if (c.start < k.end and k.start < c.end) {
overlaps = true;
break;
}
}
if (!overlaps) try kept.append(c);
}
candidates.deinit();
std.mem.sort(PiiMatch, kept.items, {}, struct {
fn lt(_: void, a: PiiMatch, b: PiiMatch) bool {
return a.start < b.start;
}
}.lt);
return kept;
}
pub const RedactOptions = struct {
mask: []const u8 = "[REDACTED]",
types: ?[]const PiiType = null,
};
/// Redact personal data from `text`, replacing every detected span with
/// `options.mask` (default `[REDACTED]`). Accepts the same `types` subset as
/// detectPii. Caller owns the returned string.
pub fn redactPii(
allocator: std.mem.Allocator,
text: []const u8,
options: RedactOptions,
) std.mem.Allocator.Error![]u8 {
const matches = try detectPii(allocator, text, options.types);
defer matches.deinit();
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
var pos: usize = 0;
for (matches.items) |m| {
try out.appendSlice(text[pos..m.start]);
try out.appendSlice(options.mask);
pos = m.end;
}
try out.appendSlice(text[pos..]);
return out.toOwnedSlice();
}
Also available in 8 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →