PII Redactor — C source
Paste text and automatically detect and mask personal data — emails, phone numbers, IP addresses, SSNs, credit card numbers, and dates.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/*
* pii-redactor — detect and redact personal data (PII) in free text.
*
* Language: C (C11, standard library only)
* Source: CosmoDev polyglot showcase port of the PII Redactor tool, ported
* from src/lib/pii-redactor.ts (the canonical TypeScript
* implementation).
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Seven personal-data types are recognised: email, phone, IPv4, IPv6, SSN,
* credit card and ISO date. Detection is two-stage, exactly as in the TS
* reference: a permissive *candidate* pattern proposes a span, then a
* structural *validator* accepts or rejects it (octet ranges, Luhn checksum,
* month/day bounds, E.164 digit budget). Overlapping candidates are resolved by
* type priority — unambiguous types claim their span before the fuzzy phone
* pattern. Nothing here can fail: bad input simply yields no matches.
*
* ISO C has no regular expressions, so each of the seven candidate patterns is
* a hand-written matcher below. They are not approximations: the three patterns
* that need it (ipv6, ipv4, phone) implement the same leftmost-greedy
* backtracking the JS engine performs, so `match_*` accepts the same language
* and returns the same span as its regex counterpart. Each matcher answers one
* question — "how many bytes does this pattern match starting at i?" (0 = no
* match) — and the scanner advances past a candidate whether or not the
* validator kept it, mirroring how a /g regex moves `lastIndex`.
*
* Build: cc -std=c11 pii-redactor.c
*/
#include <ctype.h>
#include <stdbool.h>
#include <stddef.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#define UC(c) ((unsigned char)(c))
/** Longest span copied into a match record, NUL included. */
#define PII_MAX_MATCH 128
/** Candidate ceiling for one scan (the TS version is unbounded). */
#define PII_MAX_CANDIDATES 512
/* ------------------------------------------------------------------ types --- */
/** The seven PII types the detector knows, in display order. */
typedef enum {
PII_EMAIL = 0,
PII_PHONE,
PII_IPV4,
PII_IPV6,
PII_SSN,
PII_CREDIT_CARD,
PII_DATE,
PII_TYPE_COUNT
} PiiType;
/*
* One detected item: where it is and what it was.
*
* Note on units: these offsets are byte offsets into the UTF-8 input, whereas
* the TS reference reports UTF-16 code-unit offsets. The two agree for ASCII
* text and diverge after the first non-ASCII character; the matched spans
* themselves are identical either way.
*/
typedef struct {
PiiType type;
size_t start; /* index of the first byte in the input */
size_t end; /* index one past the last byte */
char original[PII_MAX_MATCH]; /* the matched substring, verbatim */
} PiiMatch;
/** Bit for `type` in a type-subset mask; PII_ALL scans for everything. */
#define PII_BIT(t) (1u << (unsigned)(t))
#define PII_ALL ((1u << (unsigned)PII_TYPE_COUNT) - 1u)
static const char *const PII_TYPE_NAMES[PII_TYPE_COUNT] = {
"email", "phone", "ipv4", "ipv6", "ssn", "credit-card", "date",
};
const char *pii_type_name(PiiType t)
{
return (t >= 0 && t < PII_TYPE_COUNT) ? PII_TYPE_NAMES[t] : "?";
}
/* ------------------------------------------------------- character classes --- */
static bool c_digit(char c) { return isdigit(UC(c)) != 0; }
static bool c_alpha(char c) { return isalpha(UC(c)) != 0; }
static bool c_hex(char c) { return isxdigit(UC(c)) != 0; }
/* \w — [A-Za-z0-9_] */
static bool c_word(char c) { return isalnum(UC(c)) || c == '_'; }
/* The email local part: [A-Za-z0-9._%+-] */
static bool c_local(char c)
{
return isalnum(UC(c)) || c == '.' || c == '_' || c == '%' || c == '+' || c == '-';
}
/* The phone / card group separator: [ .-] */
static bool c_sep(char c) { return c == ' ' || c == '.' || c == '-'; }
/* --------------------------------------------------------------- validators --- */
/*
* Luhn checksum. `digits` must be a non-empty run of 0-9 (any separator makes
* it invalid — strip them first). Returns false otherwise.
*/
bool is_valid_luhn(const char *digits)
{
if (digits == NULL || *digits == '\0') return false;
size_t n = strlen(digits);
for (size_t i = 0; i < n; i++) {
if (!c_digit(digits[i])) return false;
}
int sum = 0;
bool twice = false;
for (size_t i = n; i-- > 0;) {
int d = digits[i] - '0';
if (twice) {
d *= 2;
if (d > 9) d -= 9;
}
sum += d;
twice = !twice;
}
return sum % 10 == 0;
}
/** Copy the decimal digits of s[0..n) into `out` (NUL-terminated). */
static size_t digits_only(const char *s, size_t n, char *out, size_t cap)
{
size_t k = 0;
for (size_t i = 0; i < n && k + 1 < cap; i++) {
if (c_digit(s[i])) out[k++] = s[i];
}
out[k] = '\0';
return k;
}
/** Octets 0-255 each; the matcher already bounds the shape to a dotted quad. */
static bool is_valid_ipv4(const char *s, size_t n)
{
unsigned octet = 0;
size_t seen = 0;
for (size_t i = 0; i <= n; i++) {
if (i == n || s[i] == '.') {
if (seen == 0 || octet > 255) return false;
octet = 0;
seen = 0;
} else {
octet = octet * 10 + (unsigned)(s[i] - '0');
seen++;
}
}
return true;
}
/** One side of a compressed IPv6 address: every ':'-group is 1-4 hex digits. */
static bool ipv6_side_ok(const char *s, size_t n, size_t *groups)
{
*groups = 0;
if (n == 0) return true; /* an empty side contributes no groups */
size_t start = 0, count = 0;
for (size_t i = 0; i <= n; i++) {
if (i == n || s[i] == ':') {
size_t len = i - start;
if (len < 1 || len > 4) return false;
for (size_t j = start; j < i; j++) {
if (!c_hex(s[j])) return false;
}
count++;
start = i + 1;
}
}
*groups = count;
return true;
}
/** Full 8-group form, or a compressed `::` form expanding to at most 8. */
static bool is_valid_ipv6(const char *s, size_t n)
{
/* Lone ":" / "::" (URL scheme separators like https://) carry no hex. */
bool has_hex = false;
for (size_t i = 0; i < n; i++) {
if (c_hex(s[i])) { has_hex = true; break; }
}
if (!has_hex) return false;
/* Does any ':'-delimited group come out empty? That means compression. */
bool empty_group = false;
size_t start = 0, groups = 0;
for (size_t i = 0; i <= n; i++) {
if (i == n || s[i] == ':') {
if (i == start) empty_group = true;
groups++;
start = i + 1;
}
}
if (!empty_group) {
if (groups != 8) return false;
size_t ignored = 0;
return ipv6_side_ok(s, n, &ignored);
}
/* Compressed: exactly one "::", and its two sides hold at most 7 groups.
* A single "::" is what a JS split('::') into two parts means; zero or
* two-plus occurrences both make the address invalid. */
size_t seps = 0, at = 0;
for (size_t i = 0; i + 1 < n;) {
if (s[i] == ':' && s[i + 1] == ':') {
if (seps == 0) at = i;
seps++;
i += 2;
} else {
i++;
}
}
if (seps != 1) return false;
size_t left = 0, right = 0;
if (!ipv6_side_ok(s, at, &left)) return false;
if (!ipv6_side_ok(s + at + 2, n - at - 2, &right)) return false;
return left + right <= 7;
}
/** ISO calendar plausibility: month 01-12, day 01-31. */
static bool is_valid_date(const char *s, size_t n)
{
if (n != 10) return false;
int month = (s[5] - '0') * 10 + (s[6] - '0');
int day = (s[8] - '0') * 10 + (s[9] - '0');
return month >= 1 && month <= 12 && day >= 1 && day <= 31;
}
/** E.164 digit budget (7-15) plus structural guards for the fuzzy shape. */
static bool is_valid_phone(const char *s, size_t n)
{
char digits[64];
size_t count = digits_only(s, n, digits, sizeof digits);
if (count < 7 || count > 15) return false;
/* A dotted quad is IP-shaped: a valid one was already claimed by the ipv4
* detector; an invalid one (999.x) is likelier a version string. */
size_t dots = 0, runs = 0, run = 0;
bool shaped = true;
for (size_t i = 0; i < n; i++) {
if (c_digit(s[i])) {
if (++run > 3) shaped = false;
} else if (s[i] == '.') {
if (run == 0) shaped = false;
dots++;
runs++;
run = 0;
} else {
shaped = false;
}
}
if (run > 0) runs++;
if (shaped && dots == 3 && runs == 4) return false;
/* YYYY-MM-DD shaped (even an impossible date) is never a phone number. */
if (n == 10 && s[4] == '-' && s[7] == '-') {
bool iso = true;
for (size_t i = 0; i < 10 && iso; i++) {
if (i == 4 || i == 7) continue;
if (!c_digit(s[i])) iso = false;
}
if (iso) return false;
}
return true;
}
/** 13-19 digits with optional space/dash grouping, plus a Luhn checksum. */
static bool is_valid_card(const char *s, size_t n)
{
char digits[64];
size_t count = digits_only(s, n, digits, sizeof digits);
if (count < 13 || count > 19) return false;
return is_valid_luhn(digits);
}
/* ------------------------------------------------------------- candidates --- */
/*
* Each matcher answers "how long is the pattern's match starting at i?" and
* returns 0 for no match. Together they stand in for these seven regexes:
*
* email /[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}/
* card /\d(?:[ -]?\d){11,}/
* ssn /\b\d{3}-\d{2}-\d{4}\b/
* ipv6 /(?<![:\w])[A-Fa-f0-9]{0,4}(?::[A-Fa-f0-9]{0,4}){1,7}(?![:\w])/
* ipv4 /(?<![\w.])(?:\d{1,3}\.){3}\d{1,3}(?!\.?\d)(?!\w)/
* date /(?<!\d)\d{4}-\d{2}-\d{2}(?!\d)/
* phone /(?<![\d(])(?:\+\d{1,3}[ .-]?)?(?:\(\d{1,4}\)|\d{1,4})(?:[ .-]?\d{2,4}){1,4}(?!\d)/
*/
static size_t match_email(const char *s, size_t n, size_t i)
{
size_t p = i;
while (p < n && c_local(s[p])) p++;
if (p == i) return 0; /* local part needs 1+ chars */
if (p >= n || s[p] != '@') return 0;
p++;
size_t dstart = p;
while (p < n && (isalnum(UC(s[p])) || s[p] == '.' || s[p] == '-')) p++;
size_t dend = p;
if (dend == dstart) return 0;
/* [A-Za-z0-9.-]+ is greedy, so the engine tries the rightmost possible
* position for the literal '.' first, then the letters run greedily. */
for (size_t k = dend; k-- > dstart + 1;) {
if (s[k] != '.') continue;
size_t t = k + 1, letters = 0;
while (t < dend && c_alpha(s[t])) { t++; letters++; }
if (letters >= 2) return t - i;
}
return 0;
}
static size_t match_card(const char *s, size_t n, size_t i)
{
if (!c_digit(s[i])) return 0;
size_t p = i + 1, reps = 0;
for (;;) {
size_t q = p;
if (q < n && (s[q] == ' ' || s[q] == '-')) q++;
if (q < n && c_digit(s[q])) {
p = q + 1;
reps++;
} else {
break;
}
}
return reps >= 11 ? p - i : 0;
}
static size_t match_ssn(const char *s, size_t n, size_t i)
{
if (i > 0 && c_word(s[i - 1])) return 0; /* leading \b */
if (i + 11 > n) return 0;
for (size_t k = 0; k < 11; k++) {
char c = s[i + k];
bool want_dash = (k == 3 || k == 6);
if (want_dash ? c != '-' : !c_digit(c)) return 0;
}
if (i + 11 < n && c_word(s[i + 11])) return 0; /* trailing \b */
return 11;
}
static size_t match_date(const char *s, size_t n, size_t i)
{
if (i > 0 && c_digit(s[i - 1])) return 0; /* (?<!\d) */
if (i + 10 > n) return 0;
for (size_t k = 0; k < 10; k++) {
char c = s[i + k];
bool want_dash = (k == 4 || k == 7);
if (want_dash ? c != '-' : !c_digit(c)) return 0;
}
if (i + 10 < n && c_digit(s[i + 10])) return 0; /* (?!\d) */
return 10;
}
/** `(?![:\w])` — the ipv6 pattern's trailing lookahead. */
static bool ipv6_tail_ok(const char *s, size_t n, size_t pos)
{
if (pos >= n) return true;
return s[pos] != ':' && !c_word(s[pos]);
}
/*
* `(?::[A-Fa-f0-9]{0,4}){1,7}` — greedy, with backtracking. The repetition
* prefers more groups and each group prefers more hex digits; only when every
* extension fails does it settle for the count it already has.
*/
static bool ipv6_rest(const char *s, size_t n, size_t pos, int groups, size_t *end)
{
if (groups < 7 && pos < n && s[pos] == ':') {
size_t run = 0;
while (pos + 1 + run < n && c_hex(s[pos + 1 + run]) && run < 4) run++;
for (size_t len = run + 1; len-- > 0;) {
if (ipv6_rest(s, n, pos + 1 + len, groups + 1, end)) return true;
}
}
if (groups >= 1 && ipv6_tail_ok(s, n, pos)) {
*end = pos;
return true;
}
return false;
}
static size_t match_ipv6(const char *s, size_t n, size_t i)
{
if (i > 0 && (s[i - 1] == ':' || c_word(s[i - 1]))) return 0; /* (?<![:\w]) */
size_t run = 0;
while (i + run < n && c_hex(s[i + run]) && run < 4) run++;
for (size_t len = run + 1; len-- > 0;) {
size_t end = 0;
if (ipv6_rest(s, n, i + len, 0, &end)) return end - i;
}
return 0;
}
/*
* `(?:\d{1,3}\.){3}\d{1,3}` plus the two trailing lookaheads. Group 3 is the
* final octet; groups 0-2 each end in a literal '.'.
*/
static bool ipv4_rest(const char *s, size_t n, size_t pos, int group, size_t *end)
{
size_t run = 0;
while (pos + run < n && c_digit(s[pos + run]) && run < 3) run++;
for (size_t len = run; len >= 1; len--) {
size_t p = pos + len;
if (group < 3) {
if (p < n && s[p] == '.' && ipv4_rest(s, n, p + 1, group + 1, end)) return true;
continue;
}
/* (?!\.?\d)(?!\w) */
if (p < n && (c_digit(s[p]) || c_word(s[p]))) continue;
if (p + 1 < n && s[p] == '.' && c_digit(s[p + 1])) continue;
*end = p;
return true;
}
return false;
}
static size_t match_ipv4(const char *s, size_t n, size_t i)
{
if (i > 0 && (c_word(s[i - 1]) || s[i - 1] == '.')) return 0; /* (?<![\w.]) */
size_t end = 0;
if (ipv4_rest(s, n, i, 0, &end)) return end - i;
return 0;
}
/* `(?:[ .-]?\d{2,4}){1,4}(?!\d)` — greedy repetition, greedy separator. */
static bool phone_reps(const char *s, size_t n, size_t pos, int reps, size_t *end)
{
if (reps < 4) {
for (int with_sep = 1; with_sep >= 0; with_sep--) {
size_t p = pos;
if (with_sep) {
if (!(p < n && c_sep(s[p]))) continue;
p++;
}
size_t run = 0;
while (p + run < n && c_digit(s[p + run]) && run < 4) run++;
for (size_t len = run; len >= 2; len--) {
if (phone_reps(s, n, p + len, reps + 1, end)) return true;
}
}
}
if (reps >= 1 && (pos >= n || !c_digit(s[pos]))) { /* (?!\d) */
*end = pos;
return true;
}
return false;
}
/* `(?:\(\d{1,4}\)|\d{1,4})` — alternation, left branch first. */
static bool phone_body(const char *s, size_t n, size_t pos, size_t *end)
{
if (pos < n && s[pos] == '(') {
size_t p = pos + 1, run = 0;
while (p + run < n && c_digit(s[p + run]) && run < 4) run++;
for (size_t len = run; len >= 1; len--) {
if (p + len < n && s[p + len] == ')' &&
phone_reps(s, n, p + len + 1, 0, end)) {
return true;
}
}
}
size_t run = 0;
while (pos + run < n && c_digit(s[pos + run]) && run < 4) run++;
for (size_t len = run; len >= 1; len--) {
if (phone_reps(s, n, pos + len, 0, end)) return true;
}
return false;
}
static size_t match_phone(const char *s, size_t n, size_t i)
{
if (i > 0 && (c_digit(s[i - 1]) || s[i - 1] == '(')) return 0; /* (?<![\d(]) */
size_t end = 0;
/* `(?:\+\d{1,3}[ .-]?)?` is greedy: try it present before absent. */
if (i < n && s[i] == '+') {
size_t p = i + 1, run = 0;
while (p + run < n && c_digit(s[p + run]) && run < 3) run++;
for (size_t len = run; len >= 1; len--) {
for (int with_sep = 1; with_sep >= 0; with_sep--) {
size_t q = p + len;
if (with_sep) {
if (!(q < n && c_sep(s[q]))) continue;
q++;
}
if (phone_body(s, n, q, &end)) return end - i;
}
}
}
if (phone_body(s, n, i, &end)) return end - i;
return 0;
}
/* ---------------------------------------------------------------- detector --- */
typedef size_t (*MatchFn)(const char *s, size_t n, size_t i);
typedef bool (*ValidateFn)(const char *s, size_t n);
typedef struct {
PiiType type;
MatchFn match;
ValidateFn validate; /* NULL when the pattern alone is decisive */
} Detector;
/* Scan order mirrors the TS DETECTORS array. */
static const Detector DETECTORS[] = {
{ PII_EMAIL, match_email, NULL },
{ PII_CREDIT_CARD, match_card, is_valid_card },
{ PII_SSN, match_ssn, NULL },
{ PII_IPV6, match_ipv6, is_valid_ipv6 },
{ PII_IPV4, match_ipv4, is_valid_ipv4 },
{ PII_DATE, match_date, is_valid_date },
{ PII_PHONE, match_phone, is_valid_phone },
};
#define DETECTOR_COUNT (sizeof DETECTORS / sizeof DETECTORS[0])
/*
* Overlap resolution: when two candidates cover the same span the more
* specific type wins. Phone is deliberately last — a date, SSN, IP or card
* number can all masquerade as one.
*/
static int priority_of(PiiType t)
{
switch (t) {
case PII_EMAIL: return 0;
case PII_CREDIT_CARD: return 1;
case PII_SSN: return 2;
case PII_IPV6: return 3;
case PII_IPV4: return 4;
case PII_DATE: return 5;
case PII_PHONE: return 6;
default: return 7;
}
}
static int by_priority(const void *a, const void *b)
{
const PiiMatch *x = a, *y = b;
int d = priority_of(x->type) - priority_of(y->type);
if (d != 0) return d;
return (x->start > y->start) - (x->start < y->start);
}
static int by_start(const void *a, const void *b)
{
const PiiMatch *x = a, *y = b;
return (x->start > y->start) - (x->start < y->start);
}
static void record(PiiMatch *m, PiiType type, const char *s, size_t start, size_t end)
{
m->type = type;
m->start = start;
m->end = end;
size_t len = end - start;
if (len >= PII_MAX_MATCH) len = PII_MAX_MATCH - 1;
memcpy(m->original, s + start, len);
m->original[len] = '\0';
}
/*
* Detect personal data in `text`. `type_mask` selects a subset (PII_ALL scans
* for everything). Fills `out` with at most `max` matches in document order,
* non-overlapping, with exact start/end indices. Returns the match count.
*/
size_t detect_pii(const char *text, unsigned type_mask, PiiMatch *out, size_t max)
{
if (text == NULL || out == NULL || max == 0) return 0;
size_t n = strlen(text);
static PiiMatch candidates[PII_MAX_CANDIDATES];
size_t count = 0;
for (size_t d = 0; d < DETECTOR_COUNT; d++) {
const Detector *det = &DETECTORS[d];
if (!(type_mask & PII_BIT(det->type))) continue;
size_t i = 0;
while (i < n) {
size_t len = det->match(text, n, i);
if (len == 0) {
i++;
continue;
}
if (det->validate == NULL || det->validate(text + i, len)) {
if (count < PII_MAX_CANDIDATES) {
record(&candidates[count++], det->type, text, i, i + len);
}
}
i += len; /* a /g regex advances past the candidate either way */
}
}
/* Highest-priority (lowest number) candidates claim their span first. */
qsort(candidates, count, sizeof candidates[0], by_priority);
size_t kept = 0;
for (size_t c = 0; c < count; c++) {
bool clash = false;
for (size_t k = 0; k < kept && !clash; k++) {
if (candidates[c].start < out[k].end && out[k].start < candidates[c].end) {
clash = true;
}
}
if (clash) continue;
if (kept == max) break;
out[kept++] = candidates[c];
}
qsort(out, kept, sizeof out[0], by_start);
return kept;
}
/*
* Redact personal data from `text`, replacing every detected span with `mask`
* (pass NULL for the default "[REDACTED]"). Writes at most `cap` bytes to
* `out`, NUL-terminated, and returns the length that a full result would need
* — so a return value >= cap means the output was truncated.
*/
size_t redact_pii(const char *text, const char *mask, unsigned type_mask,
char *out, size_t cap)
{
if (text == NULL) text = "";
if (mask == NULL) mask = "[REDACTED]";
static PiiMatch matches[PII_MAX_CANDIDATES];
size_t count = detect_pii(text, type_mask, matches, PII_MAX_CANDIDATES);
size_t mask_len = strlen(mask);
size_t need = 0, cursor = 0, written = 0;
/* Append helper semantics, inlined: always count, copy while there's room. */
#define EMIT(ptr, len) \
do { \
const char *src_ = (ptr); \
size_t len_ = (len); \
need += len_; \
if (out != NULL && cap > 0) { \
size_t room_ = (written + 1 < cap) ? cap - 1 - written : 0; \
size_t take_ = len_ < room_ ? len_ : room_; \
memcpy(out + written, src_, take_); \
written += take_; \
} \
} while (0)
for (size_t i = 0; i < count; i++) {
EMIT(text + cursor, matches[i].start - cursor);
EMIT(mask, mask_len);
cursor = matches[i].end;
}
EMIT(text + cursor, strlen(text) - cursor);
#undef EMIT
if (out != NULL && cap > 0) out[written] = '\0';
return need;
}
/* -------------------------------------------------------------------- demo --- */
int main(void)
{
const char *sample =
"Contact ada@example.com or +1 415 555 0132 (backup: (020) 7946 0958).\n"
"Card 4539 1488 0343 6467 expires 2027-11-30, SSN 123-45-6789.\n"
"Servers 192.168.1.10 and 2001:db8::8a2e:370:7334 — build v1.2.3.4 at 10:30:45.\n";
PiiMatch found[64];
size_t count = detect_pii(sample, PII_ALL, found, 64);
printf("%zu match(es):\n", count);
for (size_t i = 0; i < count; i++) {
printf(" %-12s [%3zu,%3zu) %s\n", pii_type_name(found[i].type),
found[i].start, found[i].end, found[i].original);
}
char redacted[1024];
redact_pii(sample, NULL, PII_ALL, redacted, sizeof redacted);
printf("\nredacted:\n%s", redacted);
/* A subset scan: emails only, with a custom mask. */
char emails_only[1024];
redact_pii(sample, "<email>", PII_BIT(PII_EMAIL), emails_only, sizeof emails_only);
printf("\nemail-only mask:\n%s", emails_only);
printf("\nluhn 4539148803436467: %s\n",
is_valid_luhn("4539148803436467") ? "valid" : "invalid");
return EXIT_SUCCESS;
}
Also available in 8 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →