Email Header Analyzer — Zig source
Paste raw email headers to trace the message path, detect spoofing, and check SPF/DKIM/DMARC authentication results. Runs entirely in your browser.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! email-header-analyzer — RFC 5322 header parsing + spoofing analysis.
//!
//! Language: Zig 0.14 (standard library only)
//! Ported from: src/lib/email-header-analyzer.ts (the canonical TypeScript implementation).
//! display source — part of CosmoDev's polyglot tool pages.
//!
//! Pure string parsing — no dependencies, deterministic. Timestamps are Unix
//! epoch seconds (the TS reference uses `Date`; Zig has no Date object, so
//! an `i64` carries the same information losslessly).
const std = @import("std");
pub const Header = struct {
name: []const u8,
value: []const u8,
};
pub const Hop = struct {
from: []const u8,
by: []const u8,
/// Unix epoch seconds; null when unparseable.
timestamp: ?i64,
/// Seconds elapsed since the previous (earlier) hop; null when unknown.
delay: ?f64,
protocol: []const u8,
};
pub const AuthResult = struct {
/// "pass" | "fail" | "softfail" | "neutral" | "none" | "temperror" | "permerror" | "unknown"
result: []const u8,
domain: ?[]const u8,
selector: ?[]const u8,
/// DMARC policy from the `p=` token (dmarc only).
policy: ?[]const u8,
/// The full authentication clause, when present.
detail: ?[]const u8,
};
pub const AuthResults = struct {
spf: AuthResult,
dkim: AuthResult,
dmarc: AuthResult,
};
pub const AnomalySeverity = enum { critical, warning, info };
pub const Anomaly = struct {
kind: []const u8,
severity: AnomalySeverity,
message: []const u8,
};
pub const ParsedHeaders = struct {
headers: []Header,
/// Chronological order: hops[0] is the OLDEST hop (parse bottom-up).
hops: []Hop,
auth: AuthResults,
anomalies: []Anomaly,
};
pub const Error = error{OutOfMemory};
const NO_AUTH = AuthResult{
.result = "none",
.domain = null,
.selector = null,
.policy = null,
.detail = null,
};
const SUSPICIOUS_MAILERS = [_][]const u8{
"bulk", "mass mail", "massmail", "storm", "flood", "bomber",
"grabber", "harvest", "spambot", "stealth", "anonymous", "dark", "crack",
};
// --- Small ASCII helpers -------------------------------------------------------
fn trim(s: []const u8) []const u8 {
return std.mem.trim(u8, s, " \t\r\n");
}
fn eqlIgnoreCase(a: []const u8, b: []const u8) bool {
return std.ascii.eqlIgnoreCase(a, b);
}
fn lowerAlloc(allocator: std.mem.Allocator, s: []const u8) ![]u8 {
const out = try allocator.alloc(u8, s.len);
for (s, 0..) |c, i| out[i] = std.ascii.toLower(c);
return out;
}
fn startsWithIgnoreCase(haystack: []const u8, prefix: []const u8) bool {
return haystack.len >= prefix.len and std.ascii.eqlIgnoreCase(haystack[0..prefix.len], prefix);
}
/// Index of `needle` in `haystack`, ASCII case-insensitive.
fn indexOfIgnoreCase(haystack: []const u8, needle: []const u8) ?usize {
if (needle.len == 0 or haystack.len < needle.len) return null;
var i: usize = 0;
while (i + needle.len <= haystack.len) : (i += 1) {
if (std.ascii.eqlIgnoreCase(haystack[i .. i + needle.len], needle)) return i;
}
return null;
}
fn containsIgnoreCase(haystack: []const u8, needle: []const u8) bool {
return indexOfIgnoreCase(haystack, needle) != null;
}
fn isSpace(c: u8) bool {
return c == ' ' or c == '\t' or c == '\r' or c == '\n';
}
// --- Header-block parsing ------------------------------------------------------
/// Split raw header text into unfolded name/value pairs. Stops at the first
/// empty line (body separator). Caller owns the returned slice.
pub fn parseHeaderLines(allocator: std.mem.Allocator, raw: []const u8) Error![]Header {
var headers = std.ArrayList(Header).init(allocator);
errdefer headers.deinit();
var current: ?Header = null;
var lines = std.mem.splitScalar(u8, raw, '\n');
while (lines.next()) |line_raw| {
const line = std.mem.trimRight(u8, line_raw, "\r");
if (trim(line).len == 0) break; // end of headers / blank separator
if (line.len > 0 and (line[0] == ' ' or line[0] == '\t')) {
// Continuation line (RFC 5322 folding) — append to the previous value.
if (current != null) {
const prev = current.?.value;
current.?.value = try std.fmt.allocPrint(allocator, "{s} {s}", .{ prev, trim(line) });
headers.items[headers.items.len - 1] = current.?;
}
continue;
}
const colon = std.mem.indexOfScalar(u8, line, ':') orelse continue; // not a header line — skip garbage
if (colon == 0) continue;
current = .{
.name = trim(line[0..colon]),
.value = trim(line[colon + 1 ..]),
};
try headers.append(current.?);
}
return headers.toOwnedSlice();
}
/// All values for a header name, case-insensitive, in file order.
pub fn getHeaders(allocator: std.mem.Allocator, headers: []const Header, name: []const u8) Error![][]const u8 {
var out = std.ArrayList([]const u8).init(allocator);
errdefer out.deinit();
for (headers) |h| {
if (eqlIgnoreCase(h.name, name)) try out.append(h.value);
}
return out.toOwnedSlice();
}
/// First value for a header name (case-insensitive), or "" when absent.
fn firstHeader(headers: []const Header, name: []const u8) []const u8 {
for (headers) |h| {
if (eqlIgnoreCase(h.name, name)) return h.value;
}
return "";
}
fn isAddressChar(c: u8) bool {
return switch (c) {
' ', '\t', '\r', '\n', '<', '>', ',', ';', '"', '\'' => false,
else => true,
};
}
/// Extract an email address from a header value: prefers <addr>, falls back to
/// the first bare address. Returns null when neither is present.
pub fn extractAddress(value: []const u8) ?[]const u8 {
// Prefer the first <...> group.
if (std.mem.indexOfScalar(u8, value, '<')) |open| {
if (std.mem.indexOfScalarPos(u8, value, open, '>')) |close| {
const inner = value[open + 1 .. close];
if (inner.len > 0 and std.mem.indexOfScalar(u8, inner, '@') != null) return inner;
}
}
// Fall back to the first bare address: chars up to '@', then to the first
// delimiter after it.
var i: usize = 0;
while (i < value.len) : (i += 1) {
if (!isAddressChar(value[i])) continue;
var j = i;
while (j < value.len and isAddressChar(value[j]) and value[j] != '@') j += 1;
if (j < value.len and value[j] == '@') {
var k = j + 1;
while (k < value.len and isAddressChar(value[k])) k += 1;
if (k > j + 1) return value[i..k];
}
i = j;
}
return null;
}
// --- RFC 2822 date parsing -------------------------------------------------------
const MONTHS = [_][]const u8{ "jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec" };
const WEEKDAYS = [_][]const u8{ "mon", "tue", "wed", "thu", "fri", "sat", "sun" };
fn monthFromName(s: []const u8) ?u5 {
for (MONTHS, 0..) |m, i| {
if (eqlIgnoreCase(s, m)) return @intCast(i + 1);
}
return null;
}
/// Numeric timezone offset in seconds: "+0200", "-0530", or a known name.
fn zoneOffsetSeconds(token: []const u8) ?i64 {
if (token.len == 5 and (token[0] == '+' or token[0] == '-')) {
const hh = std.fmt.parseInt(u8, token[1..3], 10) catch return null;
const mm = std.fmt.parseInt(u8, token[3..5], 10) catch return null;
const total: i64 = @as(i64, hh) * 3600 + @as(i64, mm) * 60;
return if (token[0] == '-') -total else total;
}
if (eqlIgnoreCase(token, "UT") or eqlIgnoreCase(token, "GMT") or eqlIgnoreCase(token, "Z")) return 0;
if (eqlIgnoreCase(token, "EST")) return -5 * 3600;
if (eqlIgnoreCase(token, "EDT")) return -4 * 3600;
if (eqlIgnoreCase(token, "CST")) return -6 * 3600;
if (eqlIgnoreCase(token, "CDT")) return -5 * 3600;
if (eqlIgnoreCase(token, "MST")) return -7 * 3600;
if (eqlIgnoreCase(token, "MDT")) return -6 * 3600;
if (eqlIgnoreCase(token, "PST")) return -8 * 3600;
if (eqlIgnoreCase(token, "PDT")) return -7 * 3600;
return null;
}
/// Parse an RFC 2822 date, ignoring trailing "(ZONE)" comments. Returns Unix
/// epoch seconds, or null when unparseable.
pub fn parseRfc2822Date(allocator: std.mem.Allocator, value: []const u8) Error!?i64 {
// Strip (...) comments by splicing around them, then tokenize.
var cleaned = std.ArrayList(u8).init(allocator);
defer cleaned.deinit();
var depth: usize = 0;
for (value) |c| {
if (c == '(') {
depth += 1;
continue;
}
if (c == ')') {
if (depth > 0) depth -= 1;
continue;
}
if (depth == 0) try cleaned.append(c);
}
const s = trim(cleaned.items);
if (s.len == 0) return null;
var toks = std.mem.tokenizeAny(u8, s, " \t");
var tok = toks.next() orelse return null;
// Optional leading weekday "Mon,"
var day_tok: []const u8 = tok;
if (std.mem.indexOfScalar(u8, tok, ',') != null or dayTokenIsWeekday(tok)) {
tok = toks.next() orelse return null;
day_tok = tok;
}
const day = std.fmt.parseInt(u6, trim(day_tok, ","), 10) catch return null;
const mon_str = toks.next() orelse return null;
const month = monthFromName(mon_str) orelse return null;
const year_tok = toks.next() orelse return null;
var year = std.fmt.parseInt(i64, year_tok, 10) catch return null;
if (year < 50) year += 2000 else if (year < 100) year += 1900;
const time_tok = toks.next() orelse return null;
var time_parts = std.mem.splitScalar(u8, time_tok, ':');
const hh = std.fmt.parseInt(i64, time_parts.next() orelse return null, 10) catch return null;
const mm = std.fmt.parseInt(i64, time_parts.next() orelse return null, 10) catch return null;
const ss: i64 = if (time_parts.next()) |t| (std.fmt.parseInt(i64, t, 10) catch 0) else 0;
var offset: i64 = 0;
if (toks.next()) |zone| {
if (zoneOffsetSeconds(zone)) |off| offset = off;
}
// Days-civil → epoch days (Howard Hinnant's algorithm), then to seconds.
const days = daysFromCivil(year, month, day);
const utc = days * 86400 + hh * 3600 + mm * 60 + ss - offset;
return utc;
}
fn dayTokenIsWeekday(tok: []const u8) bool {
const t = trimRightComma(tok);
if (t.len != 3) return false;
for (WEEKDAYS) |wd| {
if (eqlIgnoreCase(t, wd)) return true;
}
return false;
}
fn trimRightComma(s: []const u8) []const u8 {
var end = s.len;
while (end > 0 and (s[end - 1] == ',')) end -= 1;
return s[0..end];
}
/// Days since 1970-01-01 for a civil (year, month, day) date.
fn daysFromCivil(y_in: i64, m: u5, d: u6) i64 {
var y = y_in;
if (m <= 2) y -= 1;
const era = @divFloor(if (y >= 0) y else y - 399, 400);
const yoe: i64 = y - era * 400; // [0, 399]
const mi: i64 = m;
const di: i64 = d;
const doy = @divTrunc(153 * (mi + (if (mi > 2) @as(i64, -3) else @as(i64, 9))) + 2, 5) + di - 1;
const doe = yoe * 365 + @divTrunc(yoe, 4) - @divTrunc(yoe, 100) + doy;
return era * 146097 + doe - 719468;
}
// --- Received-header parsing -----------------------------------------------------
/// The token after a keyword like `from` / `by` / `with`: non-empty run of
/// chars up to the next whitespace, ';', '(' — the TS regex `\bKEY\s+([^\s(;]+)`.
fn keywordToken(value: []const u8, keyword: []const u8) ?[]const u8 {
var i: usize = 0;
while (indexOfIgnoreCase(value[i..], keyword)) |at| {
const abs = i + at;
const word_start = abs;
const word_end = abs + keyword.len;
const boundary_before = word_start == 0 or !isAlnumOrUnderscore(value[word_start - 1]);
const boundary_after = word_end == value.len or !isAlnumOrUnderscore(value[word_end]);
if (boundary_before and boundary_after) {
var j = word_end;
while (j < value.len and isSpace(value[j])) j += 1;
if (j < value.len and value[j] != ';' and value[j] != '(') {
var k = j;
while (k < value.len and !isSpace(value[k]) and value[k] != ';' and value[k] != '(') k += 1;
if (k > j) return value[j..k];
}
}
i = abs + keyword.len;
}
return null;
}
fn isAlnumOrUnderscore(c: u8) bool {
return std.ascii.isAlphanumeric(c) or c == '_';
}
fn stripTrailingPunct(s: []const u8) []const u8 {
var end = s.len;
while (end > 0 and (s[end - 1] == '.' or s[end - 1] == ',')) end -= 1;
return s[0..end];
}
/// Does `value` contain a date-looking token ("Mon, ...") for the no-';' fallback?
fn firstDateGuess(value: []const u8) ?[]const u8 {
var i: usize = 0;
while (i < value.len) : (i += 1) {
if (i != 0 and isAlnumOrUnderscore(value[i - 1])) continue;
for (WEEKDAYS) |wd| {
if (value.len - i < 3) continue;
if (!eqlIgnoreCase(value[i .. i + 3], wd)) continue;
if (value.len - i >= 4 and value[i + 3] == ',') {
// Take to the end of line — parseRfc2822Date re-tokenizes anyway.
return value[i..];
}
}
}
return null;
}
/// Parse one Received header value into a hop (delay filled in later).
pub fn parseReceived(allocator: std.mem.Allocator, value: []const u8) Error!Hop {
const from_raw = keywordToken(value, "from");
const by_raw = keywordToken(value, "by");
const protocol_raw = keywordToken(value, "with");
// The timestamp follows the last ';' in the value.
var timestamp: ?i64 = null;
if (std.mem.lastIndexOfScalar(u8, value, ';')) |last_semi| {
timestamp = try parseRfc2822Date(allocator, value[last_semi + 1 ..]);
}
if (timestamp == null) {
// Some clients omit the ';'. Fall back to the first date-looking token.
if (firstDateGuess(value)) |guess| {
timestamp = try parseRfc2822Date(allocator, guess);
}
}
return .{
.from = if (from_raw) |f| stripTrailingPunct(f) else "",
.by = if (by_raw) |b| stripTrailingPunct(b) else "",
.timestamp = timestamp,
.delay = null,
.protocol = protocol_raw orelse "",
};
}
// --- Authentication-Results -------------------------------------------------------
fn emptyAuth() AuthResult {
return .{
.result = NO_AUTH.result,
.domain = null,
.selector = null,
.policy = null,
.detail = null,
};
}
/// `=\s*([a-z]+)` after the kind prefix — the result word, lowercased.
fn resultWord(allocator: std.mem.Allocator, clause: []const u8) Error![]const u8 {
const eq = std.mem.indexOfScalar(u8, clause, '=') orelse return "unknown";
var i = eq + 1;
while (i < clause.len and isSpace(clause[i])) i += 1;
const start = i;
while (i < clause.len and std.ascii.isAlphabetic(clause[i])) i += 1;
if (i == start) return "unknown";
return lowerAlloc(allocator, clause[start..i]);
}
/// `\bKEY=([^\s;)]+)` — the value token for keys like `smtp.mailfrom`,
/// `header.d`, `header.s`. Trailing '.'/',' is stripped.
fn clauseToken(clause: []const u8, key: []const u8) ?[]const u8 {
var i: usize = 0;
while (indexOfIgnoreCase(clause[i..], key)) |at| {
const abs = i + at;
const after = abs + key.len;
const boundary = abs == 0 or !isAlnumOrUnderscore(clause[abs - 1]);
if (boundary and after < clause.len and clause[after] == '=') {
var j = after + 1;
while (j < clause.len and !isSpace(clause[j]) and clause[j] != ';' and clause[j] != ')') j += 1;
if (j > after + 1) return stripTrailingPunct(clause[after + 1 .. j]);
}
i = abs + key.len;
}
return null;
}
/// `p=([a-z]+)` — the DMARC policy token.
fn dmarcPolicy(clause: []const u8) ?[]const u8 {
var i: usize = 0;
while (i + 2 <= clause.len) : (i += 1) {
if (clause[i] != 'p' and clause[i] != 'P') continue;
if (i != 0 and isAlnumOrUnderscore(clause[i - 1])) continue;
if (clause[i + 1] != '=') continue;
var j = i + 2;
while (j < clause.len and std.ascii.isAlphabetic(clause[j])) j += 1;
if (j > i + 2) return clause[i + 2 .. j];
}
return null;
}
/// Parse Authentication-Results clauses. Example:
/// mx.google.com; spf=pass smtp.mailfrom=a@b.com; dkim=pass header.s=sel header.d=b.com;
/// dmarc=pass (p=REJECT) header.from=b.com
pub fn parseAuthResults(allocator: std.mem.Allocator, values: []const []const u8) Error!AuthResults {
var auth = AuthResults{ .spf = emptyAuth(), .dkim = emptyAuth(), .dmarc = emptyAuth() };
if (values.len == 0) return auth;
for (values) |value| {
var clauses = std.mem.splitScalar(u8, value, ';');
while (clauses.next()) |raw_clause| {
const clause = trim(raw_clause);
const Kind = enum { spf, dkim, dmarc };
const kind: ?Kind = blk: {
if (startsWithIgnoreCase(clause, "spf=")) break :blk .spf;
if (startsWithIgnoreCase(clause, "dkim=")) break :blk .dkim;
if (startsWithIgnoreCase(clause, "dmarc=")) break :blk .dmarc;
break :blk null;
};
const k = kind orelse continue;
const target = switch (k) {
.spf => &auth.spf,
.dkim => &auth.dkim,
.dmarc => &auth.dmarc,
};
if (!std.mem.eql(u8, target.result, "none")) continue; // first result wins
target.result = try resultWord(allocator, clause);
target.detail = try allocator.dupe(u8, clause);
switch (k) {
.spf => target.domain = clauseToken(clause, "smtp.mailfrom") orelse clauseToken(clause, "mailfrom"),
.dkim => {
target.domain = clauseToken(clause, "header.d") orelse clauseToken(clause, "header.i");
target.selector = clauseToken(clause, "header.s");
},
.dmarc => {
target.domain = clauseToken(clause, "header.from");
if (dmarcPolicy(clause)) |p| {
target.policy = try lowerAlloc(allocator, p);
}
},
}
}
}
return auth;
}
// --- Anomaly detection -------------------------------------------------------------
fn detectAnomalies(allocator: std.mem.Allocator, headers: []const Header, hops: []const Hop, auth: AuthResults) Error![]Anomaly {
var anomalies = std.ArrayList(Anomaly).init(allocator);
errdefer anomalies.deinit();
// 1. From vs Return-Path mismatch — classic spoofing signal.
const from_addr = extractAddress(firstHeader(headers, "From"));
const return_path = extractAddress(firstHeader(headers, "Return-Path"));
if (from_addr != null and return_path != null) {
const from_lower = try lowerAlloc(allocator, from_addr.?);
defer allocator.free(from_lower);
const rp_lower = try lowerAlloc(allocator, return_path.?);
defer allocator.free(rp_lower);
if (!std.mem.eql(u8, from_lower, rp_lower)) {
try anomalies.append(.{
.kind = "from-return-path-mismatch",
.severity = .critical,
.message = try std.fmt.allocPrint(allocator, "Return-Path ({s}) does not match From ({s}) — the envelope sender differs from the displayed sender. Common in spoofing and mailing-list relay.", .{ return_path.?, from_addr.? }),
});
}
}
// 2. Reply-To pointing somewhere other than From.
const reply_to = extractAddress(firstHeader(headers, "Reply-To"));
if (from_addr != null and reply_to != null) {
const from_lower = try lowerAlloc(allocator, from_addr.?);
defer allocator.free(from_lower);
const rt_lower = try lowerAlloc(allocator, reply_to.?);
defer allocator.free(rt_lower);
if (!std.mem.eql(u8, rt_lower, from_lower)) {
try anomalies.append(.{
.kind = "reply-to-mismatch",
.severity = .warning,
.message = try std.fmt.allocPrint(allocator, "Reply-To ({s}) differs from From ({s}) — replies would go to a different address than the visible sender.", .{ reply_to.?, from_addr.? }),
});
}
}
// 3. Suspicious X-Mailer / User-Agent strings.
const mailer = blk: {
const xm = firstHeader(headers, "X-Mailer");
break :blk if (xm.len > 0) xm else firstHeader(headers, "User-Agent");
};
if (mailer.len > 0) {
const mailer_lower = try lowerAlloc(allocator, mailer);
defer allocator.free(mailer_lower);
for (SUSPICIOUS_MAILERS) |s| {
if (containsIgnoreCase(mailer_lower, s)) {
try anomalies.append(.{
.kind = "suspicious-mailer",
.severity = .warning,
.message = try std.fmt.allocPrint(allocator, "Mailer string \"{s}\" contains a suspicious token (\"{s}\") often seen in bulk sending tools.", .{ mailer, s }),
});
break;
}
}
}
// 4. Received chain gaps: unparseable/missing timestamps and time going backwards.
for (hops, 0..) |hop, i| {
if (hop.timestamp == null) {
const who = if (hop.from.len > 0) hop.from else if (hop.by.len > 0) hop.by else "unknown";
try anomalies.append(.{
.kind = "hop-missing-timestamp",
.severity = .info,
.message = try std.fmt.allocPrint(allocator, "Hop {d} ({s}) has no parseable timestamp — delay for this leg cannot be computed.", .{ i + 1, who }),
});
continue;
}
if (i > 0 and hops[i - 1].timestamp != null) {
const delta: f64 = @floatFromInt(hop.timestamp.? - hops[i - 1].timestamp.?);
if (delta < 0) {
try anomalies.append(.{
.kind = "negative-delay",
.severity = .warning,
.message = try std.fmt.allocPrint(allocator, "Hop {d} is timestamped {d:.1}s BEFORE hop {d} — clock skew between servers or a forged Received header.", .{ i + 1, @abs(delta), i }),
});
}
}
}
// 5. No authentication results at all.
var has_auth = false;
for (headers) |h| {
if (eqlIgnoreCase(h.name, "Authentication-Results")) {
has_auth = true;
break;
}
}
if (!has_auth) {
try anomalies.append(.{
.kind = "no-auth-results",
.severity = .info,
.message = "No Authentication-Results header found — SPF/DKIM/DMARC status cannot be verified from this message.",
});
}
_ = auth; // auth results shape is reflected by check 5's presence/absence
return anomalies.toOwnedSlice();
}
// --- Main entry point ---------------------------------------------------------------
/// Parse raw email headers (RFC 5322) into structured data: unfolded headers,
/// chronological Received hops with delays, SPF/DKIM/DMARC results, and
/// spoofing anomalies. Caller owns every returned slice.
pub fn parseHeaders(allocator: std.mem.Allocator, raw: []const u8) Error!ParsedHeaders {
const headers = try parseHeaderLines(allocator, raw);
// Received headers are listed newest-first; parse bottom-up so hops[0] is the oldest.
const received = try getHeaders(allocator, headers, "Received");
defer allocator.free(received);
var hops = try allocator.alloc(Hop, received.len);
for (received, 0..) |value, i| {
hops[received.len - 1 - i] = try parseReceived(allocator, value);
}
// delays run chronologically: hop i's delay = t(i) - t(i-1), in seconds
for (hops, 0..) |*hop, i| {
if (i == 0 or hop.timestamp == null or hops[i - 1].timestamp == null) continue;
hop.delay = @floatFromInt(hop.timestamp.? - hops[i - 1].timestamp.?);
}
const auth_values = try getHeaders(allocator, headers, "Authentication-Results");
defer allocator.free(auth_values);
const auth = try parseAuthResults(allocator, auth_values);
const anomalies = try detectAnomalies(allocator, headers, hops, auth);
return .{ .headers = headers, .hops = hops, .auth = auth, .anomalies = anomalies };
}
Also available in 8 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →