Email Validator — Zig source
Validate email addresses one at a time or in bulk. Checks syntax, length limits, local-part and domain rules, plus-addressing, and IP-literal domains - all in your browser.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! email-validator — RFC 5321/5322-inspired email validation.
//!
//! Language: Zig (0.13, standard library only)
//! Source: CosmoDev polyglot showcase port of the Email Validator tool,
//! ported from src/lib/email-validator.ts (the canonical TypeScript
//! implementation).
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Practical, provider-friendly validation: errs on the side of deliverability
//! while still recognising the legal-but-unusual forms (quoted local parts,
//! IP-literal domains). Pure and deterministic — every malformed input becomes
//! a non-valid verdict carrying explanatory reasons. All allocations use the
//! caller's allocator; free a verdict with `EmailResult.deinit`.
//!
//! Self-contained: the character classes used by the original are implemented
//! as small byte predicates (std.ascii), avoiding any regex dependency.
const std = @import("std");
/// RFC-inspired length ceilings: local part, domain, total address.
const local_max = 64;
const domain_max = 253;
const total_max = 320;
/// ASCII whitespace trimmed from input edges: space, tab, LF, CR, VT, FF.
const whitespace = " \t\n\r\x0b\x0c";
/// The structured verdict returned by `validateEmail`. `valid` is true iff
/// `reasons` is empty; `warnings` never affect validity. `local` / `domain`
/// are empty slices when the address could not be split into parts, and
/// `normalized` is null in that case.
pub const EmailResult = struct {
valid: bool,
local: []const u8,
domain: []const u8,
normalized: ?[]const u8,
/// Blocking problems (`valid` is true iff this is empty).
reasons: std.ArrayList([]const u8),
/// Non-blocking observations (rare forms, plus-tags, ...).
warnings: std.ArrayList([]const u8),
/// Frees every string the verdict owns (all list entries plus the parts).
pub fn deinit(self: *EmailResult, allocator: std.mem.Allocator) void {
for (self.reasons.items) |msg| allocator.free(msg);
for (self.warnings.items) |msg| allocator.free(msg);
self.reasons.deinit();
self.warnings.deinit();
allocator.free(self.local);
allocator.free(self.domain);
if (self.normalized) |n| allocator.free(n);
}
};
/// Internal split result: borrowed views into the input address.
const Split = struct {
local: []const u8,
domain: []const u8,
quoted: bool,
};
// ------------------------------------------------------------- predicates --
/// True when every byte of `s` belongs to the RFC-style "atom" character set
/// (ASCII alphanumeric plus the printable specials permitted unquoted).
/// Iterating bytes is correct here because the class is strictly ASCII.
fn isAtomLocal(s: []const u8) bool {
if (s.len == 0) return false;
for (s) |c| {
if (!std.ascii.isAlphanumeric(c) and !isAtomSpecial(c)) return false;
}
return true;
}
fn isAtomSpecial(c: u8) bool {
return switch (c) {
'.', '!', '#', '$', '%', '&', '\'', '*', '+', '/', '=', '?', '^', '_', '`', '{', '|', '}', '~', '-' => true,
else => false,
};
}
/// A valid domain label: ASCII letters, digits, and hyphens (non-empty).
fn isValidLabel(s: []const u8) bool {
if (s.len == 0) return false;
for (s) |c| {
if (!std.ascii.isAlphanumeric(c) and c != '-') return false;
}
return true;
}
/// A valid TLD: two or more ASCII letters. Byte length equals char count
/// when every byte is ASCII alphabetic, so the length check is exact.
fn isValidTld(s: []const u8) bool {
if (s.len < 2) return false;
for (s) |c| {
if (!std.ascii.isAlphabetic(c)) return false;
}
return true;
}
/// An all-decimal, non-empty octet string.
fn isDecimal(s: []const u8) bool {
if (s.len == 0) return false;
for (s) |c| {
if (!std.ascii.isDigit(c)) return false;
}
return true;
}
/// True when `s` starts with "ipv6:" (case-insensitive).
fn isIpv6Literal(s: []const u8) bool {
return s.len >= 5 and std.ascii.eqlIgnoreCase(s[0..5], "ipv6:");
}
/// True when `s` is a dotted-quad: four octets, each 0-255, with no leading
/// zeros. The 3-digit length cap rejects arbitrarily long digit strings
/// before they can overflow the value parse (equivalent to the reference's
/// overflow-on-cast behaviour).
fn isIpv4(s: []const u8) bool {
var it = std.mem.splitScalar(u8, s, '.');
var count: usize = 0;
while (it.next()) |part| {
count += 1;
if (count > 4) return false;
if (part.len == 0 or part.len > 3) return false;
if (!isDecimal(part)) return false;
var value: u32 = 0;
for (part) |c| value = value * 10 + (c - '0');
if (value > 255) return false;
if (part.len > 1 and part[0] == '0') return false; // leading zero ("01")
}
return count == 4;
}
// ---------------------------------------------------------------- helpers --
/// Appends an owned copy of a literal message to a list.
fn push(list: *std.ArrayList([]const u8), allocator: std.mem.Allocator, msg: []const u8) !void {
try list.append(try allocator.dupe(u8, msg));
}
/// Appends a formatted, owned message to a list (length-limit reasons).
fn pushFmt(list: *std.ArrayList([]const u8), allocator: std.mem.Allocator,
comptime fmt: []const u8, args: anytype) !void {
try list.append(try std.fmt.allocPrint(allocator, fmt, args));
}
// ------------------------------------------------------------------ split --
/// Splits an address into local + domain, honouring a quoted ("...") local
/// part. Returns null when the address cannot be split into exactly one '@'
/// in the right place.
fn splitLocalDomain(email: []const u8) ?Split {
if (email.len > 0 and email[0] == '"') {
// Walk the quoted string; a backslash escapes the next byte (so `\"`
// does not terminate the quote). We only branch on ASCII delimiters,
// so byte-indexing is safe; all slice bounds land on ASCII chars.
var i: usize = 1;
while (i < email.len) {
const ch = email[i];
if (ch == '\\') {
i += 2;
continue;
}
if (ch == '"') break;
i += 1;
}
if (i >= email.len or email[i] != '"') return null; // unterminated quote
const at = i + 1;
if (at >= email.len or email[at] != '@') return null; // '@' must follow quote
if (std.mem.indexOfScalar(u8, email[at + 1 ..], '@') != null) return null; // stray '@'
return Split{ .local = email[0..at], .domain = email[at + 1 ..], .quoted = true };
}
const first = std.mem.indexOfScalar(u8, email, '@') orelse return null;
if (std.mem.indexOfScalar(u8, email[first + 1 ..], '@') != null) return null; // multiple '@'
return Split{ .local = email[0..first], .domain = email[first + 1 ..], .quoted = false };
}
// ---------------------------------------------------------------- domain --
/// Appends domain-level problems to `reasons` / `warnings`.
fn validateDomain(domain: []const u8, allocator: std.mem.Allocator,
reasons: *std.ArrayList([]const u8),
warnings: *std.ArrayList([]const u8)) !void {
if (domain.len == 0) {
try push(reasons, allocator, "Domain is empty");
return;
}
if (domain.len > domain_max) {
try pushFmt(reasons, allocator, "Domain exceeds {d} characters", .{domain_max});
}
// IP-literal domain: [1.2.3.4] or [IPv6:...].
if (domain[0] == '[' and domain[domain.len - 1] == ']') {
const inner = domain[1 .. domain.len - 1];
if (isIpv6Literal(inner)) {
try push(warnings, allocator, "IPv6 literal domain (uncommon; ensure your provider supports it)");
return;
}
if (isIpv4(inner)) {
try push(warnings, allocator, "IP-literal domain (uncommon; ensure your provider supports it)");
return;
}
try push(reasons, allocator, "Invalid IP-literal domain");
return;
}
if (domain[0] == '[' or domain[domain.len - 1] == ']') {
try push(reasons, allocator, "Malformed IP-literal domain (unmatched brackets)");
return;
}
if (std.mem.indexOfScalar(u8, domain, '.') == null) {
try push(reasons, allocator, "Domain must contain at least one dot (e.g. example.com)");
return;
}
var labels = std.mem.splitScalar(u8, domain, '.');
while (labels.next()) |label| {
if (label.len == 0) {
try push(reasons, allocator, "Domain contains an empty label (consecutive or trailing dots)");
continue;
}
if (label.len > 63) {
try push(reasons, allocator, "Domain label exceeds 63 characters");
}
if (!isValidLabel(label)) {
try push(reasons, allocator, "Domain label contains invalid characters");
}
if (label[0] == '-' or label[label.len - 1] == '-') {
try push(reasons, allocator, "Domain label starts or ends with a hyphen");
}
}
// The TLD is the final label; require >=2 ASCII letters so bare hostnames
// and numeric tails are rejected.
var tld_start = domain.len;
while (tld_start > 0 and domain[tld_start - 1] != '.') tld_start -= 1;
if (!isValidTld(domain[tld_start..])) {
try push(reasons, allocator, "Top-level domain must be at least two letters");
}
}
// ----------------------------------------------------------------- email --
/// Validates a single email address, returning a structured verdict.
///
/// Pure and deterministic: every malformed input becomes a non-valid result
/// carrying explanatory reasons.
pub fn validateEmail(allocator: std.mem.Allocator, raw: []const u8) !EmailResult {
var reasons = std.ArrayList([]const u8).init(allocator);
var warnings = std.ArrayList([]const u8).init(allocator);
errdefer {
for (reasons.items) |msg| allocator.free(msg);
for (warnings.items) |msg| allocator.free(msg);
reasons.deinit();
warnings.deinit();
}
const email = std.mem.trim(u8, raw, whitespace);
if (email.len == 0) {
try push(&reasons, allocator, "Email is empty");
return EmailResult{
.valid = false,
.local = try allocator.dupe(u8, ""),
.domain = try allocator.dupe(u8, ""),
.normalized = null,
.reasons = reasons,
.warnings = warnings,
};
}
if (email.len > total_max) {
try pushFmt(&reasons, allocator, "Email exceeds maximum length of {d} characters", .{total_max});
}
const split = splitLocalDomain(email) orelse {
try push(&reasons, allocator, "Email must contain exactly one \"@\" separating local part and domain");
return EmailResult{
.valid = false,
.local = try allocator.dupe(u8, ""),
.domain = try allocator.dupe(u8, ""),
.normalized = null,
.reasons = reasons,
.warnings = warnings,
};
};
const local = split.local;
const domain = split.domain;
if (split.quoted) {
// Quoted local parts are RFC-legal but almost universally rejected by
// mailbox providers — warn, and only length-check structurally.
if (local.len > local_max) {
try pushFmt(&reasons, allocator, "Local part exceeds {d} characters", .{local_max});
}
try push(&warnings, allocator, "Quoted local part (rarely supported by providers)");
} else if (local.len == 0) {
try push(&reasons, allocator, "Local part is empty");
} else {
if (local.len > local_max) {
try pushFmt(&reasons, allocator, "Local part exceeds {d} characters", .{local_max});
}
if (local[0] == '.' or local[local.len - 1] == '.') {
try push(&reasons, allocator, "Local part starts or ends with a dot");
}
if (std.mem.indexOf(u8, local, "..") != null) {
try push(&reasons, allocator, "Local part contains consecutive dots");
}
if (!isAtomLocal(local)) {
try push(&reasons, allocator, "Local part contains invalid characters");
}
}
// Plus-addressing (`user+tag@`) is valid and delivers to the base mailbox,
// but callers filtering on exact address may want to know.
if (!split.quoted and std.mem.indexOfScalar(u8, local, '+') != null) {
try push(&warnings, allocator, "Plus-addressing (tag) detected — delivers to the base mailbox");
}
try validateDomain(domain, allocator, &reasons, &warnings);
const valid = reasons.items.len == 0;
var normalized: ?[]const u8 = null;
if (local.len > 0 and domain.len > 0) {
const lowered = try allocator.dupe(u8, domain);
_ = std.ascii.lowerString(lowered, lowered);
normalized = try std.fmt.allocPrint(allocator, "{s}@{s}", .{ local, lowered });
allocator.free(lowered);
}
return EmailResult{
.valid = valid,
.local = try allocator.dupe(u8, local),
.domain = try allocator.dupe(u8, domain),
.normalized = normalized,
.reasons = reasons,
.warnings = warnings,
};
}
/// Validates many addresses — one per line. Blank or whitespace-only lines
/// are skipped. Line endings may be LF or CRLF (matching the reference's
/// `\r?\n` split). The caller owns the returned slice and each verdict.
pub fn validateBatch(allocator: std.mem.Allocator, input: []const u8) ![]EmailResult {
if (input.len == 0) return &[_]EmailResult{};
var out = std.ArrayList(EmailResult).init(allocator);
errdefer {
for (out.items) |*r| r.deinit(allocator);
out.deinit();
}
var lines = std.mem.splitScalar(u8, input, '\n');
while (lines.next()) |raw_line| {
var line = raw_line;
if (line.len > 0 and line[line.len - 1] == '\r') line = line[0 .. line.len - 1]; // CRLF
const trimmed = std.mem.trim(u8, line, whitespace);
if (trimmed.len > 0) {
try out.append(try validateEmail(allocator, trimmed));
}
}
return out.toOwnedSlice();
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →