Skip to content

Email Validator — Zig source

Validate email addresses one at a time or in bulk. Checks syntax, length limits, local-part and domain rules, plus-addressing, and IP-literal domains - all in your browser.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! email-validator — RFC 5321/5322-inspired email validation.
//!
//! Language: Zig (0.13, standard library only)
//! Source:   CosmoDev polyglot showcase port of the Email Validator tool,
//!           ported from src/lib/email-validator.ts (the canonical TypeScript
//!           implementation).
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Practical, provider-friendly validation: errs on the side of deliverability
//! while still recognising the legal-but-unusual forms (quoted local parts,
//! IP-literal domains). Pure and deterministic — every malformed input becomes
//! a non-valid verdict carrying explanatory reasons. All allocations use the
//! caller's allocator; free a verdict with `EmailResult.deinit`.
//!
//! Self-contained: the character classes used by the original are implemented
//! as small byte predicates (std.ascii), avoiding any regex dependency.

const std = @import("std");

/// RFC-inspired length ceilings: local part, domain, total address.
const local_max = 64;
const domain_max = 253;
const total_max = 320;

/// ASCII whitespace trimmed from input edges: space, tab, LF, CR, VT, FF.
const whitespace = " \t\n\r\x0b\x0c";

/// The structured verdict returned by `validateEmail`. `valid` is true iff
/// `reasons` is empty; `warnings` never affect validity. `local` / `domain`
/// are empty slices when the address could not be split into parts, and
/// `normalized` is null in that case.
pub const EmailResult = struct {
    valid: bool,
    local: []const u8,
    domain: []const u8,
    normalized: ?[]const u8,
    /// Blocking problems (`valid` is true iff this is empty).
    reasons: std.ArrayList([]const u8),
    /// Non-blocking observations (rare forms, plus-tags, ...).
    warnings: std.ArrayList([]const u8),

    /// Frees every string the verdict owns (all list entries plus the parts).
    pub fn deinit(self: *EmailResult, allocator: std.mem.Allocator) void {
        for (self.reasons.items) |msg| allocator.free(msg);
        for (self.warnings.items) |msg| allocator.free(msg);
        self.reasons.deinit();
        self.warnings.deinit();
        allocator.free(self.local);
        allocator.free(self.domain);
        if (self.normalized) |n| allocator.free(n);
    }
};

/// Internal split result: borrowed views into the input address.
const Split = struct {
    local: []const u8,
    domain: []const u8,
    quoted: bool,
};

// ------------------------------------------------------------- predicates --

/// True when every byte of `s` belongs to the RFC-style "atom" character set
/// (ASCII alphanumeric plus the printable specials permitted unquoted).
/// Iterating bytes is correct here because the class is strictly ASCII.
fn isAtomLocal(s: []const u8) bool {
    if (s.len == 0) return false;
    for (s) |c| {
        if (!std.ascii.isAlphanumeric(c) and !isAtomSpecial(c)) return false;
    }
    return true;
}

fn isAtomSpecial(c: u8) bool {
    return switch (c) {
        '.', '!', '#', '$', '%', '&', '\'', '*', '+', '/', '=', '?', '^', '_', '`', '{', '|', '}', '~', '-' => true,
        else => false,
    };
}

/// A valid domain label: ASCII letters, digits, and hyphens (non-empty).
fn isValidLabel(s: []const u8) bool {
    if (s.len == 0) return false;
    for (s) |c| {
        if (!std.ascii.isAlphanumeric(c) and c != '-') return false;
    }
    return true;
}

/// A valid TLD: two or more ASCII letters. Byte length equals char count
/// when every byte is ASCII alphabetic, so the length check is exact.
fn isValidTld(s: []const u8) bool {
    if (s.len < 2) return false;
    for (s) |c| {
        if (!std.ascii.isAlphabetic(c)) return false;
    }
    return true;
}

/// An all-decimal, non-empty octet string.
fn isDecimal(s: []const u8) bool {
    if (s.len == 0) return false;
    for (s) |c| {
        if (!std.ascii.isDigit(c)) return false;
    }
    return true;
}

/// True when `s` starts with "ipv6:" (case-insensitive).
fn isIpv6Literal(s: []const u8) bool {
    return s.len >= 5 and std.ascii.eqlIgnoreCase(s[0..5], "ipv6:");
}

/// True when `s` is a dotted-quad: four octets, each 0-255, with no leading
/// zeros. The 3-digit length cap rejects arbitrarily long digit strings
/// before they can overflow the value parse (equivalent to the reference's
/// overflow-on-cast behaviour).
fn isIpv4(s: []const u8) bool {
    var it = std.mem.splitScalar(u8, s, '.');
    var count: usize = 0;
    while (it.next()) |part| {
        count += 1;
        if (count > 4) return false;
        if (part.len == 0 or part.len > 3) return false;
        if (!isDecimal(part)) return false;
        var value: u32 = 0;
        for (part) |c| value = value * 10 + (c - '0');
        if (value > 255) return false;
        if (part.len > 1 and part[0] == '0') return false; // leading zero ("01")
    }
    return count == 4;
}

// ---------------------------------------------------------------- helpers --

/// Appends an owned copy of a literal message to a list.
fn push(list: *std.ArrayList([]const u8), allocator: std.mem.Allocator, msg: []const u8) !void {
    try list.append(try allocator.dupe(u8, msg));
}

/// Appends a formatted, owned message to a list (length-limit reasons).
fn pushFmt(list: *std.ArrayList([]const u8), allocator: std.mem.Allocator,
           comptime fmt: []const u8, args: anytype) !void {
    try list.append(try std.fmt.allocPrint(allocator, fmt, args));
}

// ------------------------------------------------------------------ split --

/// Splits an address into local + domain, honouring a quoted ("...") local
/// part. Returns null when the address cannot be split into exactly one '@'
/// in the right place.
fn splitLocalDomain(email: []const u8) ?Split {
    if (email.len > 0 and email[0] == '"') {
        // Walk the quoted string; a backslash escapes the next byte (so `\"`
        // does not terminate the quote). We only branch on ASCII delimiters,
        // so byte-indexing is safe; all slice bounds land on ASCII chars.
        var i: usize = 1;
        while (i < email.len) {
            const ch = email[i];
            if (ch == '\\') {
                i += 2;
                continue;
            }
            if (ch == '"') break;
            i += 1;
        }
        if (i >= email.len or email[i] != '"') return null; // unterminated quote
        const at = i + 1;
        if (at >= email.len or email[at] != '@') return null; // '@' must follow quote
        if (std.mem.indexOfScalar(u8, email[at + 1 ..], '@') != null) return null; // stray '@'
        return Split{ .local = email[0..at], .domain = email[at + 1 ..], .quoted = true };
    }

    const first = std.mem.indexOfScalar(u8, email, '@') orelse return null;
    if (std.mem.indexOfScalar(u8, email[first + 1 ..], '@') != null) return null; // multiple '@'
    return Split{ .local = email[0..first], .domain = email[first + 1 ..], .quoted = false };
}

// ---------------------------------------------------------------- domain --

/// Appends domain-level problems to `reasons` / `warnings`.
fn validateDomain(domain: []const u8, allocator: std.mem.Allocator,
                  reasons: *std.ArrayList([]const u8),
                  warnings: *std.ArrayList([]const u8)) !void {
    if (domain.len == 0) {
        try push(reasons, allocator, "Domain is empty");
        return;
    }
    if (domain.len > domain_max) {
        try pushFmt(reasons, allocator, "Domain exceeds {d} characters", .{domain_max});
    }

    // IP-literal domain: [1.2.3.4] or [IPv6:...].
    if (domain[0] == '[' and domain[domain.len - 1] == ']') {
        const inner = domain[1 .. domain.len - 1];
        if (isIpv6Literal(inner)) {
            try push(warnings, allocator, "IPv6 literal domain (uncommon; ensure your provider supports it)");
            return;
        }
        if (isIpv4(inner)) {
            try push(warnings, allocator, "IP-literal domain (uncommon; ensure your provider supports it)");
            return;
        }
        try push(reasons, allocator, "Invalid IP-literal domain");
        return;
    }
    if (domain[0] == '[' or domain[domain.len - 1] == ']') {
        try push(reasons, allocator, "Malformed IP-literal domain (unmatched brackets)");
        return;
    }

    if (std.mem.indexOfScalar(u8, domain, '.') == null) {
        try push(reasons, allocator, "Domain must contain at least one dot (e.g. example.com)");
        return;
    }

    var labels = std.mem.splitScalar(u8, domain, '.');
    while (labels.next()) |label| {
        if (label.len == 0) {
            try push(reasons, allocator, "Domain contains an empty label (consecutive or trailing dots)");
            continue;
        }
        if (label.len > 63) {
            try push(reasons, allocator, "Domain label exceeds 63 characters");
        }
        if (!isValidLabel(label)) {
            try push(reasons, allocator, "Domain label contains invalid characters");
        }
        if (label[0] == '-' or label[label.len - 1] == '-') {
            try push(reasons, allocator, "Domain label starts or ends with a hyphen");
        }
    }
    // The TLD is the final label; require >=2 ASCII letters so bare hostnames
    // and numeric tails are rejected.
    var tld_start = domain.len;
    while (tld_start > 0 and domain[tld_start - 1] != '.') tld_start -= 1;
    if (!isValidTld(domain[tld_start..])) {
        try push(reasons, allocator, "Top-level domain must be at least two letters");
    }
}

// ----------------------------------------------------------------- email --

/// Validates a single email address, returning a structured verdict.
///
/// Pure and deterministic: every malformed input becomes a non-valid result
/// carrying explanatory reasons.
pub fn validateEmail(allocator: std.mem.Allocator, raw: []const u8) !EmailResult {
    var reasons = std.ArrayList([]const u8).init(allocator);
    var warnings = std.ArrayList([]const u8).init(allocator);
    errdefer {
        for (reasons.items) |msg| allocator.free(msg);
        for (warnings.items) |msg| allocator.free(msg);
        reasons.deinit();
        warnings.deinit();
    }

    const email = std.mem.trim(u8, raw, whitespace);

    if (email.len == 0) {
        try push(&reasons, allocator, "Email is empty");
        return EmailResult{
            .valid = false,
            .local = try allocator.dupe(u8, ""),
            .domain = try allocator.dupe(u8, ""),
            .normalized = null,
            .reasons = reasons,
            .warnings = warnings,
        };
    }

    if (email.len > total_max) {
        try pushFmt(&reasons, allocator, "Email exceeds maximum length of {d} characters", .{total_max});
    }

    const split = splitLocalDomain(email) orelse {
        try push(&reasons, allocator, "Email must contain exactly one \"@\" separating local part and domain");
        return EmailResult{
            .valid = false,
            .local = try allocator.dupe(u8, ""),
            .domain = try allocator.dupe(u8, ""),
            .normalized = null,
            .reasons = reasons,
            .warnings = warnings,
        };
    };
    const local = split.local;
    const domain = split.domain;

    if (split.quoted) {
        // Quoted local parts are RFC-legal but almost universally rejected by
        // mailbox providers — warn, and only length-check structurally.
        if (local.len > local_max) {
            try pushFmt(&reasons, allocator, "Local part exceeds {d} characters", .{local_max});
        }
        try push(&warnings, allocator, "Quoted local part (rarely supported by providers)");
    } else if (local.len == 0) {
        try push(&reasons, allocator, "Local part is empty");
    } else {
        if (local.len > local_max) {
            try pushFmt(&reasons, allocator, "Local part exceeds {d} characters", .{local_max});
        }
        if (local[0] == '.' or local[local.len - 1] == '.') {
            try push(&reasons, allocator, "Local part starts or ends with a dot");
        }
        if (std.mem.indexOf(u8, local, "..") != null) {
            try push(&reasons, allocator, "Local part contains consecutive dots");
        }
        if (!isAtomLocal(local)) {
            try push(&reasons, allocator, "Local part contains invalid characters");
        }
    }
    // Plus-addressing (`user+tag@`) is valid and delivers to the base mailbox,
    // but callers filtering on exact address may want to know.
    if (!split.quoted and std.mem.indexOfScalar(u8, local, '+') != null) {
        try push(&warnings, allocator, "Plus-addressing (tag) detected — delivers to the base mailbox");
    }

    try validateDomain(domain, allocator, &reasons, &warnings);

    const valid = reasons.items.len == 0;
    var normalized: ?[]const u8 = null;
    if (local.len > 0 and domain.len > 0) {
        const lowered = try allocator.dupe(u8, domain);
        _ = std.ascii.lowerString(lowered, lowered);
        normalized = try std.fmt.allocPrint(allocator, "{s}@{s}", .{ local, lowered });
        allocator.free(lowered);
    }
    return EmailResult{
        .valid = valid,
        .local = try allocator.dupe(u8, local),
        .domain = try allocator.dupe(u8, domain),
        .normalized = normalized,
        .reasons = reasons,
        .warnings = warnings,
    };
}

/// Validates many addresses — one per line. Blank or whitespace-only lines
/// are skipped. Line endings may be LF or CRLF (matching the reference's
/// `\r?\n` split). The caller owns the returned slice and each verdict.
pub fn validateBatch(allocator: std.mem.Allocator, input: []const u8) ![]EmailResult {
    if (input.len == 0) return &[_]EmailResult{};

    var out = std.ArrayList(EmailResult).init(allocator);
    errdefer {
        for (out.items) |*r| r.deinit(allocator);
        out.deinit();
    }

    var lines = std.mem.splitScalar(u8, input, '\n');
    while (lines.next()) |raw_line| {
        var line = raw_line;
        if (line.len > 0 and line[line.len - 1] == '\r') line = line[0 .. line.len - 1]; // CRLF
        const trimmed = std.mem.trim(u8, line, whitespace);
        if (trimmed.len > 0) {
            try out.append(try validateEmail(allocator, trimmed));
        }
    }
    return out.toOwnedSlice();
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →