Skip to content

Email Validator — C++ source

Validate email addresses one at a time or in bulk. Checks syntax, length limits, local-part and domain rules, plus-addressing, and IP-literal domains - all in your browser.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// email-validator — RFC 5321/5322-inspired email validation.
//
// Language: C++ (C++17, standard library only)
// Source:   CosmoDev polyglot showcase port of the Email Validator tool,
//           ported from src/lib/email-validator.ts (the canonical TypeScript
//           implementation).
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Practical, provider-friendly validation: errs on the side of deliverability
// while still recognising the legal-but-unusual forms (quoted local parts,
// IP-literal domains). Pure and deterministic — every malformed input becomes
// a non-valid verdict carrying explanatory reasons; nothing below throws.
//
// Self-contained: stdlib only — the character classes used by the original
// are implemented as small byte predicates, avoiding any external regex
// dependency.

#include <cstddef>
#include <optional>
#include <string>
#include <string_view>
#include <vector>

/// RFC-inspired length ceilings: local part, domain, total address.
constexpr std::size_t kLocalMax = 64;
constexpr std::size_t kDomainMax = 253;
constexpr std::size_t kTotalMax = 320;

/// The structured verdict returned by `ValidateEmail`.
struct EmailResult {
    bool valid = false;                      ///< true when no blocking reasons were recorded
    std::string local;                       ///< part before '@'; empty when not parseable
    std::string domain;                      ///< part after '@'; empty when not parseable
    std::optional<std::string> normalized;   ///< "local@lowercased-domain", when both parts exist
    std::vector<std::string> reasons;        ///< blocking problems (`valid` is true iff empty)
    std::vector<std::string> warnings;       ///< non-blocking observations (rare forms, plus-tags)
};

namespace {

constexpr char kWhitespace[] = " \t\n\r\v\f";

/// ASCII-only letter/digit tests (std::isdigit etc. are locale-sensitive and
/// take int; direct comparisons keep the classes strictly ASCII).
constexpr bool IsAsciiAlpha(unsigned char c) {
    return (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z');
}
constexpr bool IsAsciiDigit(unsigned char c) { return c >= '0' && c <= '9'; }
constexpr bool IsAsciiAlnum(unsigned char c) { return IsAsciiAlpha(c) || IsAsciiDigit(c); }

/// True when every byte of `s` belongs to the RFC-style "atom" character set
/// (ASCII alphanumeric plus the printable specials permitted unquoted).
/// Iterating bytes is correct here because the class is strictly ASCII.
bool IsAtomLocal(std::string_view s) {
    if (s.empty()) return false;
    for (char ch : s) {
        unsigned char c = static_cast<unsigned char>(ch);
        bool special = c == '.' || c == '!' || c == '#' || c == '$' || c == '%' ||
                       c == '&' || c == '\'' || c == '*' || c == '+' || c == '/' ||
                       c == '=' || c == '?' || c == '^' || c == '_' || c == '`' ||
                       c == '{' || c == '|' || c == '}' || c == '~' || c == '-';
        if (!IsAsciiAlnum(c) && !special) return false;
    }
    return true;
}

/// A valid domain label: ASCII letters, digits, and hyphens (non-empty).
bool IsValidLabel(std::string_view s) {
    if (s.empty()) return false;
    for (char ch : s) {
        unsigned char c = static_cast<unsigned char>(ch);
        if (!IsAsciiAlnum(c) && c != '-') return false;
    }
    return true;
}

/// A valid TLD: two or more ASCII letters. Byte length equals char count when
/// every byte is ASCII alphabetic, so the length check is exact.
bool IsValidTld(std::string_view s) {
    if (s.size() < 2) return false;
    for (char ch : s) {
        if (!IsAsciiAlpha(static_cast<unsigned char>(ch))) return false;
    }
    return true;
}

/// An all-decimal, non-empty octet string.
bool IsDecimal(std::string_view s) {
    if (s.empty()) return false;
    for (char ch : s) {
        if (!IsAsciiDigit(static_cast<unsigned char>(ch))) return false;
    }
    return true;
}

/// True when `s` starts with "ipv6:" (ASCII case-insensitive).
bool IsIpv6Literal(std::string_view s) {
    return s.size() >= 5 && s.substr(0, 5) == "ipv6:";
}

/// True when `s` is a dotted-quad: four octets, each 0-255, with no leading
/// zeros. The 3-digit length cap rejects arbitrarily long digit strings
/// before they can overflow the value parse (equivalent to the reference's
/// overflow-on-cast behaviour).
bool IsIpv4(std::string_view s) {
    std::vector<std::string_view> parts;
    std::size_t start = 0;
    while (true) {
        std::size_t dot = s.find('.', start);
        if (dot == std::string_view::npos) {
            parts.push_back(s.substr(start));
            break;
        }
        parts.push_back(s.substr(start, dot - start));
        start = dot + 1;
    }
    if (parts.size() != 4) return false;
    for (std::string_view p : parts) {
        if (p.empty() || p.size() > 3 || !IsDecimal(p)) return false;
        int value = 0;
        for (char ch : p) value = value * 10 + (ch - '0');
        if (value > 255) return false;
        if (p.size() > 1 && p[0] == '0') return false;  // leading zero ("01")
    }
    return true;
}

/// Internal split result: borrowed views into the input address.
struct Split {
    std::string_view local;
    std::string_view domain;
    bool quoted;
};

/// Splits an address into local + domain, honouring a quoted ("...") local
/// part. Returns std::nullopt when the address cannot be split into exactly
/// one '@' in the right place.
std::optional<Split> SplitLocalDomain(std::string_view email) {
    if (!email.empty() && email.front() == '"') {
        // Walk the quoted string; a backslash escapes the next byte (so `\"`
        // does not terminate the quote). We only branch on ASCII delimiters,
        // so byte-indexing is safe; all slice bounds land on ASCII chars.
        std::size_t i = 1;
        while (i < email.size()) {
            char ch = email[i];
            if (ch == '\\') {
                i += 2;
                continue;
            }
            if (ch == '"') break;
            ++i;
        }
        if (i >= email.size() || email[i] != '"') return std::nullopt;  // unterminated quote
        std::size_t at = i + 1;
        if (at >= email.size() || email[at] != '@') return std::nullopt;  // '@' must follow quote
        if (email.substr(at + 1).find('@') != std::string_view::npos) {
            return std::nullopt;  // stray '@' inside the domain
        }
        return Split{email.substr(0, at), email.substr(at + 1), true};
    }
    std::size_t first = email.find('@');
    if (first == std::string_view::npos) return std::nullopt;
    if (email.substr(first + 1).find('@') != std::string_view::npos) {
        return std::nullopt;  // multiple '@'
    }
    return Split{email.substr(0, first), email.substr(first + 1), false};
}

/// Appends domain-level problems to `reasons` / `warnings`.
void ValidateDomain(std::string_view domain, std::vector<std::string>& reasons,
                    std::vector<std::string>& warnings) {
    if (domain.empty()) {
        reasons.emplace_back("Domain is empty");
        return;
    }
    if (domain.size() > kDomainMax) {
        reasons.push_back("Domain exceeds " + std::to_string(kDomainMax) + " characters");
    }

    // IP-literal domain: [1.2.3.4] or [IPv6:...].
    if (domain.front() == '[' && domain.back() == ']') {
        std::string_view inner = domain.substr(1, domain.size() - 2);
        if (IsIpv6Literal(inner)) {
            warnings.emplace_back(
                "IPv6 literal domain (uncommon; ensure your provider supports it)");
            return;
        }
        if (IsIpv4(inner)) {
            warnings.emplace_back(
                "IP-literal domain (uncommon; ensure your provider supports it)");
            return;
        }
        reasons.emplace_back("Invalid IP-literal domain");
        return;
    }
    if (domain.front() == '[' || domain.back() == ']') {
        reasons.emplace_back("Malformed IP-literal domain (unmatched brackets)");
        return;
    }

    if (domain.find('.') == std::string_view::npos) {
        reasons.emplace_back("Domain must contain at least one dot (e.g. example.com)");
        return;
    }

    std::vector<std::string_view> labels;
    std::size_t start = 0;
    while (true) {
        std::size_t dot = domain.find('.', start);
        if (dot == std::string_view::npos) {
            labels.push_back(domain.substr(start));
            break;
        }
        labels.push_back(domain.substr(start, dot - start));
        start = dot + 1;
    }
    for (std::string_view label : labels) {
        if (label.empty()) {
            reasons.emplace_back(
                "Domain contains an empty label (consecutive or trailing dots)");
            continue;
        }
        if (label.size() > 63) {
            reasons.emplace_back("Domain label exceeds 63 characters");
        }
        if (!IsValidLabel(label)) {
            reasons.emplace_back("Domain label contains invalid characters");
        }
        if (label.front() == '-' || label.back() == '-') {
            reasons.emplace_back("Domain label starts or ends with a hyphen");
        }
    }
    // The TLD is the final label; require >=2 ASCII letters so bare hostnames
    // and numeric tails are rejected.
    std::string_view tld = labels.back();
    if (!IsValidTld(tld)) {
        reasons.emplace_back("Top-level domain must be at least two letters");
    }
}

}  // namespace

/// Validates a single email address, returning a structured verdict.
///
/// Pure and deterministic: every malformed input becomes a non-valid result
/// carrying explanatory `reasons`.
EmailResult ValidateEmail(std::string_view raw) {
    EmailResult result;
    std::string_view email = raw;
    {
        std::size_t b = email.find_first_not_of(kWhitespace);
        if (b == std::string_view::npos) {
            email = std::string_view{};
        } else {
            std::size_t e = email.find_last_not_of(kWhitespace);
            email = email.substr(b, e - b + 1);
        }
    }

    if (email.empty()) {
        result.reasons.emplace_back("Email is empty");
        return result;
    }

    if (email.size() > kTotalMax) {
        result.reasons.push_back(
            "Email exceeds maximum length of " + std::to_string(kTotalMax) + " characters");
    }

    std::optional<Split> split = SplitLocalDomain(email);
    if (!split.has_value()) {
        result.reasons.emplace_back(
            "Email must contain exactly one \"@\" separating local part and domain");
        return result;
    }
    auto [local, domain, quoted] = *split;

    if (quoted) {
        // Quoted local parts are RFC-legal but almost universally rejected by
        // mailbox providers — warn, and only length-check structurally.
        if (local.size() > kLocalMax) {
            result.reasons.push_back(
                "Local part exceeds " + std::to_string(kLocalMax) + " characters");
        }
        result.warnings.emplace_back("Quoted local part (rarely supported by providers)");
    } else if (local.empty()) {
        result.reasons.emplace_back("Local part is empty");
    } else {
        if (local.size() > kLocalMax) {
            result.reasons.push_back(
                "Local part exceeds " + std::to_string(kLocalMax) + " characters");
        }
        if (local.front() == '.' || local.back() == '.') {
            result.reasons.emplace_back("Local part starts or ends with a dot");
        }
        if (local.find("..") != std::string_view::npos) {
            result.reasons.emplace_back("Local part contains consecutive dots");
        }
        if (!IsAtomLocal(local)) {
            result.reasons.emplace_back("Local part contains invalid characters");
        }
    }
    // Plus-addressing (`user+tag@`) is valid and delivers to the base mailbox,
    // but callers filtering on exact address may want to know.
    if (!quoted && local.find('+') != std::string_view::npos) {
        result.warnings.emplace_back(
            "Plus-addressing (tag) detected — delivers to the base mailbox");
    }

    ValidateDomain(domain, result.reasons, result.warnings);

    result.valid = result.reasons.empty();
    if (!local.empty() && !domain.empty()) {
        std::string lowered(domain);
        for (char& c : lowered) {
            if (c >= 'A' && c <= 'Z') c = static_cast<char>(c - 'A' + 'a');
        }
        result.normalized = std::string(local) + "@" + lowered;
    }
    result.local = std::string(local);
    result.domain = std::string(domain);
    return result;
}

/// Validates many addresses — one per line. Blank or whitespace-only lines
/// are skipped. Line endings may be LF or CRLF (matching the reference's
/// `\r?\n` split).
std::vector<EmailResult> ValidateBatch(std::string_view input) {
    std::vector<EmailResult> out;
    if (input.empty()) return out;

    std::size_t start = 0;
    while (start <= input.size()) {
        std::size_t nl = input.find('\n', start);
        std::size_t end = (nl == std::string_view::npos) ? input.size() : nl;
        if (end > start && input[end - 1] == '\r') --end;  // strip CRLF's '\r'

        std::string_view line = input.substr(start, end - start);
        std::size_t b = line.find_first_not_of(kWhitespace);
        if (b != std::string_view::npos) {
            std::size_t e = line.find_last_not_of(kWhitespace);
            out.push_back(ValidateEmail(line.substr(b, e - b + 1)));
        }
        if (nl == std::string_view::npos) break;
        start = nl + 1;
    }
    return out;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →