Email Validator — C++ source
Validate email addresses one at a time or in bulk. Checks syntax, length limits, local-part and domain rules, plus-addressing, and IP-literal domains - all in your browser.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// email-validator — RFC 5321/5322-inspired email validation.
//
// Language: C++ (C++17, standard library only)
// Source: CosmoDev polyglot showcase port of the Email Validator tool,
// ported from src/lib/email-validator.ts (the canonical TypeScript
// implementation).
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Practical, provider-friendly validation: errs on the side of deliverability
// while still recognising the legal-but-unusual forms (quoted local parts,
// IP-literal domains). Pure and deterministic — every malformed input becomes
// a non-valid verdict carrying explanatory reasons; nothing below throws.
//
// Self-contained: stdlib only — the character classes used by the original
// are implemented as small byte predicates, avoiding any external regex
// dependency.
#include <cstddef>
#include <optional>
#include <string>
#include <string_view>
#include <vector>
/// RFC-inspired length ceilings: local part, domain, total address.
constexpr std::size_t kLocalMax = 64;
constexpr std::size_t kDomainMax = 253;
constexpr std::size_t kTotalMax = 320;
/// The structured verdict returned by `ValidateEmail`.
struct EmailResult {
bool valid = false; ///< true when no blocking reasons were recorded
std::string local; ///< part before '@'; empty when not parseable
std::string domain; ///< part after '@'; empty when not parseable
std::optional<std::string> normalized; ///< "local@lowercased-domain", when both parts exist
std::vector<std::string> reasons; ///< blocking problems (`valid` is true iff empty)
std::vector<std::string> warnings; ///< non-blocking observations (rare forms, plus-tags)
};
namespace {
constexpr char kWhitespace[] = " \t\n\r\v\f";
/// ASCII-only letter/digit tests (std::isdigit etc. are locale-sensitive and
/// take int; direct comparisons keep the classes strictly ASCII).
constexpr bool IsAsciiAlpha(unsigned char c) {
return (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z');
}
constexpr bool IsAsciiDigit(unsigned char c) { return c >= '0' && c <= '9'; }
constexpr bool IsAsciiAlnum(unsigned char c) { return IsAsciiAlpha(c) || IsAsciiDigit(c); }
/// True when every byte of `s` belongs to the RFC-style "atom" character set
/// (ASCII alphanumeric plus the printable specials permitted unquoted).
/// Iterating bytes is correct here because the class is strictly ASCII.
bool IsAtomLocal(std::string_view s) {
if (s.empty()) return false;
for (char ch : s) {
unsigned char c = static_cast<unsigned char>(ch);
bool special = c == '.' || c == '!' || c == '#' || c == '$' || c == '%' ||
c == '&' || c == '\'' || c == '*' || c == '+' || c == '/' ||
c == '=' || c == '?' || c == '^' || c == '_' || c == '`' ||
c == '{' || c == '|' || c == '}' || c == '~' || c == '-';
if (!IsAsciiAlnum(c) && !special) return false;
}
return true;
}
/// A valid domain label: ASCII letters, digits, and hyphens (non-empty).
bool IsValidLabel(std::string_view s) {
if (s.empty()) return false;
for (char ch : s) {
unsigned char c = static_cast<unsigned char>(ch);
if (!IsAsciiAlnum(c) && c != '-') return false;
}
return true;
}
/// A valid TLD: two or more ASCII letters. Byte length equals char count when
/// every byte is ASCII alphabetic, so the length check is exact.
bool IsValidTld(std::string_view s) {
if (s.size() < 2) return false;
for (char ch : s) {
if (!IsAsciiAlpha(static_cast<unsigned char>(ch))) return false;
}
return true;
}
/// An all-decimal, non-empty octet string.
bool IsDecimal(std::string_view s) {
if (s.empty()) return false;
for (char ch : s) {
if (!IsAsciiDigit(static_cast<unsigned char>(ch))) return false;
}
return true;
}
/// True when `s` starts with "ipv6:" (ASCII case-insensitive).
bool IsIpv6Literal(std::string_view s) {
return s.size() >= 5 && s.substr(0, 5) == "ipv6:";
}
/// True when `s` is a dotted-quad: four octets, each 0-255, with no leading
/// zeros. The 3-digit length cap rejects arbitrarily long digit strings
/// before they can overflow the value parse (equivalent to the reference's
/// overflow-on-cast behaviour).
bool IsIpv4(std::string_view s) {
std::vector<std::string_view> parts;
std::size_t start = 0;
while (true) {
std::size_t dot = s.find('.', start);
if (dot == std::string_view::npos) {
parts.push_back(s.substr(start));
break;
}
parts.push_back(s.substr(start, dot - start));
start = dot + 1;
}
if (parts.size() != 4) return false;
for (std::string_view p : parts) {
if (p.empty() || p.size() > 3 || !IsDecimal(p)) return false;
int value = 0;
for (char ch : p) value = value * 10 + (ch - '0');
if (value > 255) return false;
if (p.size() > 1 && p[0] == '0') return false; // leading zero ("01")
}
return true;
}
/// Internal split result: borrowed views into the input address.
struct Split {
std::string_view local;
std::string_view domain;
bool quoted;
};
/// Splits an address into local + domain, honouring a quoted ("...") local
/// part. Returns std::nullopt when the address cannot be split into exactly
/// one '@' in the right place.
std::optional<Split> SplitLocalDomain(std::string_view email) {
if (!email.empty() && email.front() == '"') {
// Walk the quoted string; a backslash escapes the next byte (so `\"`
// does not terminate the quote). We only branch on ASCII delimiters,
// so byte-indexing is safe; all slice bounds land on ASCII chars.
std::size_t i = 1;
while (i < email.size()) {
char ch = email[i];
if (ch == '\\') {
i += 2;
continue;
}
if (ch == '"') break;
++i;
}
if (i >= email.size() || email[i] != '"') return std::nullopt; // unterminated quote
std::size_t at = i + 1;
if (at >= email.size() || email[at] != '@') return std::nullopt; // '@' must follow quote
if (email.substr(at + 1).find('@') != std::string_view::npos) {
return std::nullopt; // stray '@' inside the domain
}
return Split{email.substr(0, at), email.substr(at + 1), true};
}
std::size_t first = email.find('@');
if (first == std::string_view::npos) return std::nullopt;
if (email.substr(first + 1).find('@') != std::string_view::npos) {
return std::nullopt; // multiple '@'
}
return Split{email.substr(0, first), email.substr(first + 1), false};
}
/// Appends domain-level problems to `reasons` / `warnings`.
void ValidateDomain(std::string_view domain, std::vector<std::string>& reasons,
std::vector<std::string>& warnings) {
if (domain.empty()) {
reasons.emplace_back("Domain is empty");
return;
}
if (domain.size() > kDomainMax) {
reasons.push_back("Domain exceeds " + std::to_string(kDomainMax) + " characters");
}
// IP-literal domain: [1.2.3.4] or [IPv6:...].
if (domain.front() == '[' && domain.back() == ']') {
std::string_view inner = domain.substr(1, domain.size() - 2);
if (IsIpv6Literal(inner)) {
warnings.emplace_back(
"IPv6 literal domain (uncommon; ensure your provider supports it)");
return;
}
if (IsIpv4(inner)) {
warnings.emplace_back(
"IP-literal domain (uncommon; ensure your provider supports it)");
return;
}
reasons.emplace_back("Invalid IP-literal domain");
return;
}
if (domain.front() == '[' || domain.back() == ']') {
reasons.emplace_back("Malformed IP-literal domain (unmatched brackets)");
return;
}
if (domain.find('.') == std::string_view::npos) {
reasons.emplace_back("Domain must contain at least one dot (e.g. example.com)");
return;
}
std::vector<std::string_view> labels;
std::size_t start = 0;
while (true) {
std::size_t dot = domain.find('.', start);
if (dot == std::string_view::npos) {
labels.push_back(domain.substr(start));
break;
}
labels.push_back(domain.substr(start, dot - start));
start = dot + 1;
}
for (std::string_view label : labels) {
if (label.empty()) {
reasons.emplace_back(
"Domain contains an empty label (consecutive or trailing dots)");
continue;
}
if (label.size() > 63) {
reasons.emplace_back("Domain label exceeds 63 characters");
}
if (!IsValidLabel(label)) {
reasons.emplace_back("Domain label contains invalid characters");
}
if (label.front() == '-' || label.back() == '-') {
reasons.emplace_back("Domain label starts or ends with a hyphen");
}
}
// The TLD is the final label; require >=2 ASCII letters so bare hostnames
// and numeric tails are rejected.
std::string_view tld = labels.back();
if (!IsValidTld(tld)) {
reasons.emplace_back("Top-level domain must be at least two letters");
}
}
} // namespace
/// Validates a single email address, returning a structured verdict.
///
/// Pure and deterministic: every malformed input becomes a non-valid result
/// carrying explanatory `reasons`.
EmailResult ValidateEmail(std::string_view raw) {
EmailResult result;
std::string_view email = raw;
{
std::size_t b = email.find_first_not_of(kWhitespace);
if (b == std::string_view::npos) {
email = std::string_view{};
} else {
std::size_t e = email.find_last_not_of(kWhitespace);
email = email.substr(b, e - b + 1);
}
}
if (email.empty()) {
result.reasons.emplace_back("Email is empty");
return result;
}
if (email.size() > kTotalMax) {
result.reasons.push_back(
"Email exceeds maximum length of " + std::to_string(kTotalMax) + " characters");
}
std::optional<Split> split = SplitLocalDomain(email);
if (!split.has_value()) {
result.reasons.emplace_back(
"Email must contain exactly one \"@\" separating local part and domain");
return result;
}
auto [local, domain, quoted] = *split;
if (quoted) {
// Quoted local parts are RFC-legal but almost universally rejected by
// mailbox providers — warn, and only length-check structurally.
if (local.size() > kLocalMax) {
result.reasons.push_back(
"Local part exceeds " + std::to_string(kLocalMax) + " characters");
}
result.warnings.emplace_back("Quoted local part (rarely supported by providers)");
} else if (local.empty()) {
result.reasons.emplace_back("Local part is empty");
} else {
if (local.size() > kLocalMax) {
result.reasons.push_back(
"Local part exceeds " + std::to_string(kLocalMax) + " characters");
}
if (local.front() == '.' || local.back() == '.') {
result.reasons.emplace_back("Local part starts or ends with a dot");
}
if (local.find("..") != std::string_view::npos) {
result.reasons.emplace_back("Local part contains consecutive dots");
}
if (!IsAtomLocal(local)) {
result.reasons.emplace_back("Local part contains invalid characters");
}
}
// Plus-addressing (`user+tag@`) is valid and delivers to the base mailbox,
// but callers filtering on exact address may want to know.
if (!quoted && local.find('+') != std::string_view::npos) {
result.warnings.emplace_back(
"Plus-addressing (tag) detected — delivers to the base mailbox");
}
ValidateDomain(domain, result.reasons, result.warnings);
result.valid = result.reasons.empty();
if (!local.empty() && !domain.empty()) {
std::string lowered(domain);
for (char& c : lowered) {
if (c >= 'A' && c <= 'Z') c = static_cast<char>(c - 'A' + 'a');
}
result.normalized = std::string(local) + "@" + lowered;
}
result.local = std::string(local);
result.domain = std::string(domain);
return result;
}
/// Validates many addresses — one per line. Blank or whitespace-only lines
/// are skipped. Line endings may be LF or CRLF (matching the reference's
/// `\r?\n` split).
std::vector<EmailResult> ValidateBatch(std::string_view input) {
std::vector<EmailResult> out;
if (input.empty()) return out;
std::size_t start = 0;
while (start <= input.size()) {
std::size_t nl = input.find('\n', start);
std::size_t end = (nl == std::string_view::npos) ? input.size() : nl;
if (end > start && input[end - 1] == '\r') --end; // strip CRLF's '\r'
std::string_view line = input.substr(start, end - start);
std::size_t b = line.find_first_not_of(kWhitespace);
if (b != std::string_view::npos) {
std::size_t e = line.find_last_not_of(kWhitespace);
out.push_back(ValidateEmail(line.substr(b, e - b + 1)));
}
if (nl == std::string_view::npos) break;
start = nl + 1;
}
return out;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →