Skip to content

Email Validator — Python source

Validate email addresses one at a time or in bulk. Checks syntax, length limits, local-part and domain rules, plus-addressing, and IP-literal domains - all in your browser.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""email-validator — Python polyglot showcase port.

Language: Python (3.9+, or 3.7+ via ``from __future__ import annotations``).

CosmoDev polyglot showcase: the pure logic of the email-validator tool, ported
from ``src/lib/email-validator.ts``. This file is display source — part of
CosmoDev's polyglot tool pages, where each tool's logic is shown side by side
in several languages.

RFC 5321/5322-inspired email validation. Pure, deterministic, never raises.
Errs on the side of practical deliverability (provider-friendly) while still
recognising the legal-but-unusual forms (quoted local parts, IP-literal
domains).

Self-contained: standard library only (``re``).
"""

from __future__ import annotations

import re
from dataclasses import dataclass, field
from typing import List, Optional, Tuple

# RFC-inspired length ceilings: local part, domain, and total address.
LOCAL_MAX = 64
DOMAIN_MAX = 253
TOTAL_MAX = 320

# Pre-compiled character classes (compiled once at import).
_LOCAL_CHARS = re.compile(r"^[A-Za-z0-9.!#$%&'*+/=?^_`{|}~-]+$")  # printable-ASCII "atom" set
_LABEL_CHARS = re.compile(r"^[A-Za-z0-9-]+$")                      # letters, digits, hyphens
_TLD_CHARS = re.compile(r"^[A-Za-z]{2,}$")                         # TLD: >=2 ASCII letters
_DECIMAL = re.compile(r"^\d+$")                                    # all-decimal, non-empty
_IPV6_PREFIX = re.compile(r"^ipv6:", re.IGNORECASE)                # case-insensitive "ipv6:" tag


@dataclass
class EmailResult:
    """Structured verdict for an email address.

    ``valid`` is true iff ``reasons`` is empty; ``warnings`` never affect
    validity. ``local`` / ``domain`` / ``normalized`` are ``None`` when the
    address could not be split into parts.
    """

    valid: bool
    local: Optional[str] = None
    domain: Optional[str] = None
    normalized: Optional[str] = None
    reasons: List[str] = field(default_factory=list)
    warnings: List[str] = field(default_factory=list)


def _split_local_domain(email: str) -> Optional[Tuple[str, str, bool]]:
    """Split an email into (local, domain, quoted), honouring a quoted local part.

    Returns ``None`` when the address cannot be split into exactly one '@' in
    the right place.
    """
    if email.startswith('"'):
        # Walk the quoted string; a backslash escapes the next character (so
        # `\"` does not terminate the quote).
        i = 1
        n = len(email)
        while i < n:
            ch = email[i]
            if ch == "\\":
                i += 2
                continue
            if ch == '"':
                break
            i += 1
        if i >= n or email[i] != '"':
            return None  # unterminated quote
        at = i + 1
        if at >= n or email[at] != "@":
            return None  # '@' must immediately follow the closing quote
        if email.find("@", at + 1) != -1:
            return None  # stray '@' inside the domain
        return email[:at], email[at + 1:], True

    first = email.find("@")
    if first == -1:
        return None
    if email.find("@", first + 1) != -1:
        return None  # multiple '@'
    return email[:first], email[first + 1:], False


def _is_ipv4(s: str) -> bool:
    """True when ``s`` is a dotted-quad: four octets, each 0-255, no leading zeros.

    Python ints are unbounded, so huge digit strings parse cleanly but exceed
    255; comparing back to the canonical decimal form rejects leading zeros.
    """
    parts = s.split(".")
    if len(parts) != 4:
        return False
    for p in parts:
        if not _DECIMAL.match(p):
            return False
        n = int(p)
        if n < 0 or n > 255 or str(n) != p:
            return False
    return True


def _validate_domain(domain: str, reasons: List[str], warnings: List[str]) -> None:
    """Append domain-level problems to ``reasons`` / ``warnings`` in place."""
    if not domain:
        reasons.append("Domain is empty")
        return
    if len(domain) > DOMAIN_MAX:
        reasons.append(f"Domain exceeds {DOMAIN_MAX} characters")

    # IP-literal domain: [1.2.3.4] or [IPv6:...].
    if domain.startswith("[") and domain.endswith("]"):
        inner = domain[1:-1]
        if _IPV6_PREFIX.match(inner):
            warnings.append("IPv6 literal domain (uncommon; ensure your provider supports it)")
            return
        if _is_ipv4(inner):
            warnings.append("IP-literal domain (uncommon; ensure your provider supports it)")
            return
        reasons.append("Invalid IP-literal domain")
        return
    if domain.startswith("[") or domain.endswith("]"):
        reasons.append("Malformed IP-literal domain (unmatched brackets)")
        return

    if "." not in domain:
        reasons.append("Domain must contain at least one dot (e.g. example.com)")
        return

    labels = domain.split(".")
    for label in labels:
        if not label:
            reasons.append("Domain contains an empty label (consecutive or trailing dots)")
            continue
        if len(label) > 63:
            reasons.append("Domain label exceeds 63 characters")
        if not _LABEL_CHARS.match(label):
            reasons.append("Domain label contains invalid characters")
        if label.startswith("-") or label.endswith("-"):
            reasons.append("Domain label starts or ends with a hyphen")
    # The TLD is the final label; require >=2 ASCII letters so bare hostnames
    # and numeric tails are rejected.
    tld = labels[-1]
    if not _TLD_CHARS.match(tld):
        reasons.append("Top-level domain must be at least two letters")


def validate_email(raw: str) -> EmailResult:
    """Validate a single email address; returns a structured verdict, never raises."""
    reasons: List[str] = []
    warnings: List[str] = []
    email = raw.strip()

    if not email:
        return EmailResult(valid=False, reasons=["Email is empty"], warnings=warnings)

    if len(email) > TOTAL_MAX:
        reasons.append(f"Email exceeds maximum length of {TOTAL_MAX} characters")

    split = _split_local_domain(email)
    if split is None:
        reasons.append('Email must contain exactly one "@" separating local part and domain')
        return EmailResult(valid=False, reasons=reasons, warnings=warnings)

    local, domain, quoted = split

    if quoted:
        # Quoted local parts are RFC-legal but almost universally rejected by
        # mailbox providers — warn, and only length-check structurally.
        if len(local) > LOCAL_MAX:
            reasons.append(f"Local part exceeds {LOCAL_MAX} characters")
        warnings.append("Quoted local part (rarely supported by providers)")
    elif not local:
        reasons.append("Local part is empty")
    else:
        if len(local) > LOCAL_MAX:
            reasons.append(f"Local part exceeds {LOCAL_MAX} characters")
        if local.startswith(".") or local.endswith("."):
            reasons.append("Local part starts or ends with a dot")
        if ".." in local:
            reasons.append("Local part contains consecutive dots")
        if not _LOCAL_CHARS.match(local):
            reasons.append("Local part contains invalid characters")
    # Plus-addressing (`user+tag@`) is valid and delivers to the base mailbox,
    # but callers filtering on exact address may want to know.
    if not quoted and "+" in local:
        warnings.append("Plus-addressing (tag) detected — delivers to the base mailbox")

    _validate_domain(domain, reasons, warnings)

    valid = len(reasons) == 0
    return EmailResult(
        valid=valid,
        local=local,
        domain=domain,
        normalized=f"{local}@{domain.lower()}" if local and domain else None,
        reasons=reasons,
        warnings=warnings,
    )


def validate_batch(text: str) -> List[EmailResult]:
    """Validate many emails (one per line); blank/whitespace-only lines are skipped.

    Line endings may be LF or CRLF (matching the reference's ``\\r?\\n`` split).
    """
    if not text:
        return []
    out: List[EmailResult] = []
    for line in re.split(r"\r?\n", text):
        line = line.strip()
        if line:
            out.append(validate_email(line))
    return out

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →