Skip to content

Email Validator — C source

Validate email addresses one at a time or in bulk. Checks syntax, length limits, local-part and domain rules, plus-addressing, and IP-literal domains - all in your browser.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * email-validator — RFC 5321/5322-inspired email validation.
 *
 * Language: C (C11, standard library only)
 * Source:   CosmoDev polyglot showcase port of the Email Validator tool,
 *           ported from src/lib/email-validator.ts (the canonical TypeScript
 *           implementation).
 * License:  display source — part of CosmoDev's polyglot tool pages.
 *
 * Practical, provider-friendly validation: errs on the side of deliverability
 * while still recognising the legal-but-unusual forms (quoted local parts,
 * IP-literal domains). Pure and deterministic — every malformed input becomes
 * a non-valid verdict carrying explanatory reasons; no function below can
 * crash on its input.
 *
 * Self-contained: the character classes used by the original are implemented
 * as small byte predicates, avoiding any external regex dependency. ASCII
 * case-insensitive compares are written by hand because strncasecmp() is
 * POSIX, not C11.
 */

#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

/* ------------------------------------------------------------- limits --- */

/* RFC-inspired length ceilings: local part, domain, total address. */
enum { LOCAL_MAX = 64, DOMAIN_MAX = 253, TOTAL_MAX = 320 };

/* --------------------------------------------------------------- types --- */

/* Owning list of heap-allocated strings. */
typedef struct {
    char **items;
    size_t count;
} StrList;

/* The structured verdict returned by validate_email(). Free with
 * email_result_free(). `normalized` is NULL when the address could not be
 * split into parts. */
typedef struct {
    bool valid;        /* true when no blocking reasons were recorded */
    char *local;       /* part before '@'; empty string when not parseable */
    char *domain;      /* part after '@'; empty string when not parseable */
    char *normalized;  /* "local@lowercased-domain", or NULL */
    StrList reasons;   /* blocking problems (`valid` is true iff empty) */
    StrList warnings;  /* non-blocking observations (rare forms, plus-tags) */
} EmailResult;

/* Owning batch of verdicts; free with email_results_free(). */
typedef struct {
    EmailResult *items;
    size_t count;
} EmailResults;

/* ------------------------------------------------------------ helpers --- */

/* strdup() is POSIX, not C11 — provide a portable equivalent. */
static char *xstrdup(const char *s) {
    size_t n = strlen(s) + 1;
    char *copy = malloc(n);
    if (copy == NULL) {
        abort(); /* allocation failure is fatal in this display port */
    }
    memcpy(copy, s, n);
    return copy;
}

/* Length-bounded duplicate, for views into a larger buffer. */
static char *xstrndup(const char *s, size_t n) {
    char *copy = malloc(n + 1);
    if (copy == NULL) {
        abort(); /* allocation failure is fatal in this display port */
    }
    memcpy(copy, s, n);
    copy[n] = '\0';
    return copy;
}

static void str_list_push(StrList *list, const char *s) {
    char **grown = realloc(list->items, (list->count + 1) * sizeof *grown);
    if (grown == NULL) {
        abort(); /* allocation failure is fatal in this display port */
    }
    list->items = grown;
    list->items[list->count++] = xstrdup(s);
}

/* Append a formatted message (used for the length-limit reasons). */
static void str_list_push_fmt(StrList *list, const char *fmt, int limit) {
    char buf[96];
    snprintf(buf, sizeof buf, fmt, limit);
    str_list_push(list, buf);
}

void email_result_free(EmailResult *r) {
    free(r->local);
    free(r->domain);
    free(r->normalized);
    for (size_t i = 0; i < r->reasons.count; i++) free(r->reasons.items[i]);
    for (size_t i = 0; i < r->warnings.count; i++) free(r->warnings.items[i]);
    free(r->reasons.items);
    free(r->warnings.items);
    memset(r, 0, sizeof *r);
}

void email_results_free(EmailResults *rs) {
    for (size_t i = 0; i < rs->count; i++) email_result_free(&rs->items[i]);
    free(rs->items);
    rs->items = NULL;
    rs->count = 0;
}

/* True when `needle` occurs in the first `n` bytes of `hay` (memmem() is
 * POSIX, not C11, so this is hand-rolled). */
static bool contains(const char *hay, size_t n, const char *needle) {
    size_t m = strlen(needle);
    if (m == 0) return true;
    for (size_t i = 0; i + m <= n; i++) {
        if (memcmp(hay + i, needle, m) == 0) return true;
    }
    return false;
}

/* Strip ASCII leading/trailing whitespace; returns the trimmed length and
 * points *out into the original buffer. */
static size_t trim(const char *raw, const char **out) {
    static const char ws[] = " \t\n\r\v\f";
    size_t n = strlen(raw);
    size_t b = 0, e = n;
    while (b < e && raw[b] != '\0' && strchr(ws, raw[b]) != NULL) b++;
    while (e > b && strchr(ws, raw[e - 1]) != NULL) e--;
    *out = raw + b;
    return e - b;
}

/* --------------------------------------------------------- predicates --- */

/* True when every byte of `s` belongs to the RFC-style "atom" character set
 * (ASCII alphanumeric plus the printable specials permitted unquoted).
 * Iterating bytes is correct here because the class is strictly ASCII. */
static bool is_atom_local(const char *s, size_t n) {
    static const char specials[] = ".!#$%&'*+/=?^_`{|}~-";
    if (n == 0) return false;
    for (size_t i = 0; i < n; i++) {
        unsigned char c = (unsigned char)s[i];
        bool alnum = (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') ||
                     (c >= '0' && c <= '9');
        /* c is never '\0' inside the loop (inputs are NUL-terminated), so
         * strchr() cannot false-match the terminator. */
        if (!alnum && strchr(specials, c) == NULL) return false;
    }
    return true;
}

/* A valid domain label: ASCII letters, digits, and hyphens (non-empty). */
static bool is_valid_label(const char *s, size_t n) {
    if (n == 0) return false;
    for (size_t i = 0; i < n; i++) {
        unsigned char c = (unsigned char)s[i];
        bool alnum = (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') ||
                     (c >= '0' && c <= '9');
        if (!alnum && c != '-') return false;
    }
    return true;
}

/* A valid TLD: two or more ASCII letters. Byte length equals char count when
 * every byte is ASCII alphabetic, so the length check is exact. */
static bool is_valid_tld(const char *s, size_t n) {
    if (n < 2) return false;
    for (size_t i = 0; i < n; i++) {
        unsigned char c = (unsigned char)s[i];
        if (!((c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z'))) return false;
    }
    return true;
}

/* An all-decimal, non-empty octet string. */
static bool is_decimal(const char *s, size_t n) {
    if (n == 0) return false;
    for (size_t i = 0; i < n; i++) {
        if (s[i] < '0' || s[i] > '9') return false;
    }
    return true;
}

/* True when `s` starts with "ipv6:" (case-insensitive, hand-rolled because
 * strncasecmp() is POSIX, not C11). */
static bool is_ipv6_literal(const char *s, size_t n) {
    static const char tag[] = "ipv6:";
    if (n < 5) return false;
    for (int i = 0; i < 5; i++) {
        char c = s[i];
        if (c >= 'A' && c <= 'Z') c = (char)(c - 'A' + 'a');
        if (c != tag[i]) return false;
    }
    return true;
}

/* True when `s` is a dotted-quad: four octets, each 0-255, with no leading
 * zeros. The 3-digit length cap rejects arbitrarily long digit strings before
 * they can overflow the value parse (equivalent to the reference's
 * overflow-on-cast behaviour). */
static bool is_ipv4(const char *s, size_t n) {
    int parts = 1;
    for (size_t i = 0; i < n; i++) {
        if (s[i] == '.') parts++;
    }
    if (parts != 4) return false;

    size_t i = 0;
    while (i <= n) { /* one part per iteration, incl. the final partial one */
        size_t j = i;
        while (j < n && s[j] != '.') j++;
        size_t len = j - i;
        if (len == 0 || len > 3) return false;
        if (!is_decimal(s + i, len)) return false;
        int value = 0;
        for (size_t k = i; k < j; k++) value = value * 10 + (s[k] - '0');
        if (value > 255) return false;
        if (len > 1 && s[i] == '0') return false; /* leading zero ("01") */
        if (j >= n) break;
        i = j + 1;
    }
    return true;
}

/* ---------------------------------------------------------------- split --- */

/* Internal split result: offsets into the input address. */
typedef struct {
    size_t at;    /* index of the separating '@' */
    bool quoted;  /* true when the local part is a quoted string */
} Split;

/* Splits an address into local + domain around the '@' at `at`, honouring a
 * quoted ("...") local part. Returns false when the address cannot be split
 * into exactly one '@' in the right place. */
static bool split_local_domain(const char *email, size_t n, Split *out) {
    if (n > 0 && email[0] == '"') {
        /* Walk the quoted string; a backslash escapes the next byte (so `\"`
         * does not terminate the quote). We only branch on ASCII delimiters,
         * so byte-indexing is safe; all offsets land on ASCII chars. */
        size_t i = 1;
        while (i < n) {
            char ch = email[i];
            if (ch == '\\') {
                i += 2;
                continue;
            }
            if (ch == '"') break;
            i++;
        }
        if (i >= n || email[i] != '"') return false; /* unterminated quote */
        size_t at = i + 1;
        if (at >= n || email[at] != '@') return false; /* '@' must follow quote */
        if (strchr(email + at + 1, '@') != NULL) return false; /* stray '@' */
        out->at = at;
        out->quoted = true;
        return true;
    }
    const char *first = memchr(email, '@', n);
    if (first == NULL) return false;
    if (strchr(first + 1, '@') != NULL) return false; /* multiple '@' */
    out->at = (size_t)(first - email);
    out->quoted = false;
    return true;
}

/* --------------------------------------------------------------- domain --- */

/* Appends domain-level problems to `reasons` / `warnings`. */
static void validate_domain(const char *domain, size_t n, StrList *reasons,
                            StrList *warnings) {
    if (n == 0) {
        str_list_push(reasons, "Domain is empty");
        return;
    }
    if (n > DOMAIN_MAX) {
        str_list_push_fmt(reasons, "Domain exceeds %d characters", DOMAIN_MAX);
    }

    /* IP-literal domain: [1.2.3.4] or [IPv6:...]. */
    if (domain[0] == '[' && domain[n - 1] == ']') {
        const char *inner = domain + 1;
        size_t inner_n = n - 2;
        if (is_ipv6_literal(inner, inner_n)) {
            str_list_push(warnings,
                "IPv6 literal domain (uncommon; ensure your provider supports it)");
            return;
        }
        if (is_ipv4(inner, inner_n)) {
            str_list_push(warnings,
                "IP-literal domain (uncommon; ensure your provider supports it)");
            return;
        }
        str_list_push(reasons, "Invalid IP-literal domain");
        return;
    }
    if (domain[0] == '[' || domain[n - 1] == ']') {
        str_list_push(reasons, "Malformed IP-literal domain (unmatched brackets)");
        return;
    }

    if (memchr(domain, '.', n) == NULL) {
        str_list_push(reasons,
            "Domain must contain at least one dot (e.g. example.com)");
        return;
    }

    for (size_t i = 0; i <= n;) { /* one label per iteration */
        size_t j = i;
        while (j < n && domain[j] != '.') j++;
        size_t len = j - i;
        if (len == 0) {
            str_list_push(reasons,
                "Domain contains an empty label (consecutive or trailing dots)");
        } else {
            if (len > 63) {
                str_list_push(reasons, "Domain label exceeds 63 characters");
            }
            if (!is_valid_label(domain + i, len)) {
                str_list_push(reasons, "Domain label contains invalid characters");
            }
            if (domain[i] == '-' || domain[j - 1] == '-') {
                str_list_push(reasons, "Domain label starts or ends with a hyphen");
            }
        }
        if (j >= n) break;
        i = j + 1;
    }

    /* The TLD is the final label; require >=2 ASCII letters so bare hostnames
     * and numeric tails are rejected. */
    size_t tld_start = n;
    while (tld_start > 0 && domain[tld_start - 1] != '.') tld_start--;
    if (!is_valid_tld(domain + tld_start, n - tld_start)) {
        str_list_push(reasons, "Top-level domain must be at least two letters");
    }
}

/* ---------------------------------------------------------------- email --- */

/* Validates a single email address, returning a structured verdict. Pure and
 * deterministic: every malformed input becomes a non-valid result carrying
 * explanatory reasons. */
EmailResult validate_email(const char *raw) {
    EmailResult r = {false, NULL, NULL, NULL, {NULL, 0}, {NULL, 0}};
    r.local = xstrdup("");
    r.domain = xstrdup("");

    const char *email = NULL;
    size_t n = trim(raw, &email);

    if (n == 0) {
        str_list_push(&r.reasons, "Email is empty");
        return r;
    }

    if (n > TOTAL_MAX) {
        str_list_push_fmt(&r.reasons, "Email exceeds maximum length of %d characters",
                          TOTAL_MAX);
    }

    Split split;
    if (!split_local_domain(email, n, &split)) {
        str_list_push(&r.reasons,
            "Email must contain exactly one \"@\" separating local part and domain");
        return r;
    }

    const char *local = email;
    size_t local_n = split.at;
    const char *domain = email + split.at + 1;
    size_t domain_n = n - split.at - 1;

    if (split.quoted) {
        /* Quoted local parts are RFC-legal but almost universally rejected by
         * mailbox providers — warn, and only length-check structurally. */
        if (local_n > LOCAL_MAX) {
            str_list_push_fmt(&r.reasons, "Local part exceeds %d characters",
                              LOCAL_MAX);
        }
        str_list_push(&r.warnings, "Quoted local part (rarely supported by providers)");
    } else if (local_n == 0) {
        str_list_push(&r.reasons, "Local part is empty");
    } else {
        if (local_n > LOCAL_MAX) {
            str_list_push_fmt(&r.reasons, "Local part exceeds %d characters",
                              LOCAL_MAX);
        }
        if (local[0] == '.' || local[local_n - 1] == '.') {
            str_list_push(&r.reasons, "Local part starts or ends with a dot");
        }
        if (contains(local, local_n, "..")) {
            str_list_push(&r.reasons, "Local part contains consecutive dots");
        }
        if (!is_atom_local(local, local_n)) {
            str_list_push(&r.reasons, "Local part contains invalid characters");
        }
    }
    /* Plus-addressing (`user+tag@`) is valid and delivers to the base mailbox,
     * but callers filtering on exact address may want to know. */
    if (!split.quoted && memchr(local, '+', local_n) != NULL) {
        str_list_push(&r.warnings,
            "Plus-addressing (tag) detected — delivers to the base mailbox");
    }

    validate_domain(domain, domain_n, &r.reasons, &r.warnings);

    free(r.local);
    free(r.domain);
    r.local = xstrndup(local, local_n);
    r.domain = xstrndup(domain, domain_n);

    r.valid = (r.reasons.count == 0);
    if (local_n > 0 && domain_n > 0) {
        /* ASCII lowercase (domains are ASCII in practice). */
        size_t len = local_n + 1 + domain_n;
        r.normalized = malloc(len + 1);
        if (r.normalized == NULL) abort();
        memcpy(r.normalized, local, local_n);
        r.normalized[local_n] = '@';
        for (size_t i = 0; i < domain_n; i++) {
            char c = domain[i];
            if (c >= 'A' && c <= 'Z') c = (char)(c - 'A' + 'a');
            r.normalized[local_n + 1 + i] = c;
        }
        r.normalized[len] = '\0';
    }
    return r;
}

/* Validates many addresses — one per line. Blank or whitespace-only lines are
 * skipped. Line endings may be LF or CRLF (matching the reference's "\r?\n"
 * split). */
EmailResults validate_batch(const char *input) {
    EmailResults out = {NULL, 0};
    if (input == NULL || input[0] == '\0') return out;

    size_t n = strlen(input);
    size_t i = 0;
    while (i < n) {
        size_t j = i;
        while (j < n && input[j] != '\n') j++;
        size_t eol = j;
        if (eol > i && input[eol - 1] == '\r') eol--; /* strip CRLF's '\r' */

        /* Reuse validate_email() via a bounded line copy. */
        size_t cap = eol - i + 1;
        char *line = malloc(cap);
        if (line == NULL) abort();
        memcpy(line, input + i, eol - i);
        line[eol - i] = '\0';

        const char *trimmed = NULL;
        if (trim(line, &trimmed) > 0) {
            EmailResult *grown =
                realloc(out.items, (out.count + 1) * sizeof *grown);
            if (grown == NULL) abort();
            out.items = grown;
            out.items[out.count++] = validate_email(trimmed);
        }
        free(line);
        if (j >= n) break;
        i = j + 1;
    }
    return out;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →