Email Validator — C source
Validate email addresses one at a time or in bulk. Checks syntax, length limits, local-part and domain rules, plus-addressing, and IP-literal domains - all in your browser.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/*
* email-validator — RFC 5321/5322-inspired email validation.
*
* Language: C (C11, standard library only)
* Source: CosmoDev polyglot showcase port of the Email Validator tool,
* ported from src/lib/email-validator.ts (the canonical TypeScript
* implementation).
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Practical, provider-friendly validation: errs on the side of deliverability
* while still recognising the legal-but-unusual forms (quoted local parts,
* IP-literal domains). Pure and deterministic — every malformed input becomes
* a non-valid verdict carrying explanatory reasons; no function below can
* crash on its input.
*
* Self-contained: the character classes used by the original are implemented
* as small byte predicates, avoiding any external regex dependency. ASCII
* case-insensitive compares are written by hand because strncasecmp() is
* POSIX, not C11.
*/
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
/* ------------------------------------------------------------- limits --- */
/* RFC-inspired length ceilings: local part, domain, total address. */
enum { LOCAL_MAX = 64, DOMAIN_MAX = 253, TOTAL_MAX = 320 };
/* --------------------------------------------------------------- types --- */
/* Owning list of heap-allocated strings. */
typedef struct {
char **items;
size_t count;
} StrList;
/* The structured verdict returned by validate_email(). Free with
* email_result_free(). `normalized` is NULL when the address could not be
* split into parts. */
typedef struct {
bool valid; /* true when no blocking reasons were recorded */
char *local; /* part before '@'; empty string when not parseable */
char *domain; /* part after '@'; empty string when not parseable */
char *normalized; /* "local@lowercased-domain", or NULL */
StrList reasons; /* blocking problems (`valid` is true iff empty) */
StrList warnings; /* non-blocking observations (rare forms, plus-tags) */
} EmailResult;
/* Owning batch of verdicts; free with email_results_free(). */
typedef struct {
EmailResult *items;
size_t count;
} EmailResults;
/* ------------------------------------------------------------ helpers --- */
/* strdup() is POSIX, not C11 — provide a portable equivalent. */
static char *xstrdup(const char *s) {
size_t n = strlen(s) + 1;
char *copy = malloc(n);
if (copy == NULL) {
abort(); /* allocation failure is fatal in this display port */
}
memcpy(copy, s, n);
return copy;
}
/* Length-bounded duplicate, for views into a larger buffer. */
static char *xstrndup(const char *s, size_t n) {
char *copy = malloc(n + 1);
if (copy == NULL) {
abort(); /* allocation failure is fatal in this display port */
}
memcpy(copy, s, n);
copy[n] = '\0';
return copy;
}
static void str_list_push(StrList *list, const char *s) {
char **grown = realloc(list->items, (list->count + 1) * sizeof *grown);
if (grown == NULL) {
abort(); /* allocation failure is fatal in this display port */
}
list->items = grown;
list->items[list->count++] = xstrdup(s);
}
/* Append a formatted message (used for the length-limit reasons). */
static void str_list_push_fmt(StrList *list, const char *fmt, int limit) {
char buf[96];
snprintf(buf, sizeof buf, fmt, limit);
str_list_push(list, buf);
}
void email_result_free(EmailResult *r) {
free(r->local);
free(r->domain);
free(r->normalized);
for (size_t i = 0; i < r->reasons.count; i++) free(r->reasons.items[i]);
for (size_t i = 0; i < r->warnings.count; i++) free(r->warnings.items[i]);
free(r->reasons.items);
free(r->warnings.items);
memset(r, 0, sizeof *r);
}
void email_results_free(EmailResults *rs) {
for (size_t i = 0; i < rs->count; i++) email_result_free(&rs->items[i]);
free(rs->items);
rs->items = NULL;
rs->count = 0;
}
/* True when `needle` occurs in the first `n` bytes of `hay` (memmem() is
* POSIX, not C11, so this is hand-rolled). */
static bool contains(const char *hay, size_t n, const char *needle) {
size_t m = strlen(needle);
if (m == 0) return true;
for (size_t i = 0; i + m <= n; i++) {
if (memcmp(hay + i, needle, m) == 0) return true;
}
return false;
}
/* Strip ASCII leading/trailing whitespace; returns the trimmed length and
* points *out into the original buffer. */
static size_t trim(const char *raw, const char **out) {
static const char ws[] = " \t\n\r\v\f";
size_t n = strlen(raw);
size_t b = 0, e = n;
while (b < e && raw[b] != '\0' && strchr(ws, raw[b]) != NULL) b++;
while (e > b && strchr(ws, raw[e - 1]) != NULL) e--;
*out = raw + b;
return e - b;
}
/* --------------------------------------------------------- predicates --- */
/* True when every byte of `s` belongs to the RFC-style "atom" character set
* (ASCII alphanumeric plus the printable specials permitted unquoted).
* Iterating bytes is correct here because the class is strictly ASCII. */
static bool is_atom_local(const char *s, size_t n) {
static const char specials[] = ".!#$%&'*+/=?^_`{|}~-";
if (n == 0) return false;
for (size_t i = 0; i < n; i++) {
unsigned char c = (unsigned char)s[i];
bool alnum = (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') ||
(c >= '0' && c <= '9');
/* c is never '\0' inside the loop (inputs are NUL-terminated), so
* strchr() cannot false-match the terminator. */
if (!alnum && strchr(specials, c) == NULL) return false;
}
return true;
}
/* A valid domain label: ASCII letters, digits, and hyphens (non-empty). */
static bool is_valid_label(const char *s, size_t n) {
if (n == 0) return false;
for (size_t i = 0; i < n; i++) {
unsigned char c = (unsigned char)s[i];
bool alnum = (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') ||
(c >= '0' && c <= '9');
if (!alnum && c != '-') return false;
}
return true;
}
/* A valid TLD: two or more ASCII letters. Byte length equals char count when
* every byte is ASCII alphabetic, so the length check is exact. */
static bool is_valid_tld(const char *s, size_t n) {
if (n < 2) return false;
for (size_t i = 0; i < n; i++) {
unsigned char c = (unsigned char)s[i];
if (!((c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z'))) return false;
}
return true;
}
/* An all-decimal, non-empty octet string. */
static bool is_decimal(const char *s, size_t n) {
if (n == 0) return false;
for (size_t i = 0; i < n; i++) {
if (s[i] < '0' || s[i] > '9') return false;
}
return true;
}
/* True when `s` starts with "ipv6:" (case-insensitive, hand-rolled because
* strncasecmp() is POSIX, not C11). */
static bool is_ipv6_literal(const char *s, size_t n) {
static const char tag[] = "ipv6:";
if (n < 5) return false;
for (int i = 0; i < 5; i++) {
char c = s[i];
if (c >= 'A' && c <= 'Z') c = (char)(c - 'A' + 'a');
if (c != tag[i]) return false;
}
return true;
}
/* True when `s` is a dotted-quad: four octets, each 0-255, with no leading
* zeros. The 3-digit length cap rejects arbitrarily long digit strings before
* they can overflow the value parse (equivalent to the reference's
* overflow-on-cast behaviour). */
static bool is_ipv4(const char *s, size_t n) {
int parts = 1;
for (size_t i = 0; i < n; i++) {
if (s[i] == '.') parts++;
}
if (parts != 4) return false;
size_t i = 0;
while (i <= n) { /* one part per iteration, incl. the final partial one */
size_t j = i;
while (j < n && s[j] != '.') j++;
size_t len = j - i;
if (len == 0 || len > 3) return false;
if (!is_decimal(s + i, len)) return false;
int value = 0;
for (size_t k = i; k < j; k++) value = value * 10 + (s[k] - '0');
if (value > 255) return false;
if (len > 1 && s[i] == '0') return false; /* leading zero ("01") */
if (j >= n) break;
i = j + 1;
}
return true;
}
/* ---------------------------------------------------------------- split --- */
/* Internal split result: offsets into the input address. */
typedef struct {
size_t at; /* index of the separating '@' */
bool quoted; /* true when the local part is a quoted string */
} Split;
/* Splits an address into local + domain around the '@' at `at`, honouring a
* quoted ("...") local part. Returns false when the address cannot be split
* into exactly one '@' in the right place. */
static bool split_local_domain(const char *email, size_t n, Split *out) {
if (n > 0 && email[0] == '"') {
/* Walk the quoted string; a backslash escapes the next byte (so `\"`
* does not terminate the quote). We only branch on ASCII delimiters,
* so byte-indexing is safe; all offsets land on ASCII chars. */
size_t i = 1;
while (i < n) {
char ch = email[i];
if (ch == '\\') {
i += 2;
continue;
}
if (ch == '"') break;
i++;
}
if (i >= n || email[i] != '"') return false; /* unterminated quote */
size_t at = i + 1;
if (at >= n || email[at] != '@') return false; /* '@' must follow quote */
if (strchr(email + at + 1, '@') != NULL) return false; /* stray '@' */
out->at = at;
out->quoted = true;
return true;
}
const char *first = memchr(email, '@', n);
if (first == NULL) return false;
if (strchr(first + 1, '@') != NULL) return false; /* multiple '@' */
out->at = (size_t)(first - email);
out->quoted = false;
return true;
}
/* --------------------------------------------------------------- domain --- */
/* Appends domain-level problems to `reasons` / `warnings`. */
static void validate_domain(const char *domain, size_t n, StrList *reasons,
StrList *warnings) {
if (n == 0) {
str_list_push(reasons, "Domain is empty");
return;
}
if (n > DOMAIN_MAX) {
str_list_push_fmt(reasons, "Domain exceeds %d characters", DOMAIN_MAX);
}
/* IP-literal domain: [1.2.3.4] or [IPv6:...]. */
if (domain[0] == '[' && domain[n - 1] == ']') {
const char *inner = domain + 1;
size_t inner_n = n - 2;
if (is_ipv6_literal(inner, inner_n)) {
str_list_push(warnings,
"IPv6 literal domain (uncommon; ensure your provider supports it)");
return;
}
if (is_ipv4(inner, inner_n)) {
str_list_push(warnings,
"IP-literal domain (uncommon; ensure your provider supports it)");
return;
}
str_list_push(reasons, "Invalid IP-literal domain");
return;
}
if (domain[0] == '[' || domain[n - 1] == ']') {
str_list_push(reasons, "Malformed IP-literal domain (unmatched brackets)");
return;
}
if (memchr(domain, '.', n) == NULL) {
str_list_push(reasons,
"Domain must contain at least one dot (e.g. example.com)");
return;
}
for (size_t i = 0; i <= n;) { /* one label per iteration */
size_t j = i;
while (j < n && domain[j] != '.') j++;
size_t len = j - i;
if (len == 0) {
str_list_push(reasons,
"Domain contains an empty label (consecutive or trailing dots)");
} else {
if (len > 63) {
str_list_push(reasons, "Domain label exceeds 63 characters");
}
if (!is_valid_label(domain + i, len)) {
str_list_push(reasons, "Domain label contains invalid characters");
}
if (domain[i] == '-' || domain[j - 1] == '-') {
str_list_push(reasons, "Domain label starts or ends with a hyphen");
}
}
if (j >= n) break;
i = j + 1;
}
/* The TLD is the final label; require >=2 ASCII letters so bare hostnames
* and numeric tails are rejected. */
size_t tld_start = n;
while (tld_start > 0 && domain[tld_start - 1] != '.') tld_start--;
if (!is_valid_tld(domain + tld_start, n - tld_start)) {
str_list_push(reasons, "Top-level domain must be at least two letters");
}
}
/* ---------------------------------------------------------------- email --- */
/* Validates a single email address, returning a structured verdict. Pure and
* deterministic: every malformed input becomes a non-valid result carrying
* explanatory reasons. */
EmailResult validate_email(const char *raw) {
EmailResult r = {false, NULL, NULL, NULL, {NULL, 0}, {NULL, 0}};
r.local = xstrdup("");
r.domain = xstrdup("");
const char *email = NULL;
size_t n = trim(raw, &email);
if (n == 0) {
str_list_push(&r.reasons, "Email is empty");
return r;
}
if (n > TOTAL_MAX) {
str_list_push_fmt(&r.reasons, "Email exceeds maximum length of %d characters",
TOTAL_MAX);
}
Split split;
if (!split_local_domain(email, n, &split)) {
str_list_push(&r.reasons,
"Email must contain exactly one \"@\" separating local part and domain");
return r;
}
const char *local = email;
size_t local_n = split.at;
const char *domain = email + split.at + 1;
size_t domain_n = n - split.at - 1;
if (split.quoted) {
/* Quoted local parts are RFC-legal but almost universally rejected by
* mailbox providers — warn, and only length-check structurally. */
if (local_n > LOCAL_MAX) {
str_list_push_fmt(&r.reasons, "Local part exceeds %d characters",
LOCAL_MAX);
}
str_list_push(&r.warnings, "Quoted local part (rarely supported by providers)");
} else if (local_n == 0) {
str_list_push(&r.reasons, "Local part is empty");
} else {
if (local_n > LOCAL_MAX) {
str_list_push_fmt(&r.reasons, "Local part exceeds %d characters",
LOCAL_MAX);
}
if (local[0] == '.' || local[local_n - 1] == '.') {
str_list_push(&r.reasons, "Local part starts or ends with a dot");
}
if (contains(local, local_n, "..")) {
str_list_push(&r.reasons, "Local part contains consecutive dots");
}
if (!is_atom_local(local, local_n)) {
str_list_push(&r.reasons, "Local part contains invalid characters");
}
}
/* Plus-addressing (`user+tag@`) is valid and delivers to the base mailbox,
* but callers filtering on exact address may want to know. */
if (!split.quoted && memchr(local, '+', local_n) != NULL) {
str_list_push(&r.warnings,
"Plus-addressing (tag) detected — delivers to the base mailbox");
}
validate_domain(domain, domain_n, &r.reasons, &r.warnings);
free(r.local);
free(r.domain);
r.local = xstrndup(local, local_n);
r.domain = xstrndup(domain, domain_n);
r.valid = (r.reasons.count == 0);
if (local_n > 0 && domain_n > 0) {
/* ASCII lowercase (domains are ASCII in practice). */
size_t len = local_n + 1 + domain_n;
r.normalized = malloc(len + 1);
if (r.normalized == NULL) abort();
memcpy(r.normalized, local, local_n);
r.normalized[local_n] = '@';
for (size_t i = 0; i < domain_n; i++) {
char c = domain[i];
if (c >= 'A' && c <= 'Z') c = (char)(c - 'A' + 'a');
r.normalized[local_n + 1 + i] = c;
}
r.normalized[len] = '\0';
}
return r;
}
/* Validates many addresses — one per line. Blank or whitespace-only lines are
* skipped. Line endings may be LF or CRLF (matching the reference's "\r?\n"
* split). */
EmailResults validate_batch(const char *input) {
EmailResults out = {NULL, 0};
if (input == NULL || input[0] == '\0') return out;
size_t n = strlen(input);
size_t i = 0;
while (i < n) {
size_t j = i;
while (j < n && input[j] != '\n') j++;
size_t eol = j;
if (eol > i && input[eol - 1] == '\r') eol--; /* strip CRLF's '\r' */
/* Reuse validate_email() via a bounded line copy. */
size_t cap = eol - i + 1;
char *line = malloc(cap);
if (line == NULL) abort();
memcpy(line, input + i, eol - i);
line[eol - i] = '\0';
const char *trimmed = NULL;
if (trim(line, &trimmed) > 0) {
EmailResult *grown =
realloc(out.items, (out.count + 1) * sizeof *grown);
if (grown == NULL) abort();
out.items = grown;
out.items[out.count++] = validate_email(trimmed);
}
free(line);
if (j >= n) break;
i = j + 1;
}
return out;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →