Skip to content

Email Header Analyzer — C++ source

Paste raw email headers to trace the message path, detect spoofing, and check SPF/DKIM/DMARC authentication results. Runs entirely in your browser.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Email Header Analyzer — RFC 5322 header parser + spoofing / authentication
// analysis.
//
// Language: C++ (C++17, standard library only)
// Ported from src/lib/email-header-analyzer.ts (the canonical TypeScript
// implementation). display source — part of CosmoDev's polyglot tool pages.
//
// Pure string parsing — no dependencies, no DOM, deterministic. Timestamps
// are epoch seconds (int64), the natural C++ counterpart of Date.getTime().
// RFC 2822 dates are parsed by hand: C++17 has no from_stream, and the fixed
// "Tue, 12 Feb 2026 10:30:00 +0100" shape is one regex plus a
// days-from-civil conversion.

#include <algorithm>
#include <cctype>
#include <cmath>
#include <cstdint>
#include <cstdio>
#include <optional>
#include <regex>
#include <string>
#include <vector>

namespace emailheaders {

struct Header {
  std::string name;
  std::string value;
};

struct Hop {
  std::string from;
  std::string by;
  std::optional<int64_t> timestamp; // epoch seconds
  /** Seconds elapsed since the previous (earlier) hop; nullopt when unknown. */
  std::optional<double> delay;
  std::string protocol;
};

struct AuthResult {
  /** pass | fail | softfail | neutral | none | temperror | permerror | unknown */
  std::string result = "none";
  std::optional<std::string> domain;
  std::optional<std::string> selector;
  /** DMARC policy from the `p=` token (dmarc only). */
  std::optional<std::string> policy;
  /** The full authentication clause, when present. */
  std::optional<std::string> detail;
};

struct AuthResults {
  AuthResult spf;
  AuthResult dkim;
  AuthResult dmarc;
};

enum class AnomalySeverity { Critical, Warning, Info };

struct Anomaly {
  std::string type;
  AnomalySeverity severity;
  std::string message;
};

struct ParsedHeaders {
  std::vector<Header> headers;
  /** Chronological order: hops[0] is the OLDEST hop (parsed bottom-up). */
  std::vector<Hop> hops;
  AuthResults auth;
  std::vector<Anomaly> anomalies;
};

const std::vector<std::string>& suspiciousMailers() {
  static const std::vector<std::string> TOKENS = {
      "bulk",  "mass mail", "massmail", "storm", "flood", "bomber",
      "grabber", "harvest", "spambot",  "stealth", "anonymous", "dark",
      "crack",
  };
  return TOKENS;
}

// ── small helpers ──────────────────────────────────────────────────────────

static std::string toLower(const std::string& s) {
  std::string out = s;
  std::transform(out.begin(), out.end(), out.begin(),
                 [](unsigned char c) { return static_cast<char>(std::tolower(c)); });
  return out;
}

static std::string trim(const std::string& s) {
  const size_t first = s.find_first_not_of(" \t\r\n");
  if (first == std::string::npos) return "";
  const size_t last = s.find_last_not_of(" \t\r\n");
  return s.substr(first, last - first + 1);
}

static bool startsWithWhitespace(const std::string& line) {
  return !line.empty() && (line[0] == ' ' || line[0] == '\t');
}

// ── header block parsing ───────────────────────────────────────────────────

/** Split raw header text into unfolded name/value pairs. Stops at the first
 *  empty line (the body separator). */
std::vector<Header> parseHeaderLines(const std::string& raw) {
  std::vector<std::string> lines;
  {
    std::string current;
    for (char c : raw) { // normalize CRLF → LF, then split
      if (c == '\r') continue;
      if (c == '\n') {
        lines.push_back(current);
        current.clear();
      } else {
        current += c;
      }
    }
    lines.push_back(current);
  }

  std::vector<Header> headers;
  bool haveCurrent = false;
  Header current;
  for (const auto& line : lines) {
    if (trim(line).empty()) break; // end of headers / blank separator
    if (startsWithWhitespace(line) && haveCurrent) {
      // Continuation line (RFC 5322 folding) — append to the previous value.
      current.value += " " + trim(line);
      headers.back() = current;
      continue;
    }
    const size_t colon = line.find(':');
    if (colon == std::string::npos || colon == 0) continue; // garbage — skip
    current = {trim(line.substr(0, colon)), trim(line.substr(colon + 1))};
    haveCurrent = true;
    headers.push_back(current);
  }
  return headers;
}

/** All values for a header name, case-insensitive, in file order. */
std::vector<std::string> getHeaders(const std::vector<Header>& headers, const std::string& name) {
  const std::string lower = toLower(name);
  std::vector<std::string> out;
  for (const auto& h : headers) {
    if (toLower(h.name) == lower) out.push_back(h.value);
  }
  return out;
}

/** First value for a header name, or "" when absent. */
static std::string firstHeader(const std::vector<Header>& headers, const std::string& name) {
  const auto values = getHeaders(headers, name);
  return values.empty() ? std::string{} : values.front();
}

/** Extract an email address from a header value: prefers <addr>, falls back
 *  to the first bare address. */
std::optional<std::string> extractAddress(const std::string& value) {
  static const std::regex BRACKET_RE(R"(<([^<>\s]+)>)");
  std::smatch m;
  if (std::regex_search(value, m, BRACKET_RE)) return m[1].str();
  static const std::regex BARE_RE(R"([^\s<>,;"']+@[^\s<>,;"']+)");
  if (std::regex_search(value, m, BARE_RE)) return m[0].str();
  return std::nullopt;
}

// ── RFC 2822 date parsing ──────────────────────────────────────────────────

/** Days from 1970-01-01 for a civil date (proleptic Gregorian, Howard
 *  Hinnant's algorithm) — the portable timegm core. */
static int64_t daysFromCivil(int64_t y, unsigned m, unsigned d) {
  y -= m <= 2;
  const int64_t era = (y >= 0 ? y : y - 399) / 400;
  const unsigned yoe = static_cast<unsigned>(y - era * 400);            // [0, 399]
  const unsigned doy = (153u * (m + (m > 2 ? -3 : 9)) + 2) / 5 + d - 1; // [0, 365]
  const unsigned doe = yoe * 365 + yoe / 4 - yoe / 100 + doy;           // [0, 146096]
  return era * 146097 + static_cast<int64_t>(doe) - 719468;
}

/** "Feb" → 2. */
static int monthFromName(const std::string& name) {
  static const char* NAMES[] = {"Jan", "Feb", "Mar", "Apr", "May", "Jun",
                                "Jul", "Aug", "Sep", "Oct", "Nov", "Dec"};
  for (int i = 0; i < 12; i++) {
    if (toLower(name) == toLower(NAMES[i])) return i + 1;
  }
  return 0;
}

/** Parse an RFC 2822 date, ignoring trailing "(ZONE)" comments. Returns
 *  nullopt when unparseable. */
std::optional<int64_t> parseRfc2822Date(const std::string& value) {
  static const std::regex COMMENT_RE(R"(\([^)]*\))");
  const std::string cleaned = trim(std::regex_replace(value, COMMENT_RE, " "));
  if (cleaned.empty()) return std::nullopt;

  // "Tue, 12 Feb 2026 10:30:00 +0100" (day name, day, month, year, time, zone)
  static const std::regex DATE_RE(
      R"(^(?:Mon|Tue|Wed|Thu|Fri|Sat|Sun),?\s+(\d{1,2})\s+([A-Za-z]{3})\s+(\d{2,4})\s+)"
      R"((\d{2}):(\d{2})(?::(\d{2}))?\s+([+-]\d{4}|[A-Za-z]{1,5}|UT|GMT).*$)");
  std::smatch m;
  if (!std::regex_match(cleaned, m, DATE_RE)) return std::nullopt;

  const int day = std::stoi(m[1].str());
  const int month = monthFromName(m[2].str());
  int64_t year = std::stoll(m[3].str());
  if (year < 100) year += year >= 50 ? 1900 : 2000; // two-digit years
  const int hour = std::stoi(m[4].str());
  const int minute = std::stoi(m[5].str());
  const int second = m[6].matched ? std::stoi(m[6].str()) : 0;
  if (month == 0) return std::nullopt;

  int64_t offsetSeconds = 0;
  const std::string zone = m[7].str();
  if (zone.size() == 5 && (zone[0] == '+' || zone[0] == '-')) {
    const int oh = std::stoi(zone.substr(1, 2));
    const int om = std::stoi(zone.substr(3, 2));
    offsetSeconds = (oh * 3600 + om * 60) * (zone[0] == '-' ? -1 : 1);
  }
  const int64_t utc = daysFromCivil(year, static_cast<unsigned>(month), static_cast<unsigned>(day)) *
                          86400 +
                      hour * 3600 + minute * 60 + second;
  return utc - offsetSeconds;
}

// ── Received / Authentication-Results ──────────────────────────────────────

/** Parse one Received header value into a hop (delay filled in later). */
Hop parseReceived(const std::string& value) {
  static const std::regex FROM_RE(R"(\bfrom\s+([^\s(;]+))", std::regex::icase);
  static const std::regex BY_RE(R"(\bby\s+([^\s(;]+))", std::regex::icase);
  static const std::regex WITH_RE(R"(\bwith\s+([^\s;()]+))", std::regex::icase);
  static const std::regex DATE_GUESS_RE(R"(\b(?:Mon|Tue|Wed|Thu|Fri|Sat|Sun),\s+[^\n]+)");

  auto token = [&](const std::regex& re) -> std::string {
    std::smatch m;
    if (!std::regex_search(value, m, re)) return "";
    std::string tok = m[1].str();
    if (!tok.empty() && (tok.back() == '.' || tok.back() == ',')) tok.pop_back();
    return tok;
  };

  Hop hop;
  hop.from = token(FROM_RE);
  hop.by = token(BY_RE);
  hop.protocol = token(WITH_RE);

  // The timestamp follows the last ';' in the value.
  const size_t lastSemi = value.rfind(';');
  if (lastSemi != std::string::npos) {
    hop.timestamp = parseRfc2822Date(value.substr(lastSemi + 1));
  }
  if (!hop.timestamp) {
    // Some clients omit the ';'. Fall back to the first date-looking token.
    std::smatch m;
    if (std::regex_search(value, m, DATE_GUESS_RE)) {
      hop.timestamp = parseRfc2822Date(m[0].str());
    }
  }
  return hop;
}

static AuthResult emptyAuth() { return AuthResult{}; }

static std::string resultWord(const std::string& clause) {
  static const std::regex WORD_RE(R"(=\s*([a-z]+)\b)", std::regex::icase);
  std::smatch m;
  if (!std::regex_search(clause, m, WORD_RE)) return "unknown";
  return toLower(m[1].str());
}

static std::optional<std::string> clauseToken(const std::string& clause, const std::string& key) {
  const std::regex tokenRe("\\b" + key + "=([^\\s;)]+)", std::regex::icase);
  std::smatch m;
  if (!std::regex_search(clause, m, tokenRe)) return std::nullopt;
  std::string tok = m[1].str();
  if (!tok.empty() && (tok.back() == '.' || tok.back() == ',')) tok.pop_back();
  return tok;
}

/** Split `s` on every ';' (no empty pieces kept). */
static std::vector<std::string> splitOn(const std::string& s, char sep) {
  std::vector<std::string> out;
  std::string current;
  for (char c : s) {
    if (c == sep) {
      out.push_back(current);
      current.clear();
    } else {
      current += c;
    }
  }
  out.push_back(current);
  return out;
}

/**
 * Parse Authentication-Results clauses. Example:
 *   mx.google.com; spf=pass smtp.mailfrom=a@b.com; dkim=pass header.s=sel header.d=b.com;
 *   dmarc=pass (p=REJECT) header.from=b.com
 */
AuthResults parseAuthResults(const std::vector<std::string>& values) {
  AuthResults auth;
  if (values.empty()) return auth;

  for (const auto& value : values) {
    for (const auto& rawClause : splitOn(value, ';')) {
      const std::string clause = trim(rawClause);
      const std::string lower = toLower(clause);
      AuthResult* target = nullptr;
      if (lower.rfind("spf=", 0) == 0) {
        target = &auth.spf;
      } else if (lower.rfind("dkim=", 0) == 0) {
        target = &auth.dkim;
      } else if (lower.rfind("dmarc=", 0) == 0) {
        target = &auth.dmarc;
      } else {
        continue;
      }

      if (target->result != "none") continue; // first result wins
      target->result = resultWord(clause);
      target->detail = clause;

      if (target == &auth.spf) {
        target->domain = clauseToken(clause, "smtp.mailfrom").has_value()
                             ? clauseToken(clause, "smtp.mailfrom")
                             : clauseToken(clause, "mailfrom");
      } else if (target == &auth.dkim) {
        target->domain = clauseToken(clause, "header.d").has_value()
                             ? clauseToken(clause, "header.d")
                             : clauseToken(clause, "header.i");
        target->selector = clauseToken(clause, "header.s");
      } else {
        target->domain = clauseToken(clause, "header.from");
        static const std::regex POLICY_RE(R"(\bp=([a-z]+))", std::regex::icase);
        std::smatch pm;
        target->policy = std::regex_search(clause, pm, POLICY_RE)
                             ? std::optional<std::string>(toLower(pm[1].str()))
                             : std::nullopt;
      }
    }
  }
  return auth;
}

// ── anomaly detection ──────────────────────────────────────────────────────

static std::string fixed1(double v) {
  char buf[32];
  std::snprintf(buf, sizeof(buf), "%.1f", v);
  return buf;
}

std::vector<Anomaly> detectAnomalies(const std::vector<Header>& headers,
                                     const std::vector<Hop>& hops, const AuthResults& auth) {
  std::vector<Anomaly> anomalies;

  // 1. From vs Return-Path mismatch — classic spoofing signal.
  const auto fromAddr = extractAddress(firstHeader(headers, "From"));
  const auto returnPath = extractAddress(firstHeader(headers, "Return-Path"));
  if (fromAddr && returnPath && toLower(*fromAddr) != toLower(*returnPath)) {
    anomalies.push_back({
        "from-return-path-mismatch",
        AnomalySeverity::Critical,
        "Return-Path (" + *returnPath + ") does not match From (" + *fromAddr +
            ") — the envelope sender differs from the displayed sender. Common in "
            "spoofing and mailing-list relay.",
    });
  }

  // 2. Reply-To pointing somewhere other than From.
  const auto replyTo = extractAddress(firstHeader(headers, "Reply-To"));
  if (fromAddr && replyTo && toLower(*replyTo) != toLower(*fromAddr)) {
    anomalies.push_back({
        "reply-to-mismatch",
        AnomalySeverity::Warning,
        "Reply-To (" + *replyTo + ") differs from From (" + *fromAddr +
            ") — replies would go to a different address than the visible sender.",
    });
  }

  // 3. Suspicious X-Mailer / User-Agent strings.
  std::string mailer = firstHeader(headers, "X-Mailer");
  if (mailer.empty()) mailer = firstHeader(headers, "User-Agent");
  if (!mailer.empty()) {
    const std::string lower = toLower(mailer);
    for (const auto& token : suspiciousMailers()) {
      if (lower.find(token) != std::string::npos) {
        anomalies.push_back({
            "suspicious-mailer",
            AnomalySeverity::Warning,
            "Mailer string \"" + mailer + "\" contains a suspicious token (\"" + token +
                "\") often seen in bulk sending tools.",
        });
        break;
      }
    }
  }

  // 4. Received chain gaps: unparseable/missing timestamps and time going backwards.
  for (size_t i = 0; i < hops.size(); i++) {
    const Hop& hop = hops[i];
    if (!hop.timestamp) {
      const std::string who = !hop.from.empty() ? hop.from : (!hop.by.empty() ? hop.by : "unknown");
      anomalies.push_back({
          "hop-missing-timestamp",
          AnomalySeverity::Info,
          "Hop " + std::to_string(i + 1) + " (" + who +
              ") has no parseable timestamp — delay for this leg cannot be computed.",
      });
      continue;
    }
    if (i > 0 && hops[i - 1].timestamp) {
      // Timestamps here are epoch seconds (the TS reference divides a
      // millisecond difference by 1000 — unnecessary once seconds are stored).
      const double delta = static_cast<double>(*hop.timestamp - *hops[i - 1].timestamp);
      if (delta < 0) {
        anomalies.push_back({
            "negative-delay",
            AnomalySeverity::Warning,
            "Hop " + std::to_string(i + 1) + " is timestamped " + fixed1(std::abs(delta)) +
                "s BEFORE hop " + std::to_string(i) +
                " — clock skew between servers or a forged Received header.",
        });
      }
    }
  }

  // 5. No authentication results at all.
  if (getHeaders(headers, "Authentication-Results").empty()) {
    anomalies.push_back({
        "no-auth-results",
        AnomalySeverity::Info,
        "No Authentication-Results header found — SPF/DKIM/DMARC status cannot be "
        "verified from this message.",
    });
  }

  return anomalies;
}

// ── main entry point ───────────────────────────────────────────────────────

/**
 * Parse raw email headers (RFC 5322) into structured data: unfolded headers,
 * chronological Received hops with delays, SPF/DKIM/DMARC results, and
 * spoofing anomalies.
 */
ParsedHeaders parseHeaders(const std::string& raw) {
  const std::vector<Header> headers = parseHeaderLines(raw);

  // Received headers are listed newest-first; parse bottom-up so hops[0] is the oldest.
  const std::vector<std::string> received = getHeaders(headers, "Received");
  std::vector<Hop> newestFirst;
  newestFirst.reserve(received.size());
  for (const auto& value : received) newestFirst.push_back(parseReceived(value));
  std::reverse(newestFirst.begin(), newestFirst.end());

  std::vector<Hop> hops;
  hops.reserve(newestFirst.size());
  for (size_t i = 0; i < newestFirst.size(); i++) {
    Hop hop = newestFirst[i];
    if (i > 0 && hop.timestamp && hops[i - 1].timestamp) {
      hop.delay = static_cast<double>(*hop.timestamp - *hops[i - 1].timestamp);
    }
    hops.push_back(hop);
  }

  ParsedHeaders parsed;
  parsed.headers = headers;
  parsed.hops = std::move(hops);
  parsed.auth = parseAuthResults(getHeaders(headers, "Authentication-Results"));
  parsed.anomalies = detectAnomalies(parsed.headers, parsed.hops, parsed.auth);
  return parsed;
}

} // namespace emailheaders

Also available in 8 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →