Skip to content

Password Breach Checker — C source

Check if a password has appeared in known data breaches using k-anonymity. Only the first 5 characters of the SHA-1 hash are sent - your full password never leaves your browser.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * breach-checker — password breach lookup via k-anonymity (Have I Been Pwned's
 *                  Pwned Passwords range API).
 *
 * Language: C (C11, standard library + OpenSSL 3.x libcrypto for SHA-1 — C has
 *           no hashing in its standard library)
 * Source:   CosmoDev polyglot showcase port of the Breach Checker tool, ported
 *           from src/lib/breach-checker.ts (the canonical TypeScript
 *           implementation).
 * License:  display source — part of CosmoDev's polyglot tool pages.
 *
 * Scope: this port covers the pure logic — the SHA-1 hash, the k-anonymity
 * prefix/suffix split, and the range-response parsers. The HTTP fetch of the TS
 * reference's checkBreach() is deliberately omitted (no network I/O in a
 * display snippet); the caller supplies the response body it retrieved from
 * HIBP_RANGE_URL + prefix.
 *
 * k-anonymity: only the first 5 characters of the SHA-1 hash ever leave the
 * machine. The API returns one "SUFFIX:COUNT" line per hash sharing that prefix
 * (~800 candidates) and the suffix match happens locally, so the service never
 * learns which password was checked.
 *
 * Build: cc -std=c11 breach-checker.c -lcrypto
 */

#include <ctype.h>
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

#include <openssl/sha.h>

/* --------------------------------------------------------------- constants --- */

#define HIBP_RANGE_URL "https://api.pwnedpasswords.com/range/"

enum {
    SHA1_HEX_LEN = 40, /* SHA-1 is 20 bytes -> 40 hex chars */
    PREFIX_LEN   = 5,  /* the k-anonymity prefix sent to the API */
    SUFFIX_LEN   = SHA1_HEX_LEN - PREFIX_LEN
};

/* The prefix/suffix pair from splitHash(). Both are NUL-terminated. */
typedef struct {
    char prefix[PREFIX_LEN + 1];
    char suffix[SUFFIX_LEN + 1];
} breach_hash;

/* Mirrors the TS BreachResult interface. */
typedef struct {
    /* True when the exact hash suffix appeared in the candidate list. */
    bool breached;
    /* Times the password appeared in breaches. 0 = never seen, -1 = lookup failed. */
    long count;
    /* First 5 chars of the uppercase SHA-1 hex — the only part sent to the API. */
    char hash_prefix[PREFIX_LEN + 1];
    /* Remaining 35 chars, matched locally against the response. */
    char hash_suffix[SUFFIX_LEN + 1];
    /* How many candidate suffixes the response held (all checked locally). */
    long candidates;
} breach_result;

/* ------------------------------------------------------------------- sha-1 --- */

/* SHA-1 of a UTF-8 string as uppercase hex (the format HIBP expects).
 * `out` must hold SHA1_HEX_LEN + 1 bytes. */
void sha1_hex(const char *input, char out[SHA1_HEX_LEN + 1])
{
    static const char HEX[] = "0123456789ABCDEF";
    unsigned char digest[SHA_DIGEST_LENGTH];
    size_t len = (input != NULL) ? strlen(input) : 0;

    SHA1((const unsigned char *) (input != NULL ? input : ""), len, digest);

    for (size_t i = 0; i < SHA_DIGEST_LENGTH; i++) {
        out[i * 2]     = HEX[digest[i] >> 4];
        out[i * 2 + 1] = HEX[digest[i] & 0x0F];
    }
    out[SHA1_HEX_LEN] = '\0';
}

/* -------------------------------------------------------------- split hash --- */

/*
 * Split a 40-char hash into the 5-char k-anonymity prefix and the 35-char
 * suffix, upper-casing as it goes. Short input simply yields short fields,
 * matching the TS slice() semantics.
 */
breach_hash split_hash(const char *hash)
{
    breach_hash out = {{0}, {0}};
    size_t len = (hash != NULL) ? strlen(hash) : 0;
    size_t i;

    for (i = 0; i < PREFIX_LEN && i < len; i++) {
        out.prefix[i] = (char) toupper((unsigned char) hash[i]);
    }
    out.prefix[i] = '\0';

    for (i = 0; i + PREFIX_LEN < len && i < SUFFIX_LEN; i++) {
        out.suffix[i] = (char) toupper((unsigned char) hash[i + PREFIX_LEN]);
    }
    out.suffix[i] = '\0';

    return out;
}

/* ----------------------------------------------------------------- parsing --- */

/* Advance past a line, returning the start of the next one (or NULL at the end).
 * Tolerates both LF and CRLF; `*len` receives the current line's length with any
 * trailing CR removed. */
static const char *next_line(const char *p, size_t *len)
{
    const char *nl = strchr(p, '\n');
    size_t n = (nl != NULL) ? (size_t) (nl - p) : strlen(p);

    if (n > 0 && p[n - 1] == '\r') {
        n--; /* CRLF */
    }
    *len = n;
    return (nl != NULL) ? nl + 1 : NULL;
}

/* Trim ASCII whitespace from both ends of [start, start+len). */
static void trim(const char **start, size_t *len)
{
    const char *s = *start;
    size_t n = *len;

    while (n > 0 && isspace((unsigned char) s[0])) { s++; n--; }
    while (n > 0 && isspace((unsigned char) s[n - 1])) { n--; }

    *start = s;
    *len = n;
}

/*
 * Search a range response for `suffix` and return its breach count. Never
 * fails: returns 0 when the suffix is not present. Tolerates LF and CRLF line
 * endings, blank lines, and leading/trailing whitespace per line.
 */
long parse_range_body(const char *body, const char *suffix)
{
    const char *line = body;

    if (body == NULL || suffix == NULL || suffix[0] == '\0') {
        return 0;
    }

    while (line != NULL) {
        size_t line_len;
        const char *next = next_line(line, &line_len);
        const char *colon = memchr(line, ':', line_len);

        if (colon != NULL) {
            const char *name = line;
            size_t name_len = (size_t) (colon - line);

            trim(&name, &name_len);
            if (name_len == strlen(suffix) && memcmp(name, suffix, name_len) == 0) {
                /* The count runs from just past the colon to the line end. */
                const char *value = colon + 1;
                size_t value_len = line_len - (size_t) (colon - line) - 1;
                char buf[32];
                long count;
                char *end;

                trim(&value, &value_len);
                if (value_len == 0 || value_len >= sizeof buf) {
                    return 0;
                }
                memcpy(buf, value, value_len);
                buf[value_len] = '\0';

                count = strtol(buf, &end, 10);
                /* Unparseable or negative counts are reported as "not seen",
                 * exactly as the TS NaN/`< 0` guard does. */
                if (end == buf || count < 0) {
                    return 0;
                }
                return count;
            }
        }
        line = next;
    }
    return 0;
}

/* Count the "SUFFIX:COUNT" candidate lines in a range response. */
long count_candidates(const char *body)
{
    const char *line = body;
    long n = 0;

    if (body == NULL) {
        return 0;
    }

    while (line != NULL) {
        size_t line_len;
        const char *next = next_line(line, &line_len);
        const char *colon = memchr(line, ':', line_len);

        if (colon != NULL) {
            const char *name = line;
            size_t name_len = (size_t) (colon - line);

            trim(&name, &name_len);
            if (name_len > 0) {
                n++;
            }
        }
        line = next;
    }
    return n;
}

/* ------------------------------------------------------------------- check --- */

/*
 * Check a password against a range response the caller already fetched from
 * HIBP_RANGE_URL + result.hash_prefix. Split out from the network call so the
 * matching stays pure and testable — this is the half that must never leak the
 * password.
 */
breach_result check_breach(const char *password, const char *range_body)
{
    char hash[SHA1_HEX_LEN + 1];
    breach_hash split;
    breach_result result = {0};

    sha1_hex(password, hash);
    split = split_hash(hash);

    snprintf(result.hash_prefix, sizeof result.hash_prefix, "%s", split.prefix);
    snprintf(result.hash_suffix, sizeof result.hash_suffix, "%s", split.suffix);

    if (range_body == NULL) {
        /* No response body = the lookup failed; -1 distinguishes that from a
         * genuine "never seen" (0). */
        result.count = -1;
        return result;
    }

    result.count      = parse_range_body(range_body, result.hash_suffix);
    result.breached   = result.count > 0;
    result.candidates = count_candidates(range_body);
    return result;
}

/* -------------------------------------------------------------------- demo --- */

int main(void)
{
    const char *password = "password123";
    char hash[SHA1_HEX_LEN + 1];
    breach_hash split;
    breach_result result;

    sha1_hex(password, hash);
    split = split_hash(hash);

    printf("password:  %s\n", password);
    printf("sha-1:     %s\n", hash);
    printf("prefix:    %s   <- the only part sent to the API\n", split.prefix);
    printf("suffix:    %s   <- matched locally\n", split.suffix);
    printf("request:   %s%s\n\n", HIBP_RANGE_URL, split.prefix);

    /* A stand-in for the API response. The real body holds ~800 suffix lines;
     * this one embeds the suffix for "password123" so the match succeeds.
     * Blank lines and CRLF endings are here on purpose — the parser tolerates
     * both. */
    {
        char body[512];

        snprintf(body, sizeof body,
                 "0018A45C4D1DEF81644B54AB7F969B88D65:1\r\n"
                 "%s:2749416\r\n"
                 "\r\n"
                 "00D4F6E8FA6EECAD2A3AA415EEC418D38EC:2\r\n",
                 split.suffix);

        result = check_breach(password, body);
        printf("breached:   %s\n", result.breached ? "yes" : "no");
        printf("count:      %ld\n", result.count);
        printf("candidates: %ld\n", result.candidates);
    }

    /* A password absent from the corpus reports 0, not an error. */
    result = check_breach("a-passphrase-nobody-has-ever-used-4711",
                          "0018A45C4D1DEF81644B54AB7F969B88D65:1\r\n");
    printf("\nunseen password -> breached=%s count=%ld\n",
           result.breached ? "yes" : "no", result.count);

    return EXIT_SUCCESS;
}

Also available in 9 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →