Skip to content

Hash Type Identifier — C source

Identify the likely hash algorithm of a hash string by its length and character set - MD5, SHA-1/2/3, BLAKE, CRC32, NTLM, bcrypt, Argon2 and more.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * Hash-type identifier — C port.
 *
 * Language: C (C11, standard library only)
 * Source:   CosmoDev polyglot showcase port of the `hash-type-identifier`
 *           tool, ported from src/lib/hashIdentify.ts (the canonical
 *           TypeScript implementation).
 * License:  display source — part of CosmoDev's polyglot tool pages
 *           (dev.cosmolabs.org).
 *
 * Pure string classification: inspect a candidate hash's charset and length
 * to suggest likely algorithms. No hashing happens here — this is pattern
 * recognition over an already-computed digest. Deterministic; never crashes
 * on bad input, and an allocation failure degrades to an empty result.
 *
 * The standard library ships no regex engine, so the four detection rules
 * are written as small, self-contained byte matchers — the same shape as
 * the Rust port.
 */

#include <stdbool.h>
#include <stddef.h>
#include <stdlib.h>
#include <string.h>

/* The character set classification of a candidate hash string. */
typedef enum {
    HASH_HEX,
    HASH_BASE64,
    HASH_BCRYPT,
    HASH_ARGON2,
    HASH_UNKNOWN
} hash_charset;

/*
 * Lowercase identifier matching the TypeScript string literal used by the
 * canonical implementation (so serialised output agrees).
 */
const char *
hash_charset_str(hash_charset cs)
{
    switch (cs) {
    case HASH_HEX:    return "hex";
    case HASH_BASE64: return "base64";
    case HASH_BCRYPT: return "bcrypt";
    case HASH_ARGON2: return "argon2";
    default:          return "unknown";
    }
}

/* A candidate hash algorithm and its nominal bit length. */
typedef struct {
    const char *name;
    long long   bit_length; /* hex length * 4, where applicable */
} hash_match;

/* The full identification result for an input string. */
typedef struct {
    char        *input;          /* heap copy of the original argument */
    char        *cleaned;        /* heap copy of the trimmed input */
    size_t       length;         /* cleaned length */
    hash_charset charset;
    hash_match  *candidates;     /* heap array of candidate_count entries */
    size_t       candidate_count;
} hash_info;

/* Release every heap field of `info`. Safe on a zeroed struct. */
void
hash_info_free(hash_info *info)
{
    if (info == NULL)
        return;
    free(info->input);
    free(info->cleaned);
    free(info->candidates);
    memset(info, 0, sizeof *info);
}

#define ARRAY_LEN(a) (sizeof(a) / sizeof((a)[0]))

/* Hex candidates keyed by hex-string length. Each hex char encodes 4 bits,
 * so a 64-char digest implies a 256-bit algorithm such as SHA-256. */
static const char *const HEX_8[] = { "CRC32", "Adler-32" };
static const char *const HEX_16[] = { "MySQL 3.x", "CRC64" };
static const char *const HEX_32[] = { "MD5", "MD4", "NTLM", "LM", "MD2", "RIPEMD-128", "HAVAL-128" };
static const char *const HEX_40[] = { "SHA-1", "RIPEMD-160", "HAVAL-160", "MySQL 5.x (SHA1(SHA1))", "Tiger-160" };
static const char *const HEX_56[] = { "SHA-224", "SHA3-224", "BLAKE2s-224", "HAVAL-224" };
static const char *const HEX_64[] = { "SHA-256", "SHA3-256", "BLAKE2s-256", "RIPEMD-256", "Skein-256" };
static const char *const HEX_96[] = { "SHA-384", "SHA3-384", "BLAKE2b-384" };
static const char *const HEX_128[] = { "SHA-512", "SHA3-512", "BLAKE2b-512", "Whirlpool", "Skein-512" };

/* Base64 candidates keyed by encoded-string length (16-byte MD5 digest ->
 * 24 base64 chars including padding, etc.). */
static const char *const B64_24[] = { "MD5 (base64)" };
static const char *const B64_28[] = { "SHA-1 (base64)" };
static const char *const B64_44[] = { "SHA-256 (base64)" };
static const char *const B64_88[] = { "SHA-512 (base64)" };

/* Hex candidate names for a given hex-string length, or NULL. */
static const char *const *
hex_candidates(size_t len, size_t *count)
{
    switch (len) {
    case 8:   *count = ARRAY_LEN(HEX_8);   return HEX_8;
    case 16:  *count = ARRAY_LEN(HEX_16);  return HEX_16;
    case 32:  *count = ARRAY_LEN(HEX_32);  return HEX_32;
    case 40:  *count = ARRAY_LEN(HEX_40);  return HEX_40;
    case 56:  *count = ARRAY_LEN(HEX_56);  return HEX_56;
    case 64:  *count = ARRAY_LEN(HEX_64);  return HEX_64;
    case 96:  *count = ARRAY_LEN(HEX_96);  return HEX_96;
    case 128: *count = ARRAY_LEN(HEX_128); return HEX_128;
    default:  *count = 0;                  return NULL;
    }
}

/* Base64 candidate names for a given encoded-string length, or NULL. */
static const char *const *
base64_candidates(size_t len, size_t *count)
{
    switch (len) {
    case 24:  *count = ARRAY_LEN(B64_24); return B64_24;
    case 28:  *count = ARRAY_LEN(B64_28); return B64_28;
    case 44:  *count = ARRAY_LEN(B64_44); return B64_44;
    case 88:  *count = ARRAY_LEN(B64_88); return B64_88;
    default:  *count = 0;                 return NULL;
    }
}

/* ASCII predicates — locale-independent, unlike ctype.h's is*(). */
static bool
is_ascii_hexdigit(unsigned char c)
{
    return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F');
}

static bool
is_ascii_base64(unsigned char c)
{
    return (c >= '0' && c <= '9') || (c >= 'A' && c <= 'Z')
        || (c >= 'a' && c <= 'z') || c == '+' || c == '/';
}

/* The six ASCII whitespace bytes removed by the canonical trim(). */
static bool
is_ascii_space(unsigned char c)
{
    return c == ' ' || c == '\t' || c == '\n' || c == '\v' || c == '\f' || c == '\r';
}

/* Matches the bcrypt modular-crypt prefix ^\$2[abxy]?\$ — prefix match only;
 * the variable trailing payload is not inspected (mirrors the regex). */
static bool
looks_like_bcrypt(const char *s, size_t n)
{
    if (n < 3 || s[0] != '$' || s[1] != '2')
        return false;
    if (s[2] == 'a' || s[2] == 'b' || s[2] == 'x' || s[2] == 'y')
        return n >= 4 && s[3] == '$';
    return s[2] == '$';
}

/* Matches the argon2 modular-crypt prefix ^\$argon2(id|i|d)?\$. */
static bool
looks_like_argon2(const char *s, size_t n)
{
    static const char prefix[] = "$argon2";
    const size_t plen = sizeof prefix - 1;

    if (n <= plen || memcmp(s, prefix, plen) != 0)
        return false;
    s += plen;
    n -= plen;
    /* Try the two-char variant first so `id` wins over the bare `i`. */
    if (n >= 2 && s[0] == 'i' && s[1] == 'd')
        return n >= 3 && s[2] == '$';
    if (s[0] == 'i' || s[0] == 'd')
        return n >= 2 && s[1] == '$';
    return s[0] == '$';
}

/* Whole-string hex match, mirroring the `+` quantifier (non-empty body). */
static bool
looks_like_hex(const char *s, size_t n)
{
    size_t i;

    if (n == 0)
        return false;
    for (i = 0; i < n; i++)
        if (!is_ascii_hexdigit((unsigned char)s[i]))
            return false;
    return true;
}

/* Valid standard-alphabet base64 with 0–2 trailing `=` padding; the body
 * before padding must be non-empty. */
static bool
looks_like_base64(const char *s, size_t n)
{
    size_t end = n, pad = 0, i;

    while (end > 0 && s[end - 1] == '=' && pad < 2) {
        end--;
        pad++;
    }
    if (end == 0)
        return false;
    for (i = 0; i < end; i++)
        if (!is_ascii_base64((unsigned char)s[i]))
            return false;
    return true;
}

/*
 * Classify the charset of a candidate hash string.
 *
 * Order matters: hex is checked before base64 because every hex digest is
 * also a legal base64 character set, and the more specific classification
 * should win.
 */
hash_charset
detect_charset(const char *s, size_t n)
{
    if (s == NULL) {
        s = "";
        n = 0;
    }
    if (looks_like_bcrypt(s, n))
        return HASH_BCRYPT;
    if (looks_like_argon2(s, n))
        return HASH_ARGON2;
    if (looks_like_hex(s, n))
        return HASH_HEX;
    if (looks_like_base64(s, n))
        return HASH_BASE64;
    return HASH_UNKNOWN;
}

/*
 * Identify candidate hash types for an input string.
 *
 * Always returns a populated hash_info — call hash_info_free() afterwards.
 * NULL input is treated as the empty string. An empty, unrecognised, or
 * wrong-length input simply yields zero candidates — the caller decides
 * whether "no candidates" means "not a hash".
 */
hash_info
identify_hash(const char *input)
{
    hash_info info = {0};
    size_t n, start, end, count = 0;
    const char *const *names = NULL;
    const char *single_name = NULL;
    long long bits = 0;

    if (input == NULL)
        input = "";

    /* Trim ASCII whitespace from both ends into heap copies. */
    n = strlen(input);
    start = 0;
    while (start < n && is_ascii_space((unsigned char)input[start]))
        start++;
    end = n;
    while (end > start && is_ascii_space((unsigned char)input[end - 1]))
        end--;

    info.input = malloc(n + 1);
    info.cleaned = malloc(end - start + 1);
    if (info.input == NULL || info.cleaned == NULL) {
        hash_info_free(&info);
        return (hash_info){ .charset = HASH_UNKNOWN };
    }
    memcpy(info.input, input, n);
    info.input[n] = '\0';
    memcpy(info.cleaned, input + start, end - start);
    info.cleaned[end - start] = '\0';

    info.length = end - start;
    info.charset = detect_charset(info.cleaned, info.length);

    switch (info.charset) {
    case HASH_BCRYPT:
        /* bcrypt's modular-crypt token encodes a 184-bit effective hash. */
        single_name = "bcrypt";
        bits = 184;
        break;
    case HASH_ARGON2:
        /* Argon2 output length is parameter-driven, so no fixed bit length applies. */
        single_name = "Argon2";
        bits = 0;
        break;
    case HASH_HEX:
        names = hex_candidates(info.length, &count);
        /* length*4 converts hex-char count to a bit width (4 bits per nibble). */
        bits = (long long)info.length * 4;
        break;
    case HASH_BASE64:
        names = base64_candidates(info.length, &count);
        /* Each base64 char carries 6 bits; round to the nearest byte boundary
         * so the reported length lines up with the underlying digest width.
         * All table lengths divide evenly, so the division below is exact. */
        bits = (long long)(info.length * 6 / 8) * 8;
        break;
    default:
        break;
    }

    if (single_name != NULL) {
        info.candidates = malloc(sizeof *info.candidates);
        if (info.candidates != NULL) {
            info.candidates[0] = (hash_match){ single_name, bits };
            info.candidate_count = 1;
        }
        return info;
    }

    if (names != NULL && count > 0) {
        info.candidates = malloc(count * sizeof *info.candidates);
        if (info.candidates != NULL) {
            size_t i;

            for (i = 0; i < count; i++)
                info.candidates[i] = (hash_match){ names[i], bits };
            info.candidate_count = count;
        }
    }
    return info;
}

#ifdef HASH_TYPE_IDENTIFIER_DEMO
#include <stdio.h>

int
main(void)
{
    const char *samples[] = {
        "d41d8cd98f00b204e9800998ecf8427e",       /* MD5-length hex   */
        "$2b$12$KIXQeQeJzhyMQaVXcWqUqu",          /* bcrypt token     */
        "$argon2id$v=19$m=65536,t=3,p=4$c29tZXNhbHQ", /* argon2 token  */
    };
    size_t i;

    for (i = 0; i < ARRAY_LEN(samples); i++) {
        hash_info info = identify_hash(samples[i]);
        size_t c;

        printf("%s\n  charset: %s (length %zu), %zu candidate(s)\n",
               info.cleaned, hash_charset_str(info.charset), info.length,
               info.candidate_count);
        for (c = 0; c < info.candidate_count; c++)
            printf("    %-28s %lld bits\n",
                   info.candidates[c].name, info.candidates[c].bit_length);
        hash_info_free(&info);
    }
    return 0;
}
#endif

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →