Skip to content

Hash Type Identifier — C++ source

Identify the likely hash algorithm of a hash string by its length and character set - MD5, SHA-1/2/3, BLAKE, CRC32, NTLM, bcrypt, Argon2 and more.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Hash-type identifier — C++ port.
//
// Language: C++ (C++17, standard library only)
// Source:   CosmoDev polyglot showcase port of the `hash-type-identifier`
//           tool, ported from src/lib/hashIdentify.ts (the canonical
//           TypeScript implementation).
// License:  display source — part of CosmoDev's polyglot tool pages
//           (dev.cosmolabs.org).
//
// Pure string classification: inspect a candidate hash's charset and length
// to suggest likely algorithms. No hashing happens here — this is pattern
// recognition over an already-computed digest. Deterministic; never throws.
//
// The detection rules are written as small, self-contained string_view
// matchers rather than std::regex — same shape as the Rust port, without
// std::regex's compile-time and runtime cost.

#include <cstddef>
#include <string>
#include <string_view>
#include <unordered_map>
#include <vector>

namespace hash_identify {

/// The character set classification of a candidate hash string.
enum class hash_charset { hex, base64, bcrypt, argon2, unknown };

/// Lowercase identifier matching the TypeScript string literal used by the
/// canonical implementation (so serialised output agrees).
constexpr std::string_view as_string(hash_charset cs) noexcept
{
    switch (cs) {
    case hash_charset::hex:    return "hex";
    case hash_charset::base64: return "base64";
    case hash_charset::bcrypt: return "bcrypt";
    case hash_charset::argon2: return "argon2";
    default:                   return "unknown";
    }
}

/// A candidate hash algorithm and its nominal bit length.
struct hash_match {
    std::string name;
    long long bit_length;  // hex length * 4, where applicable
};

/// The full identification result for an input string.
struct hash_info {
    std::string input;
    std::string cleaned;  // trimmed input
    std::size_t length = 0;
    hash_charset charset = hash_charset::unknown;
    std::vector<hash_match> candidates;
};

namespace detail {

constexpr bool starts_with(std::string_view s, std::string_view prefix) noexcept
{
    return s.size() >= prefix.size() && s.compare(0, prefix.size(), prefix) == 0;
}

constexpr bool is_ascii_hexdigit(char c) noexcept
{
    return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F');
}

constexpr bool is_ascii_base64(char c) noexcept
{
    return (c >= '0' && c <= '9') || (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z')
        || c == '+' || c == '/';
}

/// Matches the bcrypt modular-crypt prefix `^\$2[abxy]?\$` — prefix match
/// only; the variable trailing payload is not inspected.
constexpr bool looks_like_bcrypt(std::string_view s) noexcept
{
    if (!starts_with(s, "$2") || s.size() < 3)
        return false;
    switch (s[2]) {
    case 'a': case 'b': case 'x': case 'y':
        return s.size() >= 4 && s[3] == '$';
    case '$':
        return true;
    default:
        return false;
    }
}

/// Matches the argon2 modular-crypt prefix `^\$argon2(id|i|d)?\$`.
constexpr bool looks_like_argon2(std::string_view s) noexcept
{
    constexpr std::string_view prefix = "$argon2";
    if (!starts_with(s, prefix))
        return false;
    const std::string_view rest = s.substr(prefix.size());
    // Try the two-char variant first so `id` wins over the bare `i`.
    if (starts_with(rest, "id"))
        return rest.size() >= 3 && rest[2] == '$';
    if (rest.empty())
        return false;
    switch (rest.front()) {
    case 'i': case 'd':
        return rest.size() >= 2 && rest[1] == '$';
    case '$':
        return true;
    default:
        return false;
    }
}

/// Whole-string hex match, mirroring the `+` quantifier (non-empty body).
constexpr bool looks_like_hex(std::string_view s) noexcept
{
    if (s.empty())
        return false;
    for (const char c : s)
        if (!is_ascii_hexdigit(c))
            return false;
    return true;
}

/// Valid standard-alphabet base64 with 0–2 trailing `=` padding; the body
/// before padding must be non-empty.
constexpr bool looks_like_base64(std::string_view s) noexcept
{
    std::size_t end = s.size(), pad = 0;
    while (end > 0 && s[end - 1] == '=' && pad < 2) {
        --end;
        ++pad;
    }
    if (end == 0)
        return false;
    for (std::size_t i = 0; i < end; ++i)
        if (!is_ascii_base64(s[i]))
            return false;
    return true;
}

/// Hex candidates keyed by hex-string length — each hex char encodes 4
/// bits, so a 64-char digest implies a 256-bit algorithm such as SHA-256.
/// Function-local static: immune to the static initialisation order fiasco.
const std::unordered_map<std::size_t, std::vector<const char *>> &hex_by_length()
{
    static const std::unordered_map<std::size_t, std::vector<const char *>> table = {
        {8,   {"CRC32", "Adler-32"}},
        {16,  {"MySQL 3.x", "CRC64"}},
        {32,  {"MD5", "MD4", "NTLM", "LM", "MD2", "RIPEMD-128", "HAVAL-128"}},
        {40,  {"SHA-1", "RIPEMD-160", "HAVAL-160", "MySQL 5.x (SHA1(SHA1))", "Tiger-160"}},
        {56,  {"SHA-224", "SHA3-224", "BLAKE2s-224", "HAVAL-224"}},
        {64,  {"SHA-256", "SHA3-256", "BLAKE2s-256", "RIPEMD-256", "Skein-256"}},
        {96,  {"SHA-384", "SHA3-384", "BLAKE2b-384"}},
        {128, {"SHA-512", "SHA3-512", "BLAKE2b-512", "Whirlpool", "Skein-512"}},
    };
    return table;
}

/// Base64 candidates keyed by encoded-string length (16-byte MD5 digest ->
/// 24 base64 chars including padding, etc.).
const std::unordered_map<std::size_t, std::vector<const char *>> &base64_by_length()
{
    static const std::unordered_map<std::size_t, std::vector<const char *>> table = {
        {24, {"MD5 (base64)"}},
        {28, {"SHA-1 (base64)"}},
        {44, {"SHA-256 (base64)"}},
        {88, {"SHA-512 (base64)"}},
    };
    return table;
}

}  // namespace detail

/// Classify the charset of a candidate hash string.
///
/// Order matters: hex is checked before base64 because every hex digest is
/// also a legal base64 character set, and the more specific classification
/// should win.
inline hash_charset detect_charset(std::string_view s) noexcept
{
    using namespace detail;
    if (looks_like_bcrypt(s)) return hash_charset::bcrypt;
    if (looks_like_argon2(s)) return hash_charset::argon2;
    if (looks_like_hex(s))    return hash_charset::hex;
    if (looks_like_base64(s)) return hash_charset::base64;
    return hash_charset::unknown;
}

/// Identify candidate hash types for an input string.
///
/// Always returns a fully populated hash_info; never throws. An empty,
/// unrecognised, or wrong-length input simply yields an empty candidate
/// vector — the caller decides whether "no candidates" means "not a hash".
inline hash_info identify_hash(std::string_view input)
{
    // Trim the six ASCII whitespace bytes the canonical trim() removes.
    constexpr const char *whitespace = " \t\n\v\f\r";

    hash_info info;
    info.input = std::string(input);
    const std::size_t first = input.find_first_not_of(whitespace);
    if (first == std::string_view::npos) {
        info.cleaned = std::string();
    } else {
        const std::size_t last = input.find_last_not_of(whitespace);
        info.cleaned = std::string(input.substr(first, last - first + 1));
    }
    info.length  = info.cleaned.size();
    info.charset = detect_charset(info.cleaned);

    switch (info.charset) {
    case hash_charset::bcrypt:
        // bcrypt's modular-crypt token encodes a 184-bit effective hash.
        info.candidates.push_back({"bcrypt", 184});
        break;
    case hash_charset::argon2:
        // Argon2 output length is parameter-driven, so no fixed bit length applies.
        info.candidates.push_back({"Argon2", 0});
        break;
    case hash_charset::hex: {
        const auto it = detail::hex_by_length().find(info.length);
        if (it != detail::hex_by_length().end()) {
            // length*4 converts hex-char count to a bit width (4 bits per nibble).
            const auto bits = static_cast<long long>(info.length) * 4;
            for (const char *name : it->second)
                info.candidates.push_back({name, bits});
        }
        break;
    }
    case hash_charset::base64: {
        const auto it = detail::base64_by_length().find(info.length);
        if (it != detail::base64_by_length().end()) {
            // Each base64 char carries 6 bits; round to the nearest byte
            // boundary. All table lengths divide evenly, so the integer
            // division is exact.
            const auto bits = static_cast<long long>(info.length * 6 / 8) * 8;
            for (const char *name : it->second)
                info.candidates.push_back({name, bits});
        }
        break;
    }
    case hash_charset::unknown:
        break;
    }
    return info;
}

}  // namespace hash_identify

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →