Hash Type Identifier — C++ source
Identify the likely hash algorithm of a hash string by its length and character set - MD5, SHA-1/2/3, BLAKE, CRC32, NTLM, bcrypt, Argon2 and more.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Hash-type identifier — C++ port.
//
// Language: C++ (C++17, standard library only)
// Source: CosmoDev polyglot showcase port of the `hash-type-identifier`
// tool, ported from src/lib/hashIdentify.ts (the canonical
// TypeScript implementation).
// License: display source — part of CosmoDev's polyglot tool pages
// (dev.cosmolabs.org).
//
// Pure string classification: inspect a candidate hash's charset and length
// to suggest likely algorithms. No hashing happens here — this is pattern
// recognition over an already-computed digest. Deterministic; never throws.
//
// The detection rules are written as small, self-contained string_view
// matchers rather than std::regex — same shape as the Rust port, without
// std::regex's compile-time and runtime cost.
#include <cstddef>
#include <string>
#include <string_view>
#include <unordered_map>
#include <vector>
namespace hash_identify {
/// The character set classification of a candidate hash string.
enum class hash_charset { hex, base64, bcrypt, argon2, unknown };
/// Lowercase identifier matching the TypeScript string literal used by the
/// canonical implementation (so serialised output agrees).
constexpr std::string_view as_string(hash_charset cs) noexcept
{
switch (cs) {
case hash_charset::hex: return "hex";
case hash_charset::base64: return "base64";
case hash_charset::bcrypt: return "bcrypt";
case hash_charset::argon2: return "argon2";
default: return "unknown";
}
}
/// A candidate hash algorithm and its nominal bit length.
struct hash_match {
std::string name;
long long bit_length; // hex length * 4, where applicable
};
/// The full identification result for an input string.
struct hash_info {
std::string input;
std::string cleaned; // trimmed input
std::size_t length = 0;
hash_charset charset = hash_charset::unknown;
std::vector<hash_match> candidates;
};
namespace detail {
constexpr bool starts_with(std::string_view s, std::string_view prefix) noexcept
{
return s.size() >= prefix.size() && s.compare(0, prefix.size(), prefix) == 0;
}
constexpr bool is_ascii_hexdigit(char c) noexcept
{
return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F');
}
constexpr bool is_ascii_base64(char c) noexcept
{
return (c >= '0' && c <= '9') || (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z')
|| c == '+' || c == '/';
}
/// Matches the bcrypt modular-crypt prefix `^\$2[abxy]?\$` — prefix match
/// only; the variable trailing payload is not inspected.
constexpr bool looks_like_bcrypt(std::string_view s) noexcept
{
if (!starts_with(s, "$2") || s.size() < 3)
return false;
switch (s[2]) {
case 'a': case 'b': case 'x': case 'y':
return s.size() >= 4 && s[3] == '$';
case '$':
return true;
default:
return false;
}
}
/// Matches the argon2 modular-crypt prefix `^\$argon2(id|i|d)?\$`.
constexpr bool looks_like_argon2(std::string_view s) noexcept
{
constexpr std::string_view prefix = "$argon2";
if (!starts_with(s, prefix))
return false;
const std::string_view rest = s.substr(prefix.size());
// Try the two-char variant first so `id` wins over the bare `i`.
if (starts_with(rest, "id"))
return rest.size() >= 3 && rest[2] == '$';
if (rest.empty())
return false;
switch (rest.front()) {
case 'i': case 'd':
return rest.size() >= 2 && rest[1] == '$';
case '$':
return true;
default:
return false;
}
}
/// Whole-string hex match, mirroring the `+` quantifier (non-empty body).
constexpr bool looks_like_hex(std::string_view s) noexcept
{
if (s.empty())
return false;
for (const char c : s)
if (!is_ascii_hexdigit(c))
return false;
return true;
}
/// Valid standard-alphabet base64 with 0–2 trailing `=` padding; the body
/// before padding must be non-empty.
constexpr bool looks_like_base64(std::string_view s) noexcept
{
std::size_t end = s.size(), pad = 0;
while (end > 0 && s[end - 1] == '=' && pad < 2) {
--end;
++pad;
}
if (end == 0)
return false;
for (std::size_t i = 0; i < end; ++i)
if (!is_ascii_base64(s[i]))
return false;
return true;
}
/// Hex candidates keyed by hex-string length — each hex char encodes 4
/// bits, so a 64-char digest implies a 256-bit algorithm such as SHA-256.
/// Function-local static: immune to the static initialisation order fiasco.
const std::unordered_map<std::size_t, std::vector<const char *>> &hex_by_length()
{
static const std::unordered_map<std::size_t, std::vector<const char *>> table = {
{8, {"CRC32", "Adler-32"}},
{16, {"MySQL 3.x", "CRC64"}},
{32, {"MD5", "MD4", "NTLM", "LM", "MD2", "RIPEMD-128", "HAVAL-128"}},
{40, {"SHA-1", "RIPEMD-160", "HAVAL-160", "MySQL 5.x (SHA1(SHA1))", "Tiger-160"}},
{56, {"SHA-224", "SHA3-224", "BLAKE2s-224", "HAVAL-224"}},
{64, {"SHA-256", "SHA3-256", "BLAKE2s-256", "RIPEMD-256", "Skein-256"}},
{96, {"SHA-384", "SHA3-384", "BLAKE2b-384"}},
{128, {"SHA-512", "SHA3-512", "BLAKE2b-512", "Whirlpool", "Skein-512"}},
};
return table;
}
/// Base64 candidates keyed by encoded-string length (16-byte MD5 digest ->
/// 24 base64 chars including padding, etc.).
const std::unordered_map<std::size_t, std::vector<const char *>> &base64_by_length()
{
static const std::unordered_map<std::size_t, std::vector<const char *>> table = {
{24, {"MD5 (base64)"}},
{28, {"SHA-1 (base64)"}},
{44, {"SHA-256 (base64)"}},
{88, {"SHA-512 (base64)"}},
};
return table;
}
} // namespace detail
/// Classify the charset of a candidate hash string.
///
/// Order matters: hex is checked before base64 because every hex digest is
/// also a legal base64 character set, and the more specific classification
/// should win.
inline hash_charset detect_charset(std::string_view s) noexcept
{
using namespace detail;
if (looks_like_bcrypt(s)) return hash_charset::bcrypt;
if (looks_like_argon2(s)) return hash_charset::argon2;
if (looks_like_hex(s)) return hash_charset::hex;
if (looks_like_base64(s)) return hash_charset::base64;
return hash_charset::unknown;
}
/// Identify candidate hash types for an input string.
///
/// Always returns a fully populated hash_info; never throws. An empty,
/// unrecognised, or wrong-length input simply yields an empty candidate
/// vector — the caller decides whether "no candidates" means "not a hash".
inline hash_info identify_hash(std::string_view input)
{
// Trim the six ASCII whitespace bytes the canonical trim() removes.
constexpr const char *whitespace = " \t\n\v\f\r";
hash_info info;
info.input = std::string(input);
const std::size_t first = input.find_first_not_of(whitespace);
if (first == std::string_view::npos) {
info.cleaned = std::string();
} else {
const std::size_t last = input.find_last_not_of(whitespace);
info.cleaned = std::string(input.substr(first, last - first + 1));
}
info.length = info.cleaned.size();
info.charset = detect_charset(info.cleaned);
switch (info.charset) {
case hash_charset::bcrypt:
// bcrypt's modular-crypt token encodes a 184-bit effective hash.
info.candidates.push_back({"bcrypt", 184});
break;
case hash_charset::argon2:
// Argon2 output length is parameter-driven, so no fixed bit length applies.
info.candidates.push_back({"Argon2", 0});
break;
case hash_charset::hex: {
const auto it = detail::hex_by_length().find(info.length);
if (it != detail::hex_by_length().end()) {
// length*4 converts hex-char count to a bit width (4 bits per nibble).
const auto bits = static_cast<long long>(info.length) * 4;
for (const char *name : it->second)
info.candidates.push_back({name, bits});
}
break;
}
case hash_charset::base64: {
const auto it = detail::base64_by_length().find(info.length);
if (it != detail::base64_by_length().end()) {
// Each base64 char carries 6 bits; round to the nearest byte
// boundary. All table lengths divide evenly, so the integer
// division is exact.
const auto bits = static_cast<long long>(info.length * 6 / 8) * 8;
for (const char *name : it->second)
info.candidates.push_back({name, bits});
}
break;
}
case hash_charset::unknown:
break;
}
return info;
}
} // namespace hash_identify
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →