Hash Type Identifier — Rust source
Identify the likely hash algorithm of a hash string by its length and character set - MD5, SHA-1/2/3, BLAKE, CRC32, NTLM, bcrypt, Argon2 and more.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Hash-type identifier — Rust port.
//
// Language: Rust
// CosmoDev polyglot showcase port of the `hash-type-identifier` tool.
// Ported from src/lib/hashIdentify.ts.
//
// Display source — part of CosmoDev's polyglot tool pages
// (dev.cosmolabs.org).
//
// Pure string classification: inspect a candidate hash's charset and length
// to suggest likely algorithms. No hashing happens here — this is pattern
// recognition over an already-computed digest. Deterministic; never panics.
//
// The standard library ships no regex engine, so the four detection rules
// are written as small, self-contained byte matchers — no external crates.
/// The character set classification of a candidate hash string.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum HashCharset {
Hex,
Base64,
Bcrypt,
Argon2,
Unknown,
}
impl HashCharset {
/// Lowercase identifier matching the TypeScript string literal used by
/// the canonical implementation (so the JSON/serialised output agrees).
pub fn as_str(self) -> &'static str {
match self {
HashCharset::Hex => "hex",
HashCharset::Base64 => "base64",
HashCharset::Bcrypt => "bcrypt",
HashCharset::Argon2 => "argon2",
HashCharset::Unknown => "unknown",
}
}
}
/// A candidate hash algorithm and its nominal bit length.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct HashMatch {
pub name: String,
pub bit_length: i64, // hex length * 4, where applicable
}
/// The full identification result for an input string.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct HashInfo {
pub input: String,
pub cleaned: String,
pub length: usize,
pub charset: HashCharset,
pub candidates: Vec<HashMatch>,
}
/// Hex candidates for a given hex-string length, if any.
///
/// Each hex char encodes 4 bits, so a 64-char digest implies a 256-bit
/// algorithm such as SHA-256. Returns `None` for lengths with no known
/// algorithm family.
fn hex_candidates(len: usize) -> Option<&'static [&'static str]> {
match len {
8 => Some(&["CRC32", "Adler-32"]),
16 => Some(&["MySQL 3.x", "CRC64"]),
32 => Some(&[
"MD5",
"MD4",
"NTLM",
"LM",
"MD2",
"RIPEMD-128",
"HAVAL-128",
]),
40 => Some(&[
"SHA-1",
"RIPEMD-160",
"HAVAL-160",
"MySQL 5.x (SHA1(SHA1))",
"Tiger-160",
]),
56 => Some(&["SHA-224", "SHA3-224", "BLAKE2s-224", "HAVAL-224"]),
64 => Some(&[
"SHA-256",
"SHA3-256",
"BLAKE2s-256",
"RIPEMD-256",
"Skein-256",
]),
96 => Some(&["SHA-384", "SHA3-384", "BLAKE2b-384"]),
128 => Some(&[
"SHA-512",
"SHA3-512",
"BLAKE2b-512",
"Whirlpool",
"Skein-512",
]),
_ => None,
}
}
/// Base64 candidates for a given encoded-string length, if any
/// (16-byte MD5 digest -> 24 base64 chars including padding, etc.).
fn base64_candidates(len: usize) -> Option<&'static [&'static str]> {
match len {
24 => Some(&["MD5 (base64)"]),
28 => Some(&["SHA-1 (base64)"]),
44 => Some(&["SHA-256 (base64)"]),
88 => Some(&["SHA-512 (base64)"]),
_ => None,
}
}
/// Matches the bcrypt modular-crypt prefix `^\$2[abxy]?\$`.
///
/// bcrypt tokens begin `$2`, an optional variant letter (`a`/`b`/`x`/`y`),
/// then `$`. Like the original regex this is a *prefix* match — the variable
/// trailing payload is not inspected here.
fn looks_like_bcrypt(s: &[u8]) -> bool {
if !s.starts_with(b"$2") {
return false;
}
match s.get(2) {
// `$2a$`, `$2b$`, `$2x$`, `$2y$` — variant letter then a dollar sign.
Some(b'a') | Some(b'b') | Some(b'x') | Some(b'y') => s.get(3) == Some(&b'$'),
// `$2$` — no variant letter.
Some(b'$') => true,
_ => false,
}
}
/// Matches the argon2 modular-crypt prefix `^\$argon2(id|i|d)?\$`.
///
/// Tokens begin `$argon2`, an optional variant (`id`, `i`, or `d`), then `$`.
/// Also a prefix match, mirroring the original regex.
fn looks_like_argon2(s: &[u8]) -> bool {
let rest = match s.strip_prefix(b"$argon2") {
Some(r) => r,
None => return false,
};
// Try the two-char variant first so `id` wins over the bare `i` branch.
if rest.starts_with(b"id") {
return rest.get(2) == Some(&b'$');
}
match rest.first() {
Some(b'i') | Some(b'd') => rest.get(1) == Some(&b'$'),
Some(b'$') => true,
_ => false,
}
}
/// True when every byte of `s` is an ASCII hex digit. Requires a non-empty
/// body, matching the `+` quantifier in the canonical regex.
fn looks_like_hex(s: &[u8]) -> bool {
!s.is_empty() && s.iter().all(|&b| b.is_ascii_hexdigit())
}
/// True when `s` is valid standard-alphabet base64 with 0–2 trailing `=`
/// padding. The body (before any padding) must be non-empty.
fn looks_like_base64(s: &[u8]) -> bool {
// Strip up to two trailing `=` padding characters, then require the
// remaining body to be non-empty and entirely base64 alphabet bytes.
let mut end = s.len();
let mut pad = 0;
while end > 0 && s[end - 1] == b'=' && pad < 2 {
end -= 1;
pad += 1;
}
let body = &s[..end];
!body.is_empty() && body.iter().all(|&b| b.is_ascii_alphanumeric() || b == b'+' || b == b'/')
}
/// Classify the charset of a candidate hash string.
///
/// Order matters: hex is checked before base64 because every hex digest is
/// also a legal base64 character set, and the more specific classification
/// should win.
pub fn detect_charset(s: &str) -> HashCharset {
let b = s.as_bytes();
if looks_like_bcrypt(b) {
HashCharset::Bcrypt
} else if looks_like_argon2(b) {
HashCharset::Argon2
} else if looks_like_hex(b) {
HashCharset::Hex
} else if looks_like_base64(b) {
HashCharset::Base64
} else {
HashCharset::Unknown
}
}
/// Identify candidate hash types for an input string.
///
/// Always returns a fully populated `HashInfo`; never panics. An empty,
/// unrecognised, or wrong-length input simply yields an empty candidate
/// vector — the caller decides whether "no candidates" means "not a hash".
pub fn identify_hash(input: &str) -> HashInfo {
let cleaned = input.trim();
let charset = detect_charset(cleaned);
let length = cleaned.len();
let mut candidates = Vec::new();
match charset {
HashCharset::Bcrypt => {
// bcrypt's modular-crypt token encodes a 184-bit effective hash.
candidates.push(HashMatch {
name: "bcrypt".to_string(),
bit_length: 184,
});
}
HashCharset::Argon2 => {
// Argon2 output length is parameter-driven, so no fixed bit length applies.
candidates.push(HashMatch {
name: "Argon2".to_string(),
bit_length: 0,
});
}
HashCharset::Hex => {
if let Some(names) = hex_candidates(length) {
// length*4 converts hex-char count to a bit width (4 bits per nibble).
for &name in names {
candidates.push(HashMatch {
name: name.to_string(),
bit_length: (length as i64) * 4,
});
}
}
}
HashCharset::Base64 => {
if let Some(names) = base64_candidates(length) {
// Each base64 char carries 6 bits; round to the nearest byte boundary
// so the reported length lines up with the underlying digest width.
let bits = ((length as f64) * 6.0 / 8.0).round() as i64 * 8;
for &name in names {
candidates.push(HashMatch {
name: name.to_string(),
bit_length: bits,
});
}
}
}
HashCharset::Unknown => {}
}
HashInfo {
input: input.to_string(),
cleaned: cleaned.to_string(),
length,
charset,
candidates,
}
}
#[cfg(test)]
mod tests {
use super::*;
/// Sanity check that the prefix matchers agree with the canonical intent:
/// bcrypt/argon2 are recognised by their `$...$` header alone.
#[test]
fn detects_modular_crypt_prefixes() {
assert_eq!(detect_charset("$2a$abc..."), HashCharset::Bcrypt);
assert_eq!(detect_charset("$2y$xyz..."), HashCharset::Bcrypt);
assert_eq!(detect_charset("$argon2id$foo"), HashCharset::Argon2);
assert_eq!(detect_charset("$argon2$bar"), HashCharset::Argon2);
}
/// Hex must win over base64 for an all-hex-digit string.
#[test]
fn prefers_hex_over_base64() {
assert_eq!(
detect_charset("d41d8cd98f00b204e9800998ecf8427e"),
HashCharset::Hex
);
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →