Skip to content

Hash Type Identifier — Rust source

Identify the likely hash algorithm of a hash string by its length and character set - MD5, SHA-1/2/3, BLAKE, CRC32, NTLM, bcrypt, Argon2 and more.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Hash-type identifier — Rust port.
//
// Language: Rust
// CosmoDev polyglot showcase port of the `hash-type-identifier` tool.
// Ported from src/lib/hashIdentify.ts.
//
// Display source — part of CosmoDev's polyglot tool pages
// (dev.cosmolabs.org).
//
// Pure string classification: inspect a candidate hash's charset and length
// to suggest likely algorithms. No hashing happens here — this is pattern
// recognition over an already-computed digest. Deterministic; never panics.
//
// The standard library ships no regex engine, so the four detection rules
// are written as small, self-contained byte matchers — no external crates.

/// The character set classification of a candidate hash string.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum HashCharset {
    Hex,
    Base64,
    Bcrypt,
    Argon2,
    Unknown,
}

impl HashCharset {
    /// Lowercase identifier matching the TypeScript string literal used by
    /// the canonical implementation (so the JSON/serialised output agrees).
    pub fn as_str(self) -> &'static str {
        match self {
            HashCharset::Hex => "hex",
            HashCharset::Base64 => "base64",
            HashCharset::Bcrypt => "bcrypt",
            HashCharset::Argon2 => "argon2",
            HashCharset::Unknown => "unknown",
        }
    }
}

/// A candidate hash algorithm and its nominal bit length.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct HashMatch {
    pub name: String,
    pub bit_length: i64, // hex length * 4, where applicable
}

/// The full identification result for an input string.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct HashInfo {
    pub input: String,
    pub cleaned: String,
    pub length: usize,
    pub charset: HashCharset,
    pub candidates: Vec<HashMatch>,
}

/// Hex candidates for a given hex-string length, if any.
///
/// Each hex char encodes 4 bits, so a 64-char digest implies a 256-bit
/// algorithm such as SHA-256. Returns `None` for lengths with no known
/// algorithm family.
fn hex_candidates(len: usize) -> Option<&'static [&'static str]> {
    match len {
        8 => Some(&["CRC32", "Adler-32"]),
        16 => Some(&["MySQL 3.x", "CRC64"]),
        32 => Some(&[
            "MD5",
            "MD4",
            "NTLM",
            "LM",
            "MD2",
            "RIPEMD-128",
            "HAVAL-128",
        ]),
        40 => Some(&[
            "SHA-1",
            "RIPEMD-160",
            "HAVAL-160",
            "MySQL 5.x (SHA1(SHA1))",
            "Tiger-160",
        ]),
        56 => Some(&["SHA-224", "SHA3-224", "BLAKE2s-224", "HAVAL-224"]),
        64 => Some(&[
            "SHA-256",
            "SHA3-256",
            "BLAKE2s-256",
            "RIPEMD-256",
            "Skein-256",
        ]),
        96 => Some(&["SHA-384", "SHA3-384", "BLAKE2b-384"]),
        128 => Some(&[
            "SHA-512",
            "SHA3-512",
            "BLAKE2b-512",
            "Whirlpool",
            "Skein-512",
        ]),
        _ => None,
    }
}

/// Base64 candidates for a given encoded-string length, if any
/// (16-byte MD5 digest -> 24 base64 chars including padding, etc.).
fn base64_candidates(len: usize) -> Option<&'static [&'static str]> {
    match len {
        24 => Some(&["MD5 (base64)"]),
        28 => Some(&["SHA-1 (base64)"]),
        44 => Some(&["SHA-256 (base64)"]),
        88 => Some(&["SHA-512 (base64)"]),
        _ => None,
    }
}

/// Matches the bcrypt modular-crypt prefix `^\$2[abxy]?\$`.
///
/// bcrypt tokens begin `$2`, an optional variant letter (`a`/`b`/`x`/`y`),
/// then `$`. Like the original regex this is a *prefix* match — the variable
/// trailing payload is not inspected here.
fn looks_like_bcrypt(s: &[u8]) -> bool {
    if !s.starts_with(b"$2") {
        return false;
    }
    match s.get(2) {
        // `$2a$`, `$2b$`, `$2x$`, `$2y$` — variant letter then a dollar sign.
        Some(b'a') | Some(b'b') | Some(b'x') | Some(b'y') => s.get(3) == Some(&b'$'),
        // `$2$` — no variant letter.
        Some(b'$') => true,
        _ => false,
    }
}

/// Matches the argon2 modular-crypt prefix `^\$argon2(id|i|d)?\$`.
///
/// Tokens begin `$argon2`, an optional variant (`id`, `i`, or `d`), then `$`.
/// Also a prefix match, mirroring the original regex.
fn looks_like_argon2(s: &[u8]) -> bool {
    let rest = match s.strip_prefix(b"$argon2") {
        Some(r) => r,
        None => return false,
    };
    // Try the two-char variant first so `id` wins over the bare `i` branch.
    if rest.starts_with(b"id") {
        return rest.get(2) == Some(&b'$');
    }
    match rest.first() {
        Some(b'i') | Some(b'd') => rest.get(1) == Some(&b'$'),
        Some(b'$') => true,
        _ => false,
    }
}

/// True when every byte of `s` is an ASCII hex digit. Requires a non-empty
/// body, matching the `+` quantifier in the canonical regex.
fn looks_like_hex(s: &[u8]) -> bool {
    !s.is_empty() && s.iter().all(|&b| b.is_ascii_hexdigit())
}

/// True when `s` is valid standard-alphabet base64 with 0–2 trailing `=`
/// padding. The body (before any padding) must be non-empty.
fn looks_like_base64(s: &[u8]) -> bool {
    // Strip up to two trailing `=` padding characters, then require the
    // remaining body to be non-empty and entirely base64 alphabet bytes.
    let mut end = s.len();
    let mut pad = 0;
    while end > 0 && s[end - 1] == b'=' && pad < 2 {
        end -= 1;
        pad += 1;
    }
    let body = &s[..end];
    !body.is_empty() && body.iter().all(|&b| b.is_ascii_alphanumeric() || b == b'+' || b == b'/')
}

/// Classify the charset of a candidate hash string.
///
/// Order matters: hex is checked before base64 because every hex digest is
/// also a legal base64 character set, and the more specific classification
/// should win.
pub fn detect_charset(s: &str) -> HashCharset {
    let b = s.as_bytes();
    if looks_like_bcrypt(b) {
        HashCharset::Bcrypt
    } else if looks_like_argon2(b) {
        HashCharset::Argon2
    } else if looks_like_hex(b) {
        HashCharset::Hex
    } else if looks_like_base64(b) {
        HashCharset::Base64
    } else {
        HashCharset::Unknown
    }
}

/// Identify candidate hash types for an input string.
///
/// Always returns a fully populated `HashInfo`; never panics. An empty,
/// unrecognised, or wrong-length input simply yields an empty candidate
/// vector — the caller decides whether "no candidates" means "not a hash".
pub fn identify_hash(input: &str) -> HashInfo {
    let cleaned = input.trim();
    let charset = detect_charset(cleaned);
    let length = cleaned.len();
    let mut candidates = Vec::new();

    match charset {
        HashCharset::Bcrypt => {
            // bcrypt's modular-crypt token encodes a 184-bit effective hash.
            candidates.push(HashMatch {
                name: "bcrypt".to_string(),
                bit_length: 184,
            });
        }
        HashCharset::Argon2 => {
            // Argon2 output length is parameter-driven, so no fixed bit length applies.
            candidates.push(HashMatch {
                name: "Argon2".to_string(),
                bit_length: 0,
            });
        }
        HashCharset::Hex => {
            if let Some(names) = hex_candidates(length) {
                // length*4 converts hex-char count to a bit width (4 bits per nibble).
                for &name in names {
                    candidates.push(HashMatch {
                        name: name.to_string(),
                        bit_length: (length as i64) * 4,
                    });
                }
            }
        }
        HashCharset::Base64 => {
            if let Some(names) = base64_candidates(length) {
                // Each base64 char carries 6 bits; round to the nearest byte boundary
                // so the reported length lines up with the underlying digest width.
                let bits = ((length as f64) * 6.0 / 8.0).round() as i64 * 8;
                for &name in names {
                    candidates.push(HashMatch {
                        name: name.to_string(),
                        bit_length: bits,
                    });
                }
            }
        }
        HashCharset::Unknown => {}
    }

    HashInfo {
        input: input.to_string(),
        cleaned: cleaned.to_string(),
        length,
        charset,
        candidates,
    }
}

#[cfg(test)]
mod tests {
    use super::*;

    /// Sanity check that the prefix matchers agree with the canonical intent:
    /// bcrypt/argon2 are recognised by their `$...$` header alone.
    #[test]
    fn detects_modular_crypt_prefixes() {
        assert_eq!(detect_charset("$2a$abc..."), HashCharset::Bcrypt);
        assert_eq!(detect_charset("$2y$xyz..."), HashCharset::Bcrypt);
        assert_eq!(detect_charset("$argon2id$foo"), HashCharset::Argon2);
        assert_eq!(detect_charset("$argon2$bar"), HashCharset::Argon2);
    }

    /// Hex must win over base64 for an all-hex-digit string.
    #[test]
    fn prefers_hex_over_base64() {
        assert_eq!(
            detect_charset("d41d8cd98f00b204e9800998ecf8427e"),
            HashCharset::Hex
        );
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →