Skip to content

Hash Type Identifier — Python source

Identify the likely hash algorithm of a hash string by its length and character set - MD5, SHA-1/2/3, BLAKE, CRC32, NTLM, bcrypt, Argon2 and more.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""Hash-type identifier — Python port.

Language: Python
CosmoDev polyglot showcase port of the ``hash-type-identifier`` tool.
Ported from src/lib/hashIdentify.ts.

Display source — part of CosmoDev's polyglot tool pages
(dev.cosmolabs.org).

Pure string classification: inspect a candidate hash's charset and length to
suggest likely algorithms. No hashing happens here — this is pattern
recognition over an already-computed digest. Deterministic; never raises.
"""

from __future__ import annotations

import re
from dataclasses import dataclass, field
from typing import List, Literal

# Charset label union — mirrors the TypeScript ``HashCharset`` literal union.
HashCharset = Literal["hex", "base64", "bcrypt", "argon2", "unknown"]


# Hex candidates keyed by hex-string length. Each hex char encodes 4 bits, so
# a 64-char hex digest implies a 256-bit algorithm such as SHA-256.
HEX_BY_LENGTH: dict[int, List[str]] = {
    8: ["CRC32", "Adler-32"],
    16: ["MySQL 3.x", "CRC64"],
    32: ["MD5", "MD4", "NTLM", "LM", "MD2", "RIPEMD-128", "HAVAL-128"],
    40: ["SHA-1", "RIPEMD-160", "HAVAL-160", "MySQL 5.x (SHA1(SHA1))", "Tiger-160"],
    56: ["SHA-224", "SHA3-224", "BLAKE2s-224", "HAVAL-224"],
    64: ["SHA-256", "SHA3-256", "BLAKE2s-256", "RIPEMD-256", "Skein-256"],
    96: ["SHA-384", "SHA3-384", "BLAKE2b-384"],
    128: ["SHA-512", "SHA3-512", "BLAKE2b-512", "Whirlpool", "Skein-512"],
}

# Base64 candidates keyed by encoded-string length (16-byte MD5 digest ->
# 24 base64 chars including padding, etc.).
BASE64_BY_LENGTH: dict[int, List[str]] = {
    24: ["MD5 (base64)"],
    28: ["SHA-1 (base64)"],
    44: ["SHA-256 (base64)"],
    88: ["SHA-512 (base64)"],
}


@dataclass
class HashMatch:
    """A candidate hash algorithm and its nominal bit length."""

    name: str
    bit_length: int  # hex length * 4, where applicable


@dataclass
class HashInfo:
    """The full identification result for an input string."""

    input: str
    cleaned: str  # trimmed input
    length: int
    charset: HashCharset
    candidates: List[HashMatch] = field(default_factory=list)


# Detection patterns. bcrypt and argon2 use their modular-crypt ``$...$``
# format, so they are matched by prefix (the trailing payload is variable);
# the patterns carry no end anchor. hex and base64 match the whole string,
# and hex is checked first because any hex digest is also a legal base64
# character set.
_RE_BCRYPT = re.compile(r"^\$2[abxy]?\$")
_RE_ARGON2 = re.compile(r"^\$argon2(id|i|d)?\$")
_RE_HEX = re.compile(r"^[0-9a-fA-F]+$")
_RE_BASE64 = re.compile(r"^[A-Za-z0-9+/]+={0,2}$")


def detect_charset(s: str) -> HashCharset:
    """Classify the charset of a candidate hash string."""
    if _RE_BCRYPT.match(s):
        return "bcrypt"
    if _RE_ARGON2.match(s):
        return "argon2"
    if _RE_HEX.match(s):
        return "hex"
    if _RE_BASE64.match(s):
        return "base64"
    return "unknown"


def identify_hash(input: str) -> HashInfo:
    """Identify candidate hash types for an input string.

    Always returns a :class:`HashInfo`; never raises. An empty, unrecognised,
    or wrong-length input simply yields an empty ``candidates`` list — the
    caller decides whether "no candidates" means "not a hash".
    """
    cleaned = (input or "").strip()
    charset = detect_charset(cleaned)
    length = len(cleaned)
    candidates: List[HashMatch] = []

    if charset == "bcrypt":
        # bcrypt's modular-crypt token encodes a 184-bit effective hash.
        candidates.append(HashMatch(name="bcrypt", bit_length=184))
    elif charset == "argon2":
        # Argon2 output length is parameter-driven, so no fixed bit length applies.
        candidates.append(HashMatch(name="Argon2", bit_length=0))
    elif charset == "hex":
        # length*4 converts hex-char count to a bit width (4 bits per nibble).
        for name in HEX_BY_LENGTH.get(length, []):
            candidates.append(HashMatch(name=name, bit_length=length * 4))
    elif charset == "base64":
        # Each base64 char carries 6 bits; round to the nearest byte boundary so
        # the reported length lines up with the underlying digest width. All
        # table lengths divide evenly, so rounding is exact here.
        bit_length = round((length * 6) / 8) * 8
        for name in BASE64_BY_LENGTH.get(length, []):
            candidates.append(HashMatch(name=name, bit_length=bit_length))

    return HashInfo(
        input=input or "",
        cleaned=cleaned,
        length=length,
        charset=charset,
        candidates=candidates,
    )

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →