Skip to content

Hash Type Identifier — Zig source

Identify the likely hash algorithm of a hash string by its length and character set - MD5, SHA-1/2/3, BLAKE, CRC32, NTLM, bcrypt, Argon2 and more.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! Hash-type identifier — Zig port.
//!
//! Language: Zig 0.13 (standard library only)
//! Source: CosmoDev polyglot showcase port of the `hash-type-identifier`
//!        tool, ported from src/lib/hashIdentify.ts (the canonical
//!        TypeScript implementation).
//! License: display source — part of CosmoDev's polyglot tool pages
//!        (dev.cosmolabs.org).
//!
//! Pure string classification: inspect a candidate hash's charset and length
//! to suggest likely algorithms. No hashing happens here — this is pattern
//! recognition over an already-computed digest. Deterministic; the only
//! error path is allocation failure, surfaced honestly as an error union.
//!
//! The standard library ships no regex engine, so the four detection rules
//! are written as small, self-contained byte matchers — the same shape as
//! the Rust port.

const std = @import("std");

/// The character set classification of a candidate hash string.
pub const HashCharset = enum {
    hex,
    base64,
    bcrypt,
    argon2,
    unknown,

    /// Lowercase identifier matching the TypeScript string literal used by
    /// the canonical implementation (so serialised output agrees).
    pub fn asStr(self: HashCharset) []const u8 {
        return switch (self) {
            .hex => "hex",
            .base64 => "base64",
            .bcrypt => "bcrypt",
            .argon2 => "argon2",
            .unknown => "unknown",
        };
    }
};

/// A candidate hash algorithm and its nominal bit length.
pub const HashMatch = struct {
    name: []const u8,
    bit_length: i64, // hex length * 4, where applicable
};

/// The full identification result for an input string.
pub const HashInfo = struct {
    input: []const u8,
    cleaned: []const u8, // trimmed view into `input` — borrowed, not owned
    length: usize,
    charset: HashCharset,
    candidates: []const HashMatch, // owned by the result; free with deinit

    /// Release the candidate slice. `input`/`cleaned` are borrowed views
    /// into the caller's string and are not freed.
    pub fn deinit(self: *HashInfo, allocator: std.mem.Allocator) void {
        allocator.free(self.candidates);
        self.candidates = &.{};
    }
};

/// Hex candidates for a given hex-string length, if any.
///
/// Each hex char encodes 4 bits, so a 64-char digest implies a 256-bit
/// algorithm such as SHA-256. Returns null for lengths with no known
/// algorithm family.
fn hexCandidates(len: usize) ?[]const []const u8 {
    return switch (len) {
        8 => &[_][]const u8{ "CRC32", "Adler-32" },
        16 => &[_][]const u8{ "MySQL 3.x", "CRC64" },
        32 => &[_][]const u8{ "MD5", "MD4", "NTLM", "LM", "MD2", "RIPEMD-128", "HAVAL-128" },
        40 => &[_][]const u8{ "SHA-1", "RIPEMD-160", "HAVAL-160", "MySQL 5.x (SHA1(SHA1))", "Tiger-160" },
        56 => &[_][]const u8{ "SHA-224", "SHA3-224", "BLAKE2s-224", "HAVAL-224" },
        64 => &[_][]const u8{ "SHA-256", "SHA3-256", "BLAKE2s-256", "RIPEMD-256", "Skein-256" },
        96 => &[_][]const u8{ "SHA-384", "SHA3-384", "BLAKE2b-384" },
        128 => &[_][]const u8{ "SHA-512", "SHA3-512", "BLAKE2b-512", "Whirlpool", "Skein-512" },
        else => null,
    };
}

/// Base64 candidates for a given encoded-string length, if any
/// (16-byte MD5 digest -> 24 base64 chars including padding, etc.).
fn base64Candidates(len: usize) ?[]const []const u8 {
    return switch (len) {
        24 => &[_][]const u8{"MD5 (base64)"},
        28 => &[_][]const u8{"SHA-1 (base64)"},
        44 => &[_][]const u8{"SHA-256 (base64)"},
        88 => &[_][]const u8{"SHA-512 (base64)"},
        else => null,
    };
}

/// Matches the bcrypt modular-crypt prefix `^\$2[abxy]?\$`.
///
/// bcrypt tokens begin `$2`, an optional variant letter (`a`/`b`/`x`/`y`),
/// then `$`. Like the original regex this is a *prefix* match — the variable
/// trailing payload is not inspected here.
fn looksLikeBcrypt(s: []const u8) bool {
    if (!std.mem.startsWith(u8, s, "$2") or s.len < 3) return false;
    return switch (s[2]) {
        // `$2a$`, `$2b$`, `$2x$`, `$2y$` — variant letter then a dollar sign.
        'a', 'b', 'x', 'y' => s.len >= 4 and s[3] == '$',
        // `$2$` — no variant letter.
        '$' => true,
        else => false,
    };
}

/// Matches the argon2 modular-crypt prefix `^\$argon2(id|i|d)?\$`.
///
/// Tokens begin `$argon2`, an optional variant (`id`, `i`, or `d`), then `$`.
/// Also a prefix match, mirroring the original regex.
fn looksLikeArgon2(s: []const u8) bool {
    if (!std.mem.startsWith(u8, s, "$argon2")) return false;
    const rest = s["$argon2".len..];
    if (rest.len == 0) return false;
    // Try the two-char variant first so `id` wins over the bare `i` branch.
    if (rest.len >= 2 and rest[0] == 'i' and rest[1] == 'd')
        return rest.len >= 3 and rest[2] == '$';
    return switch (rest[0]) {
        'i', 'd' => rest.len >= 2 and rest[1] == '$',
        '$' => true,
        else => false,
    };
}

/// True when every byte of `s` is an ASCII hex digit. Requires a non-empty
/// body, matching the `+` quantifier in the canonical regex.
fn looksLikeHex(s: []const u8) bool {
    if (s.len == 0) return false;
    for (s) |c| {
        if (!std.ascii.isHex(c)) return false;
    }
    return true;
}

/// True when `s` is valid standard-alphabet base64 with 0–2 trailing `=`
/// padding. The body (before any padding) must be non-empty.
fn looksLikeBase64(s: []const u8) bool {
    // Strip up to two trailing `=` padding characters, then require the
    // remaining body to be non-empty and entirely base64 alphabet bytes.
    var end = s.len;
    var pad: usize = 0;
    while (end > 0 and s[end - 1] == '=' and pad < 2) {
        end -= 1;
        pad += 1;
    }
    if (end == 0) return false;
    for (s[0..end]) |c| {
        if (!std.ascii.isAlphanumeric(c) and c != '+' and c != '/') return false;
    }
    return true;
}

/// Classify the charset of a candidate hash string.
///
/// Order matters: hex is checked before base64 because every hex digest is
/// also a legal base64 character set, and the more specific classification
/// should win.
pub fn detectCharset(s: []const u8) HashCharset {
    if (looksLikeBcrypt(s)) return .bcrypt;
    if (looksLikeArgon2(s)) return .argon2;
    if (looksLikeHex(s)) return .hex;
    if (looksLikeBase64(s)) return .base64;
    return .unknown;
}

/// Identify candidate hash types for an input string.
///
/// Always returns a fully populated `HashInfo`; the only error is
/// `error.OutOfMemory`. An empty, unrecognised, or wrong-length input
/// simply yields an empty candidate list — the caller decides whether "no
/// candidates" means "not a hash".
pub fn identifyHash(allocator: std.mem.Allocator, input: []const u8) std.mem.Allocator.Error!HashInfo {
    // Trim the six ASCII whitespace bytes Python's str.strip() removes.
    const cleaned = std.mem.trim(u8, input, " \t\n\r\x0b\x0c");
    const charset = detectCharset(cleaned);
    const length = cleaned.len;

    var list = std.ArrayList(HashMatch).init(allocator);
    errdefer list.deinit();

    switch (charset) {
        .bcrypt => {
            // bcrypt's modular-crypt token encodes a 184-bit effective hash.
            try list.append(.{ .name = "bcrypt", .bit_length = 184 });
        },
        .argon2 => {
            // Argon2 output length is parameter-driven, so no fixed bit length applies.
            try list.append(.{ .name = "Argon2", .bit_length = 0 });
        },
        .hex => {
            if (hexCandidates(length)) |names| {
                // length*4 converts hex-char count to a bit width (4 bits per nibble).
                for (names) |name|
                    try list.append(.{ .name = name, .bit_length = @intCast(length * 4) });
            }
        },
        .base64 => {
            if (base64Candidates(length)) |names| {
                // Each base64 char carries 6 bits; round to the nearest byte
                // boundary so the reported length lines up with the underlying
                // digest width. All table lengths divide evenly, so the integer
                // division is exact.
                const bits: i64 = @intCast(length * 6 / 8 * 8);
                for (names) |name|
                    try list.append(.{ .name = name, .bit_length = bits });
            }
        },
        .unknown => {},
    }

    return .{
        .input = input,
        .cleaned = cleaned,
        .length = length,
        .charset = charset,
        .candidates = try list.toOwnedSlice(),
    };
}

test "detects modular-crypt prefixes" {
    // Sanity check that the prefix matchers agree with the canonical intent:
    // bcrypt/argon2 are recognised by their `$...$` header alone.
    try std.testing.expectEqual(HashCharset.bcrypt, detectCharset("$2a$abc..."));
    try std.testing.expectEqual(HashCharset.bcrypt, detectCharset("$2y$xyz..."));
    try std.testing.expectEqual(HashCharset.argon2, detectCharset("$argon2id$foo"));
    try std.testing.expectEqual(HashCharset.argon2, detectCharset("$argon2$bar"));
}

test "prefers hex over base64" {
    // Hex must win over base64 for an all-hex-digit string.
    try std.testing.expectEqual(HashCharset.hex, detectCharset("d41d8cd98f00b204e9800998ecf8427e"));
}

test "identify maps length to candidates" {
    const md5_hex = "d41d8cd98f00b204e9800998ecf8427e";
    var info = try identifyHash(std.testing.allocator, md5_hex);
    defer info.deinit(std.testing.allocator);

    try std.testing.expectEqual(@as(usize, 32), info.length);
    try std.testing.expectEqual(@as(usize, 7), info.candidates.len);
    try std.testing.expectEqualStrings("MD5", info.candidates[0].name);
    try std.testing.expectEqual(@as(i64, 128), info.candidates[0].bit_length);
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →