Hash Type Identifier — Zig source
Identify the likely hash algorithm of a hash string by its length and character set - MD5, SHA-1/2/3, BLAKE, CRC32, NTLM, bcrypt, Argon2 and more.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! Hash-type identifier — Zig port.
//!
//! Language: Zig 0.13 (standard library only)
//! Source: CosmoDev polyglot showcase port of the `hash-type-identifier`
//! tool, ported from src/lib/hashIdentify.ts (the canonical
//! TypeScript implementation).
//! License: display source — part of CosmoDev's polyglot tool pages
//! (dev.cosmolabs.org).
//!
//! Pure string classification: inspect a candidate hash's charset and length
//! to suggest likely algorithms. No hashing happens here — this is pattern
//! recognition over an already-computed digest. Deterministic; the only
//! error path is allocation failure, surfaced honestly as an error union.
//!
//! The standard library ships no regex engine, so the four detection rules
//! are written as small, self-contained byte matchers — the same shape as
//! the Rust port.
const std = @import("std");
/// The character set classification of a candidate hash string.
pub const HashCharset = enum {
hex,
base64,
bcrypt,
argon2,
unknown,
/// Lowercase identifier matching the TypeScript string literal used by
/// the canonical implementation (so serialised output agrees).
pub fn asStr(self: HashCharset) []const u8 {
return switch (self) {
.hex => "hex",
.base64 => "base64",
.bcrypt => "bcrypt",
.argon2 => "argon2",
.unknown => "unknown",
};
}
};
/// A candidate hash algorithm and its nominal bit length.
pub const HashMatch = struct {
name: []const u8,
bit_length: i64, // hex length * 4, where applicable
};
/// The full identification result for an input string.
pub const HashInfo = struct {
input: []const u8,
cleaned: []const u8, // trimmed view into `input` — borrowed, not owned
length: usize,
charset: HashCharset,
candidates: []const HashMatch, // owned by the result; free with deinit
/// Release the candidate slice. `input`/`cleaned` are borrowed views
/// into the caller's string and are not freed.
pub fn deinit(self: *HashInfo, allocator: std.mem.Allocator) void {
allocator.free(self.candidates);
self.candidates = &.{};
}
};
/// Hex candidates for a given hex-string length, if any.
///
/// Each hex char encodes 4 bits, so a 64-char digest implies a 256-bit
/// algorithm such as SHA-256. Returns null for lengths with no known
/// algorithm family.
fn hexCandidates(len: usize) ?[]const []const u8 {
return switch (len) {
8 => &[_][]const u8{ "CRC32", "Adler-32" },
16 => &[_][]const u8{ "MySQL 3.x", "CRC64" },
32 => &[_][]const u8{ "MD5", "MD4", "NTLM", "LM", "MD2", "RIPEMD-128", "HAVAL-128" },
40 => &[_][]const u8{ "SHA-1", "RIPEMD-160", "HAVAL-160", "MySQL 5.x (SHA1(SHA1))", "Tiger-160" },
56 => &[_][]const u8{ "SHA-224", "SHA3-224", "BLAKE2s-224", "HAVAL-224" },
64 => &[_][]const u8{ "SHA-256", "SHA3-256", "BLAKE2s-256", "RIPEMD-256", "Skein-256" },
96 => &[_][]const u8{ "SHA-384", "SHA3-384", "BLAKE2b-384" },
128 => &[_][]const u8{ "SHA-512", "SHA3-512", "BLAKE2b-512", "Whirlpool", "Skein-512" },
else => null,
};
}
/// Base64 candidates for a given encoded-string length, if any
/// (16-byte MD5 digest -> 24 base64 chars including padding, etc.).
fn base64Candidates(len: usize) ?[]const []const u8 {
return switch (len) {
24 => &[_][]const u8{"MD5 (base64)"},
28 => &[_][]const u8{"SHA-1 (base64)"},
44 => &[_][]const u8{"SHA-256 (base64)"},
88 => &[_][]const u8{"SHA-512 (base64)"},
else => null,
};
}
/// Matches the bcrypt modular-crypt prefix `^\$2[abxy]?\$`.
///
/// bcrypt tokens begin `$2`, an optional variant letter (`a`/`b`/`x`/`y`),
/// then `$`. Like the original regex this is a *prefix* match — the variable
/// trailing payload is not inspected here.
fn looksLikeBcrypt(s: []const u8) bool {
if (!std.mem.startsWith(u8, s, "$2") or s.len < 3) return false;
return switch (s[2]) {
// `$2a$`, `$2b$`, `$2x$`, `$2y$` — variant letter then a dollar sign.
'a', 'b', 'x', 'y' => s.len >= 4 and s[3] == '$',
// `$2$` — no variant letter.
'$' => true,
else => false,
};
}
/// Matches the argon2 modular-crypt prefix `^\$argon2(id|i|d)?\$`.
///
/// Tokens begin `$argon2`, an optional variant (`id`, `i`, or `d`), then `$`.
/// Also a prefix match, mirroring the original regex.
fn looksLikeArgon2(s: []const u8) bool {
if (!std.mem.startsWith(u8, s, "$argon2")) return false;
const rest = s["$argon2".len..];
if (rest.len == 0) return false;
// Try the two-char variant first so `id` wins over the bare `i` branch.
if (rest.len >= 2 and rest[0] == 'i' and rest[1] == 'd')
return rest.len >= 3 and rest[2] == '$';
return switch (rest[0]) {
'i', 'd' => rest.len >= 2 and rest[1] == '$',
'$' => true,
else => false,
};
}
/// True when every byte of `s` is an ASCII hex digit. Requires a non-empty
/// body, matching the `+` quantifier in the canonical regex.
fn looksLikeHex(s: []const u8) bool {
if (s.len == 0) return false;
for (s) |c| {
if (!std.ascii.isHex(c)) return false;
}
return true;
}
/// True when `s` is valid standard-alphabet base64 with 0–2 trailing `=`
/// padding. The body (before any padding) must be non-empty.
fn looksLikeBase64(s: []const u8) bool {
// Strip up to two trailing `=` padding characters, then require the
// remaining body to be non-empty and entirely base64 alphabet bytes.
var end = s.len;
var pad: usize = 0;
while (end > 0 and s[end - 1] == '=' and pad < 2) {
end -= 1;
pad += 1;
}
if (end == 0) return false;
for (s[0..end]) |c| {
if (!std.ascii.isAlphanumeric(c) and c != '+' and c != '/') return false;
}
return true;
}
/// Classify the charset of a candidate hash string.
///
/// Order matters: hex is checked before base64 because every hex digest is
/// also a legal base64 character set, and the more specific classification
/// should win.
pub fn detectCharset(s: []const u8) HashCharset {
if (looksLikeBcrypt(s)) return .bcrypt;
if (looksLikeArgon2(s)) return .argon2;
if (looksLikeHex(s)) return .hex;
if (looksLikeBase64(s)) return .base64;
return .unknown;
}
/// Identify candidate hash types for an input string.
///
/// Always returns a fully populated `HashInfo`; the only error is
/// `error.OutOfMemory`. An empty, unrecognised, or wrong-length input
/// simply yields an empty candidate list — the caller decides whether "no
/// candidates" means "not a hash".
pub fn identifyHash(allocator: std.mem.Allocator, input: []const u8) std.mem.Allocator.Error!HashInfo {
// Trim the six ASCII whitespace bytes Python's str.strip() removes.
const cleaned = std.mem.trim(u8, input, " \t\n\r\x0b\x0c");
const charset = detectCharset(cleaned);
const length = cleaned.len;
var list = std.ArrayList(HashMatch).init(allocator);
errdefer list.deinit();
switch (charset) {
.bcrypt => {
// bcrypt's modular-crypt token encodes a 184-bit effective hash.
try list.append(.{ .name = "bcrypt", .bit_length = 184 });
},
.argon2 => {
// Argon2 output length is parameter-driven, so no fixed bit length applies.
try list.append(.{ .name = "Argon2", .bit_length = 0 });
},
.hex => {
if (hexCandidates(length)) |names| {
// length*4 converts hex-char count to a bit width (4 bits per nibble).
for (names) |name|
try list.append(.{ .name = name, .bit_length = @intCast(length * 4) });
}
},
.base64 => {
if (base64Candidates(length)) |names| {
// Each base64 char carries 6 bits; round to the nearest byte
// boundary so the reported length lines up with the underlying
// digest width. All table lengths divide evenly, so the integer
// division is exact.
const bits: i64 = @intCast(length * 6 / 8 * 8);
for (names) |name|
try list.append(.{ .name = name, .bit_length = bits });
}
},
.unknown => {},
}
return .{
.input = input,
.cleaned = cleaned,
.length = length,
.charset = charset,
.candidates = try list.toOwnedSlice(),
};
}
test "detects modular-crypt prefixes" {
// Sanity check that the prefix matchers agree with the canonical intent:
// bcrypt/argon2 are recognised by their `$...$` header alone.
try std.testing.expectEqual(HashCharset.bcrypt, detectCharset("$2a$abc..."));
try std.testing.expectEqual(HashCharset.bcrypt, detectCharset("$2y$xyz..."));
try std.testing.expectEqual(HashCharset.argon2, detectCharset("$argon2id$foo"));
try std.testing.expectEqual(HashCharset.argon2, detectCharset("$argon2$bar"));
}
test "prefers hex over base64" {
// Hex must win over base64 for an all-hex-digit string.
try std.testing.expectEqual(HashCharset.hex, detectCharset("d41d8cd98f00b204e9800998ecf8427e"));
}
test "identify maps length to candidates" {
const md5_hex = "d41d8cd98f00b204e9800998ecf8427e";
var info = try identifyHash(std.testing.allocator, md5_hex);
defer info.deinit(std.testing.allocator);
try std.testing.expectEqual(@as(usize, 32), info.length);
try std.testing.expectEqual(@as(usize, 7), info.candidates.len);
try std.testing.expectEqualStrings("MD5", info.candidates[0].name);
try std.testing.expectEqual(@as(i64, 128), info.candidates[0].bit_length);
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →