Skip to content

Hash Type Identifier — Go source

Identify the likely hash algorithm of a hash string by its length and character set - MD5, SHA-1/2/3, BLAKE, CRC32, NTLM, bcrypt, Argon2 and more.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package hashtype is the Go twin of CosmoDev's src/lib/hashIdentify.ts (dual
// source: the web lib is TypeScript, the CLI lib is Go — kept in lock-step).
// Pure + deterministic, never panics. The table-driven tests in
// hash-type-identifier_test.go share vectors with src/lib/hashIdentify.test.ts
// so the two implementations are held to the same contract.
//
// The TS lib classifies a string's charset and suggests likely hash algorithms
// by length; it does NOT hash anything (no WebCrypto / crypto.subtle). This twin
// mirrors that exactly: charset detection + length-keyed candidate lookup, with
// identical regexes, candidate tables, and bit-length arithmetic.
package hashtype

import (
	"math"
	"regexp"
	"strings"
)

// HashCharset is the detected character set of an input string. It mirrors the
// TS HashCharset union; constants carry the same lowercase string values used by
// the TS lib so the two surfaces are directly comparable.
type HashCharset string

const (
	CharsetHex     HashCharset = "hex"
	CharsetBase64  HashCharset = "base64"
	CharsetBcrypt  HashCharset = "bcrypt"
	CharsetArgon2  HashCharset = "argon2"
	CharsetUnknown HashCharset = "unknown"
)

// HashMatch is a candidate hash algorithm. BitLength is the hex length × 4
// where applicable, mirroring the TS HashMatch interface.
type HashMatch struct {
	Name      string
	BitLength int
}

// HashInfo is the full identification result, mirroring the TS HashInfo
// interface field-for-field.
type HashInfo struct {
	Input      string      // original input ("" for the TS null/undefined guard)
	Cleaned    string      // trimmed input
	Length     int         // length of cleaned (bytes == chars; hash strings are ASCII)
	Charset    HashCharset
	Candidates []HashMatch // never nil — empty when no candidates (matches TS [])
}

// hexByLength holds the hex-string candidate names keyed by hex length. Same
// entries as HEX_BY_LENGTH in src/lib/hashIdentify.ts.
var hexByLength = map[int][]string{
	8:   {"CRC32", "Adler-32"},
	16:  {"MySQL 3.x", "CRC64"},
	32:  {"MD5", "MD4", "NTLM", "LM", "MD2", "RIPEMD-128", "HAVAL-128"},
	40:  {"SHA-1", "RIPEMD-160", "HAVAL-160", "MySQL 5.x (SHA1(SHA1))", "Tiger-160"},
	56:  {"SHA-224", "SHA3-224", "BLAKE2s-224", "HAVAL-224"},
	64:  {"SHA-256", "SHA3-256", "BLAKE2s-256", "RIPEMD-256", "Skein-256"},
	96:  {"SHA-384", "SHA3-384", "BLAKE2b-384"},
	128: {"SHA-512", "SHA3-512", "BLAKE2b-512", "Whirlpool", "Skein-512"},
}

// base64ByLength holds the base64-string candidate names keyed by encoded
// length. Same entries as BASE64_BY_LENGTH in src/lib/hashIdentify.ts.
var base64ByLength = map[int][]string{
	24: {"MD5 (base64)"},
	28: {"SHA-1 (base64)"},
	44: {"SHA-256 (base64)"},
	88: {"SHA-512 (base64)"},
}

// Detection patterns — verbatim translations of the regex literals in the TS
// detectCharset(). bcrypt/argon2 are anchored only at the start (the trailing
// \$ is a literal dollar, not an end anchor), exactly like the JS source; hex
// and base64 are anchored at both ends. MatchString is an unanchored search, so
// the ^ / $ anchors carry the same meaning as JS RegExp.test (single-line).
var (
	reBcrypt = regexp.MustCompile(`^\$2[abxy]?\$`)
	reArgon2 = regexp.MustCompile(`^\$argon2(id|i|d)?\$`)
	reHex    = regexp.MustCompile(`^[0-9a-fA-F]+$`)
	reBase64 = regexp.MustCompile(`^[A-Za-z0-9+/]+={0,2}$`)
)

// DetectCharset classifies a string's charset. It mirrors detectCharset() in
// src/lib/hashIdentify.ts, including the detection order (bcrypt → argon2 →
// hex → base64 → unknown). Never panics.
func DetectCharset(s string) HashCharset {
	switch {
	case reBcrypt.MatchString(s):
		return CharsetBcrypt
	case reArgon2.MatchString(s):
		return CharsetArgon2
	case reHex.MatchString(s):
		return CharsetHex
	case reBase64.MatchString(s):
		return CharsetBase64
	default:
		return CharsetUnknown
	}
}

// base64BitLength replicates the TS expression Math.round((length * 6) / 8) * 8.
// math.Round rounds half-away-from-zero; for the positive lengths used here that
// is identical to JS Math.round (round half toward +Inf). For every known
// base64 hash length (24/28/44/88) the division is exact, so no rounding edge
// case actually arises.
func base64BitLength(length int) int {
	return int(math.Round(float64(length*6)/8.0)) * 8
}

// IdentifyHash identifies candidate hash types for an input string. It is the Go
// twin of identifyHash() in src/lib/hashIdentify.ts and must agree with it on
// every shared vector. Always returns a populated HashInfo; never panics.
func IdentifyHash(input string) HashInfo {
	cleaned := strings.TrimSpace(input)
	charset := DetectCharset(cleaned)
	length := len(cleaned)

	candidates := []HashMatch{} // non-nil empty matches TS []
	switch charset {
	case CharsetBcrypt:
		candidates = []HashMatch{{Name: "bcrypt", BitLength: 184}}
	case CharsetArgon2:
		candidates = []HashMatch{{Name: "Argon2", BitLength: 0}}
	case CharsetHex:
		if names, ok := hexByLength[length]; ok {
			candidates = make([]HashMatch, len(names))
			for i, name := range names {
				candidates[i] = HashMatch{Name: name, BitLength: length * 4}
			}
		}
	case CharsetBase64:
		if names, ok := base64ByLength[length]; ok {
			candidates = make([]HashMatch, len(names))
			for i, name := range names {
				candidates[i] = HashMatch{Name: name, BitLength: base64BitLength(length)}
			}
		}
	}

	return HashInfo{
		Input:      input,
		Cleaned:    cleaned,
		Length:     length,
		Charset:    charset,
		Candidates: candidates,
	}
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →