Skip to content

Password Strength Analyser — Python source

Estimate password strength with zxcvbn - realistic dictionary and pattern cracking with crack-time estimates and improvement suggestions. Runs entirely in your browser.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""password-strength-analyser — password strength scoring & feedback.

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the Password Strength Analyser
          tool, ported from cli/password-strength-analyser/password-strength-
          analyser.go (the live Go twin, which wraps zxcvbn) and
          src/lib/password-strength.ts (the canonical TS lib).
License:  display source — part of CosmoDev's polyglot tool pages.

Design goals:
  - Pure + deterministic; never raises on odd input.
  - Self-contained: stdlib only (no pip packages — no `zxcvbn`, no `passlib`).
  - Mirrors the Go twin's public API and result shape: ``analyse`` returns a
    ``score`` (0–4), a human ``label``, the entropy in bits, and a single
    crack-time display string.

Parity note: the live Go and TS twins both delegate scoring to ``zxcvbn``, a
dictionary + pattern estimator whose ranking data is far too large to vendor in
a stdlib display port. This snippet reproduces the SAME 0–4 scale and the SAME
public contract with a self-contained heuristic estimator:
  (a) a curated common-password blocklist — forces the obvious weak passwords
      (password, 123456, qwerty, ...) to score 0, exactly as zxcvbn's
      dictionary does;
  (b) a character-pool entropy estimate (length × log2(pool));
  (c) penalties for low symbol variety, keyboard / numeric sequences, and
      user-supplied related tokens (the ``userInputs`` seed zxcvbn accepts).
It agrees with zxcvbn on the clear cases — common passwords score 0, long
diverse passphrases score 3–4 — and is a reasonable approximation in between.
This is a deliberate, documented trade-off to keep the port dependency-free;
for exact scores use the live Go/TS twin.
"""

from __future__ import annotations

import math
from dataclasses import dataclass
from typing import Sequence

__all__ = ["analyse", "Analysis"]

# Strength labels indexed by score. Identical to the Go/TS twins.
LABELS = ("Very weak", "Weak", "Fair", "Good", "Strong")

# Curated common passwords (lowercased). An exact match forces score 0 — the
# observable effect of zxcvbn's dictionary for the most leaked passwords. A
# small, hand-picked set rather than zxcvbn's ~30k-entry list: enough to make
# the showcase behaviour meaningful without vendoring a data file.
COMMON: frozenset[str] = frozenset({
    "1234", "12345", "123123", "123456", "12345678", "123456789", "1234567890",
    "000000", "111111", "654321", "666666", "abc123", "admin", "baseball",
    "batman", "dragon", "football", "iloveyou", "letmein", "login", "master",
    "monkey", "passw0rd", "password", "password1", "princess", "qwerty", "root",
    "shadow", "superman", "sunshine", "trustno1", "welcome",
})

# Keyboard rows and numeric / alphabetic runs. Any length-4 window of these
# appearing inside the password signals an easy-to-guess sequence and incurs an
# entropy penalty.
SEQUENCES = ("qwertyuiop", "asdfghjkl", "zxcvbnm", "1234567890", "0987654321", "abcdefg", "gfedcba")

# Min length for a sequence window to count as a match (avoids trivial 2–3 char
# coincidences inside long passphrases).
SEQ_MIN_LEN = 4

# Entropy thresholds mapping a bit count to the 0–4 score. Shared across the
# whole polyglot showcase so every port lands on the same bucket.
_THRESHOLDS = (28.0, 36.0, 60.0, 128.0)


@dataclass
class Analysis:
    """Strength breakdown. Mirrors the Go twin's ``pwstrength.Result`` and the
    TS ``PasswordAnalysis`` (the four universally-available fields)."""

    score: int
    """0 (worst) … 4 (best), identical scale to zxcvbn."""

    label: str
    """Human-readable strength label."""

    entropy: float
    """Estimated guess entropy, in bits."""

    crack_time_display: str
    """Single human-readable crack-time string (offline-fast model)."""


def _pool_size(password: str) -> int:
    """Tally the distinct character-class buckets present and return the
    combined guess-pool size (lower, upper, digit, ASCII symbols, other)."""
    lower = upper = digit = sym = other = False
    for c in password:
        if c.isascii():
            if c.islower():
                lower = True
            elif c.isupper():
                upper = True
            elif c.isdigit():
                digit = True
            else:
                sym = True  # ASCII punctuation or whitespace (not alphanumeric)
        else:
            other = True  # any non-ASCII code point
    pool = 0
    if lower:
        pool += 26
    if upper:
        pool += 26
    if digit:
        pool += 10
    if sym:
        pool += 33  # printable ASCII non-alnum: 32 punctuation + space
    if other:
        pool += 128  # broad bucket for the rest of Unicode
    return pool


def _contains_sequence(lower: str, seq: str, min_len: int = SEQ_MIN_LEN) -> bool:
    """Report whether ``lower`` contains any length-``min_len`` window drawn
    from ``seq`` — i.e. a keyboard/numeric run of consecutive symbols."""
    if len(seq) < min_len:
        return False
    return any(seq[i:i + min_len] in lower for i in range(len(seq) - min_len + 1))


def _crack_time_display(entropy: float) -> str:
    """Human-readable crack-time for an offline fast attack (1e10 guesses/sec),
    mirroring zxcvbn's ``offline_fast_hashing_1e10_per_second`` display. Works
    in log-space so large entropies (2^200 ≈ 1e60 guesses) stay finite."""
    if entropy <= 0.0:
        return "instant"
    log10_seconds = entropy * math.log10(2) - 10.0  # log10(guesses) − log10(1e10)
    if log10_seconds < 0.0:
        return "instant"
    seconds = 10.0 ** log10_seconds
    if seconds < 1.0:
        return "instant"
    minute, hour, day = 60.0, 3_600.0, 86_400.0
    month, year = 2_592_000.0, 31_536_000.0  # 30 days / 365 days
    if seconds < minute:
        return f"{round(seconds)} seconds"
    if seconds < hour:
        return f"{round(seconds / minute)} minutes"
    if seconds < day:
        return f"{round(seconds / hour)} hours"
    if seconds < month:
        return f"{round(seconds / day)} days"
    if seconds < year:
        return f"{round(seconds / month)} months"
    years = seconds / year
    if years < 1_000.0:
        return f"{round(years)} years"
    if years < 1_000_000.0:
        return f"{round(years / 1_000.0)} thousand years"
    if years < 1_000_000_000.0:
        return f"{round(years / 1_000_000.0)} million years"
    return "centuries"


def analyse(password: str, user_inputs: Sequence[str] = ()) -> Analysis:
    """Score a password's strength. ``user_inputs`` seeds the estimator with
    related tokens (username, site name, ...) — a substring match weakens the
    score, mirroring the ``userInputs`` parameter of the Go/TS twins."""
    # Empty → trivially weak.
    if not password:
        return Analysis(0, LABELS[0], 0.0, "instant")

    length = len(password)  # code-point count
    pool = _pool_size(password)
    log_pool = math.log2(pool) if pool > 1 else 0.0

    # Base entropy: length × log2(pool), capped by symbol variety so heavy
    # repetition ("aaaaaaaa") collapses instead of accruing length credit.
    raw = length * log_pool
    unique = len(set(password))
    if unique < length and pool > 1:
        # Repeats earn only a quarter credit: U full picks + 0.25× the repeats.
        variety = (unique + (length - unique) * 0.25) * log_pool
    else:
        variety = raw
    entropy = min(raw, variety) if pool > 1 else 0.0

    # Lowercased view for dictionary / sequence / user-input matching.
    lower = password.lower()

    # (a) common-password blocklist forces score 0 and zero entropy, the way
    #     zxcvbn's dictionary collapses a known leak.
    common = lower in COMMON
    if common:
        entropy = 0.0
    else:
        # (b) keyboard / numeric sequence penalty.
        for seq in SEQUENCES:
            if _contains_sequence(lower, seq):
                entropy -= 12.0
        # (c) user-input relatedness penalty.
        for ui in user_inputs:
            l = ui.lower()
            if len(l) >= 3 and l in lower:
                entropy -= 10.0
        entropy = max(0.0, entropy)

    # Map entropy (and a minimum-length floor) to the 0–4 bucket.
    if length < 4 or entropy < _THRESHOLDS[0]:
        score = 0
    elif entropy < _THRESHOLDS[1]:
        score = 1
    elif entropy < _THRESHOLDS[2]:
        score = 2
    elif entropy < _THRESHOLDS[3]:
        score = 3
    else:
        score = 4

    return Analysis(score, LABELS[score], entropy, _crack_time_display(entropy))


# ---------- showcase tests (run with: python python.py) ----------
def _showcase_tests() -> None:
    assert analyse("password").score == 0
    assert analyse("password").label == "Very weak"
    assert analyse("correct horse battery staple").score >= 3
    assert analyse("123456").score <= analyse("correct horse battery staple").score
    assert analyse("aB3$xK9p").entropy > analyse("aaaaaaaa").entropy
    for pw in ("password", "12345678", "monkey", "Tr0ub4dour&3", "correct horse battery staple", "u2#9Xq!Lp$7wZ"):
        r = analyse(pw)
        assert 0 <= r.score <= 4
        assert r.label
        assert r.crack_time_display
    assert analyse("cosmolabs2024", ("cosmolabs",)).score <= analyse("cosmolabs2024").score


if __name__ == "__main__":
    _showcase_tests()
    print("all showcase tests passed")

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →