Skip to content

Text Extractor — Python source

Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
out of arbitrary text (logs, headers, config).

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the Extract tool, ported from
          src/lib/extract.ts (the canonical TypeScript implementation) and
          held in lock-step with its Go twin cli/extract/extract.go.
License:  display source — part of CosmoDev's polyglot tool pages.

Design goals:
  - Pure + deterministic; never raises.
  - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
  - Self-contained: stdlib only (``re`` ships in the language; no pip packages).

Behavior: pull URLs, emails, IPv4/IPv6 addresses, hashes
(md5/sha1/sha256/sha512 by length), and domains out of arbitrary text. Matches
are deduped per type preserving first-occurrence order; an email also
contributes its domain to the domain list when both email and domain are
selected. A zero-length types list defaults to all six kinds.
"""

from __future__ import annotations

import re
from dataclasses import dataclass, field
from typing import Dict, List, Literal, Optional, Sequence

__all__ = ["extract", "ExtractResult", "EXTRACT_TYPES", "ExtractType"]

ExtractType = Literal["url", "email", "ipv4", "ipv6", "hash", "domain"]

#: Canonical extraction kinds, in display order. Mirrors EXTRACT_TYPES in TS.
EXTRACT_TYPES: List[ExtractType] = ["url", "email", "ipv4", "ipv6", "hash", "domain"]

# Per-kind patterns, compiled once at import. These mirror the RE record in
# src/lib/extract.ts (and the Go twin) verbatim: same anchors, classes, and
# counted repetition. Every group is non-capturing (?:...) so findall returns
# the full match string, not group tuples. The IPv6 pattern is intentionally
# permissive — a hex/colon run — and is post-filtered by is_ipv6 so bare hex
# words, times, and MACs are rejected, exactly as in the TS lib.
_RE: Dict[ExtractType, "re.Pattern[str]"] = {
    "url": re.compile(r"https?://[^\s]+"),
    "email": re.compile(r"[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}"),
    "ipv4": re.compile(r"\b(?:\d{1,3}\.){3}\d{1,3}\b"),
    "ipv6": re.compile(r"[0-9a-fA-F:]+"),
    # md5 (32) / sha1 (40) / sha256 (64) / sha512 (128). `\b` keeps each length
    # honest, so a 64-char run does not also match as a leading 32-char hash.
    "hash": re.compile(
        r"\b[a-fA-F0-9]{32}\b|\b[a-fA-F0-9]{40}\b|"
        r"\b[a-fA-F0-9]{64}\b|\b[a-fA-F0-9]{128}\b"
    ),
    "domain": re.compile(
        r"\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b"
    ),
}

# Validates a single 1–4 hex-digit IPv6 hextet (full-string match == ^...$).
_RE_IPV6_GROUP = re.compile(r"[0-9a-fA-F]{1,4}")


@dataclass
class ExtractResult:
    """One field per kind — always all six, populated only for the selected
    types (unselected kinds stay empty lists). Mirrors the TS
    ``Record<ExtractType, string[]>`` and Go's ``Result`` struct. The dataclass
    default factories give us the "always all six keys, empty when unselected"
    contract for free.
    """

    url: List[str] = field(default_factory=list)
    email: List[str] = field(default_factory=list)
    ipv4: List[str] = field(default_factory=list)
    ipv6: List[str] = field(default_factory=list)
    hash: List[str] = field(default_factory=list)
    domain: List[str] = field(default_factory=list)


def _uniq(values: List[str]) -> List[str]:
    """Deduplicate values, preserving first-occurrence order. The Go twin of
    the ``uniq()`` helper in src/lib/extract.ts."""
    seen: set = set()
    out: List[str] = []
    for v in values:
        if v not in seen:
            seen.add(v)
            out.append(v)
    return out


def _all_matches(pattern: "re.Pattern[str]", text: str) -> List[str]:
    """Every non-overlapping match of ``pattern`` in ``text``. ``findall``
    returns the full match for patterns without capturing groups (all of ours
    use ``(?:...)``), and an empty list when there are no matches."""
    return pattern.findall(text)


def is_ipv6(run: str) -> bool:
    """A hex/colon run is a plausible IPv6: it has a colon AND either contains
    ``::`` (a compressed zero-run) or is exactly eight groups of 1–4 hex
    digits. Mirrors ``isIpv6()`` in src/lib/extract.ts."""
    if ":" not in run:
        return False
    if "::" in run:
        return True
    groups = run.split(":")
    return len(groups) == 8 and all(_RE_IPV6_GROUP.fullmatch(g) is not None for g in groups)


def _domain_of(email: str) -> str:
    """Domain part (after the last ``@``) of a matched email. ``rsplit`` from
    the right with a maxsplit of 1 mirrors ``lastIndexOf('@')`` exactly."""
    return email.rsplit("@", 1)[-1]


def extract(
    input: Optional[str],
    types: Optional[Sequence[ExtractType]] = None,
) -> ExtractResult:
    """Extract every occurrence of the given ``types`` (default: all six) from
    ``input``. Returns an :class:`ExtractResult` with one field per kind —
    always all six, populated only for the selected types. Matches are deduped
    per kind, preserving first-occurrence order. An email also contributes its
    domain to the ``domain`` list when both ``email`` and ``domain`` are
    selected. Never raises; ``None`` input is treated as empty text."""
    # TS does `const text = input ?? ''`; Python mirrors it explicitly.
    text = input if input is not None else ""
    # TS defaults the param to EXTRACT_TYPES AND re-defaults an empty array;
    # `if types` covers both None and [].
    selected = list(types) if types else list(EXTRACT_TYPES)

    def want(t: ExtractType) -> bool:
        return t in selected

    out = ExtractResult()
    if want("url"):
        out.url = _uniq(_all_matches(_RE["url"], text))
    if want("email"):
        out.email = _uniq(_all_matches(_RE["email"], text))
    if want("ipv4"):
        out.ipv4 = _uniq(_all_matches(_RE["ipv4"], text))
    if want("ipv6"):
        out.ipv6 = _uniq([r for r in _all_matches(_RE["ipv6"], text) if is_ipv6(r)])
    if want("hash"):
        out.hash = _uniq(_all_matches(_RE["hash"], text))
    if want("domain"):
        combined = _all_matches(_RE["domain"], text)
        # Cross-rule: an email also yields its domain in the domain list.
        if want("email"):
            combined += [_domain_of(e) for e in _all_matches(_RE["email"], text)]
        out.domain = _uniq(combined)
    return out

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →