Skip to content

Text Extractor — TypeScript source

Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Text Extractor - pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
// out of arbitrary text (logs, headers, config). Pure, zero deps, deterministic.

export type ExtractType = 'url' | 'email' | 'ipv4' | 'ipv6' | 'hash' | 'domain';

/** Canonical extraction kinds, in display order. */
export const EXTRACT_TYPES: ExtractType[] = ['url', 'email', 'ipv4', 'ipv6', 'hash', 'domain'];

const RE: Record<ExtractType, RegExp> = {
  url: /https?:\/\/[^\s]+/g,
  email: /[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}/g,
  ipv4: /\b(?:\d{1,3}\.){3}\d{1,3}\b/g,
  // IPv6 is matched permissively as a hex/colon run, then post-filtered by isIpv6
  // so we don't over-match bare hex words, times (12:30:45), or MAC addresses.
  ipv6: /[0-9a-fA-F:]+/g,
  // md5 (32) / sha1 (40) / sha256 (64) / sha512 (128). `\b` keeps each length honest.
  hash: /\b[a-fA-F0-9]{32}\b|\b[a-fA-F0-9]{40}\b|\b[a-fA-F0-9]{64}\b|\b[a-fA-F0-9]{128}\b/g,
  domain: /\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b/g,
};

/** Dedupe, preserving first-occurrence order. */
function uniq(values: string[]): string[] {
  const seen = new Set<string>();
  const out: string[] = [];
  for (const v of values) {
    if (!seen.has(v)) {
      seen.add(v);
      out.push(v);
    }
  }
  return out;
}

/** A hex/colon run is a plausible IPv6: it has a colon AND either contains `::`
 *  (a compressed zero-run) or is exactly 8 groups of 1-4 hex digits. */
function isIpv6(run: string): boolean {
  if (!run.includes(':')) return false;
  if (run.includes('::')) return true;
  const groups = run.split(':');
  return groups.length === 8 && groups.every((g) => /^[0-9a-fA-F]{1,4}$/.test(g));
}

/** Domain part (after the last `@`) of a matched email. */
function domainOf(email: string): string {
  return email.slice(email.lastIndexOf('@') + 1);
}

function emptyResult(): Record<ExtractType, string[]> {
  return { url: [], email: [], ipv4: [], ipv6: [], hash: [], domain: [] };
}

/**
 * Extract every occurrence of the given `types` (default: all six) from `input`.
 * Returns an object with a key per type - always all six keys, populated only
 * for the selected types (unselected types are empty arrays). Matches are
 * deduped per type, preserving first-occurrence order. An email also contributes
 * its domain to the `domain` list (when both `email` and `domain` are selected).
 */
export function extract(
  input: string,
  types: ExtractType[] = EXTRACT_TYPES,
): Record<ExtractType, string[]> {
  const text = input ?? '';
  const selected = types.length > 0 ? types : EXTRACT_TYPES;
  const want = (t: ExtractType): boolean => selected.includes(t);
  const out = emptyResult();

  if (want('url')) out.url = uniq(text.match(RE.url) ?? []);
  if (want('email')) out.email = uniq(text.match(RE.email) ?? []);
  if (want('ipv4')) out.ipv4 = uniq(text.match(RE.ipv4) ?? []);
  if (want('ipv6')) out.ipv6 = uniq((text.match(RE.ipv6) ?? []).filter(isIpv6));
  if (want('hash')) out.hash = uniq(text.match(RE.hash) ?? []);
  if (want('domain')) {
    const fromRegex = text.match(RE.domain) ?? [];
    // Cross-rule: an email also yields its domain in the domain list.
    const fromEmail = want('email') ? (text.match(RE.email) ?? []).map(domainOf) : [];
    out.domain = uniq([...fromRegex, ...fromEmail]);
  }

  return out;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →