Skip to content

Text Extractor — JavaScript source

Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.

This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

/**
 * extract - pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
 * out of arbitrary text (logs, headers, config).
 *
 * Language:   JavaScript (ES2020+, runs unmodified in Node 16+ and modern browsers)
 * Source:     CosmoDev polyglot showcase port of the Extract tool, ported from
 *             src/lib/extract.ts (the canonical TypeScript implementation) and
 *             held in lock-step with its Go twin cli/extract/extract.go.
 * License:    display source - part of CosmoDev's polyglot tool pages.
 *
 * Design goals:
 *   - Pure + deterministic; never throws.
 *   - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
 *   - Self-contained: stdlib only (no npm dependencies).
 *
 * Behavior: matches are deduped per type preserving first-occurrence order; an
 * email also contributes its domain to the domain list when both email and
 * domain are selected. A zero-length types list defaults to all six kinds.
 */

'use strict';

/**
 * Canonical extraction kind.
 * @typedef {('url' | 'email' | 'ipv4' | 'ipv6' | 'hash' | 'domain')} ExtractType
 */

/**
 * Extraction result - always all six keys, populated only for the selected
 * types (unselected types are empty arrays).
 * @typedef {Record<ExtractType, string[]>} ExtractResult
 */

/** Canonical extraction kinds, in display order. Mirrors EXTRACT_TYPES in TS. */
const EXTRACT_TYPES = ['url', 'email', 'ipv4', 'ipv6', 'hash', 'domain'];

// Per-kind patterns. These mirror the RE record in src/lib/extract.ts (and the
// Go twin) verbatim. The global flag makes String.prototype.match return every
// match at once (and resets lastIndex, so each call is independent and
// stateless). The IPv6 pattern is intentionally permissive - a hex/colon run
// - post-filtered by isIpv6() so bare hex words, times, and MACs are rejected.
const RE = {
  url: /https?:\/\/[^\s]+/g,
  email: /[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}/g,
  ipv4: /\b(?:\d{1,3}\.){3}\d{1,3}\b/g,
  // IPv6 is matched permissively as a hex/colon run, then post-filtered.
  ipv6: /[0-9a-fA-F:]+/g,
  // md5 (32) / sha1 (40) / sha256 (64) / sha512 (128). `\b` keeps each length
  // honest, so a 64-char run does not also match as a leading 32-char hash.
  hash: /\b[a-fA-F0-9]{32}\b|\b[a-fA-F0-9]{40}\b|\b[a-fA-F0-9]{64}\b|\b[a-fA-F0-9]{128}\b/g,
  domain: /\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b/g,
};

/** Validates a single 1-4 hex-digit IPv6 hextet (anchored full match). */
const RE_IPV6_GROUP = /^[0-9a-fA-F]{1,4}$/;

/**
 * Dedupe, preserving first-occurrence order. Mirrors uniq() in the TS lib.
 * @param {string[]} values
 * @returns {string[]}
 */
function uniq(values) {
  const seen = new Set();
  const out = [];
  for (const v of values) {
    if (!seen.has(v)) {
      seen.add(v);
      out.push(v);
    }
  }
  return out;
}

/**
 * A hex/colon run is a plausible IPv6: it has a colon AND either contains `::`
 * (a compressed zero-run) or is exactly 8 groups of 1-4 hex digits. Mirrors
 * isIpv6() in src/lib/extract.ts.
 * @param {string} run
 * @returns {boolean}
 */
function isIpv6(run) {
  if (!run.includes(':')) return false;
  if (run.includes('::')) return true;
  const groups = run.split(':');
  return groups.length === 8 && groups.every((g) => RE_IPV6_GROUP.test(g));
}

/** Domain part (after the last `@`) of a matched email. Mirrors domainOf(). */
function domainOf(email) {
  return email.slice(email.lastIndexOf('@') + 1);
}

/**
 * Extract every occurrence of the given `types` (default: all six) from `input`.
 * Returns an object with a key per type - always all six keys, populated only
 * for the selected types (unselected types are empty arrays). Matches are
 * deduped per type, preserving first-occurrence order. An email also
 * contributes its domain to the `domain` list when both `email` and `domain`
 * are selected.
 *
 * @param {string} input
 * @param {ExtractType[]} [types] Extraction kinds to pull; omitted/empty = all six.
 * @returns {ExtractResult}
 */
function extract(input, types) {
  // TS does `const text = input ?? ''`; JS mirrors it via nullish coalescing.
  const text = input ?? '';
  // TS defaults the param AND re-defaults an empty array.
  const selected = types && types.length > 0 ? types : EXTRACT_TYPES;
  const want = (t) => selected.includes(t);

  const out = { url: [], email: [], ipv4: [], ipv6: [], hash: [], domain: [] };

  if (want('url')) out.url = uniq(text.match(RE.url) ?? []);
  if (want('email')) out.email = uniq(text.match(RE.email) ?? []);
  if (want('ipv4')) out.ipv4 = uniq(text.match(RE.ipv4) ?? []);
  if (want('ipv6')) out.ipv6 = uniq((text.match(RE.ipv6) ?? []).filter(isIpv6));
  if (want('hash')) out.hash = uniq(text.match(RE.hash) ?? []);
  if (want('domain')) {
    const fromRegex = text.match(RE.domain) ?? [];
    // Cross-rule: an email also yields its domain in the domain list.
    const fromEmail = want('email') ? (text.match(RE.email) ?? []).map(domainOf) : [];
    out.domain = uniq([...fromRegex, ...fromEmail]);
  }

  return out;
}

// CommonJS export so the file is consumable from Node without a build step,
// while staying dependency-free and framework-agnostic.
module.exports = { extract, EXTRACT_TYPES, uniq, isIpv6, domainOf };

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →