Text Extractor — TypeScript source
Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Text Extractor - pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
// out of arbitrary text (logs, headers, config). Pure, zero deps, deterministic.
export type ExtractType = 'url' | 'email' | 'ipv4' | 'ipv6' | 'hash' | 'domain';
/** Canonical extraction kinds, in display order. */
export const EXTRACT_TYPES: ExtractType[] = ['url', 'email', 'ipv4', 'ipv6', 'hash', 'domain'];
const RE: Record<ExtractType, RegExp> = {
url: /https?:\/\/[^\s]+/g,
email: /[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}/g,
ipv4: /\b(?:\d{1,3}\.){3}\d{1,3}\b/g,
// IPv6 is matched permissively as a hex/colon run, then post-filtered by isIpv6
// so we don't over-match bare hex words, times (12:30:45), or MAC addresses.
ipv6: /[0-9a-fA-F:]+/g,
// md5 (32) / sha1 (40) / sha256 (64) / sha512 (128). `\b` keeps each length honest.
hash: /\b[a-fA-F0-9]{32}\b|\b[a-fA-F0-9]{40}\b|\b[a-fA-F0-9]{64}\b|\b[a-fA-F0-9]{128}\b/g,
domain: /\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b/g,
};
/** Dedupe, preserving first-occurrence order. */
function uniq(values: string[]): string[] {
const seen = new Set<string>();
const out: string[] = [];
for (const v of values) {
if (!seen.has(v)) {
seen.add(v);
out.push(v);
}
}
return out;
}
/** A hex/colon run is a plausible IPv6: it has a colon AND either contains `::`
* (a compressed zero-run) or is exactly 8 groups of 1-4 hex digits. */
function isIpv6(run: string): boolean {
if (!run.includes(':')) return false;
if (run.includes('::')) return true;
const groups = run.split(':');
return groups.length === 8 && groups.every((g) => /^[0-9a-fA-F]{1,4}$/.test(g));
}
/** Domain part (after the last `@`) of a matched email. */
function domainOf(email: string): string {
return email.slice(email.lastIndexOf('@') + 1);
}
function emptyResult(): Record<ExtractType, string[]> {
return { url: [], email: [], ipv4: [], ipv6: [], hash: [], domain: [] };
}
/**
* Extract every occurrence of the given `types` (default: all six) from `input`.
* Returns an object with a key per type - always all six keys, populated only
* for the selected types (unselected types are empty arrays). Matches are
* deduped per type, preserving first-occurrence order. An email also contributes
* its domain to the `domain` list (when both `email` and `domain` are selected).
*/
export function extract(
input: string,
types: ExtractType[] = EXTRACT_TYPES,
): Record<ExtractType, string[]> {
const text = input ?? '';
const selected = types.length > 0 ? types : EXTRACT_TYPES;
const want = (t: ExtractType): boolean => selected.includes(t);
const out = emptyResult();
if (want('url')) out.url = uniq(text.match(RE.url) ?? []);
if (want('email')) out.email = uniq(text.match(RE.email) ?? []);
if (want('ipv4')) out.ipv4 = uniq(text.match(RE.ipv4) ?? []);
if (want('ipv6')) out.ipv6 = uniq((text.match(RE.ipv6) ?? []).filter(isIpv6));
if (want('hash')) out.hash = uniq(text.match(RE.hash) ?? []);
if (want('domain')) {
const fromRegex = text.match(RE.domain) ?? [];
// Cross-rule: an email also yields its domain in the domain list.
const fromEmail = want('email') ? (text.match(RE.email) ?? []).map(domainOf) : [];
out.domain = uniq([...fromRegex, ...fromEmail]);
}
return out;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →