Text Extractor — C# source
Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.
This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.
// extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
// out of arbitrary text (logs, headers, config).
//
// Language: C# 12 (.NET 8, standard library only)
// Source: CosmoDev polyglot showcase port of the Extract tool, ported from
// src/lib/extract.ts (the canonical TypeScript implementation) and
// held in lock-step with its Go twin cli/extract/extract.go.
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; never throws.
// - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
// - The six per-kind patterns are reproduced VERBATIM from the Go twin so the
// "lock-step contract" between the two implementations is auditable at a
// glance.
//
// Engine note: System.Text.RegularExpressions is a backtracking engine while
// Go's regexp is RE2, but both implement leftmost-first (Perl-order)
// alternation and none of the patterns hinge on a backtracking-only or
// RE2-only subtlety, so matches agree on every input. Verbatim (@"...")
// literals keep the pattern text character-for-character identical to the Go
// twin. (On .NET 7+ the same patterns could equally be compiled by the
// [GeneratedRegex] source generator; plain static Regex keeps the file
// copy-runnable without a partial class.)
using System;
using System.Collections.Generic;
using System.Linq;
using System.Text.RegularExpressions;
namespace CosmoDev.Extract;
/// <summary>One of the six canonical extraction kinds. Mirrors the TS
/// <c>ExtractType</c> union ('url' | 'email' | 'ipv4' | 'ipv6' | 'hash' |
/// 'domain') and Go's <c>Type</c>.</summary>
public enum Kind
{
Url, Email, Ipv4, Ipv6, Hash, Domain
}
public static class KindExtensions
{
/// <summary>Canonical spelling used in the TS/Go string-literal contract.</summary>
public static string AsStr(this Kind kind) => kind switch
{
Kind.Url => "url",
Kind.Email => "email",
Kind.Ipv4 => "ipv4",
Kind.Ipv6 => "ipv6",
Kind.Hash => "hash",
Kind.Domain => "domain",
_ => throw new ArgumentOutOfRangeException(nameof(kind)),
};
}
/// <summary>The C# twin of the TS <c>Record<ExtractType, string[]></c>.
/// Every field is always present (an empty list, never null): unselected kinds
/// and empty matches are empty, matching the TS lib's "always all six keys"
/// contract.</summary>
public sealed class ExtractResult
{
public List<string> Url { get; } = [];
public List<string> Email { get; } = [];
public List<string> Ipv4 { get; } = [];
public List<string> Ipv6 { get; } = [];
public List<string> Hash { get; } = [];
public List<string> Domain { get; } = [];
}
public static class Extractor
{
/// <summary>The canonical kinds in display order — Go's
/// <c>ExtractTypes</c> / TS's <c>EXTRACT_TYPES</c>.</summary>
public static readonly Kind[] AllKinds =
[
Kind.Url, Kind.Email, Kind.Ipv4, Kind.Ipv6, Kind.Hash, Kind.Domain,
];
// The per-kind patterns, compiled once and reused. They mirror the `RE`
// record in src/lib/extract.ts and the `re*` vars in the Go twin VERBATIM:
// same anchors (\b), character classes, and counted repetition.
private static readonly Regex ReUrl = new(@"https?://[^\s]+");
private static readonly Regex ReEmail =
new(@"[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}");
private static readonly Regex ReIpv4 = new(@"\b(?:\d{1,3}\.){3}\d{1,3}\b");
/// <summary>Intentionally permissive — a hex/colon run — post-filtered by
/// <see cref="IsIpv6"/>.</summary>
private static readonly Regex ReIpv6 = new(@"[0-9a-fA-F:]+");
/// <summary>md5 (32) / sha1 (40) / sha256 (64) / sha512 (128). `\b` keeps
/// each length honest, so a 64-char run does not also match as a leading
/// 32-char hash.</summary>
private static readonly Regex ReHash = new(
@"\b[a-fA-F0-9]{32}\b|\b[a-fA-F0-9]{40}\b|\b[a-fA-F0-9]{64}\b|\b[a-fA-F0-9]{128}\b");
private static readonly Regex ReDomain = new(
@"\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b");
/// <summary>Validates a single 1–4 hex-digit IPv6 hextet; the ^/$ anchors
/// pin both ends, exactly as in the Go twin.</summary>
private static readonly Regex ReIpv6Group = new(@"^[0-9a-fA-F]{1,4}$");
/// <summary>Deduplicate values, preserving first-occurrence order. The C#
/// twin of the <c>uniq()</c> helper in src/lib/extract.ts.
/// <see cref="HashSet{T}.Add"/> returns false on a repeat, so order is
/// kept without a separate contains scan.</summary>
private static List<string> Uniq(IEnumerable<string> values)
{
HashSet<string> seen = [];
List<string> result = [];
foreach (string v in values)
{
if (seen.Add(v)) result.Add(v);
}
return result;
}
/// <summary>Every non-overlapping match of <paramref name="re"/> in
/// <paramref name="text"/>, left to right — the .NET equivalent of Go's
/// <c>regexp.FindAllString</c> and JS's <c>String.prototype.match</c> with
/// the global flag.</summary>
private static List<string> AllMatches(Regex re, string text) =>
re.Matches(text).Select(m => m.Value).ToList();
/// <summary>Reports whether a hex/colon run is a plausible IPv6 address:
/// it must contain a colon AND either hold a compressed zero-run
/// (<c>::</c>) or be exactly eight groups of 1–4 hex digits. Mirrors
/// <c>isIpv6()</c> in src/lib/extract.ts.</summary>
public static bool IsIpv6(string run)
{
if (!run.Contains(':')) return false;
if (run.Contains("::")) return true;
// string.Split(char) keeps empty segments at both ends, exactly like
// split(':') in the TS/Go/Python siblings.
string[] groups = run.Split(':');
return groups.Length == 8 && groups.All(g => ReIpv6Group.IsMatch(g));
}
/// <summary>The domain part (after the last <c>@</c>) of a matched email.
/// Mirrors <c>domainOf()</c> in src/lib/extract.ts. When no <c>@</c> is
/// present, LastIndexOf returns -1 and the +1 offsets to 0, yielding the
/// whole input — the same "no separator" default as rsplit/@-1.</summary>
public static string DomainOf(string email) =>
email[(email.LastIndexOf('@') + 1)..];
/// <summary>Extract pulls every occurrence of the given
/// <paramref name="types"/> (default: all six) from
/// <paramref name="input"/>. It returns an <see cref="ExtractResult"/>
/// with one field per kind — always all six, populated only for the
/// selected types. Matches are deduped per kind, preserving
/// first-occurrence order. An email also contributes its domain to the
/// <see cref="ExtractResult.Domain"/> list when both Email and Domain are
/// selected. It is the C# twin of <c>extract()</c> in
/// src/lib/extract.ts.</summary>
public static ExtractResult Extract(string? input = null, Kind[]? types = null)
{
string text = input ?? ""; // TS does `const text = input ?? ''`
Kind[] selected = types is { Length: > 0 } ? types : AllKinds;
bool Want(Kind k) => selected.Contains(k);
ExtractResult result = new();
if (Want(Kind.Url)) result.Url.AddRange(Uniq(AllMatches(ReUrl, text)));
if (Want(Kind.Email)) result.Email.AddRange(Uniq(AllMatches(ReEmail, text)));
if (Want(Kind.Ipv4)) result.Ipv4.AddRange(Uniq(AllMatches(ReIpv4, text)));
if (Want(Kind.Ipv6)) result.Ipv6.AddRange(Uniq(AllMatches(ReIpv6, text).Where(IsIpv6)));
if (Want(Kind.Hash)) result.Hash.AddRange(Uniq(AllMatches(ReHash, text)));
if (Want(Kind.Domain))
{
List<string> combined = AllMatches(ReDomain, text);
// Cross-rule: an email also yields its domain in the domain list.
if (Want(Kind.Email))
combined.AddRange(AllMatches(ReEmail, text).Select(DomainOf));
result.Domain.AddRange(Uniq(combined));
}
return result;
}
}
// ---------- showcase tests (the canonical suite lives in src/lib) ----------
// Compile with -define:EXTRACT_SELFTEST (dotnet build with
// DefineConstants=EXTRACT_SELFTEST) for a runnable five-vector check — the C#
// analogue of the Rust sibling's #[cfg(test)] module.
#if EXTRACT_SELFTEST
public static class ExtractSelfTest
{
private static void Expect(List<string> actual, string[] expected, string field)
{
if (!actual.SequenceEqual(expected))
throw new InvalidOperationException(
$"{field}: expected [{string.Join(", ", expected)}], got [{string.Join(", ", actual)}]");
}
public static void Main()
{
// Well-known digests of the empty string (real hash values), shared
// with src/lib/extract.test.ts so the showcase uses identical vectors.
const string md5Empty = "d41d8cd98f00b204e9800998ecf8427e"; // 32
const string sha256Empty =
"e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"; // 64
// Extracts and dedupes URLs, keeping first-occurrence order.
Extractor.Extract("a https://x.com b https://y.com c https://x.com") is { } r
? Expect(r.Url, ["https://x.com", "https://y.com"], "url")
: throw new InvalidOperationException("unreachable");
// Plus-tags and multi-part-TLD domains are matched; an email also
// contributes its domain when email AND domain are both selected.
r = Extractor.Extract("reach a.b+tag@mail.example.co.uk please");
Expect(r.Email, ["a.b+tag@mail.example.co.uk"], "email");
Expect(r.Domain, ["mail.example.co.uk"], "domain");
// `::1` is a compressed zero-run; `12:30:45` has no `::` and only
// three groups, so it is rejected as a clock, not an address.
r = Extractor.Extract("loopback ::1 and time 12:30:45 now");
Expect(r.Ipv6, ["::1"], "ipv6");
// Hashes by length: md5 (32) and sha256 (64).
r = Extractor.Extract($"m {md5Empty} s {sha256Empty}");
Expect(r.Hash, [md5Empty, sha256Empty], "hash");
// Type selection: unselected kinds stay empty.
r = Extractor.Extract("https://x.com and a@b.com", [Kind.Url]);
Expect(r.Url, ["https://x.com"], "url");
if (r.Email.Count != 0) throw new InvalidOperationException("email was not selected");
Console.WriteLine("extract: all showcase tests passed");
}
}
#endif
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →