Skip to content

URL Inspector — Zig source

Break any URL into its components - protocol, host, port, path, query params, hash, and credentials. Detects default ports and security at a glance, with a decode toggle for query values. Runs entirely in your browser.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// url-inspector — break any URL into components (protocol, credentials, host, port, path, query params, fragment), detecting default ports and security at a glance. Language: Zig (0.13, standard library only). Port of src/lib/url-inspector.ts — same contract as this dir's javascript.js; the parser is hand-rolled (no platform URL lib) with the same authority walk the TS rawPort() helper does (userinfo, IPv6 literals, explicit ports).
const std = @import("std");

const invalid_msg = "Invalid URL - could not be parsed (include the scheme, e.g. https://)";

/// A single decoded query parameter, in insertion order (duplicates preserved).
const UrlParam = struct { key: []const u8, value: []const u8 };

/// Flat decomposition; optional fields mirror the TS undefined-when-absent shape.
/// Decoded strings live in `scratch`, so the report is filled in place via a
/// pointer — a by-value copy would leave the slices pointing at the original
/// (a classic Zig self-referential-struct trap).
const UrlReport = struct {
    valid: bool = false,
    warnings: [4][]const u8 = undefined,
    nwarnings: usize = 0,
    protocol: []const u8 = "",        // WHATWG form: "https:"
    username: ?[]const u8 = null,
    password: ?[]const u8 = null,
    host: []const u8 = "",            // drops a scheme-default port
    hostname: []const u8 = "",        // IPv6 literals keep their brackets
    port: ?[]const u8 = null,         // explicitly-written port only
    pathname: []const u8 = "",
    search: ?[]const u8 = null,       // "?..." — null when absent
    hash: ?[]const u8 = null,         // "#..." — null when absent
    params: [16]UrlParam = undefined,
    nparams: usize = 0,
    origin: ?[]const u8 = null,       // null = opaque origin (file:, data:, ...)
    is_secure: bool = false,
    default_port: ?bool = null,

    /// Backing storage for decoded / formatted strings + a bump cursor.
    scratch: [2048]u8 = undefined,
    used: usize = 0,

    fn warn(self: *UrlReport, msg: []const u8) void {
        if (self.nwarnings < self.warnings.len) {
            self.warnings[self.nwarnings] = msg;
            self.nwarnings += 1;
        }
    }

    /// Copy `bytes` into scratch and return a slice to the copy (truncating at
    /// the arena end keeps the report safe on oversized inputs).
    fn put(self: *UrlReport, bytes: []const u8) []const u8 {
        const start = self.used;
        const n = @min(bytes.len, self.scratch.len - start);
        @memcpy(self.scratch[start..][0..n], bytes[0..n]);
        self.used = start + n;
        return self.scratch[start..][0..n];
    }

    /// put() + lowercase — schemes and hosts serialise lowercased (WHATWG).
    fn putLower(self: *UrlReport, bytes: []const u8) []const u8 {
        const start = self.used;
        const n = @min(bytes.len, self.scratch.len - start);
        for (bytes[0..n], 0..) |c, j| self.scratch[start + j] = std.ascii.toLower(c);
        self.used = start + n;
        return self.scratch[start..][0..n];
    }
};

/// Well-known default ports per scheme (the same table as the TS lib), keyed
/// WHATWG-style with the ':'.
fn defaultPort(proto: []const u8) ?[]const u8 {
    if (std.mem.eql(u8, proto, "http:") or std.mem.eql(u8, proto, "ws:")) return "80";
    if (std.mem.eql(u8, proto, "https:") or std.mem.eql(u8, proto, "wss:")) return "443";
    if (std.mem.eql(u8, proto, "ftp:")) return "21";
    return null;
}

fn allDigits(s: []const u8) bool {
    if (s.len == 0) return false;
    for (s) |c| if (!std.ascii.isDigit(c)) return false;
    return true;
}

fn hexv(c: u8) ?u4 {
    return switch (c) {
        '0'...'9' => @as(u4, @intCast(c - '0')),
        'a'...'f' => @as(u4, @intCast(c - 'a' + 10)),
        'A'...'F' => @as(u4, @intCast(c - 'A' + 10)),
        else => null,
    };
}

/// Percent-decode `s` into `buf`, '+' as a space. Errors on a malformed escape
/// or overflow — the caller falls back to the raw input, like the TS try/catch.
fn percentDecode(buf: []u8, s: []const u8) ![]const u8 {
    var n: usize = 0;
    var i: usize = 0;
    while (i < s.len) : (i += 1) {
        if (n == buf.len) return error.NoSpaceLeft;
        if (s[i] == '+') {
            buf[n] = ' ';
            n += 1;
            continue;
        }
        if (s[i] == '%') {
            if (i + 2 >= s.len) return error.MalformedEscape; // needs two hex digits
            const hi = hexv(s[i + 1]) orelse return error.MalformedEscape;
            const lo = hexv(s[i + 2]) orelse return error.MalformedEscape;
            buf[n] = (@as(u8, hi) << 4) | @as(u8, lo);
            n += 1;
            i += 2;
            continue;
        }
        buf[n] = s[i];
        n += 1;
    }
    return buf[0..n];
}

/// Parse and decompose a URL into `r`; never fails hard — an unparseable input
/// leaves r.valid == false with the reason in r.warnings.
pub fn inspectUrl(raw: []const u8, r: *UrlReport) void {
    const trimmed = std.mem.trim(u8, raw, " \t\r\n");
    if (trimmed.len == 0) {
        r.warn("URL is empty");
        return;
    }

    // scheme: [A-Za-z][A-Za-z0-9+.-]* then "://"
    if (!std.ascii.isAlphabetic(trimmed[0])) {
        r.warn(invalid_msg);
        return;
    }
    var i: usize = 1;
    while (i < trimmed.len and (std.ascii.isAlphanumeric(trimmed[i]) or
        trimmed[i] == '+' or trimmed[i] == '-' or trimmed[i] == '.')) : (i += 1)
    {}
    if (i + 3 > trimmed.len or trimmed[i] != ':' or trimmed[i + 1] != '/' or trimmed[i + 2] != '/') {
        r.warn(invalid_msg);
        return;
    }
    r.protocol = r.putLower(trimmed[0 .. i + 1]); // scheme + ':' — the WHATWG "https:" form
    const rest = trimmed[i + 3 ..];

    // authority runs until the first '/', '?' or '#'
    const aend = std.mem.indexOfAny(u8, rest, "/?#") orelse rest.len;
    const authority = rest[0..aend];

    // userinfo: split at the FIRST ':' up to the LAST '@' in the authority
    var hostport = authority;
    if (std.mem.lastIndexOfScalar(u8, authority, '@')) |at| {
        hostport = authority[at + 1 ..];
        const userinfo = authority[0..at];
        if (std.mem.indexOfScalar(u8, userinfo, ':')) |c| {
            const un = userinfo[0..c];
            const pw = userinfo[c + 1 ..];
            if (un.len > 0) {
                r.username = r.put(un);
                r.warn("URL contains a username credential");
            }
            if (pw.len > 0) {
                r.password = r.put(pw);
                r.warn("URL contains a password credential");
            }
        } else if (userinfo.len > 0) {
            r.username = r.put(userinfo);
            r.warn("URL contains a username credential");
        }
    }

    // host[:port] — IPv6 literals keep their brackets, host is lowercased
    var port: ?[]const u8 = null;
    if (hostport.len > 0 and hostport[0] == '[') {
        const close = std.mem.indexOfScalar(u8, hostport, ']') orelse {
            r.warn(invalid_msg); // unterminated IPv6 literal
            return;
        };
        r.hostname = r.putLower(hostport[0 .. close + 1]);
        if (close + 1 < hostport.len and hostport[close + 1] == ':') {
            const cand = hostport[close + 2 ..];
            if (allDigits(cand)) port = cand;
        }
    } else if (std.mem.indexOfScalar(u8, hostport, ':')) |c| {
        r.hostname = r.putLower(hostport[0..c]);
        const cand = hostport[c + 1 ..];
        if (allDigits(cand)) port = cand;
    } else {
        r.hostname = r.putLower(hostport);
    }
    if (r.hostname.len == 0) {
        r.warn(invalid_msg); // empty host
        return;
    }

    // Explicit port, flagged when it equals the scheme default.
    const dp = defaultPort(r.protocol);
    if (port) |pt| {
        r.port = pt;
        const is_default = dp != null and std.mem.eql(u8, dp.?, pt);
        r.default_port = is_default;
        if (is_default) {
            var tmp: [96]u8 = undefined;
            if (std.fmt.bufPrint(&tmp, "Port {s} is the default for {s}", .{ pt, r.protocol })) |w|
                r.warn(r.put(w))
            else |_| {}
        }
    }

    // host drops a scheme-default port (WHATWG serialisation).
    if (port != null and !(r.default_port orelse false)) {
        var tmp: [224]u8 = undefined;
        if (std.fmt.bufPrint(&tmp, "{s}:{s}", .{ r.hostname, port.? })) |h|
            r.host = r.put(h)
        else |_| {}
    } else r.host = r.hostname;

    // path / query / fragment from the first delimiter on; a '?' after the '#'
    // belongs to the fragment.
    const tail = rest[aend..];
    const hashpos = std.mem.indexOfScalar(u8, tail, '#');
    var qmark = std.mem.indexOfScalar(u8, tail, '?');
    if (qmark != null and hashpos != null and qmark.? > hashpos.?) qmark = null;
    const pend = qmark orelse (hashpos orelse tail.len);
    r.pathname = if (pend == 0) "/" else tail[0..pend]; // WHATWG: empty path -> '/'
    if (qmark) |q| {
        const qe = hashpos orelse tail.len;
        var tmp: [288]u8 = undefined;
        if (std.fmt.bufPrint(&tmp, "?{s}", .{tail[q + 1 .. qe]})) |s|
            r.search = r.put(s)
        else |_| {}
    }
    if (hashpos) |h| {
        var tmp: [160]u8 = undefined;
        if (std.fmt.bufPrint(&tmp, "#{s}", .{tail[h + 1 ..]})) |s|
            r.hash = r.put(s)
        else |_| {}
    }

    // query params: split '&', split at the FIRST '=', decode '+' and %XX
    if (r.search) |s| {
        if (s.len > 1) {
            var it = std.mem.splitScalar(u8, s[1..], '&');
            while (it.next()) |pair| {
                if (pair.len == 0 or r.nparams == r.params.len) continue;
                const eq = std.mem.indexOfScalar(u8, pair, '=');
                const raw_k = if (eq) |x| pair[0..x] else pair;
                const raw_v = if (eq) |x| pair[x + 1 ..] else "";
                var kbuf: [128]u8 = undefined;
                var vbuf: [128]u8 = undefined;
                const k = percentDecode(&kbuf, raw_k) catch raw_k;
                const v = percentDecode(&vbuf, raw_v) catch raw_v;
                r.params[r.nparams] = .{ .key = r.put(k), .value = r.put(v) };
                r.nparams += 1;
            }
        }
    }

    if (std.mem.eql(u8, r.pathname, "/") and
        (r.search == null or std.mem.eql(u8, r.search.?, "?")) and r.nparams == 0)
        r.warn("URL points to the site root (no path or query)");

    r.is_secure = std.mem.eql(u8, r.protocol, "https:") or std.mem.eql(u8, r.protocol, "wss:");
    if (dp != null) { // only the special schemes yield a non-opaque origin
        var tmp: [224]u8 = undefined;
        if (std.fmt.bufPrint(&tmp, "{s}://{s}", .{ r.protocol[0 .. r.protocol.len - 1], r.host })) |o|
            r.origin = r.put(o)
        else |_| {}
    }
    r.valid = true;
}

pub fn main() void {
    const out = std.io.getStdOut().writer();

    var r: UrlReport = .{};
    inspectUrl("https://user:pass@example.com:8443/docs/api?q=hello+world&tags=a&tags=b&path=%2Fhome#section", &r);
    out.print("protocol  {s}  secure={}\n", .{ r.protocol, r.is_secure }) catch {};
    out.print("creds     {s}:{s}\n", .{ r.username orelse "-", r.password orelse "-" }) catch {};
    out.print("host      {s}  (port {s}, default={})\n", .{ r.host, r.port orelse "-", r.default_port }) catch {};
    out.print("path      {s}  search {s}  hash {s}\n", .{ r.pathname, r.search orelse "-", r.hash orelse "-" }) catch {};
    var i: usize = 0;
    while (i < r.nparams) : (i += 1)
        out.print("param     {s} = {s}\n", .{ r.params[i].key, r.params[i].value }) catch {};
    out.print("origin    {s}\nwarnings  {d}\n", .{ r.origin orelse "(opaque)", r.nwarnings }) catch {};

    var d: UrlReport = .{};
    inspectUrl("http://example.com:80/", &d);
    out.print("\nhttp://example.com:80/ ->\n", .{}) catch {};
    var j: usize = 0;
    while (j < d.nwarnings) : (j += 1) out.print("  - {s}\n", .{d.warnings[j]}) catch {};

    var e: UrlReport = .{};
    inspectUrl("not a url", &e);
    out.print("\n'not a url' -> valid={} ({s})\n", .{ e.valid, e.warnings[0] }) catch {};
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →