Skip to content

URL Inspector — C source

Break any URL into its components - protocol, host, port, path, query params, hash, and credentials. Detects default ports and security at a glance, with a decode toggle for query values. Runs entirely in your browser.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/* url-inspector — break any URL into components (protocol, credentials, host, port, path, query params, fragment), detecting default ports and security at a glance. Language: C (C11, stdlib only). Port of src/lib/url-inspector.ts — same contract as this dir's javascript.js; C has no URL library, so the parser is hand-rolled with the same authority walk the TS rawPort() helper does (userinfo, IPv6 literals, explicit ports). */
#include <ctype.h>
#include <stdio.h>
#include <string.h>

#define MAXP 16 /* max decoded query parameters kept */

typedef struct { char key[64], value[64]; } UrlParam;

typedef struct {
    int valid;                            /* 1 when the URL parsed            */
    char warnings[4][80]; int nwarn;
    char protocol[16];                    /* WHATWG form: "https:"            */
    char username[64], password[64];      /* empty string = absent            */
    char host[192], hostname[160];        /* host drops a scheme-default port */
    char port[8];                         /* explicit port only, "" = absent  */
    char pathname[256], search[256], hash[128];
    UrlParam params[MAXP]; int nparams;
    char origin[224];                     /* "" = opaque origin (file:, data:)*/
    int is_secure, default_port;          /* default_port: 0 no / 1 yes / -1 n/a */
} UrlReport;

static void set_str(char *dst, size_t n, const char *src, size_t len) {
    if (len >= n) len = n - 1;
    memcpy(dst, src, len); dst[len] = 0;
}

static void add_warn(UrlReport *r, const char *msg) {
    if (r->nwarn < 4) set_str(r->warnings[r->nwarn++], 80, msg, strlen(msg));
}

static UrlReport invalid(const char *msg) {
    UrlReport r; memset(&r, 0, sizeof r); r.default_port = -1;
    add_warn(&r, msg);
    return r;
}

/* Well-known default ports per scheme (the same table as the TS lib). */
static const char *default_port(const char *proto) {
    if (!strcmp(proto, "http:") || !strcmp(proto, "ws:")) return "80";
    if (!strcmp(proto, "https:") || !strcmp(proto, "wss:")) return "443";
    if (!strcmp(proto, "ftp:")) return "21";
    return NULL;
}

static int hexv(char c) {
    return c >= '0' && c <= '9' ? c - '0' : c >= 'a' && c <= 'f' ? c - 'a' + 10
         : c >= 'A' && c <= 'F' ? c - 'A' + 10 : -1;
}

/* Percent-decode s in place ('+' -> space). Returns 0 on a malformed escape so
   the caller can keep the raw original — the TS try/catch fallback. */
static int percent_decode(char *s) {
    char *w = s;
    for (char *p = s; *p; p++) {
        if (*p == '+') { *w++ = ' '; continue; }
        if (*p == '%') {
            int hi = hexv(p[1]), lo = hexv(p[2]); /* hexv('\0') = -1 */
            if (hi < 0 || lo < 0) return 0;
            *w++ = (char)(hi * 16 + lo); p += 2;
        } else *w++ = *p;
    }
    *w = 0; return 1;
}

/* Parse and decompose a URL into a structured report; never fails hard — an
   unparseable input yields valid == 0 with the reason in warnings. */
static UrlReport inspect_url(const char *raw) {
    UrlReport r; memset(&r, 0, sizeof r); r.default_port = -1;

    const char *s = raw; while (isspace((unsigned char)*s)) s++;
    const char *e = s + strlen(s);
    while (e > s && isspace((unsigned char)e[-1])) e--;
    if (s == e) return invalid("URL is empty");

    /* scheme: [A-Za-z][A-Za-z0-9+.-]* then "://" */
    if (!isalpha((unsigned char)*s))
        return invalid("Invalid URL - could not be parsed (include the scheme, e.g. https://)");
    const char *p = s + 1;
    while (p < e && (isalnum((unsigned char)*p) || *p == '+' || *p == '-' || *p == '.')) p++;
    if (e - p < 3 || p[0] != ':' || p[1] != '/' || p[2] != '/')
        return invalid("Invalid URL - could not be parsed (include the scheme, e.g. https://)");
    for (size_t i = 0; i < (size_t)(p - s); i++) /* scheme lowercased, WHATWG */
        r.protocol[i] = (char)tolower((unsigned char)s[i]);
    r.protocol[p - s] = ':'; /* WHATWG form: "https:" */
    r.protocol[p - s + 1] = '\0';
    const char *rest = p + 3;

    /* authority runs until the first '/', '?' or '#' */
    const char *aend = e;
    for (const char *q = rest; q < e; q++)
        if (*q == '/' || *q == '?' || *q == '#') { aend = q; break; }

    /* userinfo: split at the FIRST ':' up to the LAST '@' in the authority */
    const char *hp = rest;
    const char *at = NULL;
    for (const char *q = rest; q < aend; q++) if (*q == '@') at = q;
    if (at) {
        const char *colon = NULL;
        for (const char *q = rest; q < at; q++) if (*q == ':') { colon = q; break; }
        if (colon) {
            set_str(r.username, sizeof r.username, rest, (size_t)(colon - rest));
            set_str(r.password, sizeof r.password, colon + 1, (size_t)(at - colon - 1));
        } else set_str(r.username, sizeof r.username, rest, (size_t)(at - rest));
        if (r.username[0]) add_warn(&r, "URL contains a username credential");
        if (r.password[0]) add_warn(&r, "URL contains a password credential");
        hp = at + 1;
    }

    /* host[:port] — IPv6 literals keep their brackets, host is lowercased */
    const char *portstart = NULL;
    if (*hp == '[') {
        const char *close = memchr(hp, ']', (size_t)(aend - hp));
        if (!close) return invalid("Invalid URL - could not be parsed (include the scheme, e.g. https://)");
        set_str(r.hostname, sizeof r.hostname, hp, (size_t)(close - hp + 1));
        if (close + 1 < aend && close[1] == ':') portstart = close + 2;
    } else {
        const char *colon = memchr(hp, ':', (size_t)(aend - hp));
        if (colon) { set_str(r.hostname, sizeof r.hostname, hp, (size_t)(colon - hp)); portstart = colon + 1; }
        else set_str(r.hostname, sizeof r.hostname, hp, (size_t)(aend - hp));
    }
    if (!r.hostname[0])
        return invalid("Invalid URL - could not be parsed (include the scheme, e.g. https://)");
    for (char *c = r.hostname; *c; c++) *c = (char)tolower((unsigned char)*c);

    /* explicit port: digits only, then flagged when it equals the scheme default */
    if (portstart && portstart < aend) {
        int ok = 1;
        for (const char *q = portstart; q < aend; q++) if (!isdigit((unsigned char)*q)) { ok = 0; break; }
        if (ok) set_str(r.port, sizeof r.port, portstart, (size_t)(aend - portstart));
    }
    const char *dp = default_port(r.protocol);
    if (r.port[0]) {
        r.default_port = dp && !strcmp(dp, r.port);
        if (r.default_port) {
            char w[80]; snprintf(w, sizeof w, "Port %s is the default for %s", r.port, r.protocol);
            add_warn(&r, w);
        }
    }

    /* host drops a scheme-default port (WHATWG serialisation) */
    if (r.port[0] && !r.default_port) snprintf(r.host, sizeof r.host, "%s:%s", r.hostname, r.port);
    else set_str(r.host, sizeof r.host, r.hostname, strlen(r.hostname));

    /* path / query / fragment from the first delimiter on; a '?' after the '#'
       belongs to the fragment */
    const char *tail = aend;
    const char *hashpos = memchr(tail, '#', (size_t)(e - tail));
    const char *qmark = memchr(tail, '?', (size_t)(e - tail));
    if (qmark && hashpos && qmark > hashpos) qmark = NULL;
    const char *pend = qmark ? qmark : (hashpos ? hashpos : e);
    if (pend == tail) set_str(r.pathname, sizeof r.pathname, "/", 1); /* WHATWG: empty path -> '/' */
    else set_str(r.pathname, sizeof r.pathname, tail, (size_t)(pend - tail));
    if (qmark) set_str(r.search, sizeof r.search, qmark, (size_t)((hashpos ? hashpos : e) - qmark));
    if (hashpos) set_str(r.hash, sizeof r.hash, hashpos, (size_t)(e - hashpos));

    /* query params: split '&', split at the FIRST '=', decode '+' and %XX */
    if (r.search[0]) {
        char buf[256]; set_str(buf, sizeof buf, r.search + 1, strlen(r.search) - 1);
        char *save = NULL;
        for (char *pair = strtok_r(buf, "&", &save); pair; pair = strtok_r(NULL, "&", &save)) {
            if (r.nparams >= MAXP) break;
            char k[64] = "", v[64] = "";
            char *eq = strchr(pair, '=');
            if (eq) {
                set_str(k, sizeof k, pair, (size_t)(eq - pair));
                set_str(v, sizeof v, eq + 1, strlen(eq + 1));
                if (!percent_decode(k)) set_str(k, sizeof k, pair, (size_t)(eq - pair));
                if (!percent_decode(v)) set_str(v, sizeof v, eq + 1, strlen(eq + 1));
            } else {
                set_str(k, sizeof k, pair, strlen(pair));
                if (!percent_decode(k)) set_str(k, sizeof k, pair, strlen(pair));
            }
            set_str(r.params[r.nparams].key, 64, k, strlen(k));
            set_str(r.params[r.nparams].value, 64, v, strlen(v));
            r.nparams++;
        }
    }

    if (!strcmp(r.pathname, "/") && !r.search[0] && !r.nparams)
        add_warn(&r, "URL points to the site root (no path or query)");

    r.is_secure = !strcmp(r.protocol, "https:") || !strcmp(r.protocol, "wss:");
    if (dp) snprintf(r.origin, sizeof r.origin, "%s//%s", r.protocol, r.host); /* special schemes; protocol ends with ':' */
    r.valid = 1;
    return r;
}

int main(void) {
    UrlReport r = inspect_url("https://user:pass@example.com:8443/docs/api?q=hello+world&tags=a&tags=b&path=%2Fhome#section");
    printf("protocol  %s  secure=%d\n", r.protocol, r.is_secure);
    printf("creds     %s:%s\n", r.username, r.password);
    printf("host      %s  (port %s, default=%d)\n", r.host, r.port[0] ? r.port : "-", r.default_port);
    printf("path      %s  search %s  hash %s\n", r.pathname, r.search[0] ? r.search : "-", r.hash[0] ? r.hash : "-");
    for (int i = 0; i < r.nparams; i++) printf("param     %s = %s\n", r.params[i].key, r.params[i].value);
    printf("origin    %s\nwarnings  %d\n", r.origin, r.nwarn);

    UrlReport d = inspect_url("http://example.com:80/");
    printf("\nhttp://example.com:80/ ->\n");
    for (int i = 0; i < d.nwarn; i++) printf("  - %s\n", d.warnings[i]);

    UrlReport e = inspect_url("not a url");
    printf("\n'not a url' -> valid=%d (%s)\n", e.valid, e.warnings[0]);
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →