URL Inspector — C source
Break any URL into its components - protocol, host, port, path, query params, hash, and credentials. Detects default ports and security at a glance, with a decode toggle for query values. Runs entirely in your browser.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/* url-inspector — break any URL into components (protocol, credentials, host, port, path, query params, fragment), detecting default ports and security at a glance. Language: C (C11, stdlib only). Port of src/lib/url-inspector.ts — same contract as this dir's javascript.js; C has no URL library, so the parser is hand-rolled with the same authority walk the TS rawPort() helper does (userinfo, IPv6 literals, explicit ports). */
#include <ctype.h>
#include <stdio.h>
#include <string.h>
#define MAXP 16 /* max decoded query parameters kept */
typedef struct { char key[64], value[64]; } UrlParam;
typedef struct {
int valid; /* 1 when the URL parsed */
char warnings[4][80]; int nwarn;
char protocol[16]; /* WHATWG form: "https:" */
char username[64], password[64]; /* empty string = absent */
char host[192], hostname[160]; /* host drops a scheme-default port */
char port[8]; /* explicit port only, "" = absent */
char pathname[256], search[256], hash[128];
UrlParam params[MAXP]; int nparams;
char origin[224]; /* "" = opaque origin (file:, data:)*/
int is_secure, default_port; /* default_port: 0 no / 1 yes / -1 n/a */
} UrlReport;
static void set_str(char *dst, size_t n, const char *src, size_t len) {
if (len >= n) len = n - 1;
memcpy(dst, src, len); dst[len] = 0;
}
static void add_warn(UrlReport *r, const char *msg) {
if (r->nwarn < 4) set_str(r->warnings[r->nwarn++], 80, msg, strlen(msg));
}
static UrlReport invalid(const char *msg) {
UrlReport r; memset(&r, 0, sizeof r); r.default_port = -1;
add_warn(&r, msg);
return r;
}
/* Well-known default ports per scheme (the same table as the TS lib). */
static const char *default_port(const char *proto) {
if (!strcmp(proto, "http:") || !strcmp(proto, "ws:")) return "80";
if (!strcmp(proto, "https:") || !strcmp(proto, "wss:")) return "443";
if (!strcmp(proto, "ftp:")) return "21";
return NULL;
}
static int hexv(char c) {
return c >= '0' && c <= '9' ? c - '0' : c >= 'a' && c <= 'f' ? c - 'a' + 10
: c >= 'A' && c <= 'F' ? c - 'A' + 10 : -1;
}
/* Percent-decode s in place ('+' -> space). Returns 0 on a malformed escape so
the caller can keep the raw original — the TS try/catch fallback. */
static int percent_decode(char *s) {
char *w = s;
for (char *p = s; *p; p++) {
if (*p == '+') { *w++ = ' '; continue; }
if (*p == '%') {
int hi = hexv(p[1]), lo = hexv(p[2]); /* hexv('\0') = -1 */
if (hi < 0 || lo < 0) return 0;
*w++ = (char)(hi * 16 + lo); p += 2;
} else *w++ = *p;
}
*w = 0; return 1;
}
/* Parse and decompose a URL into a structured report; never fails hard — an
unparseable input yields valid == 0 with the reason in warnings. */
static UrlReport inspect_url(const char *raw) {
UrlReport r; memset(&r, 0, sizeof r); r.default_port = -1;
const char *s = raw; while (isspace((unsigned char)*s)) s++;
const char *e = s + strlen(s);
while (e > s && isspace((unsigned char)e[-1])) e--;
if (s == e) return invalid("URL is empty");
/* scheme: [A-Za-z][A-Za-z0-9+.-]* then "://" */
if (!isalpha((unsigned char)*s))
return invalid("Invalid URL - could not be parsed (include the scheme, e.g. https://)");
const char *p = s + 1;
while (p < e && (isalnum((unsigned char)*p) || *p == '+' || *p == '-' || *p == '.')) p++;
if (e - p < 3 || p[0] != ':' || p[1] != '/' || p[2] != '/')
return invalid("Invalid URL - could not be parsed (include the scheme, e.g. https://)");
for (size_t i = 0; i < (size_t)(p - s); i++) /* scheme lowercased, WHATWG */
r.protocol[i] = (char)tolower((unsigned char)s[i]);
r.protocol[p - s] = ':'; /* WHATWG form: "https:" */
r.protocol[p - s + 1] = '\0';
const char *rest = p + 3;
/* authority runs until the first '/', '?' or '#' */
const char *aend = e;
for (const char *q = rest; q < e; q++)
if (*q == '/' || *q == '?' || *q == '#') { aend = q; break; }
/* userinfo: split at the FIRST ':' up to the LAST '@' in the authority */
const char *hp = rest;
const char *at = NULL;
for (const char *q = rest; q < aend; q++) if (*q == '@') at = q;
if (at) {
const char *colon = NULL;
for (const char *q = rest; q < at; q++) if (*q == ':') { colon = q; break; }
if (colon) {
set_str(r.username, sizeof r.username, rest, (size_t)(colon - rest));
set_str(r.password, sizeof r.password, colon + 1, (size_t)(at - colon - 1));
} else set_str(r.username, sizeof r.username, rest, (size_t)(at - rest));
if (r.username[0]) add_warn(&r, "URL contains a username credential");
if (r.password[0]) add_warn(&r, "URL contains a password credential");
hp = at + 1;
}
/* host[:port] — IPv6 literals keep their brackets, host is lowercased */
const char *portstart = NULL;
if (*hp == '[') {
const char *close = memchr(hp, ']', (size_t)(aend - hp));
if (!close) return invalid("Invalid URL - could not be parsed (include the scheme, e.g. https://)");
set_str(r.hostname, sizeof r.hostname, hp, (size_t)(close - hp + 1));
if (close + 1 < aend && close[1] == ':') portstart = close + 2;
} else {
const char *colon = memchr(hp, ':', (size_t)(aend - hp));
if (colon) { set_str(r.hostname, sizeof r.hostname, hp, (size_t)(colon - hp)); portstart = colon + 1; }
else set_str(r.hostname, sizeof r.hostname, hp, (size_t)(aend - hp));
}
if (!r.hostname[0])
return invalid("Invalid URL - could not be parsed (include the scheme, e.g. https://)");
for (char *c = r.hostname; *c; c++) *c = (char)tolower((unsigned char)*c);
/* explicit port: digits only, then flagged when it equals the scheme default */
if (portstart && portstart < aend) {
int ok = 1;
for (const char *q = portstart; q < aend; q++) if (!isdigit((unsigned char)*q)) { ok = 0; break; }
if (ok) set_str(r.port, sizeof r.port, portstart, (size_t)(aend - portstart));
}
const char *dp = default_port(r.protocol);
if (r.port[0]) {
r.default_port = dp && !strcmp(dp, r.port);
if (r.default_port) {
char w[80]; snprintf(w, sizeof w, "Port %s is the default for %s", r.port, r.protocol);
add_warn(&r, w);
}
}
/* host drops a scheme-default port (WHATWG serialisation) */
if (r.port[0] && !r.default_port) snprintf(r.host, sizeof r.host, "%s:%s", r.hostname, r.port);
else set_str(r.host, sizeof r.host, r.hostname, strlen(r.hostname));
/* path / query / fragment from the first delimiter on; a '?' after the '#'
belongs to the fragment */
const char *tail = aend;
const char *hashpos = memchr(tail, '#', (size_t)(e - tail));
const char *qmark = memchr(tail, '?', (size_t)(e - tail));
if (qmark && hashpos && qmark > hashpos) qmark = NULL;
const char *pend = qmark ? qmark : (hashpos ? hashpos : e);
if (pend == tail) set_str(r.pathname, sizeof r.pathname, "/", 1); /* WHATWG: empty path -> '/' */
else set_str(r.pathname, sizeof r.pathname, tail, (size_t)(pend - tail));
if (qmark) set_str(r.search, sizeof r.search, qmark, (size_t)((hashpos ? hashpos : e) - qmark));
if (hashpos) set_str(r.hash, sizeof r.hash, hashpos, (size_t)(e - hashpos));
/* query params: split '&', split at the FIRST '=', decode '+' and %XX */
if (r.search[0]) {
char buf[256]; set_str(buf, sizeof buf, r.search + 1, strlen(r.search) - 1);
char *save = NULL;
for (char *pair = strtok_r(buf, "&", &save); pair; pair = strtok_r(NULL, "&", &save)) {
if (r.nparams >= MAXP) break;
char k[64] = "", v[64] = "";
char *eq = strchr(pair, '=');
if (eq) {
set_str(k, sizeof k, pair, (size_t)(eq - pair));
set_str(v, sizeof v, eq + 1, strlen(eq + 1));
if (!percent_decode(k)) set_str(k, sizeof k, pair, (size_t)(eq - pair));
if (!percent_decode(v)) set_str(v, sizeof v, eq + 1, strlen(eq + 1));
} else {
set_str(k, sizeof k, pair, strlen(pair));
if (!percent_decode(k)) set_str(k, sizeof k, pair, strlen(pair));
}
set_str(r.params[r.nparams].key, 64, k, strlen(k));
set_str(r.params[r.nparams].value, 64, v, strlen(v));
r.nparams++;
}
}
if (!strcmp(r.pathname, "/") && !r.search[0] && !r.nparams)
add_warn(&r, "URL points to the site root (no path or query)");
r.is_secure = !strcmp(r.protocol, "https:") || !strcmp(r.protocol, "wss:");
if (dp) snprintf(r.origin, sizeof r.origin, "%s//%s", r.protocol, r.host); /* special schemes; protocol ends with ':' */
r.valid = 1;
return r;
}
int main(void) {
UrlReport r = inspect_url("https://user:pass@example.com:8443/docs/api?q=hello+world&tags=a&tags=b&path=%2Fhome#section");
printf("protocol %s secure=%d\n", r.protocol, r.is_secure);
printf("creds %s:%s\n", r.username, r.password);
printf("host %s (port %s, default=%d)\n", r.host, r.port[0] ? r.port : "-", r.default_port);
printf("path %s search %s hash %s\n", r.pathname, r.search[0] ? r.search : "-", r.hash[0] ? r.hash : "-");
for (int i = 0; i < r.nparams; i++) printf("param %s = %s\n", r.params[i].key, r.params[i].value);
printf("origin %s\nwarnings %d\n", r.origin, r.nwarn);
UrlReport d = inspect_url("http://example.com:80/");
printf("\nhttp://example.com:80/ ->\n");
for (int i = 0; i < d.nwarn; i++) printf(" - %s\n", d.warnings[i]);
UrlReport e = inspect_url("not a url");
printf("\n'not a url' -> valid=%d (%s)\n", e.valid, e.warnings[0]);
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →