Skip to content

URL Inspector — C++ source

Break any URL into its components - protocol, host, port, path, query params, hash, and credentials. Detects default ports and security at a glance, with a decode toggle for query values. Runs entirely in your browser.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// url-inspector — break any URL into components (protocol, credentials, host, port, path, query params, fragment), detecting default ports and security at a glance. Language: C++ (C++17, stdlib only). Port of src/lib/url-inspector.ts — same contract as this dir's javascript.js; the parser is hand-rolled (no platform URL lib) with the same authority walk the TS rawPort() helper does (userinfo, IPv6 literals, explicit ports).
#include <cctype>
#include <iostream>
#include <optional>
#include <string>
#include <vector>

namespace urlinspector {

struct UrlParam {
    std::string key, value;
};

// Flat, serialisable decomposition; nullopt mirrors the TS undefined-when-absent.
// Field names follow the TS report surface (camelCase) so the ports stay
// line-comparable.
struct UrlReport {
    bool valid = false;
    std::vector<std::string> warnings;
    std::string protocol;                       // WHATWG form: "https:"
    std::optional<std::string> username, password, port, search, hash, origin;
    std::string host, hostname, pathname;
    std::vector<UrlParam> searchParams;
    bool isSecure = false;
    std::optional<bool> defaultPort;

    static UrlReport invalid(const std::string& warning) {
        UrlReport r;
        r.warnings.push_back(warning);
        return r;
    }
};

inline const std::string INVALID =
    "Invalid URL - could not be parsed (include the scheme, e.g. https://)";

// Well-known default ports per scheme (the same table as the TS lib), keyed
// WHATWG-style with the ':'.
inline std::optional<std::string> default_port(const std::string& proto) {
    if (proto == "http:" || proto == "ws:") return std::string("80");
    if (proto == "https:" || proto == "wss:") return std::string("443");
    if (proto == "ftp:") return std::string("21");
    return std::nullopt;
}

inline bool is_digits(const std::string& s) {
    if (s.empty()) return false;
    for (char c : s)
        if (!std::isdigit(static_cast<unsigned char>(c))) return false;
    return true;
}

inline std::string to_lower(std::string s) {
    for (char& c : s) c = static_cast<char>(std::tolower(static_cast<unsigned char>(c)));
    return s;
}

inline int hexv(char c) {
    return c >= '0' && c <= '9' ? c - '0' : c >= 'a' && c <= 'f' ? c - 'a' + 10
         : c >= 'A' && c <= 'F' ? c - 'A' + 10 : -1;
}

// Percent-decode a query value, '+' as a space. nullopt on a malformed escape
// so the caller keeps the raw original — the TS try/catch fallback.
inline std::optional<std::string> percent_decode(const std::string& s) {
    std::string out;
    for (size_t i = 0; i < s.size(); i++) {
        if (s[i] == '+') { out += ' '; continue; }
        if (s[i] == '%') {
            int hi = i + 1 < s.size() ? hexv(s[i + 1]) : -1;
            int lo = i + 2 < s.size() ? hexv(s[i + 2]) : -1;
            if (hi < 0 || lo < 0) return std::nullopt;
            out += static_cast<char>(hi * 16 + lo);
            i += 2;
        } else out += s[i];
    }
    return out;
}

inline std::string decode_or_raw(const std::string& s) {
    return percent_decode(s).value_or(s);
}

// Decode a raw query string into ordered pairs, preserving duplicates.
inline std::vector<UrlParam> parse_query(const std::string& raw) {
    std::vector<UrlParam> out;
    size_t start = 0;
    while (start <= raw.size()) {                    // split on '&' (empty pairs skipped)
        size_t amp = raw.find('&', start);
        std::string pair = raw.substr(start, amp == std::string::npos ? std::string::npos : amp - start);
        if (!pair.empty()) {
            size_t eq = pair.find('=');
            if (eq == std::string::npos)
                out.push_back({decode_or_raw(pair), ""});
            else
                out.push_back({decode_or_raw(pair.substr(0, eq)), decode_or_raw(pair.substr(eq + 1))});
        }
        if (amp == std::string::npos) break;
        start = amp + 1;
    }
    return out;
}

// Parse and decompose a URL into a structured report; never throws.
inline UrlReport inspect_url(const std::string& raw) {
    size_t b = raw.find_first_not_of(" \t\r\n");
    size_t e = raw.find_last_not_of(" \t\r\n");
    std::string trimmed = b == std::string::npos ? "" : raw.substr(b, e - b + 1);
    if (trimmed.empty()) return UrlReport::invalid("URL is empty");

    // scheme: [A-Za-z][A-Za-z0-9+.-]* then "://"
    auto bad_scheme = [] { return UrlReport::invalid(INVALID); };
    if (trimmed.empty() || !std::isalpha(static_cast<unsigned char>(trimmed[0]))) return bad_scheme();
    size_t p = 1;
    while (p < trimmed.size() && (std::isalnum(static_cast<unsigned char>(trimmed[p])) ||
                                  trimmed[p] == '+' || trimmed[p] == '-' || trimmed[p] == '.'))
        p++;
    if (trimmed.compare(p, 3, "://") != 0) return bad_scheme();
    std::string proto = to_lower(trimmed.substr(0, p)) + ":";
    std::string rest = trimmed.substr(p + 3);

    UrlReport r;
    r.protocol = proto;

    // authority runs until the first '/', '?' or '#'
    size_t aend = rest.find_first_of("/?#");
    std::string authority = aend == std::string::npos ? rest : rest.substr(0, aend);

    // userinfo: split at the FIRST ':' up to the LAST '@' in the authority
    std::string hostport = authority;
    if (size_t at = authority.rfind('@'); at != std::string::npos) {
        std::string userinfo = authority.substr(0, at);
        size_t colon = userinfo.find(':');
        r.username = colon == std::string::npos ? userinfo : userinfo.substr(0, colon);
        if (colon != std::string::npos) r.password = userinfo.substr(colon + 1);
        if (r.username && !r.username->empty()) r.warnings.push_back("URL contains a username credential");
        if (r.password && !r.password->empty()) r.warnings.push_back("URL contains a password credential");
        hostport = authority.substr(at + 1);
    }

    // host[:port] — IPv6 literals keep their brackets, host is lowercased
    std::optional<std::string> port;
    if (!hostport.empty() && hostport.front() == '[') {
        size_t close = hostport.find(']');
        if (close == std::string::npos) return bad_scheme();   // unterminated IPv6 literal
        r.hostname = to_lower(hostport.substr(0, close + 1));
        if (close + 1 < hostport.size() && hostport[close + 1] == ':') {
            std::string cand = hostport.substr(close + 2);
            if (is_digits(cand)) port = cand;
        }
    } else if (size_t colon = hostport.find(':'); colon != std::string::npos) {
        r.hostname = to_lower(hostport.substr(0, colon));
        std::string cand = hostport.substr(colon + 1);
        if (is_digits(cand)) port = cand;
    } else {
        r.hostname = to_lower(hostport);
    }
    if (r.hostname.empty()) return bad_scheme();

    // Explicit port, flagged when it equals the scheme default.
    auto dp = default_port(proto);
    if (port) {
        r.port = port;
        r.defaultPort = dp && *dp == *port;
        if (*r.defaultPort)
            r.warnings.push_back("Port " + *port + " is the default for " + proto);
    }

    // host drops a scheme-default port (WHATWG serialisation).
    r.host = port && !r.defaultPort.value_or(false) ? r.hostname + ":" + *port : r.hostname;

    // path / query / fragment; a '?' after the '#' belongs to the fragment
    std::string tail = aend == std::string::npos ? "" : rest.substr(aend);
    size_t hashpos = tail.find('#');
    size_t qmark = tail.find('?');
    if (qmark != std::string::npos && hashpos != std::string::npos && qmark > hashpos) qmark = std::string::npos;
    size_t pend = qmark != std::string::npos ? qmark : (hashpos != std::string::npos ? hashpos : tail.size());
    r.pathname = tail.substr(0, pend);                       // WHATWG: empty path -> '/'
    if (r.pathname.empty()) r.pathname = "/";
    if (qmark != std::string::npos)
        r.search = "?" + tail.substr(qmark + 1, (hashpos != std::string::npos ? hashpos : tail.size()) - qmark - 1);
    if (hashpos != std::string::npos) r.hash = "#" + tail.substr(hashpos + 1);

    // Query params, decoded in insertion order with duplicates preserved.
    if (r.search && r.search->size() > 1) r.searchParams = parse_query(r.search->substr(1));

    if (r.pathname == "/" && (!r.search || r.search == "?") && r.searchParams.empty())
        r.warnings.push_back("URL points to the site root (no path or query)");

    r.isSecure = proto == "https:" || proto == "wss:";
    // Origin: only the special schemes yield a non-opaque origin ("scheme://host",
    // default port already dropped from host).
    if (dp) r.origin = proto.substr(0, proto.size() - 1) + "://" + r.host;
    r.valid = true;
    return r;
}

}  // namespace urlinspector

int main() {
    using namespace urlinspector;
    UrlReport r = inspect_url(
        "https://user:pass@example.com:8443/docs/api?q=hello+world&tags=a&tags=b&path=%2Fhome#section");
    std::cout << "protocol  " << r.protocol << "  secure=" << r.isSecure << "\n";
    std::cout << "creds     " << r.username.value_or("-") << ":" << r.password.value_or("-") << "\n";
    std::cout << "host      " << r.host << "  (port " << r.port.value_or("-")
              << ", default=" << (r.defaultPort ? std::to_string(*r.defaultPort) : "n/a") << ")\n";
    std::cout << "path      " << r.pathname << "  search " << r.search.value_or("-")
              << "  hash " << r.hash.value_or("-") << "\n";
    for (const auto& prm : r.searchParams) std::cout << "param     " << prm.key << " = " << prm.value << "\n";
    std::cout << "origin    " << r.origin.value_or("(opaque)") << "\n";
    std::cout << "warnings  " << (r.warnings.empty() ? "(none)" : r.warnings[0]) << "\n";

    UrlReport d = inspect_url("http://example.com:80/");
    std::cout << "\nhttp://example.com:80/ ->\n";
    for (const auto& w : d.warnings) std::cout << "  - " << w << "\n";

    UrlReport e = inspect_url("not a url");
    std::cout << "\n'not a url' -> valid=" << e.valid << " (" << e.warnings[0] << ")\n";
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →