Skip to content

URL Encode / Decode — Python source

Percent-encode or decode URLs and query parameters. Choose component (encodeURIComponent) or full-URI (encodeURI) mode. 100% client-side.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

#!/usr/bin/env python3
# URL encode / decode — component-level (encodeURIComponent) and full URI (encodeURI).
#
# Language: Python
# CosmoDev polyglot showcase port of the `url-encode` tool.
# Ported from src/tools/UrlEncodeTool.tsx — display source, part of CosmoDev's
# polyglot tool pages.
#
# Explicit percent-encoding matching JavaScript's encodeURIComponent/encodeURI.
# Python str is Unicode, so we encode to UTF-8 first and walk the bytes: each
# non-safe byte becomes %XX (uppercase hex). Component scope leaves
# A-Za-z0-9-_.!~*'() unescaped; full URI scope additionally leaves the
# reserved set ;,/?:@&=+$# unescaped. Decode reverses this, raising
# ValueError on malformed % sequences or invalid UTF-8; full URI leaves
# encoded reserved bytes intact. urllib.parse.quote differs from
# encodeURIComponent on !*'(), so this is hand-rolled for exact parity.

_COMPONENT_SAFE = set(
    b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_.!~*'()"
)
_URI_EXTRA = set(b";,/?:@&=+$#")
_RESERVED = set(b";/?:@&=+$#")
_HEXDIGITS = set("0123456789abcdefABCDEF")


def _unhex_pair(s: str, i: int) -> int | None:
    """Return the byte value of the two hex digits after '%' at index i, or None."""
    if i + 2 >= len(s) or s[i + 1] not in _HEXDIGITS or s[i + 2] not in _HEXDIGITS:
        return None
    return int(s[i + 1:i + 3], 16)


def encode(s: str, full_uri: bool = False) -> str:
    """Percent-encode ``s``. ``full_uri`` selects encodeURI vs encodeURIComponent."""
    out = []
    for b in s.encode("utf-8"):
        if b in _COMPONENT_SAFE or (full_uri and b in _URI_EXTRA):
            out.append(chr(b))
        else:
            out.append("%%%02X" % b)
    return "".join(out)


def decode(s: str, full_uri: bool = False) -> str:
    """Percent-decode ``s``. Raises ValueError on malformed sequences.

    ``full_uri`` selects decodeURI (encoded reserved chars preserved) vs
    decodeURIComponent.
    """
    out = bytearray()
    i, n = 0, len(s)
    while i < n:
        if s[i] != "%":
            out.extend(s[i].encode("utf-8"))
            i += 1
            continue
        b = _unhex_pair(s, i)
        if b is None:
            raise ValueError("malformed URI sequence")
        if b < 0x80:
            if full_uri and b in _RESERVED:
                out.extend(s[i:i + 3].encode("utf-8"))
            else:
                out.append(b)
            i += 3
            continue
        # Multi-byte UTF-8 lead byte.
        if b & 0xE0 == 0xC0:
            seq_len = 2
        elif b & 0xF0 == 0xE0:
            seq_len = 3
        elif b & 0xF8 == 0xF0:
            seq_len = 4
        else:
            raise ValueError("malformed URI sequence")
        out.append(b)
        for k in range(1, seq_len):
            pi = i + 3 * k
            if pi >= n or s[pi] != "%":
                raise ValueError("malformed URI sequence")
            cb = _unhex_pair(s, pi)
            if cb is None or cb & 0xC0 != 0x80:
                raise ValueError("malformed URI sequence")
            out.append(cb)
        i += 3 * seq_len
    try:
        return out.decode("utf-8")
    except UnicodeDecodeError as exc:
        raise ValueError("malformed URI sequence") from exc


if __name__ == "__main__":
    print(encode("hello world & café", False))
    print(decode("hello%20world%20%26%20caf%C3%A9", False))

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →