URL Encode / Decode — Python source
Percent-encode or decode URLs and query parameters. Choose component (encodeURIComponent) or full-URI (encodeURI) mode. 100% client-side.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
#!/usr/bin/env python3
# URL encode / decode — component-level (encodeURIComponent) and full URI (encodeURI).
#
# Language: Python
# CosmoDev polyglot showcase port of the `url-encode` tool.
# Ported from src/tools/UrlEncodeTool.tsx — display source, part of CosmoDev's
# polyglot tool pages.
#
# Explicit percent-encoding matching JavaScript's encodeURIComponent/encodeURI.
# Python str is Unicode, so we encode to UTF-8 first and walk the bytes: each
# non-safe byte becomes %XX (uppercase hex). Component scope leaves
# A-Za-z0-9-_.!~*'() unescaped; full URI scope additionally leaves the
# reserved set ;,/?:@&=+$# unescaped. Decode reverses this, raising
# ValueError on malformed % sequences or invalid UTF-8; full URI leaves
# encoded reserved bytes intact. urllib.parse.quote differs from
# encodeURIComponent on !*'(), so this is hand-rolled for exact parity.
_COMPONENT_SAFE = set(
b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_.!~*'()"
)
_URI_EXTRA = set(b";,/?:@&=+$#")
_RESERVED = set(b";/?:@&=+$#")
_HEXDIGITS = set("0123456789abcdefABCDEF")
def _unhex_pair(s: str, i: int) -> int | None:
"""Return the byte value of the two hex digits after '%' at index i, or None."""
if i + 2 >= len(s) or s[i + 1] not in _HEXDIGITS or s[i + 2] not in _HEXDIGITS:
return None
return int(s[i + 1:i + 3], 16)
def encode(s: str, full_uri: bool = False) -> str:
"""Percent-encode ``s``. ``full_uri`` selects encodeURI vs encodeURIComponent."""
out = []
for b in s.encode("utf-8"):
if b in _COMPONENT_SAFE or (full_uri and b in _URI_EXTRA):
out.append(chr(b))
else:
out.append("%%%02X" % b)
return "".join(out)
def decode(s: str, full_uri: bool = False) -> str:
"""Percent-decode ``s``. Raises ValueError on malformed sequences.
``full_uri`` selects decodeURI (encoded reserved chars preserved) vs
decodeURIComponent.
"""
out = bytearray()
i, n = 0, len(s)
while i < n:
if s[i] != "%":
out.extend(s[i].encode("utf-8"))
i += 1
continue
b = _unhex_pair(s, i)
if b is None:
raise ValueError("malformed URI sequence")
if b < 0x80:
if full_uri and b in _RESERVED:
out.extend(s[i:i + 3].encode("utf-8"))
else:
out.append(b)
i += 3
continue
# Multi-byte UTF-8 lead byte.
if b & 0xE0 == 0xC0:
seq_len = 2
elif b & 0xF0 == 0xE0:
seq_len = 3
elif b & 0xF8 == 0xF0:
seq_len = 4
else:
raise ValueError("malformed URI sequence")
out.append(b)
for k in range(1, seq_len):
pi = i + 3 * k
if pi >= n or s[pi] != "%":
raise ValueError("malformed URI sequence")
cb = _unhex_pair(s, pi)
if cb is None or cb & 0xC0 != 0x80:
raise ValueError("malformed URI sequence")
out.append(cb)
i += 3 * seq_len
try:
return out.decode("utf-8")
except UnicodeDecodeError as exc:
raise ValueError("malformed URI sequence") from exc
if __name__ == "__main__":
print(encode("hello world & café", False))
print(decode("hello%20world%20%26%20caf%C3%A9", False))
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →