Hex ↔ Text Converter — Swift source
Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.
This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.
// hex-converter — pure hex ↔ text conversion.
//
// Language: Swift 5.9+ (standard library + Foundation for string conveniences)
// Source: CosmoDev polyglot showcase port of the hex-converter tool,
// ported from src/lib/hexText.ts (the canonical TypeScript
// implementation).
// License: display source — part of CosmoDev's polyglot tool pages
// (dev.cosmolabs.org). Deterministic, side-effect free; invalid
// byte sequences decode to U+FFFD, matching the canonical logic.
import Foundation
/// How encoded bytes are joined when rendered as a hex string.
enum Delimiter {
/// No separator: "48656c6c6f".
case none
/// Single space between bytes: "48 65 6c 6c 6f".
case space
/// Each byte prefixed with "0x", space-separated.
case prefix0x
/// Each byte prefixed with "\x", no separator (C-style).
case backslashX
}
/// Outcome of decoding hex back to text. Mirrors the canonical TS surface:
/// `ok`, `text`, and `error` (nil when ok).
struct DecodeResult: Equatable {
let ok: Bool
let text: String
let error: String?
static func okText(_ text: String) -> DecodeResult {
DecodeResult(ok: true, text: text, error: nil)
}
static func fail(_ message: String) -> DecodeResult {
DecodeResult(ok: false, text: "", error: message)
}
}
/// U+FFFD, substituted for malformed UTF-8 on decode.
private let replacementChar = "\u{FFFD}"
/// UTF-8 encode a Swift string into an array of bytes.
///
/// Hand-rolled so every language in the polyglot showcase produces
/// byte-identical output. `unicodeScalars` yields Unicode scalar values, so
/// astral characters encode as 4-byte sequences.
func utf8Encode(_ text: String) -> [UInt8] {
var bytes: [UInt8] = []
for scalar in text.unicodeScalars {
let cp = scalar.value
if cp <= 0x7F {
bytes.append(UInt8(cp))
} else if cp <= 0x7FF {
bytes.append(0xC0 | UInt8(cp >> 6))
bytes.append(0x80 | UInt8(cp & 0x3F))
} else if cp <= 0xFFFF {
bytes.append(0xE0 | UInt8(cp >> 12))
bytes.append(0x80 | UInt8((cp >> 6) & 0x3F))
bytes.append(0x80 | UInt8(cp & 0x3F))
} else {
bytes.append(0xF0 | UInt8(cp >> 18))
bytes.append(0x80 | UInt8((cp >> 12) & 0x3F))
bytes.append(0x80 | UInt8((cp >> 6) & 0x3F))
bytes.append(0x80 | UInt8(cp & 0x3F))
}
}
return bytes
}
/// UTF-8 decode a byte array into a String. Truncated or invalid sequences
/// yield U+FFFD; missing continuation bytes are taken as 0, matching the
/// canonical decoder's lenient consumption. `Unicode.Scalar.init` returns nil
/// for surrogates / out-of-range values, which we also map to U+FFFD so the
/// decoder is total.
func utf8Decode(_ bytes: [UInt8]) -> String {
var out = ""
var i = 0
// Reads past the end return 0 — the canonical decoder's behavior.
func nextByte() -> UInt8 {
if i >= bytes.count { return 0 }
defer { i += 1 }
return bytes[i]
}
while i < bytes.count {
let b = bytes[i]
i += 1
let cp: UInt32
if b <= 0x7F {
cp = UInt32(b)
} else if b >> 5 == 0b110 {
let b1 = UInt32(nextByte())
cp = (UInt32(b) & 0x1F) << 6 | (b1 & 0x3F)
} else if b >> 4 == 0b1110 {
let b1 = UInt32(nextByte())
let b2 = UInt32(nextByte())
cp = (UInt32(b) & 0x0F) << 12 | (b1 & 0x3F) << 6 | (b2 & 0x3F)
} else if b >> 3 == 0b11110 {
let b1 = UInt32(nextByte())
let b2 = UInt32(nextByte())
let b3 = UInt32(nextByte())
cp = (UInt32(b) & 0x07) << 18 | (b1 & 0x3F) << 12 | (b2 & 0x3F) << 6 | (b3 & 0x3F)
} else {
cp = 0xFFFD
}
if let scalar = Unicode.Scalar(cp) {
out.unicodeScalars.append(scalar)
} else {
out += replacementChar
}
}
return out
}
/// Render text as a hex string.
///
/// `delimiter` controls how per-byte hex pairs are joined:
/// - .none -> "48656c6c6f"
/// - .space -> "48 65 6c 6c 6f"
/// - .prefix0x -> "0x48 0x65 ..."
/// - .backslashX -> "\x48\x65..." (no separators, C-style)
func textToHex(_ text: String, delimiter: Delimiter = .none, uppercase: Bool = false) -> String {
var hexes = utf8Encode(text).map { String(format: "%02x", $0) }
if uppercase {
hexes = hexes.map { $0.uppercased() }
}
switch delimiter {
case .none: return hexes.joined()
case .space: return hexes.joined(separator: " ")
case .prefix0x: return hexes.map { "0x\($0)" }.joined(separator: " ")
case .backslashX: return hexes.map { "\\x\($0)" }.joined()
}
}
/// Strip common affixes users paste alongside hex — `0x` and `\x` literals,
/// whitespace, commas, and colons (MAC-style "aa:bb:cc") — then lowercase.
/// Safe on empty input. `Character.isWhitespace` is Unicode-aware, matching
/// the canonical TS `\s`.
func sanitizeHex(_ input: String?) -> String {
guard let input else { return "" }
let noMarkers = input
.replacingOccurrences(of: "0x", with: "", options: .caseInsensitive)
.replacingOccurrences(of: "\\x", with: "", options: .caseInsensitive)
let cleaned = noMarkers.filter { !($0.isWhitespace || $0 == "," || $0 == ":") }
return cleaned.lowercased()
}
/// Decode a (possibly decorated) hex string back to text. Invalid characters
/// and odd lengths are reported via `error`; valid input containing malformed
/// UTF-8 still decodes with U+FFFD substitution.
func hexToText(_ hex: String?) -> DecodeResult {
let cleaned = sanitizeHex(hex)
if cleaned.isEmpty { return .okText("") }
// After sanitizing + lowercasing, every char must be in [0-9a-f].
let hexDigits = Set("0123456789abcdef")
if !cleaned.allSatisfy({ hexDigits.contains($0) }) {
return .fail("Hex strings may only contain 0-9 and a-f.")
}
if cleaned.count % 2 != 0 {
return .fail("Hex must have an even number of digits.")
}
let chars = Array(cleaned)
var bytes: [UInt8] = []
bytes.reserveCapacity(chars.count / 2)
for idx in stride(from: 0, to: chars.count, by: 2) {
let pair = String(chars[idx]) + String(chars[idx + 1])
// Pre-validated above, so radix parsing cannot fail here.
bytes.append(UInt8(pair, radix: 16)!)
}
return .okText(utf8Decode(bytes))
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →