Skip to content

Hex ↔ Text Converter — JavaScript source

Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.

This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

/**
 * hex-converter - pure hex ↔ text conversion.
 * Language: JavaScript (ES module).
 *
 * CosmoDev polyglot showcase port of the `hex-converter` tool.
 * Ported from src/lib/hexText.ts (the canonical TypeScript implementation).
 *
 * Display source - part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
 * Deterministic, side-effect free, never throws: invalid byte sequences
 * decode to U+FFFD, matching the canonical logic.
 */

/**
 * @typedef {'none' | 'space' | '0x' | 'backslash-x'} Delimiter
 * @typedef {{ ok: boolean, text: string, error: string|null }} DecodeResult
 */

/** U+FFFD - substituted for malformed UTF-8 on decode. */
const REPLACEMENT_CHAR = 0xfffd;

/**
 * Render any Unicode code point as a string, substituting U+FFFD when the
 * value is not a valid Unicode scalar (surrogates or out of range). Keeps the
 * decoder total / non-throwing on malformed input.
 * @param {number} cp
 * @returns {string}
 */
function fromCodePointSafe(cp) {
  const isSurrogate = cp >= 0xd800 && cp <= 0xdfff;
  const inRange = cp >= 0 && cp <= 0x10ffff;
  return inRange && !isSurrogate
    ? String.fromCodePoint(cp)
    : String.fromCodePoint(REPLACEMENT_CHAR);
}

/**
 * UTF-8 encode a JS string into an array of byte values (0..255).
 *
 * Hand-rolled (rather than TextEncoder) so every language in the polyglot
 * showcase produces byte-identical output. `for...of` iterates by Unicode
 * code point, so astral characters (> U+FFFF) encode as 4-byte sequences.
 * @param {string} str
 * @returns {number[]}
 */
export function utf8Encode(str) {
  const bytes = [];
  for (const ch of str) {
    const cp = ch.codePointAt(0);
    if (cp <= 0x7f) {
      bytes.push(cp);
    } else if (cp <= 0x7ff) {
      bytes.push(0xc0 | (cp >> 6), 0x80 | (cp & 0x3f));
    } else if (cp <= 0xffff) {
      bytes.push(
        0xe0 | (cp >> 12),
        0x80 | ((cp >> 6) & 0x3f),
        0x80 | (cp & 0x3f),
      );
    } else {
      bytes.push(
        0xf0 | (cp >> 18),
        0x80 | ((cp >> 12) & 0x3f),
        0x80 | ((cp >> 6) & 0x3f),
        0x80 | (cp & 0x3f),
      );
    }
  }
  return bytes;
}

/**
 * UTF-8 decode a byte array into a JS string. Truncated or invalid sequences
 * yield U+FFFD; missing continuation bytes default to 0, mirroring the
 * canonical decoder's lenient consumption.
 * @param {number[]} bytes
 * @returns {string}
 */
export function utf8Decode(bytes) {
  let out = '';
  let i = 0;
  while (i < bytes.length) {
    const b = bytes[i++];
    let cp;
    if (b <= 0x7f) {
      cp = b;
    } else if (b >> 5 === 0b110) {
      const b1 = bytes[i++] ?? 0;
      cp = ((b & 0x1f) << 6) | (b1 & 0x3f);
    } else if (b >> 4 === 0b1110) {
      const b1 = bytes[i++] ?? 0;
      const b2 = bytes[i++] ?? 0;
      cp = ((b & 0x0f) << 12) | ((b1 & 0x3f) << 6) | (b2 & 0x3f);
    } else if (b >> 3 === 0b11110) {
      const b1 = bytes[i++] ?? 0;
      const b2 = bytes[i++] ?? 0;
      const b3 = bytes[i++] ?? 0;
      cp = ((b & 0x07) << 18) | ((b1 & 0x3f) << 12) | ((b2 & 0x3f) << 6) | (b3 & 0x3f);
    } else {
      cp = REPLACEMENT_CHAR;
    }
    out += fromCodePointSafe(cp);
  }
  return out;
}

/**
 * Render text as a hex string.
 *
 * `delimiter` controls how per-byte hex pairs are joined:
 *   - 'none'         → "48656c6c6f"
 *   - 'space'        → "48 65 6c 6c 6f"
 *   - '0x'           → "0x48 0x65 ..."
 *   - 'backslash-x'  → "\x48\x65..." (no separators, C-style)
 *
 * @param {string} text
 * @param {Delimiter} [delimiter='none']
 * @param {boolean} [uppercase=false]
 * @returns {string}
 */
export function textToHex(text, delimiter = 'none', uppercase = false) {
  let hexes = utf8Encode(text).map((b) => b.toString(16).padStart(2, '0'));
  if (uppercase) hexes = hexes.map((h) => h.toUpperCase());
  switch (delimiter) {
    case 'none':
      return hexes.join('');
    case 'space':
      return hexes.join(' ');
    case '0x':
      return hexes.map((h) => `0x${h}`).join(' ');
    case 'backslash-x':
      return hexes.map((h) => `\\x${h}`).join('');
    default:
      // JS has no exhaustiveness check; unknown delimiters fall back to "none".
      return hexes.join('');
  }
}

/**
 * Normalize a hex string prior to decoding.
 *
 * Strips the common affixes users paste alongside hex - `0x` and `\x` literals,
 * whitespace, commas, colons (MAC-style "aa:bb:cc") - then lowercases.
 * @param {string} input
 * @returns {string}
 */
export function sanitizeHex(input) {
  return (input || '')
    .replace(/0x/gi, '')
    .replace(/\\x/gi, '')
    .replace(/[\s,:]/g, '')
    .toLowerCase();
}

/**
 * Decode a hex string back to text.
 *
 * Returns a result object rather than throwing: invalid characters and odd
 * lengths are reported via `error`, while valid (possibly malformed-UTF-8)
 * input decodes with U+FFFD substitution.
 *
 * @param {string} hex
 * @param {Delimiter} [_delimiter='none']  Currently unused; kept for API parity.
 * @returns {DecodeResult}
 */
export function hexToText(hex, _delimiter = 'none') {
  const cleaned = sanitizeHex(hex);
  if (cleaned.length === 0) {
    return { ok: true, text: '', error: null };
  }
  if (!/^[0-9a-f]+$/.test(cleaned)) {
    return { ok: false, text: '', error: 'Hex strings may only contain 0-9 and a-f.' };
  }
  if (cleaned.length % 2 !== 0) {
    return { ok: false, text: '', error: 'Hex must have an even number of digits.' };
  }
  const bytes = [];
  for (let i = 0; i < cleaned.length; i += 2) {
    bytes.push(parseInt(cleaned.slice(i, i + 2), 16));
  }
  return { ok: true, text: utf8Decode(bytes), error: null };
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →