Skip to content

Hex ↔ Text Converter — Java source

Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.

This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.

// hex-converter — pure hex ↔ text conversion.
//
// Language: Java (17+, standard library only)
// Source:   CosmoDev polyglot showcase port of the hex-converter tool,
//           ported from src/lib/hexText.ts (the canonical TypeScript
//           implementation).
// License:  display source — part of CosmoDev's polyglot tool pages
//           (dev.cosmolabs.org). Deterministic, side-effect free; invalid
//           byte sequences decode to U+FFFD, matching the canonical logic.

import java.util.ArrayList;
import java.util.List;
import java.util.Locale;
import java.util.stream.Collectors;

public final class HexConverter {

    /** How encoded bytes are joined when rendered as a hex string. */
    public enum Delimiter {
        /** No separator: "48656c6c6f". */
        NONE,
        /** Single space between bytes: "48 65 6c 6c 6f". */
        SPACE,
        /** Each byte prefixed with "0x", space-separated. */
        PREFIX_0X,
        /** Each byte prefixed with "\x", no separator (C-style). */
        BACKSLASH_X
    }

    /**
     * Outcome of decoding hex back to text. Mirrors the canonical TS surface:
     * {@code ok}, {@code text}, and {@code error} (null when ok).
     */
    public record DecodeResult(boolean ok, String text, String error) {
        static DecodeResult okText(String text) {
            return new DecodeResult(true, text, null);
        }

        static DecodeResult fail(String message) {
            return new DecodeResult(false, "", message);
        }
    }

    /** U+FFFD, substituted for malformed UTF-8 on decode. */
    private static final String REPLACEMENT_CHAR = "�";

    private HexConverter() {
    }

    /**
     * UTF-8 encode a string into a list of byte values (0..255).
     *
     * Hand-rolled for byte-exact parity across every showcase language.
     * {@link String#codePointAt} iterates by Unicode scalar value (surrogate
     * pairs join), so astral characters encode as 4-byte sequences.
     */
    public static List<Integer> utf8Encode(String text) {
        List<Integer> bytes = new ArrayList<>();
        for (int i = 0; i < text.length(); ) {
            int cp = text.codePointAt(i);
            i += Character.charCount(cp);
            if (cp <= 0x7F) {
                bytes.add(cp);
            } else if (cp <= 0x7FF) {
                bytes.add(0xC0 | (cp >> 6));
                bytes.add(0x80 | (cp & 0x3F));
            } else if (cp <= 0xFFFF) {
                bytes.add(0xE0 | (cp >> 12));
                bytes.add(0x80 | ((cp >> 6) & 0x3F));
                bytes.add(0x80 | (cp & 0x3F));
            } else {
                bytes.add(0xF0 | (cp >> 18));
                bytes.add(0x80 | ((cp >> 12) & 0x3F));
                bytes.add(0x80 | ((cp >> 6) & 0x3F));
                bytes.add(0x80 | (cp & 0x3F));
            }
        }
        return bytes;
    }

    /**
     * UTF-8 decode a list of bytes into a string. Truncated or invalid
     * sequences yield U+FFFD; missing continuation bytes are taken as 0,
     * matching the canonical decoder's lenient consumption. Surrogate /
     * out-of-range code points are also mapped to U+FFFD so the decoder is
     * total.
     */
    public static String utf8Decode(List<Integer> bytes) {
        StringBuilder out = new StringBuilder();
        ByteCursor cursor = new ByteCursor(bytes);
        while (cursor.hasMore()) {
            int b = cursor.read();
            int cp;
            if (b <= 0x7F) {
                cp = b;
            } else if ((b >> 5) == 0b110) {
                int b1 = cursor.read();
                cp = ((b & 0x1F) << 6) | (b1 & 0x3F);
            } else if ((b >> 4) == 0b1110) {
                int b1 = cursor.read();
                int b2 = cursor.read();
                cp = ((b & 0x0F) << 12) | ((b1 & 0x3F) << 6) | (b2 & 0x3F);
            } else if ((b >> 3) == 0b11110) {
                int b1 = cursor.read();
                int b2 = cursor.read();
                int b3 = cursor.read();
                cp = ((b & 0x07) << 18) | ((b1 & 0x3F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F);
            } else {
                cp = 0xFFFD;
            }
            out.append(charFromCodePoint(cp));
        }
        return out.toString();
    }

    /** Cursor over the byte list; reads past the end return 0, matching the
     *  canonical decoder's lenient consumption. */
    private static final class ByteCursor {
        private final List<Integer> bytes;
        private int i;

        ByteCursor(List<Integer> bytes) {
            this.bytes = bytes;
        }

        boolean hasMore() {
            return i < bytes.size();
        }

        int read() {
            if (i >= bytes.size()) return 0;
            return bytes.get(i++);
        }
    }

    /** Render a single code point as a string, substituting U+FFFD for any
     *  value that is not a valid Unicode scalar (surrogates or out of range).
     *  {@link Character#toChars} cannot render such values; substituting keeps
     *  the decoder total, consistent with its "invalid → U+FFFD" contract. */
    private static String charFromCodePoint(int cp) {
        if (cp < 0 || cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)) {
            return REPLACEMENT_CHAR;
        }
        return new String(Character.toChars(cp));
    }

    /**
     * Render text as a hex string.
     *
     * delimiter controls how per-byte hex pairs are joined:
     *   - NONE         -> "48656c6c6f"
     *   - SPACE        -> "48 65 6c 6c 6f"
     *   - PREFIX_0X    -> "0x48 0x65 ..."
     *   - BACKSLASH_X  -> "\x48\x65..." (no separators, C-style)
     */
    public static String textToHex(String text, Delimiter delimiter, boolean uppercase) {
        List<String> hexes = new ArrayList<>();
        for (int b : utf8Encode(text)) {
            hexes.add(String.format("%02x", b));
        }
        if (uppercase) {
            hexes.replaceAll(h -> h.toUpperCase(Locale.ROOT));
        }
        return switch (delimiter) {
            case SPACE -> String.join(" ", hexes);
            case PREFIX_0X -> hexes.stream().map(h -> "0x" + h).collect(Collectors.joining(" "));
            case BACKSLASH_X -> hexes.stream().map(h -> "\\x" + h).collect(Collectors.joining(""));
            case NONE -> String.join("", hexes);
        };
    }

    /**
     * Strip common affixes users paste alongside hex — {@code 0x} and
     * {@code \x} markers (case-insensitive, anywhere), whitespace, commas, and
     * colons (MAC-style "aa:bb:cc") — then lowercase. The {@code (?U)} flag
     * makes {@code \s} Unicode-aware, matching the canonical TS regex.
     */
    public static String sanitizeHex(String input) {
        if (input == null) return "";
        String noMarkers = input.replaceAll("(?i)0x", "").replaceAll("(?i)\\\\x", "");
        return noMarkers.replaceAll("(?U)[\\s,:]", "").toLowerCase(Locale.ROOT);
    }

    /**
     * Decode a (possibly decorated) hex string back to text. Invalid
     * characters and odd lengths are reported via {@code error}; valid input
     * containing malformed UTF-8 still decodes with U+FFFD substitution.
     */
    public static DecodeResult hexToText(String hexStr) {
        String cleaned = sanitizeHex(hexStr);
        if (cleaned.isEmpty()) return DecodeResult.okText("");
        // After sanitizing + lowercasing, every char must be in [0-9a-f].
        if (!cleaned.matches("[0-9a-f]+")) {
            return DecodeResult.fail("Hex strings may only contain 0-9 and a-f.");
        }
        if (cleaned.length() % 2 != 0) {
            return DecodeResult.fail("Hex must have an even number of digits.");
        }
        List<Integer> bytes = new ArrayList<>();
        for (int i = 0; i < cleaned.length(); i += 2) {
            bytes.add(Integer.parseInt(cleaned.substring(i, i + 2), 16));
        }
        return DecodeResult.okText(utf8Decode(bytes));
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →