Skip to content

ASCII Table — Zig source

A searchable, filterable reference for all 128 ASCII code points. See the decimal, hex, octal, and binary form of every character, narrow to printable characters only, clamp the code-point range, and click any row to copy. Runs 100% client-side.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! ASCII Table — pure logic (Zig port): the 128-entry ASCII reference table with
//! search and range filters.
//!
//! Language: Zig 0.13 — standard library only (std).
//! Source: CosmoDev polyglot showcase port; canonical = src/lib/ascii-table.ts
//!         + this tool's python.py / rust.rs.
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Each of the 128 ASCII code points is described by its decimal / hex / octal /
//! binary forms, a display glyph, a human name, whether it is printable, whether
//! it is a control character, and (for control chars that have one) a C-style
//! escape sequence.
//!
//! Zig has no garbage collector, so every derived string is allocated from a
//! caller-supplied allocator. Pass an arena that lives as long as you need the
//! table and free it in one step — the usual pattern for immutable,
//! build-once data.

const std = @import("std");
const Allocator = std.mem.Allocator;

/// A single ASCII code point and its derived representations.
pub const Entry = struct {
    /// Decimal code point, 0–127.
    dec: u8,
    /// Uppercase hex, e.g. "0x41".
    hex: []const u8,
    /// Zero-padded 3-digit octal, e.g. "101".
    oct: []const u8,
    /// Zero-padded 8-bit binary, e.g. "01000001".
    binary: []const u8,
    /// Display glyph: the literal char for printables, a control-picture glyph otherwise.
    char: []const u8,
    /// Human-readable name, e.g. "Line Feed (LF)", "Uppercase A", "Space".
    name: []const u8,
    /// True for 32–126 (visible or space).
    printable: bool,
    /// True for 0–31 and 127.
    control: bool,
    /// C-style escape for control chars that have one (e.g. "\\n"); null otherwise.
    escape: ?[]const u8 = null,
};

/// Official ASCII control-character names for 0–31 and 127.
///
/// Expressed as a `switch` over every control code point; the else arm is the
/// defensive guard that documents the precondition (Rust's port uses
/// `unreachable!()` for the same case).
fn controlName(code: u8) []const u8 {
    return switch (code) {
        0 => "Null (NUL)",
        1 => "Start of Heading (SOH)",
        2 => "Start of Text (STX)",
        3 => "End of Text (ETX)",
        4 => "End of Transmission (EOT)",
        5 => "Enquiry (ENQ)",
        6 => "Acknowledge (ACK)",
        7 => "Bell (BEL)",
        8 => "Backspace (BS)",
        9 => "Horizontal Tab (HT)",
        10 => "Line Feed (LF)",
        11 => "Vertical Tab (VT)",
        12 => "Form Feed (FF)",
        13 => "Carriage Return (CR)",
        14 => "Shift Out (SO)",
        15 => "Shift In (SI)",
        16 => "Data Link Escape (DLE)",
        17 => "Device Control 1 (DC1)",
        18 => "Device Control 2 (DC2)",
        19 => "Device Control 3 (DC3)",
        20 => "Device Control 4 (DC4)",
        21 => "Negative Acknowledge (NAK)",
        22 => "Synchronous Idle (SYN)",
        23 => "End of Transmission Block (ETB)",
        24 => "Cancel (CAN)",
        25 => "End of Medium (EM)",
        26 => "Substitute (SUB)",
        27 => "Escape (ESC)",
        28 => "File Separator (FS)",
        29 => "Group Separator (GS)",
        30 => "Record Separator (RS)",
        31 => "Unit Separator (US)",
        127 => "Delete (DEL)",
        // Only called with control codes; defensive guard for misuse.
        else => "",
    };
}

/// C-style escape for the control chars that own a standard/common one.
/// Returns the literal 2-character source token (backslash + letter), or null.
fn controlEscape(code: u8) ?[]const u8 {
    return switch (code) {
        0 => "\\0",
        7 => "\\a",
        8 => "\\b",
        9 => "\\t",
        10 => "\\n",
        11 => "\\v",
        12 => "\\f",
        13 => "\\r",
        27 => "\\e",
        else => null,
    };
}

/// Name for a printable punctuation / symbol glyph, or null.
/// Letters and digits are derived in `letterName`.
fn symbolName(code: u8) ?[]const u8 {
    return switch (code) {
        32 => "Space",
        33 => "Exclamation mark",
        34 => "Quotation mark",
        35 => "Number sign",
        36 => "Dollar sign",
        37 => "Percent sign",
        38 => "Ampersand",
        39 => "Apostrophe",
        40 => "Left parenthesis",
        41 => "Right parenthesis",
        42 => "Asterisk",
        43 => "Plus sign",
        44 => "Comma",
        45 => "Hyphen / Minus",
        46 => "Full stop",
        47 => "Slash",
        58 => "Colon",
        59 => "Semicolon",
        60 => "Less-than sign",
        61 => "Equals sign",
        62 => "Greater-than sign",
        63 => "Question mark",
        64 => "At sign",
        91 => "Left bracket",
        92 => "Backslash",
        93 => "Right bracket",
        94 => "Circumflex / Caret",
        95 => "Underscore",
        96 => "Grave accent",
        123 => "Left brace",
        124 => "Vertical bar",
        125 => "Right brace",
        126 => "Tilde",
        else => null,
    };
}

/// Derive a name for a printable letter or digit.
fn letterName(allocator: Allocator, code: u8) Allocator.Error![]const u8 {
    return switch (code) {
        48...57 => try std.fmt.allocPrint(allocator, "Digit {d}", .{code - 48}), // '0'..'9'
        65...90 => try std.fmt.allocPrint(allocator, "Uppercase {c}", .{code}), // 'A'..'Z'
        97...122 => try std.fmt.allocPrint(allocator, "Lowercase {c}", .{code}), // 'a'..'z'
        else => try allocator.dupe(u8, &[_]u8{code}),
    };
}

/// Encode a Unicode scalar as UTF-8 into `buf`, returning the written bytes.
/// std.unicode.utf8Encode is failable; every code point used here (all of ASCII
/// plus U+2400..U+2421) is a valid scalar, so we mirror the Rust port's
/// defensive empty-string default rather than propagate the error.
fn utf8Encode(cp: u21, buf: []u8) []const u8 {
    const written = std.unicode.utf8Encode(cp, buf) catch return "";
    return buf[0..written];
}

/// Build a single entry from its code point (0–127).
fn makeEntry(allocator: Allocator, code: u8) Allocator.Error!Entry {
    const control = code <= 31 or code == 127;

    // Control-picture glyph: U+2400 for 0–31, U+2421 ("symbol for delete") for DEL.
    var buf: [4]u8 = undefined;
    const char: []const u8 = if (control)
        try allocator.dupe(u8, utf8Encode(if (code == 127) 0x2421 else 0x2400 + @as(u21, code), &buf))
    else
        try allocator.dupe(u8, &[_]u8{code});

    const name: []const u8 = if (control)
        controlName(code) // string literal — no allocation needed
    else if (symbolName(code)) |sym|
        sym
    else
        try letterName(allocator, code);

    return .{
        .dec = code,
        .hex = try std.fmt.allocPrint(allocator, "0x{X:0>2}", .{code}),
        .oct = try std.fmt.allocPrint(allocator, "{o:0>3}", .{code}),
        .binary = try std.fmt.allocPrint(allocator, "{b:0>8}", .{code}),
        .char = char,
        .name = name,
        .printable = !control,
        .control = control,
        .escape = if (control) controlEscape(code) else null,
    };
}

// The 128-entry table is built once and cached for the process lifetime. Zig
// has no process-global allocator, so the FIRST caller supplies one and its
// arena must outlive every use of the table. The cache itself is a plain
// module-level optional — callers race on the first call, wrap the
// initialization in a std.Thread.Mutex.
var cached_entries: ?[]const Entry = null;

/// The full 128-entry ASCII table, indexed by code point.
pub fn entries(allocator: Allocator) Allocator.Error![]const Entry {
    if (cached_entries) |table| return table;

    var list = std.ArrayList(Entry).init(allocator);
    errdefer list.deinit();

    // u16 counter: a u8 loop would overflow on the final `code += 1` at 127.
    var code: u16 = 0;
    while (code < 128) : (code += 1) {
        try list.append(try makeEntry(allocator, @intCast(code)));
    }
    cached_entries = try list.toOwnedSlice();
    return cached_entries.?;
}

/// Look up a single entry by code point.
///
/// Returns null for out-of-range input. The TypeScript variant returns null;
/// Zig's idiomatic equivalent is an optional pointer. The error union covers
/// the lazy first-call table build.
pub fn getEntry(allocator: Allocator, code: i32) Allocator.Error!?*const Entry {
    if (code < 0 or code > 127) return null;
    const table = try entries(allocator);
    return &table[@intCast(code)];
}

/// Optional parameters for `filterTable`. All fields defaulted.
pub const FilterOpts = struct {
    /// Case-insensitive substring matched against dec / hex / oct / binary / name / char / escape.
    query: []const u8 = "",
    /// Hide control characters.
    printable_only: bool = false,
    /// Lower decimal bound (default 0).
    min: u8 = 0,
    /// Upper decimal bound (default 127).
    max: u8 = 127,
};

/// Case-insensitive ASCII substring test: haystack and needle bytes are
/// lowered one-by-one — equivalent to the reference implementations, which
/// lowercase the haystack before matching.
fn containsIgnoreCase(haystack: []const u8, needle: []const u8) bool {
    if (needle.len == 0) return true;
    if (haystack.len < needle.len) return false;
    var i: usize = 0;
    while (i + needle.len <= haystack.len) : (i += 1) {
        var j: usize = 0;
        while (j < needle.len and
            std.ascii.toLower(haystack[i + j]) == std.ascii.toLower(needle[j])) : (j += 1)
        {}
        if (j == needle.len) return true;
    }
    return false;
}

/// Filter the table.
///
/// `opts.query` is a case-insensitive substring matched against dec / hex /
/// oct / binary / name / char / escape (surrounding whitespace is trimmed).
/// `opts.printable_only` hides controls. `opts.min` / `opts.max` clamp the
/// decimal range (defaults 0–127). Returns pointers into the shared table, so
/// filtering allocates nothing beyond the result slice.
pub fn filterTable(allocator: Allocator, opts: FilterOpts) Allocator.Error![]const *const Entry {
    const q = std.mem.trim(u8, opts.query, " \t\r\n");

    var out = std.ArrayList(*const Entry).init(allocator);
    errdefer out.deinit();

    for (try entries(allocator)) |*e| {
        if (opts.printable_only and !e.printable) continue;
        if (e.dec < opts.min or e.dec > opts.max) continue;
        if (q.len > 0) {
            // "127" is the widest decimal the u8 code point can produce.
            var dec_buf: [3]u8 = undefined;
            const dec = std.fmt.bufPrint(&dec_buf, "{d}", .{e.dec}) catch unreachable;
            const esc = e.escape orelse "";
            const hit = containsIgnoreCase(dec, q) or containsIgnoreCase(e.hex, q) or
                containsIgnoreCase(e.oct, q) or containsIgnoreCase(e.binary, q) or
                containsIgnoreCase(e.name, q) or containsIgnoreCase(e.char, q) or
                containsIgnoreCase(esc, q);
            if (!hit) continue;
        }
        try out.append(e);
    }
    return out.toOwnedSlice();
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →