Skip to content

Roman Numeral Converter — Zig source

Convert integers up to 3,999,999 to Roman numerals and back. Vinculum overline above 3,999, canonical-form validation, a step-by-step greedy breakdown, and 14 language sources. Runs entirely in your browser.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// roman-numeral-converter - Roman <-> Arabic (vinculum, 1..3,999,999).
//
// Language: Zig (0.14+, standard library only)
// Source:   CosmoDev polyglot showcase port of the Roman Numeral Converter tool,
//           ported from src/lib/roman-numeral.ts (the canonical TypeScript
//           implementation); kept in lock-step with the Go twin at
//           cli/roman-numeral-converter/roman-numeral-converter.go.
// License:  display source - part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; never panics on bad input (returns "" / null).
//   - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
//   - Self-contained: std only, caller-provided buffers (no allocator needed).
//
// Algorithm: one ordered (value, symbol) table for 1..3,999 drives both
// directions. toRoman greedily subtracts the largest fitting symbol; above
// 3,999 the thousands part is rendered with the same table and each glyph
// gains a combining overline (U+0305, the 2 UTF-8 bytes 0xCC 0x85) meaning
// x 1,000. fromRoman scans left-to-right where a smaller letter before a
// larger one subtracts (IV = 4, CM = 900), then RE-RENDERS the parsed total
// and rejects anything that doesn't round-trip - that one check enforces
// canonical form (rejecting "IIII", "VV", "IC", plain "MMMM" for 4,000).

const std = @import("std");

pub const max_roman: i64 = 3_999_999;

/// U+0305 combining overline as UTF-8 bytes (vinculum: value x 1,000).
const mark = [2]u8{ 0xCC, 0x85 };

const Entry = struct { val: i64, sym: []const u8 };
const base = [13]Entry{
    .{ .val = 1000, .sym = "M" },   .{ .val = 900, .sym = "CM" },
    .{ .val = 500, .sym = "D" },    .{ .val = 400, .sym = "CD" },
    .{ .val = 100, .sym = "C" },    .{ .val = 90, .sym = "XC" },
    .{ .val = 50, .sym = "L" },     .{ .val = 40, .sym = "XL" },
    .{ .val = 10, .sym = "X" },     .{ .val = 9, .sym = "IX" },
    .{ .val = 5, .sym = "V" },      .{ .val = 4, .sym = "IV" },
    .{ .val = 1, .sym = "I" },
};

fn letterVal(c: u8) i64 {
    return switch (c) {
        'I' => 1,
        'V' => 5,
        'X' => 10,
        'L' => 50,
        'C' => 100,
        'D' => 500,
        'M' => 1000,
        else => 0,
    };
}

fn toUpper(c: u8) u8 {
    return if (c >= 'a' and c <= 'z') c - 32 else c;
}

fn isSpace(c: u8) bool {
    return c == ' ' or c == '\t' or c == '\n' or c == '\r';
}

fn isMark(bytes: []const u8) bool {
    return bytes.len >= 2 and bytes[0] == 0xCC and bytes[1] == 0x85;
}

/// Greedy render of 1..3,999 into the start of `buf` (needs >= 16 bytes).
fn toRomanBase(buf: []u8, v_in: i64) []u8 {
    var n: usize = 0;
    var v = v_in;
    for (base) |e| {
        while (v >= e.val) {
            @memcpy(buf[n .. n + e.sym.len], e.sym);
            n += e.sym.len;
            v -= e.val;
        }
    }
    return buf[0..n];
}

/// Converts 1..3,999,999 ("" when out of range). Above 3,999 the thousands
/// part carries a combining overline per glyph. `out` needs >= 64 bytes.
pub fn toRoman(out: []u8, n: i64) []const u8 {
    if (n < 1 or n > max_roman) return out[0..0];
    if (n <= 3999) return toRomanBase(out, n);

    var tmp: [16]u8 = undefined;
    const thousands = toRomanBase(&tmp, @divTrunc(n, 1000));
    var n2: usize = 0;
    for (thousands) |c| {
        out[n2] = c;
        out[n2 + 1] = mark[0];
        out[n2 + 2] = mark[1];
        n2 += 3;
    }
    const rest = @rem(n, 1000);
    if (rest > 0) {
        const r = toRomanBase(out[n2..], rest);
        n2 += r.len;
    }
    return out[0..n2];
}

/// One left-to-right pass where a smaller letter before a larger one
/// subtracts. Returns junk for non-canonical strings - the round-trip in
/// fromRoman is the canonicality gate.
fn scanValue(s: []const u8) i64 {
    var total: i64 = 0;
    for (s, 0..) |c, i| {
        const v = letterVal(c);
        const next: i64 = if (i + 1 < s.len) letterVal(s[i + 1]) else 0;
        total += if (next > v) -v else v;
    }
    return total;
}

/// Parses a canonical numeral (plain or vinculum), or null. Trimmed and
/// uppercased first; a pasted macron (U+0304) counts as the overline mark.
pub fn fromRoman(s: []const u8) ?i64 {
    // Normalize: trim ends, uppercase, macron -> overline.
    var buf: [128]u8 = undefined;
    var len: usize = 0;
    var a: usize = 0;
    var b: usize = s.len;
    while (a < b and isSpace(s[a])) a += 1;
    while (b > a and isSpace(s[b - 1])) b -= 1;
    var i = a;
    while (i < b) {
        if (i + 1 < b and s[i] == 0xCC and s[i + 1] == 0x84) {
            buf[len] = mark[0];
            buf[len + 1] = mark[1];
            len += 2;
            i += 2;
        } else {
            buf[len] = toUpper(s[i]);
            len += 1;
            i += 1;
        }
    }
    const input = buf[0..len];

    // Split into overlined glyphs (letter + mark) and plain letters.
    var over: [64]u8 = undefined;
    var plain: [64]u8 = undefined;
    var no: usize = 0;
    var np: usize = 0;
    var j: usize = 0;
    while (j < input.len) {
        const c = input[j];
        if (letterVal(c) == 0) return null;
        if (j + 2 < input.len and isMark(input[j + 1 ..])) {
            over[no] = c;
            no += 1;
            j += 3;
        } else {
            plain[np] = c;
            np += 1;
            j += 1;
        }
    }

    var total: i64 = 0;
    if (no > 0) total += scanValue(over[0..no]) * 1000;
    if (np > 0) total += scanValue(plain[0..np]);
    if (total < 1 or total > max_roman) return null;

    var rt: [64]u8 = undefined;
    if (!std.mem.eql(u8, toRoman(&rt, total), input)) return null;
    return total;
}

// ---------- showcase (run: zig run zig.zig) ----------
pub fn main() !void {
    var out: [64]u8 = undefined;
    // toRoman - known values, both scales
    try std.testing.expect(std.mem.eql(u8, toRoman(&out, 1), "I"));
    try std.testing.expect(std.mem.eql(u8, toRoman(&out, 1994), "MCMXCIV"));
    try std.testing.expect(std.mem.eql(u8, toRoman(&out, 3999), "MMMCMXCIX"));
    try std.testing.expect(std.mem.eql(u8, toRoman(&out, 4000), "I\xCC\x85V\xCC\x85"));
    try std.testing.expect(std.mem.eql(u8, toRoman(&out, 4001), "I\xCC\x85V\xCC\x85I"));
    try std.testing.expect(std.mem.eql(u8, toRoman(&out, 3_999_999),
        "M\xCC\x85M\xCC\x85M\xCC\x85C\xCC\x85M\xCC\x85X\xCC\x85C\xCC\x85I\xCC\x85X\xCC\x85CMXCIX"));
    // toRoman - out of range
    try std.testing.expect(toRoman(&out, 0).len == 0);
    try std.testing.expect(toRoman(&out, 4_000_000).len == 0);
    // fromRoman - canonical, with case/whitespace/macron tolerance
    try std.testing.expect(fromRoman("MCMXCIV") == 1994);
    try std.testing.expect(fromRoman("  mcmxciv  ") == 1994);
    try std.testing.expect(fromRoman("I\xCC\x85V\xCC\x85") == 4000);
    try std.testing.expect(fromRoman("I\xCC\x84V\xCC\x84") == 4000); // macron
    // fromRoman - non-canonical / invalid
    try std.testing.expect(fromRoman("IIII") == null);
    try std.testing.expect(fromRoman("VV") == null);
    try std.testing.expect(fromRoman("IC") == null);
    try std.testing.expect(fromRoman("MMMM") == null); // 4,000 must be vinculum
    try std.testing.expect(fromRoman("ABC") == null);
    try std.testing.expect(fromRoman("") == null);
    std.debug.print("all showcase assertions passed\n", .{});
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →