Roman Numeral Converter — Zig source
Convert integers up to 3,999,999 to Roman numerals and back. Vinculum overline above 3,999, canonical-form validation, a step-by-step greedy breakdown, and 14 language sources. Runs entirely in your browser.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
// roman-numeral-converter - Roman <-> Arabic (vinculum, 1..3,999,999).
//
// Language: Zig (0.14+, standard library only)
// Source: CosmoDev polyglot showcase port of the Roman Numeral Converter tool,
// ported from src/lib/roman-numeral.ts (the canonical TypeScript
// implementation); kept in lock-step with the Go twin at
// cli/roman-numeral-converter/roman-numeral-converter.go.
// License: display source - part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; never panics on bad input (returns "" / null).
// - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
// - Self-contained: std only, caller-provided buffers (no allocator needed).
//
// Algorithm: one ordered (value, symbol) table for 1..3,999 drives both
// directions. toRoman greedily subtracts the largest fitting symbol; above
// 3,999 the thousands part is rendered with the same table and each glyph
// gains a combining overline (U+0305, the 2 UTF-8 bytes 0xCC 0x85) meaning
// x 1,000. fromRoman scans left-to-right where a smaller letter before a
// larger one subtracts (IV = 4, CM = 900), then RE-RENDERS the parsed total
// and rejects anything that doesn't round-trip - that one check enforces
// canonical form (rejecting "IIII", "VV", "IC", plain "MMMM" for 4,000).
const std = @import("std");
pub const max_roman: i64 = 3_999_999;
/// U+0305 combining overline as UTF-8 bytes (vinculum: value x 1,000).
const mark = [2]u8{ 0xCC, 0x85 };
const Entry = struct { val: i64, sym: []const u8 };
const base = [13]Entry{
.{ .val = 1000, .sym = "M" }, .{ .val = 900, .sym = "CM" },
.{ .val = 500, .sym = "D" }, .{ .val = 400, .sym = "CD" },
.{ .val = 100, .sym = "C" }, .{ .val = 90, .sym = "XC" },
.{ .val = 50, .sym = "L" }, .{ .val = 40, .sym = "XL" },
.{ .val = 10, .sym = "X" }, .{ .val = 9, .sym = "IX" },
.{ .val = 5, .sym = "V" }, .{ .val = 4, .sym = "IV" },
.{ .val = 1, .sym = "I" },
};
fn letterVal(c: u8) i64 {
return switch (c) {
'I' => 1,
'V' => 5,
'X' => 10,
'L' => 50,
'C' => 100,
'D' => 500,
'M' => 1000,
else => 0,
};
}
fn toUpper(c: u8) u8 {
return if (c >= 'a' and c <= 'z') c - 32 else c;
}
fn isSpace(c: u8) bool {
return c == ' ' or c == '\t' or c == '\n' or c == '\r';
}
fn isMark(bytes: []const u8) bool {
return bytes.len >= 2 and bytes[0] == 0xCC and bytes[1] == 0x85;
}
/// Greedy render of 1..3,999 into the start of `buf` (needs >= 16 bytes).
fn toRomanBase(buf: []u8, v_in: i64) []u8 {
var n: usize = 0;
var v = v_in;
for (base) |e| {
while (v >= e.val) {
@memcpy(buf[n .. n + e.sym.len], e.sym);
n += e.sym.len;
v -= e.val;
}
}
return buf[0..n];
}
/// Converts 1..3,999,999 ("" when out of range). Above 3,999 the thousands
/// part carries a combining overline per glyph. `out` needs >= 64 bytes.
pub fn toRoman(out: []u8, n: i64) []const u8 {
if (n < 1 or n > max_roman) return out[0..0];
if (n <= 3999) return toRomanBase(out, n);
var tmp: [16]u8 = undefined;
const thousands = toRomanBase(&tmp, @divTrunc(n, 1000));
var n2: usize = 0;
for (thousands) |c| {
out[n2] = c;
out[n2 + 1] = mark[0];
out[n2 + 2] = mark[1];
n2 += 3;
}
const rest = @rem(n, 1000);
if (rest > 0) {
const r = toRomanBase(out[n2..], rest);
n2 += r.len;
}
return out[0..n2];
}
/// One left-to-right pass where a smaller letter before a larger one
/// subtracts. Returns junk for non-canonical strings - the round-trip in
/// fromRoman is the canonicality gate.
fn scanValue(s: []const u8) i64 {
var total: i64 = 0;
for (s, 0..) |c, i| {
const v = letterVal(c);
const next: i64 = if (i + 1 < s.len) letterVal(s[i + 1]) else 0;
total += if (next > v) -v else v;
}
return total;
}
/// Parses a canonical numeral (plain or vinculum), or null. Trimmed and
/// uppercased first; a pasted macron (U+0304) counts as the overline mark.
pub fn fromRoman(s: []const u8) ?i64 {
// Normalize: trim ends, uppercase, macron -> overline.
var buf: [128]u8 = undefined;
var len: usize = 0;
var a: usize = 0;
var b: usize = s.len;
while (a < b and isSpace(s[a])) a += 1;
while (b > a and isSpace(s[b - 1])) b -= 1;
var i = a;
while (i < b) {
if (i + 1 < b and s[i] == 0xCC and s[i + 1] == 0x84) {
buf[len] = mark[0];
buf[len + 1] = mark[1];
len += 2;
i += 2;
} else {
buf[len] = toUpper(s[i]);
len += 1;
i += 1;
}
}
const input = buf[0..len];
// Split into overlined glyphs (letter + mark) and plain letters.
var over: [64]u8 = undefined;
var plain: [64]u8 = undefined;
var no: usize = 0;
var np: usize = 0;
var j: usize = 0;
while (j < input.len) {
const c = input[j];
if (letterVal(c) == 0) return null;
if (j + 2 < input.len and isMark(input[j + 1 ..])) {
over[no] = c;
no += 1;
j += 3;
} else {
plain[np] = c;
np += 1;
j += 1;
}
}
var total: i64 = 0;
if (no > 0) total += scanValue(over[0..no]) * 1000;
if (np > 0) total += scanValue(plain[0..np]);
if (total < 1 or total > max_roman) return null;
var rt: [64]u8 = undefined;
if (!std.mem.eql(u8, toRoman(&rt, total), input)) return null;
return total;
}
// ---------- showcase (run: zig run zig.zig) ----------
pub fn main() !void {
var out: [64]u8 = undefined;
// toRoman - known values, both scales
try std.testing.expect(std.mem.eql(u8, toRoman(&out, 1), "I"));
try std.testing.expect(std.mem.eql(u8, toRoman(&out, 1994), "MCMXCIV"));
try std.testing.expect(std.mem.eql(u8, toRoman(&out, 3999), "MMMCMXCIX"));
try std.testing.expect(std.mem.eql(u8, toRoman(&out, 4000), "I\xCC\x85V\xCC\x85"));
try std.testing.expect(std.mem.eql(u8, toRoman(&out, 4001), "I\xCC\x85V\xCC\x85I"));
try std.testing.expect(std.mem.eql(u8, toRoman(&out, 3_999_999),
"M\xCC\x85M\xCC\x85M\xCC\x85C\xCC\x85M\xCC\x85X\xCC\x85C\xCC\x85I\xCC\x85X\xCC\x85CMXCIX"));
// toRoman - out of range
try std.testing.expect(toRoman(&out, 0).len == 0);
try std.testing.expect(toRoman(&out, 4_000_000).len == 0);
// fromRoman - canonical, with case/whitespace/macron tolerance
try std.testing.expect(fromRoman("MCMXCIV") == 1994);
try std.testing.expect(fromRoman(" mcmxciv ") == 1994);
try std.testing.expect(fromRoman("I\xCC\x85V\xCC\x85") == 4000);
try std.testing.expect(fromRoman("I\xCC\x84V\xCC\x84") == 4000); // macron
// fromRoman - non-canonical / invalid
try std.testing.expect(fromRoman("IIII") == null);
try std.testing.expect(fromRoman("VV") == null);
try std.testing.expect(fromRoman("IC") == null);
try std.testing.expect(fromRoman("MMMM") == null); // 4,000 must be vinculum
try std.testing.expect(fromRoman("ABC") == null);
try std.testing.expect(fromRoman("") == null);
std.debug.print("all showcase assertions passed\n", .{});
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →