Roman Numeral Converter — C source
Convert integers up to 3,999,999 to Roman numerals and back. Vinculum overline above 3,999, canonical-form validation, a step-by-step greedy breakdown, and 14 language sources. Runs entirely in your browser.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/* roman-numeral-converter - Roman <-> Arabic (vinculum, 1..3,999,999).
Language: C (C99, standard library only)
Source: CosmoDev polyglot showcase port of the Roman Numeral Converter tool,
ported from src/lib/roman-numeral.ts (the canonical TypeScript
implementation); kept in lock-step with the Go twin at
cli/roman-numeral-converter/roman-numeral-converter.go.
License: display source - part of CosmoDev's polyglot tool pages.
Design goals:
- Pure + deterministic; never crashes on bad input (returns "" / false).
- Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
- Self-contained: stdlib only.
Algorithm: one ordered (value, symbol) table for 1..3,999 drives both
directions. to_roman greedily subtracts the largest fitting symbol; above
3,999 the thousands part is rendered with the same table and each glyph
gains a combining overline (U+0305 in UTF-8: 0xCC 0x85) meaning x 1,000.
from_roman scans left-to-right where a smaller letter before a larger one
subtracts (IV = 4, CM = 900), then RE-RENDERS the parsed total and rejects
anything that doesn't round-trip - that one check enforces canonical form
(rejecting "IIII", "VV", "IC", plain "MMMM" for 4,000).
*/
#include <assert.h>
#include <ctype.h>
#include <stdbool.h>
#include <string.h>
#define MAX_ROMAN 3999999L
/* U+0305 combining overline and U+0304 macron, as UTF-8 byte pairs. */
static const char MARK[] = { (char)0xCC, (char)0x85 };
static const char *BASE_SYM[13] = {
"M", "CM", "D", "CD", "C", "XC", "L", "XL", "X", "IX", "V", "IV", "I",
};
static const long BASE_VAL[13] = { 1000, 900, 500, 400, 100, 90, 50, 40, 10, 9, 5, 4, 1 };
/* Face value per ASCII letter; 0 for anything else (lookup by byte). */
static long letter_val(unsigned char c) {
switch (c) {
case 'I': return 1;
case 'V': return 5;
case 'X': return 10;
case 'L': return 50;
case 'C': return 100;
case 'D': return 500;
case 'M': return 1000;
default: return 0;
}
}
/* Greedy render of 1..3,999 into buf (needs >= 16 bytes). */
static void to_roman_base(long v, char *buf) {
char *p = buf;
for (int i = 0; i < 13; i++) {
while (v >= BASE_VAL[i]) {
size_t len = strlen(BASE_SYM[i]);
memcpy(p, BASE_SYM[i], len);
p += len;
v -= BASE_VAL[i];
}
}
*p = '\0';
}
/* to_roman converts 1..3,999,999 ("" when out of range). Above 3,999 the
thousands part carries a combining overline per glyph. buf: >= 64 bytes. */
void to_roman(long n, char *buf) {
if (n < 1 || n > MAX_ROMAN) { buf[0] = '\0'; return; }
if (n <= 3999) { to_roman_base(n, buf); return; }
char base[16];
char *p = buf;
to_roman_base(n / 1000, base);
for (const char *c = base; *c; c++) {
*p++ = *c;
*p++ = MARK[0];
*p++ = MARK[1];
}
if (n % 1000) {
to_roman_base(n % 1000, base);
strcpy(p, base);
} else {
*p = '\0';
}
}
/* scan_value: one left-to-right pass where a smaller letter before a larger
one subtracts. Returns junk for non-canonical strings - the round-trip in
from_roman is the canonicality gate. */
static long scan_value(const char *s) {
long total = 0;
for (size_t i = 0; s[i]; i++) {
long v = letter_val((unsigned char)s[i]);
long next = letter_val((unsigned char)s[i + 1]); /* 0 at the end */
total += (next > v) ? -v : v;
}
return total;
}
static bool is_mark(const char *p) {
return (unsigned char)p[0] == 0xCC && (unsigned char)p[1] == 0x85;
}
/* from_roman parses a canonical numeral (plain or vinculum). Trimmed and
uppercased first; a pasted macron (U+0304) counts as the overline mark. */
bool from_roman(const char *s, long *out) {
/* Normalize: trim ends, uppercase, macron -> overline. */
char buf[128];
size_t len = 0;
const char *a = s;
while (*a && isspace((unsigned char)*a)) a++;
const char *b = a + strlen(a);
while (b > a && isspace((unsigned char)b[-1])) b--;
for (const char *p = a; p < b;) {
if ((unsigned char)p[0] == 0xCC && (unsigned char)p[1] == 0x84) {
buf[len++] = MARK[0]; buf[len++] = MARK[1]; p += 2;
} else {
buf[len++] = (char)toupper((unsigned char)*p++);
}
}
buf[len] = '\0';
/* Split into overlined glyphs (letter + mark) and plain letters. */
char over[64], plain[64];
size_t no = 0, np = 0;
for (size_t i = 0; i < len;) {
if (letter_val((unsigned char)buf[i]) == 0) return false;
if (i + 2 < len && is_mark(buf + i + 1)) {
over[no++] = buf[i];
i += 3;
} else {
plain[np++] = buf[i];
i++;
}
}
over[no] = '\0';
plain[np] = '\0';
long total = 0;
if (no) total += scan_value(over) * 1000;
if (np) total += scan_value(plain);
if (total < 1 || total > MAX_ROMAN) return false;
char rt[64];
to_roman(total, rt);
if (strcmp(rt, buf) != 0) return false;
*out = total;
return true;
}
/* ---------- showcase (run: cc c.c && ./a.out) ---------- */
#include <stdio.h>
int main(void) {
char out[64];
/* to_roman - known values, both scales */
to_roman(1, out); assert(strcmp(out, "I") == 0);
to_roman(1994, out); assert(strcmp(out, "MCMXCIV") == 0);
to_roman(3999, out); assert(strcmp(out, "MMMCMXCIX") == 0);
to_roman(4000, out); assert(strcmp(out, "I\314\205V\314\205") == 0);
to_roman(4001, out); assert(strcmp(out, "I\314\205V\314\205I") == 0);
to_roman(3999999, out); assert(strcmp(out, "M\314\205M\314\205M\314\205C\314\205M\314\205X\314\205C\314\205I\314\205X\314\205CMXCIX") == 0);
/* to_roman - out of range */
to_roman(0, out); assert(out[0] == '\0');
to_roman(4000000, out); assert(out[0] == '\0');
/* from_roman - canonical, with case/whitespace/macron tolerance */
long v = 0;
assert(from_roman("MCMXCIV", &v) && v == 1994);
assert(from_roman(" mcmxciv ", &v) && v == 1994);
assert(from_roman("I\314\205V\314\205", &v) && v == 4000);
assert(from_roman("I\314\204V\314\204", &v) && v == 4000); /* macron */
/* from_roman - non-canonical / invalid */
assert(!from_roman("IIII", &v));
assert(!from_roman("VV", &v));
assert(!from_roman("IC", &v));
assert(!from_roman("MMMM", &v)); /* 4,000 must be vinculum */
assert(!from_roman("ABC", &v));
assert(!from_roman("", &v));
puts("all showcase assertions passed");
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →