Skip to content

Roman Numeral Converter — C source

Convert integers up to 3,999,999 to Roman numerals and back. Vinculum overline above 3,999, canonical-form validation, a step-by-step greedy breakdown, and 14 language sources. Runs entirely in your browser.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/* roman-numeral-converter - Roman <-> Arabic (vinculum, 1..3,999,999).

   Language: C (C99, standard library only)
   Source:   CosmoDev polyglot showcase port of the Roman Numeral Converter tool,
             ported from src/lib/roman-numeral.ts (the canonical TypeScript
             implementation); kept in lock-step with the Go twin at
             cli/roman-numeral-converter/roman-numeral-converter.go.
   License:  display source - part of CosmoDev's polyglot tool pages.

   Design goals:
     - Pure + deterministic; never crashes on bad input (returns "" / false).
     - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
     - Self-contained: stdlib only.

   Algorithm: one ordered (value, symbol) table for 1..3,999 drives both
   directions. to_roman greedily subtracts the largest fitting symbol; above
   3,999 the thousands part is rendered with the same table and each glyph
   gains a combining overline (U+0305 in UTF-8: 0xCC 0x85) meaning x 1,000.
   from_roman scans left-to-right where a smaller letter before a larger one
   subtracts (IV = 4, CM = 900), then RE-RENDERS the parsed total and rejects
   anything that doesn't round-trip - that one check enforces canonical form
   (rejecting "IIII", "VV", "IC", plain "MMMM" for 4,000).
*/

#include <assert.h>
#include <ctype.h>
#include <stdbool.h>
#include <string.h>

#define MAX_ROMAN 3999999L

/* U+0305 combining overline and U+0304 macron, as UTF-8 byte pairs. */
static const char MARK[] = { (char)0xCC, (char)0x85 };

static const char *BASE_SYM[13] = {
    "M", "CM", "D", "CD", "C", "XC", "L", "XL", "X", "IX", "V", "IV", "I",
};
static const long BASE_VAL[13] = { 1000, 900, 500, 400, 100, 90, 50, 40, 10, 9, 5, 4, 1 };

/* Face value per ASCII letter; 0 for anything else (lookup by byte). */
static long letter_val(unsigned char c) {
    switch (c) {
        case 'I': return 1;
        case 'V': return 5;
        case 'X': return 10;
        case 'L': return 50;
        case 'C': return 100;
        case 'D': return 500;
        case 'M': return 1000;
        default: return 0;
    }
}

/* Greedy render of 1..3,999 into buf (needs >= 16 bytes). */
static void to_roman_base(long v, char *buf) {
    char *p = buf;
    for (int i = 0; i < 13; i++) {
        while (v >= BASE_VAL[i]) {
            size_t len = strlen(BASE_SYM[i]);
            memcpy(p, BASE_SYM[i], len);
            p += len;
            v -= BASE_VAL[i];
        }
    }
    *p = '\0';
}

/* to_roman converts 1..3,999,999 ("" when out of range). Above 3,999 the
   thousands part carries a combining overline per glyph. buf: >= 64 bytes. */
void to_roman(long n, char *buf) {
    if (n < 1 || n > MAX_ROMAN) { buf[0] = '\0'; return; }
    if (n <= 3999) { to_roman_base(n, buf); return; }
    char base[16];
    char *p = buf;
    to_roman_base(n / 1000, base);
    for (const char *c = base; *c; c++) {
        *p++ = *c;
        *p++ = MARK[0];
        *p++ = MARK[1];
    }
    if (n % 1000) {
        to_roman_base(n % 1000, base);
        strcpy(p, base);
    } else {
        *p = '\0';
    }
}

/* scan_value: one left-to-right pass where a smaller letter before a larger
   one subtracts. Returns junk for non-canonical strings - the round-trip in
   from_roman is the canonicality gate. */
static long scan_value(const char *s) {
    long total = 0;
    for (size_t i = 0; s[i]; i++) {
        long v = letter_val((unsigned char)s[i]);
        long next = letter_val((unsigned char)s[i + 1]); /* 0 at the end */
        total += (next > v) ? -v : v;
    }
    return total;
}

static bool is_mark(const char *p) {
    return (unsigned char)p[0] == 0xCC && (unsigned char)p[1] == 0x85;
}

/* from_roman parses a canonical numeral (plain or vinculum). Trimmed and
   uppercased first; a pasted macron (U+0304) counts as the overline mark. */
bool from_roman(const char *s, long *out) {
    /* Normalize: trim ends, uppercase, macron -> overline. */
    char buf[128];
    size_t len = 0;
    const char *a = s;
    while (*a && isspace((unsigned char)*a)) a++;
    const char *b = a + strlen(a);
    while (b > a && isspace((unsigned char)b[-1])) b--;
    for (const char *p = a; p < b;) {
        if ((unsigned char)p[0] == 0xCC && (unsigned char)p[1] == 0x84) {
            buf[len++] = MARK[0]; buf[len++] = MARK[1]; p += 2;
        } else {
            buf[len++] = (char)toupper((unsigned char)*p++);
        }
    }
    buf[len] = '\0';

    /* Split into overlined glyphs (letter + mark) and plain letters. */
    char over[64], plain[64];
    size_t no = 0, np = 0;
    for (size_t i = 0; i < len;) {
        if (letter_val((unsigned char)buf[i]) == 0) return false;
        if (i + 2 < len && is_mark(buf + i + 1)) {
            over[no++] = buf[i];
            i += 3;
        } else {
            plain[np++] = buf[i];
            i++;
        }
    }
    over[no] = '\0';
    plain[np] = '\0';

    long total = 0;
    if (no) total += scan_value(over) * 1000;
    if (np) total += scan_value(plain);
    if (total < 1 || total > MAX_ROMAN) return false;

    char rt[64];
    to_roman(total, rt);
    if (strcmp(rt, buf) != 0) return false;
    *out = total;
    return true;
}

/* ---------- showcase (run: cc c.c && ./a.out) ---------- */
#include <stdio.h>

int main(void) {
    char out[64];
    /* to_roman - known values, both scales */
    to_roman(1, out);       assert(strcmp(out, "I") == 0);
    to_roman(1994, out);    assert(strcmp(out, "MCMXCIV") == 0);
    to_roman(3999, out);    assert(strcmp(out, "MMMCMXCIX") == 0);
    to_roman(4000, out);    assert(strcmp(out, "I\314\205V\314\205") == 0);
    to_roman(4001, out);    assert(strcmp(out, "I\314\205V\314\205I") == 0);
    to_roman(3999999, out); assert(strcmp(out, "M\314\205M\314\205M\314\205C\314\205M\314\205X\314\205C\314\205I\314\205X\314\205CMXCIX") == 0);
    /* to_roman - out of range */
    to_roman(0, out);       assert(out[0] == '\0');
    to_roman(4000000, out); assert(out[0] == '\0');
    /* from_roman - canonical, with case/whitespace/macron tolerance */
    long v = 0;
    assert(from_roman("MCMXCIV", &v) && v == 1994);
    assert(from_roman("  mcmxciv  ", &v) && v == 1994);
    assert(from_roman("I\314\205V\314\205", &v) && v == 4000);
    assert(from_roman("I\314\204V\314\204", &v) && v == 4000); /* macron */
    /* from_roman - non-canonical / invalid */
    assert(!from_roman("IIII", &v));
    assert(!from_roman("VV", &v));
    assert(!from_roman("IC", &v));
    assert(!from_roman("MMMM", &v)); /* 4,000 must be vinculum */
    assert(!from_roman("ABC", &v));
    assert(!from_roman("", &v));
    puts("all showcase assertions passed");
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →