ER/Schema Visualizer — TypeScript source
Paste CREATE TABLE DDL and get an ER diagram as SVG: tables with typed columns, primary keys, and foreign-key arrows in a deterministic layered layout. Pan and zoom the live diagram; export the SVG.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Pure CREATE TABLE DDL → schema parser / layered layout / SVG renderer - no
// React, no DOM, deterministic. The unit-test surface for the ER/Schema
// Visualizer tool (FEAT-072). Parsing is a tolerant common subset (Postgres /
// MySQL / SQLite): unparseable statements degrade to notes[] entries, never
// an error — tables.length === 0 is a valid empty state. Layout emits INTEGER
// coordinates only (TS/Go byte-parity; no float formatting anywhere). The
// shared V1..V14 vectors in schema-visualizer.test.ts and the Go twin
// (cli/schema-visualizer) hold both implementations to one contract.
export interface ColumnDef { name: string; type: string; nullable: boolean; isPrimaryKey: boolean }
export interface ForeignKey { fromTable: string; fromColumn: string; toTable: string; toColumn: string }
export interface TableDef { name: string; columns: ColumnDef[] }
export interface ParsedSchema { tables: TableDef[]; foreignKeys: ForeignKey[]; notes: string[] }
export interface Box { x: number; y: number; w: number; h: number }
export interface Geometry {
width: number; height: number;
tables: Array<{ table: TableDef; box: Box; titleBar: Box; columnRows: Box[] }>;
edges: Array<{ fk: ForeignKey; path: string; label: string }>;
}
export interface LayoutOptions {
rowHeight?: number; charWidth?: number; padding?: number; layerGap?: number; columnGap?: number;
}
export interface RenderOptions { title?: string; viewBox?: string }
const DEFAULT_LAYOUT = { rowHeight: 24, charWidth: 7, padding: 8, layerGap: 60, columnGap: 40 };
// ---------------------------------------------------------------------------
// Tokenizer
// ---------------------------------------------------------------------------
type TokenKind = 'word' | 'qident' | 'string' | 'punct';
interface Token { text: string; kind: TokenKind }
/** Tokenize a DDL fragment: quoted identifiers ("x", `x`, [x]) and strings
* ('x', '' escape) become single tokens (qident/string, quotes stripped);
* ( ) , . are punct; everything else is a word. */
function tokenize(s: string): Token[] {
const tokens: Token[] = [];
let i = 0;
while (i < s.length) {
const ch = s[i];
if (/\s/.test(ch)) { i++; continue; }
if (ch === "'" || ch === '"' || ch === '`' || ch === '[') {
const close = ch === '[' ? ']' : ch;
let text = '';
i++;
while (i < s.length) {
if (s[i] === close) {
// '' inside a single-quoted string is an escaped quote.
if (close === "'" && s[i + 1] === "'") { text += "'"; i += 2; continue; }
break;
}
text += s[i++];
}
i++; // consume the closer (or run off the end - tolerant)
tokens.push({ text, kind: ch === "'" ? 'string' : 'qident' });
continue;
}
if (ch === '(' || ch === ')' || ch === ',' || ch === '.') {
tokens.push({ text: ch, kind: 'punct' });
i++;
continue;
}
let word = '';
while (i < s.length && !/[\s"',().`\[\]]/.test(s[i])) word += s[i++];
tokens.push({ text: word, kind: 'word' });
}
return tokens;
}
const isPunct = (t: Token | undefined, p: string) => !!t && t.kind === 'punct' && t.text === p;
const kw = (t: Token | undefined, word: string) =>
!!t && t.kind === 'word' && t.text.toUpperCase() === word;
/** Split DDL text into statements on `;` outside strings and quoted
* identifiers. Depth is deliberately NOT tracked: an unterminated paren is
* the common breakage, and letting its `;` still split keeps one broken
* statement from swallowing the good ones after it. Literal `;` inside
* parens never appears in valid CREATE TABLE DDL. */
function splitStatements(ddl: string): string[] {
const out: string[] = [];
let current = '';
let i = 0;
while (i < ddl.length) {
const ch = ddl[i];
if (ch === "'" || ch === '"' || ch === '`' || ch === '[') {
const close = ch === '[' ? ']' : ch;
current += ch;
i++;
while (i < ddl.length) {
current += ddl[i];
if (ddl[i] === close) {
if (close === "'" && ddl[i + 1] === "'") { current += ddl[i + 1]; i += 2; continue; }
break;
}
i++;
}
i++;
continue;
}
if (ch === ';') { out.push(current); current = ''; i++; continue; }
current += ch;
i++;
}
if (current.trim()) out.push(current);
return out;
}
// ---------------------------------------------------------------------------
// Statement parsing
// ---------------------------------------------------------------------------
const MODIFIERS = new Set([
'NOT', 'NULL', 'PRIMARY', 'KEY', 'UNIQUE', 'DEFAULT', 'REFERENCES',
'AUTO_INCREMENT', 'AUTOINCREMENT', 'ON', 'COMMENT', 'CHECK', 'CONSTRAINT',
]);
const isModifier = (t: Token) => t.kind === 'word' && MODIFIERS.has(t.text.toUpperCase());
/** Consume an identifier (quoted or bare, dotted) from tokens at index i.
* Returns [name, nextIndex] or null when no identifier is present. */
function takeName(tokens: Token[], i: number): [string, number] | null {
const first = tokens[i];
if (!first || (first.kind !== 'qident' && first.kind !== 'word')) return null;
let name = first.text;
let j = i + 1;
while (isPunct(tokens[j], '.') && tokens[j + 1] && (tokens[j + 1].kind === 'qident' || tokens[j + 1].kind === 'word')) {
name += '.' + tokens[j + 1].text;
j += 2;
}
return [name, j];
}
/** Collect the comma-separated identifiers inside a paren group starting at
* tokens[i] === '('. Returns [names, nextIndex] or null when malformed. */
function takeParenList(tokens: Token[], i: number): [string[], number] | null {
if (!isPunct(tokens[i], '(')) return null;
const names: string[] = [];
let j = i + 1;
for (;;) {
const name = takeName(tokens, j);
if (!name) return null;
names.push(name[0]);
j = name[1];
if (isPunct(tokens[j], ',')) { j++; continue; }
if (isPunct(tokens[j], ')')) return [names, j + 1];
return null;
}
}
/** Rejoin type tokens: words + nested paren args, uppercased, spaces removed
* adjacent to parens/commas (VARCHAR ( 100 ) -> VARCHAR(100)). */
function joinType(toks: Token[]): string {
const raw = toks.map((t) => t.text).join(' ');
return raw
.replace(/\s*\(\s*/g, '(')
.replace(/\s*\)\s*/g, ')')
.replace(/\s*,\s*/g, ',')
.trim()
.toUpperCase();
}
interface PendingFk { fromTable: string; fromColumn: string; toTable: string; toColumn: string | null }
/** Parse one CREATE TABLE statement's body into columns + pending FKs.
* Throws on structural breakage (caller degrades to a note). */
function parseTableBody(
tableName: string,
bodyTokens: Token[],
columns: ColumnDef[],
fks: PendingFk[],
notes: string[],
): void {
// Split into top-level comma-separated lines.
const lines: Token[][] = [];
let line: Token[] = [];
let depth = 0;
for (const t of bodyTokens) {
if (isPunct(t, '(')) depth++;
if (isPunct(t, ')')) depth--;
if (isPunct(t, ',') && depth === 0) { lines.push(line); line = []; continue; }
line.push(t);
}
if (line.length > 0) lines.push(line);
for (const toks of lines) {
if (toks.length === 0) continue;
const first = toks[0];
const U = first.kind === 'word' ? first.text.toUpperCase() : '';
if (U === 'PRIMARY' && kw(toks[1], 'KEY')) {
const list = takeParenList(toks, 2);
if (list) {
for (const name of list[0]) {
const col = columns.find((c) => c.name === name);
if (col) { col.isPrimaryKey = true; col.nullable = false; }
}
}
continue;
}
if (U === 'FOREIGN' && kw(toks[1], 'KEY')) { parseForeignKeyLine(tableName, toks, 2, fks); continue; }
if (U === 'CONSTRAINT') {
// CONSTRAINT <name> <constraint kind> ... - find the kind (its index
// varies: the name may be quoted or absent) and re-dispatch on it.
const fkIdx = toks.findIndex((t, k) => k >= 1 && kw(t, 'FOREIGN') && kw(toks[k + 1], 'KEY'));
if (fkIdx >= 0) { parseForeignKeyLine(tableName, toks, fkIdx + 2, fks); continue; }
const pkIdx = toks.findIndex((t, k) => k >= 1 && kw(t, 'PRIMARY') && kw(toks[k + 1], 'KEY'));
if (pkIdx >= 0) {
const list = takeParenList(toks, pkIdx + 2);
if (list) {
for (const name of list[0]) {
const col = columns.find((c) => c.name === name);
if (col) { col.isPrimaryKey = true; col.nullable = false; }
}
}
}
// UNIQUE / CHECK / EXCLUDE constraint bodies: skipped silently.
continue;
}
if (U === 'UNIQUE' || U === 'KEY' || U === 'INDEX' || U === 'CHECK' || U === 'EXCLUDE' ||
U === 'FULLTEXT' || U === 'SPATIAL') {
continue; // table-level options - skipped
}
parseColumnLine(tableName, toks, columns, fks, notes);
}
}
/** FOREIGN KEY (a[,b]) REFERENCES t [(c[,d])] starting at index i (== the
* token after KEY). Pairs columns positionally; missing target columns
* resolve in a post-pass. */
function parseForeignKeyLine(tableName: string, toks: Token[], i: number, fks: PendingFk[]): void {
const from = takeParenList(toks, i);
if (!from) return;
let j = from[1];
if (!kw(toks[j], 'REFERENCES')) return;
j++;
const target = takeName(toks, j);
if (!target) return;
j = target[1];
let toCols: string[] | null = null;
if (isPunct(toks[j], '(')) {
const to = takeParenList(toks, j);
if (to) { toCols = to[0]; j = to[1]; }
}
from[0].forEach((fromCol, idx) => {
fks.push({
fromTable: tableName,
fromColumn: fromCol,
toTable: target[0],
toColumn: toCols ? (toCols[idx] ?? toCols[toCols.length - 1]) : null,
});
});
}
/** <name> <type tokens…> [modifiers…] - unknown modifiers ignored. */
function parseColumnLine(
tableName: string,
toks: Token[],
columns: ColumnDef[],
fks: PendingFk[],
notes: string[],
): void {
const name = takeName(toks, 0);
if (!name) {
notes.push(`Skipped a column line with no name in "${tableName}".`);
return;
}
let i = name[1];
// Type: consume until a modifier keyword (or end). Nested paren args ride along.
const typeToks: Token[] = [];
while (i < toks.length && !isModifier(toks[i])) typeToks.push(toks[i++]);
let nullable = true;
let isPrimaryKey = false;
while (i < toks.length) {
const t = toks[i];
if (kw(t, 'NOT') && kw(toks[i + 1], 'NULL')) { nullable = false; i += 2; continue; }
if (kw(t, 'NULL')) { nullable = true; i++; continue; }
if (kw(t, 'PRIMARY') && kw(toks[i + 1], 'KEY')) { isPrimaryKey = true; nullable = false; i += 2; continue; }
if (kw(t, 'UNIQUE')) { i++; continue; }
if (kw(t, 'AUTO_INCREMENT') || kw(t, 'AUTOINCREMENT')) { i++; continue; }
if (kw(t, 'DEFAULT')) {
i++;
if (isPunct(toks[i], '(')) { // skip ( expr ) by depth
let depth = 0;
do { if (isPunct(toks[i], '(')) depth++; if (isPunct(toks[i], ')')) depth--; i++; } while (i < toks.length && depth > 0);
} else if (toks[i]) { i++; }
continue;
}
if (kw(t, 'COMMENT')) { i++; if (toks[i]?.kind === 'string') i++; continue; }
if (kw(t, 'ON')) {
// ON DELETE|UPDATE <action>: CASCADE | RESTRICT | SET NULL/DEFAULT | NO ACTION
i += 2; // ON + DELETE/UPDATE
if (kw(toks[i], 'SET') || kw(toks[i], 'NO')) i += 2;
else if (toks[i]) i++;
continue;
}
if (kw(t, 'REFERENCES')) {
i++;
const target = takeName(toks, i);
if (target) {
i = target[1];
let toCol: string | null = null;
if (isPunct(toks[i], '(')) {
const list = takeParenList(toks, i);
if (list) { toCol = list[0][0]; i = list[1]; }
}
fks.push({ fromTable: tableName, fromColumn: name[0], toTable: target[0], toColumn: toCol });
}
continue;
}
i++; // unknown modifier token - tolerated
}
columns.push({ name: name[0], type: joinType(typeToks), nullable, isPrimaryKey });
}
function notePrefix(stmt: string): string {
const trimmed = stmt.trim().replace(/\s+/g, ' ');
return trimmed.length > 40 ? trimmed.slice(0, 40) + '…' : trimmed;
}
export function parseDdl(ddl: string): ParsedSchema {
const tables: TableDef[] = [];
const pendingFks: PendingFk[] = [];
const notes: string[] = [];
if (!ddl.trim()) return { tables: [], foreignKeys: [], notes: ['No DDL input.'] };
for (const stmt of splitStatements(ddl)) {
if (!stmt.trim()) continue;
try {
const toks = tokenize(stmt);
let i = 0;
if (!kw(toks[i], 'CREATE')) throw new Error('not a CREATE statement');
i++;
while (kw(toks[i], 'TEMP') || kw(toks[i], 'TEMPORARY') || kw(toks[i], 'UNLOGGED')) i++;
if (!kw(toks[i], 'TABLE')) {
notes.push(`Skipped non-table statement starting "${notePrefix(stmt)}".`);
continue;
}
i++;
if (kw(toks[i], 'IF') && kw(toks[i + 1], 'NOT') && kw(toks[i + 2], 'EXISTS')) i += 3;
const name = takeName(toks, i);
if (!name) throw new Error('missing table name');
i = name[1];
if (!isPunct(toks[i], '(')) throw new Error('missing column list');
// Body = everything inside the OUTER parens (depth-aware slice).
const body: Token[] = [];
let depth = 0;
i++;
for (; i < toks.length; i++) {
if (isPunct(toks[i], '(')) depth++;
if (isPunct(toks[i], ')')) {
if (depth === 0) break;
depth--;
}
body.push(toks[i]);
}
if (i >= toks.length) throw new Error('unterminated column list');
const table: TableDef = { name: name[0], columns: [] };
tables.push(table);
parseTableBody(name[0], body, table.columns, pendingFks, notes);
} catch {
notes.push(`Skipped unparseable statement starting "${notePrefix(stmt)}".`);
}
}
// Post-pass: resolve omitted FK target columns to the referenced table's
// first primary key (or "id" when unknown/unmarked).
const foreignKeys: ForeignKey[] = pendingFks.map((fk) => {
if (fk.toColumn) return fk as ForeignKey;
const target = tables.find((t) => t.name === fk.toTable);
const pk = target?.columns.find((c) => c.isPrimaryKey);
return { ...fk, toColumn: pk?.name ?? 'id' };
});
return { tables, foreignKeys, notes };
}
// ---------------------------------------------------------------------------
// Layout (deterministic, INTEGER geometry only — TS/Go byte-parity)
// ---------------------------------------------------------------------------
/** Layered layout: referenced tables above referencing ones. Layer numbers
* come from longest-path relaxation over the FK graph, capped at
* |tables| passes (cycle fallback — every table still gets a layer). FKs to
* tables absent from the DDL and self-FKs do not drive layering. */
export function layoutSchema(schema: ParsedSchema, opts?: LayoutOptions): Geometry {
const o = { ...DEFAULT_LAYOUT, ...opts };
const tables = schema.tables;
if (tables.length === 0) return { width: 0, height: 0, tables: [], edges: [] };
// First-occurrence index by table name (duplicate names share one node).
const index = new Map<string, number>();
tables.forEach((t, i) => { if (!index.has(t.name)) index.set(t.name, i); });
// Box sizes: width from the longest rendered line, height from rows.
const boxes: Box[] = tables.map((t) => {
const textLen = Math.max(
t.name.length,
...t.columns.map((c) => `${c.name} ${c.type}`.length),
1,
);
const w = Math.round(textLen * o.charWidth + 2 * o.padding);
const h = Math.round(o.rowHeight * (1 + t.columns.length) + o.padding);
return { x: 0, y: 0, w, h };
});
// Longest-path layering with a pass cap (cycle fallback).
const layerOf = tables.map(() => 0);
for (let pass = 0; pass < tables.length; pass++) {
let changed = false;
for (const fk of schema.foreignKeys) {
const ti = index.get(fk.fromTable);
const tj = index.get(fk.toTable);
if (ti === undefined || tj === undefined || ti === tj) continue;
if (layerOf[ti] < layerOf[tj] + 1) { layerOf[ti] = layerOf[tj] + 1; changed = true; }
}
if (!changed) break;
}
// Stack layers top-to-bottom; tables within a layer left-to-right in
// first-seen (parse) order.
const layerCount = Math.max(...layerOf) + 1;
const layers: number[][] = Array.from({ length: layerCount }, () => []);
tables.forEach((_, i) => layers[layerOf[i]].push(i));
let yCursor = 0;
let width = 0;
let height = 0;
for (const layer of layers) {
let xCursor = 0;
let layerH = 0;
for (const i of layer) {
boxes[i].x = xCursor;
boxes[i].y = yCursor;
xCursor += boxes[i].w + o.columnGap;
layerH = Math.max(layerH, boxes[i].h);
}
width = Math.max(width, xCursor - o.columnGap);
height = Math.max(height, yCursor + layerH);
yCursor += layerH + o.layerGap;
}
const laidOut: Geometry['tables'] = tables.map((table, i) => {
const box = boxes[i];
const titleBar: Box = { x: box.x, y: box.y, w: box.w, h: Math.round(o.rowHeight) };
const columnRows: Box[] = table.columns.map((_, ci) => ({
x: box.x,
y: box.y + Math.round(o.rowHeight * (1 + ci)),
w: box.w,
h: Math.round(o.rowHeight),
}));
return { table, box, titleBar, columnRows };
});
// Orthogonal FK elbows: referenced bottom-center -> referencing top-center.
const boxOf = (name: string): Box | undefined => {
const i = index.get(name);
return i === undefined ? undefined : boxes[i];
};
const edges: Geometry['edges'] = [];
for (const fk of schema.foreignKeys) {
const from = boxOf(fk.fromTable);
const to = boxOf(fk.toTable);
if (!from || !to) continue;
const x1 = to.x + Math.round(to.w / 2);
const y1 = to.y + to.h;
const x2 = from.x + Math.round(from.w / 2);
const y2 = from.y;
const midY = Math.round((y1 + y2) / 2);
edges.push({
fk,
path: `M ${x1} ${y1} V ${midY} H ${x2} V ${y2}`,
label: `${fk.fromColumn} → ${fk.toColumn}`,
});
}
return { width: Math.round(width), height: Math.round(height), tables: laidOut, edges };
}
// ---------------------------------------------------------------------------
// SVG renderer (pure string builder — no DOM, theme-agnostic classes)
// ---------------------------------------------------------------------------
const XML_ESC: Record<string, string> = { '&': '&', '<': '<', '>': '>', '"': '"' };
/** Escape the five XML text characters. The ONLY escaping the renderer does —
* every DDL-derived string passes through here (V13 contract). */
export function escapeXmlText(s: string): string {
return s.replace(/[&<>"]/g, (c) => XML_ESC[c]);
}
/** Render a layout as an SVG string. Edges sit under the table groups; each
* FK edge carries a path plus a small arrowhead polygon at the referencing
* table's top-center. Styling is class-based (sv-root, sv-table, sv-title,
* sv-pk, sv-edge, sv-arrow) — the page/island owns colors. */
export function renderSvg(geo: Geometry, opts?: RenderOptions): string {
const title = escapeXmlText(opts?.title ?? 'Schema diagram');
const viewBox = opts?.viewBox ?? `0 0 ${geo.width} ${geo.height}`;
const out: string[] = [];
out.push(`<svg xmlns="http://www.w3.org/2000/svg" viewBox="${viewBox}" class="sv-root" role="img">`);
out.push(`<title>${title}</title>`);
// Edges first (visually under the tables). The arrowhead lands at the
// referencing table's top-center — recomputed from its box, same formula
// the layout used.
const boxByName = new Map<string, Box>();
for (const t of geo.tables) if (!boxByName.has(t.table.name)) boxByName.set(t.table.name, t.box);
for (const e of geo.edges) {
const fromBox = boxByName.get(e.fk.fromTable);
if (!fromBox) continue;
const ax = fromBox.x + Math.round(fromBox.w / 2);
const ay = fromBox.y;
out.push(`<path class="sv-edge" d="${e.path}"/>`);
out.push(
`<polygon class="sv-arrow" points="${ax - 5},${ay - 8} ${ax + 5},${ay - 8} ${ax},${ay}"/>`,
);
}
for (const t of geo.tables) {
const { table, box, titleBar, columnRows } = t;
out.push('<g class="sv-table">');
out.push(`<rect class="sv-box" x="${box.x}" y="${box.y}" width="${box.w}" height="${box.h}" rx="6"/>`);
out.push(`<rect class="sv-titlebar" x="${titleBar.x}" y="${titleBar.y}" width="${titleBar.w}" height="${titleBar.h}" rx="6"/>`);
const titleY = titleBar.y + Math.round(titleBar.h * 0.7);
out.push(`<text class="sv-title" x="${titleBar.x + 8}" y="${titleY}">${escapeXmlText(table.name)}</text>`);
table.columns.forEach((col, ci) => {
const row = columnRows[ci];
const textY = row.y + Math.round(row.h * 0.7);
out.push(
`<text class="${col.isPrimaryKey ? 'sv-pk' : 'sv-col'}" x="${row.x + 8}" y="${textY}">` +
`${escapeXmlText(col.name)} ${escapeXmlText(col.type)}</text>`,
);
});
out.push('</g>');
}
out.push('</svg>');
return out.join('');
}
/** Convenience: parse + layout + render in one call (the island's initial
* state uses this at BUILD time — that prerendered markup is the zero-JS
* first paint). */
export function ddlToSvg(
ddl: string,
opts?: LayoutOptions & RenderOptions,
): { svg: string; schema: ParsedSchema } {
const schema = parseDdl(ddl);
return { svg: renderSvg(layoutSchema(schema, opts), opts), schema };
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →