Markdown Table Generator — Python source
Turn pipe, CSV, tab, semicolon, or space-separated data into a clean GitHub-Flavored Markdown table. Auto-detects the delimiter, pads columns, escapes pipes, and supports per-column alignment - all in your browser.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""Markdown Table Generator — pure logic, Python polyglot showcase port.
Language: Python (3.10+)
Origin: CosmoDev polyglot showcase port of the ``markdown-table`` tool.
Ported from: src/lib/markdown-table.ts (the canonical, live TypeScript lib).
Parsing and rendering are deterministic and depend only on their inputs.
This file is display source — part of CosmoDev's polyglot tool pages, where
the same pure logic is shown side-by-side across languages.
"""
from __future__ import annotations
from typing import Literal, Sequence
# --- Type aliases mirroring the TypeScript union types -----------------------
Delimiter = Literal["|", ",", "\t", ";", " "]
Align = Literal["left", "center", "right", "none"]
# Delimiter candidates considered during auto-detection, in priority order.
# Structural delimiters (pipe, tab) outrank punctuation (`,` / `;`) outrank space.
_CANDIDATES: tuple[str, ...] = ("|", "\t", ";", ",", " ")
_WEIGHT: dict[str, int] = {"|": 3, "\t": 3, ";": 2, ",": 2, " ": 1}
class ToMarkdownOptions:
"""Options for :func:`to_markdown`.
header: when ``True``, row 0 is the table header; when ``False``, a blank
header row is synthesized so the output is still valid GFM.
align: per-column alignment; entries beyond the column count are ignored,
missing entries default to ``"none"``.
"""
def __init__(self, header: bool, align: Sequence[Align] | None = None) -> None:
self.header = header
self.align: list[Align] = list(align) if align is not None else []
def _split_line(line: str, delimiter: Delimiter) -> list[str]:
"""Split a single line by *delimiter*, trimming each resulting cell.
- ``|`` strips one leading/trailing pipe (so ``| a | b |`` works) then splits.
- ``<space>`` splits on runs of whitespace.
- ``,`` ``\\t`` ``;`` split on the literal character.
"""
if delimiter == "|":
s = line.strip()
if s.startswith("|"):
s = s[1:]
if s.endswith("|"):
s = s[:-1]
# A fully-empty line collapses to a single empty cell, not zero cells.
if s == "":
return [""]
return [c.strip() for c in s.split("|")]
if delimiter == " ":
# str.split() with no args trims and splits on runs of whitespace,
# matching /\s+/. parse_table only passes non-empty lines here.
return line.split()
return [c.strip() for c in line.split(delimiter)]
def parse_table(input: str, delimiter: Delimiter) -> list[list[str]]:
"""Parse *input* into a 2-D grid of trimmed cells. Blank lines are skipped."""
lines = [ln.strip() for ln in input.splitlines()]
return [_split_line(ln, delimiter) for ln in lines if ln]
def _count_occurrences(line: str, delimiter: Delimiter) -> int:
"""Count delimiter occurrences in *line*.
Whitespace counts *runs* of whitespace, not individual space characters.
"""
if delimiter == " ":
return len(line.split()) - 1
return line.count(delimiter)
def detect_delimiter(sample: str) -> Delimiter:
"""Heuristic delimiter detection.
Each candidate is scored by ``frequency x cross-line consistency x structural
weight`` and the best wins. Falls back to ``","`` when nothing scores
(single column or empty input).
"""
lines = [ln.strip() for ln in sample.splitlines() if ln.strip()]
if not lines:
return ","
best: Delimiter = ","
best_score = 0.0
for d in _CANDIDATES:
counts = [float(_count_occurrences(ln, d)) for ln in lines]
avg = sum(counts) / len(counts)
if avg == 0.0:
continue
# population variance across lines -> lower is more consistent
variance = sum((c - avg) ** 2 for c in counts) / len(counts)
consistency = 1.0 / (1.0 + variance)
score = avg * consistency * _WEIGHT[d]
if score > best_score:
best_score = score
best = d
return best
def _escape_cell(cell: str) -> str:
"""Collapse newlines (CRLF or LF) to a single space, then escape literal pipes."""
collapsed = cell.replace("\r\n", " ").replace("\n", " ")
return collapsed.replace("|", "\\|")
def _pad(cell: str, width: int, align: Align) -> str:
"""Pad *cell* to *width* honoring alignment.
Center splits the slack with the floor on the left. ``len()`` counts code
points, which lines multibyte cells up correctly.
"""
diff = width - len(cell)
if diff <= 0:
return cell
if align == "right":
return " " * diff + cell
if align == "center":
left = diff // 2
return " " * left + cell + " " * (diff - left)
return cell + " " * diff # "left" | "none"
def _sep_cell(align: Align, width: int) -> str:
"""Render a separator cell (``---``, ``:--``, ``--:``, ``:-:``) of >= 3 dashes."""
w = max(3, width)
if align == "center":
return ":" + "-" * (w - 2) + ":"
if align == "right":
return "-" * (w - 1) + ":"
if align == "left":
return ":" + "-" * (w - 1)
return "-" * w
def to_markdown(rows: list[list[str]], opts: ToMarkdownOptions) -> str:
"""Render a 2-D grid as a GitHub-Flavored Markdown table.
Cells are padded to equal column widths (computed from the escaped text),
literal ``|`` is escaped, and the separator row carries the per-column
alignment. Returns ``""`` for an empty grid.
"""
if not rows:
return ""
# Column count is the longest row.
cols = max(len(r) for r in rows)
# Escape every cell and normalize each row to the column count.
grid = [[_escape_cell(c) for c in r] + [""] * (cols - len(r)) for r in rows]
# Per-column alignment: missing entries default to "none"; surplus ignored.
aligns: list[Align] = [
opts.align[i] if i < len(opts.align) else "none" for i in range(cols)
]
# Per-column width: at least 3 (GFM separator minimum), grown to fit the
# widest escaped cell in the column.
widths: list[int] = []
for c in range(cols):
col_max = max((len(row[c]) for row in grid), default=0)
widths.append(max(3, col_max))
# Frame one row of cells with the GFM pipe scaffolding.
def render_line(cells: list[str]) -> str:
padded = [_pad(cells[i], widths[i], aligns[i]) for i in range(len(cells))]
return "| " + " | ".join(padded) + " |"
separator = "| " + " | ".join(
_sep_cell(aligns[i], widths[i]) for i in range(cols)
) + " |"
# When there is no header, synthesize a blank header row so the table is
# still valid GFM.
if opts.header:
header = render_line(grid[0])
data_start = 1
else:
header = render_line([""] * cols)
data_start = 0
out = [header, separator]
out.extend(render_line(r) for r in grid[data_start:])
return "\n".join(out)
def transpose(rows: list[list[str]]) -> list[list[str]]:
"""Transpose a grid (rows <-> columns). Jagged grids are filled with ``""``."""
if not rows:
return []
cols = max(len(r) for r in rows)
return [[(r[c] if c < len(r) else "") for r in rows] for c in range(cols)]
if __name__ == "__main__":
# Small end-to-end demo so this file is runnable as a showcase.
sample = "name,role,team\nAda,engineer,platform\nLin,designer,brand"
d = detect_delimiter(sample)
grid = parse_table(sample, d)
md = to_markdown(grid, ToMarkdownOptions(header=True, align=["left", "left", "center"]))
print(f"Detected delimiter: {d!r}")
print(md)
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →