Skip to content

Markdown Table Generator — Python source

Turn pipe, CSV, tab, semicolon, or space-separated data into a clean GitHub-Flavored Markdown table. Auto-detects the delimiter, pads columns, escapes pipes, and supports per-column alignment - all in your browser.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""Markdown Table Generator — pure logic, Python polyglot showcase port.

Language:    Python (3.10+)
Origin:      CosmoDev polyglot showcase port of the ``markdown-table`` tool.
Ported from: src/lib/markdown-table.ts (the canonical, live TypeScript lib).

Parsing and rendering are deterministic and depend only on their inputs.
This file is display source — part of CosmoDev's polyglot tool pages, where
the same pure logic is shown side-by-side across languages.
"""

from __future__ import annotations

from typing import Literal, Sequence

# --- Type aliases mirroring the TypeScript union types -----------------------
Delimiter = Literal["|", ",", "\t", ";", " "]
Align = Literal["left", "center", "right", "none"]

# Delimiter candidates considered during auto-detection, in priority order.
# Structural delimiters (pipe, tab) outrank punctuation (`,` / `;`) outrank space.
_CANDIDATES: tuple[str, ...] = ("|", "\t", ";", ",", " ")
_WEIGHT: dict[str, int] = {"|": 3, "\t": 3, ";": 2, ",": 2, " ": 1}


class ToMarkdownOptions:
    """Options for :func:`to_markdown`.

    header: when ``True``, row 0 is the table header; when ``False``, a blank
            header row is synthesized so the output is still valid GFM.
    align:  per-column alignment; entries beyond the column count are ignored,
            missing entries default to ``"none"``.
    """

    def __init__(self, header: bool, align: Sequence[Align] | None = None) -> None:
        self.header = header
        self.align: list[Align] = list(align) if align is not None else []


def _split_line(line: str, delimiter: Delimiter) -> list[str]:
    """Split a single line by *delimiter*, trimming each resulting cell.

    - ``|`` strips one leading/trailing pipe (so ``| a | b |`` works) then splits.
    - ``<space>`` splits on runs of whitespace.
    - ``,`` ``\\t`` ``;`` split on the literal character.
    """
    if delimiter == "|":
        s = line.strip()
        if s.startswith("|"):
            s = s[1:]
        if s.endswith("|"):
            s = s[:-1]
        # A fully-empty line collapses to a single empty cell, not zero cells.
        if s == "":
            return [""]
        return [c.strip() for c in s.split("|")]
    if delimiter == " ":
        # str.split() with no args trims and splits on runs of whitespace,
        # matching /\s+/. parse_table only passes non-empty lines here.
        return line.split()
    return [c.strip() for c in line.split(delimiter)]


def parse_table(input: str, delimiter: Delimiter) -> list[list[str]]:
    """Parse *input* into a 2-D grid of trimmed cells. Blank lines are skipped."""
    lines = [ln.strip() for ln in input.splitlines()]
    return [_split_line(ln, delimiter) for ln in lines if ln]


def _count_occurrences(line: str, delimiter: Delimiter) -> int:
    """Count delimiter occurrences in *line*.

    Whitespace counts *runs* of whitespace, not individual space characters.
    """
    if delimiter == " ":
        return len(line.split()) - 1
    return line.count(delimiter)


def detect_delimiter(sample: str) -> Delimiter:
    """Heuristic delimiter detection.

    Each candidate is scored by ``frequency x cross-line consistency x structural
    weight`` and the best wins. Falls back to ``","`` when nothing scores
    (single column or empty input).
    """
    lines = [ln.strip() for ln in sample.splitlines() if ln.strip()]
    if not lines:
        return ","

    best: Delimiter = ","
    best_score = 0.0
    for d in _CANDIDATES:
        counts = [float(_count_occurrences(ln, d)) for ln in lines]
        avg = sum(counts) / len(counts)
        if avg == 0.0:
            continue
        # population variance across lines -> lower is more consistent
        variance = sum((c - avg) ** 2 for c in counts) / len(counts)
        consistency = 1.0 / (1.0 + variance)
        score = avg * consistency * _WEIGHT[d]
        if score > best_score:
            best_score = score
            best = d
    return best


def _escape_cell(cell: str) -> str:
    """Collapse newlines (CRLF or LF) to a single space, then escape literal pipes."""
    collapsed = cell.replace("\r\n", " ").replace("\n", " ")
    return collapsed.replace("|", "\\|")


def _pad(cell: str, width: int, align: Align) -> str:
    """Pad *cell* to *width* honoring alignment.

    Center splits the slack with the floor on the left. ``len()`` counts code
    points, which lines multibyte cells up correctly.
    """
    diff = width - len(cell)
    if diff <= 0:
        return cell
    if align == "right":
        return " " * diff + cell
    if align == "center":
        left = diff // 2
        return " " * left + cell + " " * (diff - left)
    return cell + " " * diff  # "left" | "none"


def _sep_cell(align: Align, width: int) -> str:
    """Render a separator cell (``---``, ``:--``, ``--:``, ``:-:``) of >= 3 dashes."""
    w = max(3, width)
    if align == "center":
        return ":" + "-" * (w - 2) + ":"
    if align == "right":
        return "-" * (w - 1) + ":"
    if align == "left":
        return ":" + "-" * (w - 1)
    return "-" * w


def to_markdown(rows: list[list[str]], opts: ToMarkdownOptions) -> str:
    """Render a 2-D grid as a GitHub-Flavored Markdown table.

    Cells are padded to equal column widths (computed from the escaped text),
    literal ``|`` is escaped, and the separator row carries the per-column
    alignment. Returns ``""`` for an empty grid.
    """
    if not rows:
        return ""

    # Column count is the longest row.
    cols = max(len(r) for r in rows)
    # Escape every cell and normalize each row to the column count.
    grid = [[_escape_cell(c) for c in r] + [""] * (cols - len(r)) for r in rows]

    # Per-column alignment: missing entries default to "none"; surplus ignored.
    aligns: list[Align] = [
        opts.align[i] if i < len(opts.align) else "none" for i in range(cols)
    ]

    # Per-column width: at least 3 (GFM separator minimum), grown to fit the
    # widest escaped cell in the column.
    widths: list[int] = []
    for c in range(cols):
        col_max = max((len(row[c]) for row in grid), default=0)
        widths.append(max(3, col_max))

    # Frame one row of cells with the GFM pipe scaffolding.
    def render_line(cells: list[str]) -> str:
        padded = [_pad(cells[i], widths[i], aligns[i]) for i in range(len(cells))]
        return "| " + " | ".join(padded) + " |"

    separator = "| " + " | ".join(
        _sep_cell(aligns[i], widths[i]) for i in range(cols)
    ) + " |"

    # When there is no header, synthesize a blank header row so the table is
    # still valid GFM.
    if opts.header:
        header = render_line(grid[0])
        data_start = 1
    else:
        header = render_line([""] * cols)
        data_start = 0

    out = [header, separator]
    out.extend(render_line(r) for r in grid[data_start:])
    return "\n".join(out)


def transpose(rows: list[list[str]]) -> list[list[str]]:
    """Transpose a grid (rows <-> columns). Jagged grids are filled with ``""``."""
    if not rows:
        return []
    cols = max(len(r) for r in rows)
    return [[(r[c] if c < len(r) else "") for r in rows] for c in range(cols)]


if __name__ == "__main__":
    # Small end-to-end demo so this file is runnable as a showcase.
    sample = "name,role,team\nAda,engineer,platform\nLin,designer,brand"
    d = detect_delimiter(sample)
    grid = parse_table(sample, d)
    md = to_markdown(grid, ToMarkdownOptions(header=True, align=["left", "left", "center"]))
    print(f"Detected delimiter: {d!r}")
    print(md)

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →