Skip to content

Chat Format Converter — Python source

Convert chat transcripts between OpenAI messages[], Anthropic system+messages, Gemini contents[], and plain Markdown. Roles map faithfully, tool calls are preserved where possible, and anything unmappable is flagged — never dropped silently. Runs entirely in your browser.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""chat-format-converter — Python polyglot showcase port.

Converts one chat transcript between provider shapes through a shared internal
message list:

- openai    — {"messages": [{"role": "system|user|assistant|tool", ...}]}
- anthropic — {"system": "...", "messages": [{"role": "user|assistant", ...}]}
- gemini    — {"contents": [{"role": "user|model", "parts": [...]}]}
- markdown  — a plain ``**role**: text`` transcript

Roles map faithfully (Gemini has no assistant — it is ``model``; Anthropic
system lives top-level). Fields a target format cannot represent (e.g. tool
calls in Markdown) are flagged as warnings, never dropped silently.

This is the Python sibling of src/lib/chatFormatConverter.ts (the canonical
TypeScript that powers the live tool). Standard library only (json + re +
dataclasses); parse failures raise ValueError.

Display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
"""

from __future__ import annotations

import json
import re
from dataclasses import dataclass, field

CHAT_FORMATS = ["openai", "anthropic", "gemini", "markdown"]


@dataclass
class ToolCall:
    """One assistant-emitted tool call, provider-agnostic. ``args`` is a JSON string."""
    id: str = ""
    name: str = ""
    args: str = ""


@dataclass
class InternalMessage:
    """The shared internal shape every format parses into and serializes from."""
    role: str  # 'system' | 'user' | 'assistant' | 'tool'
    content: str = ""
    name: str | None = None
    tool_calls: list[ToolCall] | None = None
    tool_call_id: str | None = None


@dataclass
class ConversionResult:
    output: str
    warnings: list[str] = field(default_factory=list)


def _parse_json(text: str, label: str):
    try:
        return json.loads(text)
    except json.JSONDecodeError as e:
        raise ValueError(f"{label}: invalid JSON — {e.msg}") from e


def _parse_args_or_empty(args: str) -> dict:
    """Parse a JSON-string argument back to a dict, ``{}`` when it is not a JSON object."""
    try:
        v = json.loads(args)
    except (json.JSONDecodeError, TypeError):
        return {}
    return v if isinstance(v, dict) else {}


def _as_record(v) -> dict | None:
    return v if isinstance(v, dict) else None


# ---------------------------------- parse ----------------------------------


def parse_openai(text: str, warnings: list[str]) -> list[InternalMessage]:
    """OpenAI messages[] (bare list, or wrapped in ``{"messages": [...]}``)."""
    parsed = _parse_json(text, "OpenAI transcript")
    root = _as_record(parsed)
    raw = root["messages"] if root is not None and isinstance(root.get("messages"), list) else parsed
    if not isinstance(raw, list):
        raise ValueError('OpenAI transcript must be a messages[] array or an object with a "messages" array')
    out: list[InternalMessage] = []
    for i, entry in enumerate(raw):
        m = _as_record(entry)
        if m is None:
            raise ValueError(f"messages[{i}] is not an object")
        role = m.get("role")
        if role in ("system", "developer"):
            out.append(InternalMessage("system", _content_text(m.get("content"), f"messages[{i}]", warnings)))
            continue
        if role in ("user", "assistant", "tool"):
            msg = InternalMessage(role, _content_text(m.get("content"), f"messages[{i}]", warnings))
            if isinstance(m.get("name"), str):
                msg.name = m["name"]
            if isinstance(m.get("tool_call_id"), str):
                msg.tool_call_id = m["tool_call_id"]
            calls = _parse_openai_tool_calls(m.get("tool_calls"))
            if calls is not None:
                msg.tool_calls = calls
            out.append(msg)
            continue
        raise ValueError(f"messages[{i}] has unsupported role: {json.dumps(role)}")
    return out


def _content_text(content, where: str, warnings: list[str]) -> str:
    """OpenAI content → plain text (string, or text parts joined; other parts flagged)."""
    if isinstance(content, str):
        return content
    if content is None:
        return ""
    if isinstance(content, list):
        texts: list[str] = []
        dropped = 0
        for part in content:
            p = _as_record(part)
            if p is not None and p.get("type") == "text" and isinstance(p.get("text"), str):
                texts.append(p["text"])
            else:
                dropped += 1
        if dropped > 0:
            warnings.append(f"{where}: dropped {dropped} non-text content part(s)")
        return "".join(texts)
    raise ValueError(f"{where}: content must be a string, an array of parts, or null")


def _parse_openai_tool_calls(raw) -> list[ToolCall] | None:
    if raw is None:
        return None
    if not isinstance(raw, list):
        raise ValueError("tool_calls must be an array")
    calls = []
    for entry in raw:
        tc = _as_record(entry)
        fn = _as_record(tc.get("function")) if tc is not None else None
        calls.append(ToolCall(
            id=tc.get("id") if tc is not None and isinstance(tc.get("id"), str) else "",
            name=fn.get("name") if fn is not None and isinstance(fn.get("name"), str) else "",
            args=fn.get("arguments") if fn is not None and isinstance(fn.get("arguments"), str) else "",
        ))
    return calls


def parse_anthropic(text: str, warnings: list[str]) -> list[InternalMessage]:
    """Anthropic messages[] (+ top-level ``system``)."""
    root = _as_record(_parse_json(text, "Anthropic transcript"))
    if root is None or not isinstance(root.get("messages"), list):
        raise ValueError('Anthropic transcript must be an object with a "messages" array')
    out: list[InternalMessage] = []
    if "system" in root:
        out.append(InternalMessage("system", _blocks_to_text(root["system"], "system", warnings)))

    for i, entry in enumerate(root["messages"]):
        m = _as_record(entry)
        if m is None:
            raise ValueError(f"messages[{i}] is not an object")
        where = f"messages[{i}]"
        if m.get("role") == "assistant":
            texts: list[str] = []
            calls: list[ToolCall] = []
            for block in _content_blocks(m.get("content"), where):
                if block[0] == "text":
                    texts.append(block[1])
                elif block[0] == "tool_use":
                    raw = block[1]
                    calls.append(ToolCall(
                        id=raw["id"] if isinstance(raw.get("id"), str) else "",
                        name=raw["name"] if isinstance(raw.get("name"), str) else "",
                        args=json.dumps(raw.get("input") if raw.get("input") is not None else {}),
                    ))
                elif block[0] == "other":
                    warnings.append(f"{where}: dropped unsupported {_block_type_name(block[1])} block")
            msg = InternalMessage("assistant", "\n".join(texts))
            if calls:
                msg.tool_calls = calls
            out.append(msg)
            continue
        if m.get("role") == "user":
            texts = []

            def flush():
                if texts:
                    out.append(InternalMessage("user", "\n".join(texts)))
                    texts.clear()

            for block in _content_blocks(m.get("content"), where):
                if block[0] == "text":
                    texts.append(block[1])
                elif block[0] == "tool_use":
                    warnings.append(f"{where}: tool_use block inside a user message moved to an assistant tool call")
                    flush()
                    raw = block[1]
                    out.append(InternalMessage("assistant", "", tool_calls=[ToolCall(
                        id=raw["id"] if isinstance(raw.get("id"), str) else "",
                        name=raw["name"] if isinstance(raw.get("name"), str) else "",
                        args=json.dumps(raw.get("input") if raw.get("input") is not None else {}),
                    )]))
                elif block[0] == "tool_result":
                    flush()
                    raw = block[1]
                    out.append(InternalMessage(
                        "tool",
                        _blocks_to_text(raw.get("content") or "", where, warnings),
                        tool_call_id=raw["tool_use_id"] if isinstance(raw.get("tool_use_id"), str) else "",
                    ))
                else:
                    warnings.append(f"{where}: dropped unsupported {_block_type_name(block[1])} block")
            flush()
            continue
        raise ValueError(f"{where} has unsupported role: {json.dumps(m.get('role'))}")
    return out


def _content_blocks(content, where: str) -> list[tuple]:
    """Normalize Anthropic content (string | block list) into typed tuples:
    ('text', text) | ('tool_use' | 'tool_result' | 'other', raw_dict)."""
    if isinstance(content, str):
        return [("text", content)]
    if not isinstance(content, list):
        raise ValueError(f"{where}: content must be a string or an array of blocks")
    blocks = []
    for block in content:
        b = _as_record(block)
        if b is not None and b.get("type") == "text" and isinstance(b.get("text"), str):
            blocks.append(("text", b["text"]))
        elif b is not None and b.get("type") in ("tool_use", "tool_result"):
            blocks.append((b["type"], b))
        else:
            blocks.append(("other", b if b is not None else {}))
    return blocks


def _block_type_name(raw: dict) -> str:
    t = raw.get("type")
    return str(t) if t is not None else "content"


def _blocks_to_text(content, where: str, warnings: list[str]) -> str:
    """Anthropic string-or-block-list content → plain text."""
    if isinstance(content, str):
        return content
    texts = []
    for kind, raw in _content_blocks(content, where):
        if kind == "text":
            texts.append(raw)
        elif kind == "other":
            warnings.append(f"{where}: dropped unsupported {_block_type_name(raw)} block")
        else:
            warnings.append(f"{where}: dropped {kind} block from text-only content")
    return "\n".join(texts)


def parse_gemini(text: str, warnings: list[str]) -> list[InternalMessage]:
    """Gemini contents[] (+ optional systemInstruction)."""
    root = _as_record(_parse_json(text, "Gemini transcript"))
    if root is None or not isinstance(root.get("contents"), list):
        raise ValueError('Gemini transcript must be an object with a "contents" array')
    out: list[InternalMessage] = []
    if "systemInstruction" in root:
        out.append(InternalMessage("system", _gemini_text(root["systemInstruction"], "systemInstruction", warnings)))

    for i, entry in enumerate(root["contents"]):
        m = _as_record(entry)
        if m is None:
            raise ValueError(f"contents[{i}] is not an object")
        where = f"contents[{i}]"
        if m.get("role") not in ("user", "model"):
            raise ValueError(
                f'{where} has unsupported role: {json.dumps(m.get("role"))} (Gemini uses "user" or "model")'
            )
        role = "assistant" if m["role"] == "model" else "user"
        parts = m.get("parts")
        if not isinstance(parts, list):
            raise ValueError(f"{where}: parts must be an array")

        if role == "assistant":
            # A model turn keeps its text and function calls in ONE message,
            # mirroring an OpenAI assistant message with tool_calls.
            texts: list[str] = []
            calls: list[ToolCall] = []
            for part in parts:
                p = _as_record(part)
                if p is not None and isinstance(p.get("text"), str):
                    texts.append(p["text"])
                    continue
                fc = _as_record(p.get("functionCall")) if p is not None else None
                if fc is not None:
                    calls.append(ToolCall(
                        name=fc["name"] if isinstance(fc.get("name"), str) else "",
                        args=json.dumps(fc.get("args") if fc.get("args") is not None else {}),
                    ))
                    continue
                warnings.append(f"{where}: dropped unsupported part (inlineData or similar)")
            msg = InternalMessage("assistant", "\n".join(texts))
            if calls:
                msg.tool_calls = calls
            out.append(msg)
            continue

        texts = []

        def flush_text():
            if texts:
                out.append(InternalMessage("user", "\n".join(texts)))
                texts.clear()

        for part in parts:
            p = _as_record(part)
            if p is not None and isinstance(p.get("text"), str):
                texts.append(p["text"])
                continue
            fc = _as_record(p.get("functionCall")) if p is not None else None
            if fc is not None:
                flush_text()
                out.append(InternalMessage("assistant", "", tool_calls=[ToolCall(
                    name=fc["name"] if isinstance(fc.get("name"), str) else "",
                    args=json.dumps(fc.get("args") if fc.get("args") is not None else {}),
                )]))
                continue
            fr = _as_record(p.get("functionResponse")) if p is not None else None
            if fr is not None:
                flush_text()
                name = fr["name"] if isinstance(fr.get("name"), str) else ""
                out.append(InternalMessage(
                    "tool",
                    json.dumps(fr.get("response") if fr.get("response") is not None else {}),
                    name=name,
                    tool_call_id=name,
                ))
                continue
            warnings.append(f"{where}: dropped unsupported part (inlineData or similar)")
        flush_text()
    return out


def _gemini_text(v, where: str, warnings: list[str]) -> str:
    """Gemini systemInstruction (string or {parts}) → plain text."""
    if isinstance(v, str):
        return v
    rec = _as_record(v)
    if rec is not None:
        if isinstance(rec.get("text"), str):
            return rec["text"]
        if isinstance(rec.get("parts"), list):
            return "\n".join(
                p["text"] if (p := _as_record(part)) is not None and isinstance(p.get("text"), str) else ""
                for part in rec["parts"]
            )
    warnings.append(f"{where}: unsupported systemInstruction shape, treated as empty")
    return ""


_MARKDOWN_HEADER = re.compile(r"^\*\*(system|user|assistant|model|tool)\*\*:\s*(.*)$")


def parse_markdown(text: str, _warnings: list[str]) -> list[InternalMessage]:
    """Markdown transcript: ``**role**: text`` header lines with continuation
    lines belonging to the same message. ``model`` maps to assistant."""
    out: list[InternalMessage] = []
    current: tuple[str, list[str]] | None = None
    for line in text.split("\n"):
        match = _MARKDOWN_HEADER.match(line)
        if match is not None:
            if current is not None:
                out.append(_finish_message(current))
            current = ("assistant" if match.group(1) == "model" else match.group(1), [match.group(2)])
            continue
        if current is not None:
            current[1].append(line)
        elif line.strip() != "":
            raise ValueError("Markdown transcript must start with a `**role**:` header line")
    if current is not None:
        out.append(_finish_message(current))
    return out


def _finish_message(current: tuple[str, list[str]]) -> InternalMessage:
    # Trim leading/trailing blank continuation lines but keep inner blank lines.
    lines = current[1]
    while len(lines) > 1 and lines[0].strip() == "":
        lines.pop(0)
    while len(lines) > 1 and lines[-1].strip() == "":
        lines.pop()
    return InternalMessage(current[0], "\n".join(lines))


# -------------------------------- serialize --------------------------------


def serialize_openai(messages: list[InternalMessage]) -> str:
    arr = []
    for m in messages:
        if m.role == "tool":
            o = {"role": "tool", "content": m.content, "tool_call_id": m.tool_call_id or ""}
            if m.name is not None:
                o["name"] = m.name
            arr.append(o)
            continue
        if m.role == "assistant" and m.tool_calls is not None:
            o = {
                "role": "assistant",
                "content": None if m.content == "" else m.content,
                "tool_calls": [
                    {"id": tc.id, "type": "function", "function": {"name": tc.name, "arguments": tc.args}}
                    for tc in m.tool_calls
                ],
            }
            if m.name is not None:
                o["name"] = m.name
            arr.append(o)
            continue
        o = {"role": m.role, "content": m.content}
        if m.name is not None:
            o["name"] = m.name
        arr.append(o)
    return json.dumps({"messages": arr}, indent=2, ensure_ascii=False) + "\n"


def serialize_anthropic(messages: list[InternalMessage], warnings: list[str]) -> str:
    system = [m.content for m in messages if m.role == "system"]
    out = []
    for i, m in enumerate(messages):
        if m.role == "system":
            continue
        if m.name is not None and m.role != "tool":
            warnings.append(f'message {i}: "name" has no Anthropic equivalent and was dropped')
        if m.role == "user":
            out.append({"role": "user", "content": m.content})
        elif m.role == "assistant":
            if m.tool_calls is not None:
                blocks = []
                if m.content != "":
                    blocks.append({"type": "text", "text": m.content})
                for tc in m.tool_calls:
                    blocks.append({"type": "tool_use", "id": tc.id, "name": tc.name,
                                   "input": _parse_args_or_empty(tc.args)})
                out.append({"role": "assistant", "content": blocks})
            else:
                out.append({"role": "assistant", "content": m.content})
        else:
            out.append({
                "role": "user",
                "content": [{"type": "tool_result", "tool_use_id": m.tool_call_id or "", "content": m.content}],
            })
    root = {"messages": out} if out else {}
    if system:
        root["system"] = "\n\n".join(system)
    return json.dumps(root, indent=2, ensure_ascii=False) + "\n"


def serialize_gemini(messages: list[InternalMessage], warnings: list[str]) -> str:
    system = [m.content for m in messages if m.role == "system"]
    contents: list[dict] = []

    def push(role: str, part: dict):
        if contents and contents[-1]["role"] == role:
            contents[-1]["parts"].append(part)
        else:
            contents.append({"role": role, "parts": [part]})

    for i, m in enumerate(messages):
        if m.role == "system":
            continue
        if m.name is not None and m.role != "tool":
            warnings.append(f'message {i}: "name" has no Gemini equivalent and was dropped')
        if m.role == "user":
            push("user", {"text": m.content})
        elif m.role == "assistant":
            if m.content != "":
                push("model", {"text": m.content})
            for tc in m.tool_calls or []:
                push("model", {"functionCall": {"name": tc.name, "args": _parse_args_or_empty(tc.args)}})
        else:
            push("user", {
                "functionResponse": {
                    "name": m.name or m.tool_call_id or "",
                    "response": _parse_args_or_empty(m.content),
                }
            })
    root = {"contents": contents} if contents else {}
    if system:
        root["systemInstruction"] = {"parts": [{"text": "\n\n".join(system)}]}
    return json.dumps(root, indent=2, ensure_ascii=False) + "\n"


def serialize_markdown(messages: list[InternalMessage], warnings: list[str]) -> str:
    for i, m in enumerate(messages):
        if m.tool_calls:
            warnings.append(f"message {i}: tool calls are not representable in Markdown and were dropped")
        if m.role == "tool" and m.tool_call_id is not None:
            warnings.append(f"message {i}: tool result id is not representable in Markdown and was dropped")
    return "\n\n".join(f"**{m.role}**: {m.content}" for m in messages) + "\n"


PARSERS = {
    "openai": parse_openai,
    "anthropic": parse_anthropic,
    "gemini": parse_gemini,
    "markdown": parse_markdown,
}

SERIALIZERS = {
    "openai": serialize_openai,
    "anthropic": serialize_anthropic,
    "gemini": serialize_gemini,
    "markdown": serialize_markdown,
}


def convert(transcript: str, from_format: str, to_format: str) -> ConversionResult:
    """Convert a transcript between chat formats through the shared internal shape.

    Raises ValueError when the transcript is empty, the source format fails to
    parse, or ``from_format``/``to_format`` is not a known format.
    """
    if not isinstance(transcript, str) or transcript.strip() == "":
        raise ValueError("Transcript is empty — paste a transcript first")
    parser = PARSERS.get(from_format)
    if parser is None:
        raise ValueError(f"Unknown source format: {from_format}")
    serializer = SERIALIZERS.get(to_format)
    if serializer is None:
        raise ValueError(f"Unknown target format: {to_format}")
    warnings: list[str] = []
    messages = parser(transcript, warnings)
    output = serializer(messages, warnings)
    return ConversionResult(output, warnings)

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →