Chat Format Converter — Python source
Convert chat transcripts between OpenAI messages[], Anthropic system+messages, Gemini contents[], and plain Markdown. Roles map faithfully, tool calls are preserved where possible, and anything unmappable is flagged — never dropped silently. Runs entirely in your browser.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""chat-format-converter — Python polyglot showcase port.
Converts one chat transcript between provider shapes through a shared internal
message list:
- openai — {"messages": [{"role": "system|user|assistant|tool", ...}]}
- anthropic — {"system": "...", "messages": [{"role": "user|assistant", ...}]}
- gemini — {"contents": [{"role": "user|model", "parts": [...]}]}
- markdown — a plain ``**role**: text`` transcript
Roles map faithfully (Gemini has no assistant — it is ``model``; Anthropic
system lives top-level). Fields a target format cannot represent (e.g. tool
calls in Markdown) are flagged as warnings, never dropped silently.
This is the Python sibling of src/lib/chatFormatConverter.ts (the canonical
TypeScript that powers the live tool). Standard library only (json + re +
dataclasses); parse failures raise ValueError.
Display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
"""
from __future__ import annotations
import json
import re
from dataclasses import dataclass, field
CHAT_FORMATS = ["openai", "anthropic", "gemini", "markdown"]
@dataclass
class ToolCall:
"""One assistant-emitted tool call, provider-agnostic. ``args`` is a JSON string."""
id: str = ""
name: str = ""
args: str = ""
@dataclass
class InternalMessage:
"""The shared internal shape every format parses into and serializes from."""
role: str # 'system' | 'user' | 'assistant' | 'tool'
content: str = ""
name: str | None = None
tool_calls: list[ToolCall] | None = None
tool_call_id: str | None = None
@dataclass
class ConversionResult:
output: str
warnings: list[str] = field(default_factory=list)
def _parse_json(text: str, label: str):
try:
return json.loads(text)
except json.JSONDecodeError as e:
raise ValueError(f"{label}: invalid JSON — {e.msg}") from e
def _parse_args_or_empty(args: str) -> dict:
"""Parse a JSON-string argument back to a dict, ``{}`` when it is not a JSON object."""
try:
v = json.loads(args)
except (json.JSONDecodeError, TypeError):
return {}
return v if isinstance(v, dict) else {}
def _as_record(v) -> dict | None:
return v if isinstance(v, dict) else None
# ---------------------------------- parse ----------------------------------
def parse_openai(text: str, warnings: list[str]) -> list[InternalMessage]:
"""OpenAI messages[] (bare list, or wrapped in ``{"messages": [...]}``)."""
parsed = _parse_json(text, "OpenAI transcript")
root = _as_record(parsed)
raw = root["messages"] if root is not None and isinstance(root.get("messages"), list) else parsed
if not isinstance(raw, list):
raise ValueError('OpenAI transcript must be a messages[] array or an object with a "messages" array')
out: list[InternalMessage] = []
for i, entry in enumerate(raw):
m = _as_record(entry)
if m is None:
raise ValueError(f"messages[{i}] is not an object")
role = m.get("role")
if role in ("system", "developer"):
out.append(InternalMessage("system", _content_text(m.get("content"), f"messages[{i}]", warnings)))
continue
if role in ("user", "assistant", "tool"):
msg = InternalMessage(role, _content_text(m.get("content"), f"messages[{i}]", warnings))
if isinstance(m.get("name"), str):
msg.name = m["name"]
if isinstance(m.get("tool_call_id"), str):
msg.tool_call_id = m["tool_call_id"]
calls = _parse_openai_tool_calls(m.get("tool_calls"))
if calls is not None:
msg.tool_calls = calls
out.append(msg)
continue
raise ValueError(f"messages[{i}] has unsupported role: {json.dumps(role)}")
return out
def _content_text(content, where: str, warnings: list[str]) -> str:
"""OpenAI content → plain text (string, or text parts joined; other parts flagged)."""
if isinstance(content, str):
return content
if content is None:
return ""
if isinstance(content, list):
texts: list[str] = []
dropped = 0
for part in content:
p = _as_record(part)
if p is not None and p.get("type") == "text" and isinstance(p.get("text"), str):
texts.append(p["text"])
else:
dropped += 1
if dropped > 0:
warnings.append(f"{where}: dropped {dropped} non-text content part(s)")
return "".join(texts)
raise ValueError(f"{where}: content must be a string, an array of parts, or null")
def _parse_openai_tool_calls(raw) -> list[ToolCall] | None:
if raw is None:
return None
if not isinstance(raw, list):
raise ValueError("tool_calls must be an array")
calls = []
for entry in raw:
tc = _as_record(entry)
fn = _as_record(tc.get("function")) if tc is not None else None
calls.append(ToolCall(
id=tc.get("id") if tc is not None and isinstance(tc.get("id"), str) else "",
name=fn.get("name") if fn is not None and isinstance(fn.get("name"), str) else "",
args=fn.get("arguments") if fn is not None and isinstance(fn.get("arguments"), str) else "",
))
return calls
def parse_anthropic(text: str, warnings: list[str]) -> list[InternalMessage]:
"""Anthropic messages[] (+ top-level ``system``)."""
root = _as_record(_parse_json(text, "Anthropic transcript"))
if root is None or not isinstance(root.get("messages"), list):
raise ValueError('Anthropic transcript must be an object with a "messages" array')
out: list[InternalMessage] = []
if "system" in root:
out.append(InternalMessage("system", _blocks_to_text(root["system"], "system", warnings)))
for i, entry in enumerate(root["messages"]):
m = _as_record(entry)
if m is None:
raise ValueError(f"messages[{i}] is not an object")
where = f"messages[{i}]"
if m.get("role") == "assistant":
texts: list[str] = []
calls: list[ToolCall] = []
for block in _content_blocks(m.get("content"), where):
if block[0] == "text":
texts.append(block[1])
elif block[0] == "tool_use":
raw = block[1]
calls.append(ToolCall(
id=raw["id"] if isinstance(raw.get("id"), str) else "",
name=raw["name"] if isinstance(raw.get("name"), str) else "",
args=json.dumps(raw.get("input") if raw.get("input") is not None else {}),
))
elif block[0] == "other":
warnings.append(f"{where}: dropped unsupported {_block_type_name(block[1])} block")
msg = InternalMessage("assistant", "\n".join(texts))
if calls:
msg.tool_calls = calls
out.append(msg)
continue
if m.get("role") == "user":
texts = []
def flush():
if texts:
out.append(InternalMessage("user", "\n".join(texts)))
texts.clear()
for block in _content_blocks(m.get("content"), where):
if block[0] == "text":
texts.append(block[1])
elif block[0] == "tool_use":
warnings.append(f"{where}: tool_use block inside a user message moved to an assistant tool call")
flush()
raw = block[1]
out.append(InternalMessage("assistant", "", tool_calls=[ToolCall(
id=raw["id"] if isinstance(raw.get("id"), str) else "",
name=raw["name"] if isinstance(raw.get("name"), str) else "",
args=json.dumps(raw.get("input") if raw.get("input") is not None else {}),
)]))
elif block[0] == "tool_result":
flush()
raw = block[1]
out.append(InternalMessage(
"tool",
_blocks_to_text(raw.get("content") or "", where, warnings),
tool_call_id=raw["tool_use_id"] if isinstance(raw.get("tool_use_id"), str) else "",
))
else:
warnings.append(f"{where}: dropped unsupported {_block_type_name(block[1])} block")
flush()
continue
raise ValueError(f"{where} has unsupported role: {json.dumps(m.get('role'))}")
return out
def _content_blocks(content, where: str) -> list[tuple]:
"""Normalize Anthropic content (string | block list) into typed tuples:
('text', text) | ('tool_use' | 'tool_result' | 'other', raw_dict)."""
if isinstance(content, str):
return [("text", content)]
if not isinstance(content, list):
raise ValueError(f"{where}: content must be a string or an array of blocks")
blocks = []
for block in content:
b = _as_record(block)
if b is not None and b.get("type") == "text" and isinstance(b.get("text"), str):
blocks.append(("text", b["text"]))
elif b is not None and b.get("type") in ("tool_use", "tool_result"):
blocks.append((b["type"], b))
else:
blocks.append(("other", b if b is not None else {}))
return blocks
def _block_type_name(raw: dict) -> str:
t = raw.get("type")
return str(t) if t is not None else "content"
def _blocks_to_text(content, where: str, warnings: list[str]) -> str:
"""Anthropic string-or-block-list content → plain text."""
if isinstance(content, str):
return content
texts = []
for kind, raw in _content_blocks(content, where):
if kind == "text":
texts.append(raw)
elif kind == "other":
warnings.append(f"{where}: dropped unsupported {_block_type_name(raw)} block")
else:
warnings.append(f"{where}: dropped {kind} block from text-only content")
return "\n".join(texts)
def parse_gemini(text: str, warnings: list[str]) -> list[InternalMessage]:
"""Gemini contents[] (+ optional systemInstruction)."""
root = _as_record(_parse_json(text, "Gemini transcript"))
if root is None or not isinstance(root.get("contents"), list):
raise ValueError('Gemini transcript must be an object with a "contents" array')
out: list[InternalMessage] = []
if "systemInstruction" in root:
out.append(InternalMessage("system", _gemini_text(root["systemInstruction"], "systemInstruction", warnings)))
for i, entry in enumerate(root["contents"]):
m = _as_record(entry)
if m is None:
raise ValueError(f"contents[{i}] is not an object")
where = f"contents[{i}]"
if m.get("role") not in ("user", "model"):
raise ValueError(
f'{where} has unsupported role: {json.dumps(m.get("role"))} (Gemini uses "user" or "model")'
)
role = "assistant" if m["role"] == "model" else "user"
parts = m.get("parts")
if not isinstance(parts, list):
raise ValueError(f"{where}: parts must be an array")
if role == "assistant":
# A model turn keeps its text and function calls in ONE message,
# mirroring an OpenAI assistant message with tool_calls.
texts: list[str] = []
calls: list[ToolCall] = []
for part in parts:
p = _as_record(part)
if p is not None and isinstance(p.get("text"), str):
texts.append(p["text"])
continue
fc = _as_record(p.get("functionCall")) if p is not None else None
if fc is not None:
calls.append(ToolCall(
name=fc["name"] if isinstance(fc.get("name"), str) else "",
args=json.dumps(fc.get("args") if fc.get("args") is not None else {}),
))
continue
warnings.append(f"{where}: dropped unsupported part (inlineData or similar)")
msg = InternalMessage("assistant", "\n".join(texts))
if calls:
msg.tool_calls = calls
out.append(msg)
continue
texts = []
def flush_text():
if texts:
out.append(InternalMessage("user", "\n".join(texts)))
texts.clear()
for part in parts:
p = _as_record(part)
if p is not None and isinstance(p.get("text"), str):
texts.append(p["text"])
continue
fc = _as_record(p.get("functionCall")) if p is not None else None
if fc is not None:
flush_text()
out.append(InternalMessage("assistant", "", tool_calls=[ToolCall(
name=fc["name"] if isinstance(fc.get("name"), str) else "",
args=json.dumps(fc.get("args") if fc.get("args") is not None else {}),
)]))
continue
fr = _as_record(p.get("functionResponse")) if p is not None else None
if fr is not None:
flush_text()
name = fr["name"] if isinstance(fr.get("name"), str) else ""
out.append(InternalMessage(
"tool",
json.dumps(fr.get("response") if fr.get("response") is not None else {}),
name=name,
tool_call_id=name,
))
continue
warnings.append(f"{where}: dropped unsupported part (inlineData or similar)")
flush_text()
return out
def _gemini_text(v, where: str, warnings: list[str]) -> str:
"""Gemini systemInstruction (string or {parts}) → plain text."""
if isinstance(v, str):
return v
rec = _as_record(v)
if rec is not None:
if isinstance(rec.get("text"), str):
return rec["text"]
if isinstance(rec.get("parts"), list):
return "\n".join(
p["text"] if (p := _as_record(part)) is not None and isinstance(p.get("text"), str) else ""
for part in rec["parts"]
)
warnings.append(f"{where}: unsupported systemInstruction shape, treated as empty")
return ""
_MARKDOWN_HEADER = re.compile(r"^\*\*(system|user|assistant|model|tool)\*\*:\s*(.*)$")
def parse_markdown(text: str, _warnings: list[str]) -> list[InternalMessage]:
"""Markdown transcript: ``**role**: text`` header lines with continuation
lines belonging to the same message. ``model`` maps to assistant."""
out: list[InternalMessage] = []
current: tuple[str, list[str]] | None = None
for line in text.split("\n"):
match = _MARKDOWN_HEADER.match(line)
if match is not None:
if current is not None:
out.append(_finish_message(current))
current = ("assistant" if match.group(1) == "model" else match.group(1), [match.group(2)])
continue
if current is not None:
current[1].append(line)
elif line.strip() != "":
raise ValueError("Markdown transcript must start with a `**role**:` header line")
if current is not None:
out.append(_finish_message(current))
return out
def _finish_message(current: tuple[str, list[str]]) -> InternalMessage:
# Trim leading/trailing blank continuation lines but keep inner blank lines.
lines = current[1]
while len(lines) > 1 and lines[0].strip() == "":
lines.pop(0)
while len(lines) > 1 and lines[-1].strip() == "":
lines.pop()
return InternalMessage(current[0], "\n".join(lines))
# -------------------------------- serialize --------------------------------
def serialize_openai(messages: list[InternalMessage]) -> str:
arr = []
for m in messages:
if m.role == "tool":
o = {"role": "tool", "content": m.content, "tool_call_id": m.tool_call_id or ""}
if m.name is not None:
o["name"] = m.name
arr.append(o)
continue
if m.role == "assistant" and m.tool_calls is not None:
o = {
"role": "assistant",
"content": None if m.content == "" else m.content,
"tool_calls": [
{"id": tc.id, "type": "function", "function": {"name": tc.name, "arguments": tc.args}}
for tc in m.tool_calls
],
}
if m.name is not None:
o["name"] = m.name
arr.append(o)
continue
o = {"role": m.role, "content": m.content}
if m.name is not None:
o["name"] = m.name
arr.append(o)
return json.dumps({"messages": arr}, indent=2, ensure_ascii=False) + "\n"
def serialize_anthropic(messages: list[InternalMessage], warnings: list[str]) -> str:
system = [m.content for m in messages if m.role == "system"]
out = []
for i, m in enumerate(messages):
if m.role == "system":
continue
if m.name is not None and m.role != "tool":
warnings.append(f'message {i}: "name" has no Anthropic equivalent and was dropped')
if m.role == "user":
out.append({"role": "user", "content": m.content})
elif m.role == "assistant":
if m.tool_calls is not None:
blocks = []
if m.content != "":
blocks.append({"type": "text", "text": m.content})
for tc in m.tool_calls:
blocks.append({"type": "tool_use", "id": tc.id, "name": tc.name,
"input": _parse_args_or_empty(tc.args)})
out.append({"role": "assistant", "content": blocks})
else:
out.append({"role": "assistant", "content": m.content})
else:
out.append({
"role": "user",
"content": [{"type": "tool_result", "tool_use_id": m.tool_call_id or "", "content": m.content}],
})
root = {"messages": out} if out else {}
if system:
root["system"] = "\n\n".join(system)
return json.dumps(root, indent=2, ensure_ascii=False) + "\n"
def serialize_gemini(messages: list[InternalMessage], warnings: list[str]) -> str:
system = [m.content for m in messages if m.role == "system"]
contents: list[dict] = []
def push(role: str, part: dict):
if contents and contents[-1]["role"] == role:
contents[-1]["parts"].append(part)
else:
contents.append({"role": role, "parts": [part]})
for i, m in enumerate(messages):
if m.role == "system":
continue
if m.name is not None and m.role != "tool":
warnings.append(f'message {i}: "name" has no Gemini equivalent and was dropped')
if m.role == "user":
push("user", {"text": m.content})
elif m.role == "assistant":
if m.content != "":
push("model", {"text": m.content})
for tc in m.tool_calls or []:
push("model", {"functionCall": {"name": tc.name, "args": _parse_args_or_empty(tc.args)}})
else:
push("user", {
"functionResponse": {
"name": m.name or m.tool_call_id or "",
"response": _parse_args_or_empty(m.content),
}
})
root = {"contents": contents} if contents else {}
if system:
root["systemInstruction"] = {"parts": [{"text": "\n\n".join(system)}]}
return json.dumps(root, indent=2, ensure_ascii=False) + "\n"
def serialize_markdown(messages: list[InternalMessage], warnings: list[str]) -> str:
for i, m in enumerate(messages):
if m.tool_calls:
warnings.append(f"message {i}: tool calls are not representable in Markdown and were dropped")
if m.role == "tool" and m.tool_call_id is not None:
warnings.append(f"message {i}: tool result id is not representable in Markdown and was dropped")
return "\n\n".join(f"**{m.role}**: {m.content}" for m in messages) + "\n"
PARSERS = {
"openai": parse_openai,
"anthropic": parse_anthropic,
"gemini": parse_gemini,
"markdown": parse_markdown,
}
SERIALIZERS = {
"openai": serialize_openai,
"anthropic": serialize_anthropic,
"gemini": serialize_gemini,
"markdown": serialize_markdown,
}
def convert(transcript: str, from_format: str, to_format: str) -> ConversionResult:
"""Convert a transcript between chat formats through the shared internal shape.
Raises ValueError when the transcript is empty, the source format fails to
parse, or ``from_format``/``to_format`` is not a known format.
"""
if not isinstance(transcript, str) or transcript.strip() == "":
raise ValueError("Transcript is empty — paste a transcript first")
parser = PARSERS.get(from_format)
if parser is None:
raise ValueError(f"Unknown source format: {from_format}")
serializer = SERIALIZERS.get(to_format)
if serializer is None:
raise ValueError(f"Unknown target format: {to_format}")
warnings: list[str] = []
messages = parser(transcript, warnings)
output = serializer(messages, warnings)
return ConversionResult(output, warnings)
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →