System Prompt Builder — Python source
Assemble a system prompt from ordered blocks — role, context, constraints, output format — with a live token count, soft-limit warnings, and a shareable URL. 100% client-side.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""System Prompt Builder — assemble an ordered list of prompt blocks into a
markdown-structured system prompt, with pure list operations, presets,
warnings, and a compact URL codec for shareable state.
Language: Python (3.9+, standard library only)
Source: CosmoDev polyglot showcase port of the System Prompt Builder tool,
ported from src/lib/systemPromptBuilder.ts (the canonical
TypeScript implementation).
Tool page: https://dev.cosmolabs.org/tools/system-prompt-builder
License: display source — part of CosmoDev's polyglot tool pages.
Token counting inlines the chars-per-token heuristic from
src/lib/tokenEstimator.ts (the original imports it); the soft-limit constant
lives here as the domain rule. All list operations are pure — they return new
lists and never mutate their input.
"""
from __future__ import annotations
import base64
import json
import math
from dataclasses import dataclass, field, replace
from typing import Dict, List, Optional, Sequence, Tuple, Union
@dataclass(frozen=True)
class PromptBlock:
"""One editable section of the system prompt."""
id: str
title: str
content: str
enabled: bool
@dataclass(frozen=True)
class PromptPreset:
"""A starter template from the recommended prompt skeleton."""
id: str
title: str
description: str
content: str
@dataclass
class PromptReport:
"""Assemble + count + lint in one pass — the island's live report."""
assembled: str
tokens: int
warnings: List[str] = field(default_factory=list)
#: Blocks whose assembled size starts crowding the context on most models.
SYSTEM_PROMPT_SOFT_LIMIT_TOKENS: int = 2000
#: Ordered starter templates — the recommended skeleton of a system prompt.
SYSTEM_PROMPT_PRESETS: Tuple[PromptPreset, ...] = (
PromptPreset(
id="role",
title="Role",
description="Who the model is and what it optimizes for.",
content=(
"You are a senior software engineer. You give correct, concise "
"answers and say so plainly when you are unsure."
),
),
PromptPreset(
id="context",
title="Context",
description="The situation the model is working in.",
content=(
"The user is a developer working in a TypeScript codebase. Prefer "
"runnable examples over prose when both work."
),
),
PromptPreset(
id="constraints",
title="Constraints",
description="Hard rules the model must not break.",
content=(
"- Never invent library APIs; use only the ones in the provided code.\n"
"- Keep answers under 300 words unless asked for more."
),
),
PromptPreset(
id="output-format",
title="Output format",
description="The exact shape of the answer.",
content=(
"Respond with: 1) a one-line summary, 2) a fenced code block, "
"3) any caveats as bullet points."
),
),
PromptPreset(
id="examples",
title="Examples",
description="Few-shot demonstrations of the desired behavior.",
content='Input: reverse "abc"\nOutput: "cba"',
),
PromptPreset(
id="tone",
title="Tone",
description="Voice and register.",
content="Direct and friendly. No filler openers, no apologies.",
),
PromptPreset(
id="refusal",
title="Refusal policy",
description="How to handle out-of-scope requests.",
content=(
"If a request is outside your scope, say so in one sentence and "
"suggest the closest thing you can do."
),
),
PromptPreset(
id="safety",
title="Safety",
description="Guardrails for sensitive content.",
content=(
"Refuse requests that could cause harm, and never echo secrets, "
"keys, or credentials back in full."
),
),
)
# ---- token estimate (tokens figure only, from tokenEstimator.ts) ------------
#: Average characters per token, by content type.
CHARS_PER_TOKEN: Dict[str, float] = {"prose": 4.0, "code": 3.5, "json": 3.0, "cjk": 1.5}
ContentType = Union[str, None] # 'prose' | 'code' | 'json' | 'cjk' | 'auto' | None
def _round_half_up(x: float) -> int:
"""JS-style Math.round (half away from zero for the positives used here)."""
return int(math.floor(x + 0.5))
def _detect_line_type(line: str) -> str:
"""Classify a single line by its shape. Order: json, cjk, code, prose."""
trimmed = line.strip()
# JSON-ish: opens like a JSON fragment AND carries a separator.
if trimmed[:1] in "{[}\"" and (":" in line or "," in line):
return "json"
# CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
if any(
0x4E00 <= ord(ch) <= 0x9FFF or 0x3040 <= ord(ch) <= 0x30FF or 0xAC00 <= ord(ch) <= 0xD82F
for ch in line
):
return "cjk"
# Code: symbol-dense, or a statement terminator / block opener at EOL.
density = sum(ch in "{}();=<>[]#" for ch in line) / len(line) if line else 0.0
if density > 0.08 or trimmed.endswith((";", "{", "}")):
return "code"
return "prose"
def _estimate_tokens(text: str, content_type: ContentType = "prose") -> int:
"""Sum of per-line token estimates (excludes chat framing). 'auto'
classifies per line, with a document that parses as JSON counted as json
throughout; any other value forces one type on every line."""
forced = content_type if content_type and content_type != "auto" else None
whole_text_json = False
if forced is None and text.strip():
try:
json.loads(text)
whole_text_json = True
except ValueError:
pass # not JSON — classify line by line
tokens = 0
for line in text.splitlines():
if not line.strip():
continue
line_type = forced if forced is not None else ("json" if whole_text_json else _detect_line_type(line))
tokens += max(1, _round_half_up(len(line) / CHARS_PER_TOKEN[line_type]))
return tokens
def assemble_prompt(blocks: Sequence[PromptBlock], headers: bool = True) -> str:
"""Render enabled, non-empty blocks (in order) as one markdown-structured
prompt (drop the '## Title' headers by passing headers=False)."""
rendered = [
(f"## {b.title.strip() or 'Untitled'}\n{b.content.strip()}" if headers else b.content.strip())
for b in blocks
if b.enabled and b.content.strip()
]
return "\n\n".join(rendered).strip()
def add_block(
blocks: Sequence[PromptBlock],
id: str,
title: str,
content: str = "",
enabled: bool = True,
) -> List[PromptBlock]:
"""Append a block (caller supplies the id so the lib stays pure)."""
return [*blocks, PromptBlock(id=id, title=title, content=content, enabled=enabled)]
def update_block(
blocks: Sequence[PromptBlock], id: str, patch: Optional[dict] = None
) -> List[PromptBlock]:
"""Patch one block by id (patch keys: title/content/enabled); unknown ids
leave the list unchanged."""
patch = patch or {}
return [replace(b, **patch) if b.id == id else b for b in blocks]
def toggle_block(blocks: Sequence[PromptBlock], id: str) -> List[PromptBlock]:
"""Flip one block's enabled flag by id."""
return [replace(b, enabled=not b.enabled) if b.id == id else b for b in blocks]
def remove_block(blocks: Sequence[PromptBlock], id: str) -> List[PromptBlock]:
"""Remove one block by id."""
return [b for b in blocks if b.id != id]
def move_block(blocks: Sequence[PromptBlock], from_index: int, to_index: int) -> List[PromptBlock]:
"""Move a block (clamped; no-op when indexes are out of range or equal)."""
n = len(blocks)
if from_index < 0 or from_index >= n or to_index < 0 or to_index >= n or from_index == to_index:
return list(blocks)
nxt = list(blocks)
nxt.insert(to_index, nxt.pop(from_index))
return nxt
def build_report(blocks: Sequence[PromptBlock], content_type: str = "prose") -> PromptReport:
"""Assemble + count + lint in one pass — the island's live report."""
assembled = assemble_prompt(blocks)
tokens = _estimate_tokens(assembled, content_type) if assembled else 0
warnings: List[str] = []
if tokens > SYSTEM_PROMPT_SOFT_LIMIT_TOKENS:
warnings.append(
f"Assembled prompt is ~{tokens:,} tokens — beyond "
f"{SYSTEM_PROMPT_SOFT_LIMIT_TOKENS:,} it starts crowding the "
f"context window on most models."
)
if len(blocks) > 0 and not any(b.enabled and b.title.strip().lower() == "role" for b in blocks):
warnings.append(
'No enabled "Role" block — stating who the model is tends to '
"anchor every following instruction."
)
if len(blocks) > 0 and assembled == "":
warnings.append("Every block is disabled or empty — the assembled prompt is empty.")
return PromptReport(assembled=assembled, tokens=tokens, warnings=warnings)
# ---- shareable state codec (URL-safe, compact) ------------------------------
# Triples of [enabled(0/1), title, content] keep URLs far smaller than the
# full object shape; ids are regenerated on decode (they are UI-local).
_MAX_ENCODED_LENGTH = 4000
def encode_blocks(blocks: Sequence[PromptBlock]) -> str:
"""Encode blocks to a compact base64url string; '' when blocks are empty."""
if len(blocks) == 0:
return ""
compact = [[1 if b.enabled else 0, b.title, b.content] for b in blocks]
payload = json.dumps(compact, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
return base64.urlsafe_b64encode(payload).rstrip(b"=").decode("ascii")
def encoded_too_long(encoded: str) -> bool:
"""True when the encoded form would make an uncomfortably long URL."""
return len(encoded) > _MAX_ENCODED_LENGTH
def decode_blocks(encoded: str) -> Optional[List[PromptBlock]]:
"""Decode `encode_blocks` output; regenerates ids (b1, b2, …). Returns
None on malformed input — never raises; '' decodes to []."""
if not encoded:
return []
try:
padded = encoded + "=" * ((4 - len(encoded) % 4) % 4)
raw = json.loads(base64.urlsafe_b64decode(padded).decode("utf-8"))
except (ValueError, UnicodeDecodeError):
return None
if not isinstance(raw, list):
return None
blocks: List[PromptBlock] = []
for i, entry in enumerate(raw):
if not isinstance(entry, list) or len(entry) != 3:
return None
enabled, title, content = entry
if not isinstance(enabled, int) or isinstance(enabled, bool):
return None
if not isinstance(title, str) or not isinstance(content, str):
return None
blocks.append(
PromptBlock(id=f"b{i + 1}", title=title, content=content, enabled=enabled == 1)
)
return blocks
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →