Skip to content

System Prompt Builder — Python source

Assemble a system prompt from ordered blocks — role, context, constraints, output format — with a live token count, soft-limit warnings, and a shareable URL. 100% client-side.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""System Prompt Builder — assemble an ordered list of prompt blocks into a
markdown-structured system prompt, with pure list operations, presets,
warnings, and a compact URL codec for shareable state.

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the System Prompt Builder tool,
          ported from src/lib/systemPromptBuilder.ts (the canonical
          TypeScript implementation).
Tool page: https://dev.cosmolabs.org/tools/system-prompt-builder
License:  display source — part of CosmoDev's polyglot tool pages.

Token counting inlines the chars-per-token heuristic from
src/lib/tokenEstimator.ts (the original imports it); the soft-limit constant
lives here as the domain rule. All list operations are pure — they return new
lists and never mutate their input.
"""

from __future__ import annotations

import base64
import json
import math
from dataclasses import dataclass, field, replace
from typing import Dict, List, Optional, Sequence, Tuple, Union


@dataclass(frozen=True)
class PromptBlock:
    """One editable section of the system prompt."""

    id: str
    title: str
    content: str
    enabled: bool


@dataclass(frozen=True)
class PromptPreset:
    """A starter template from the recommended prompt skeleton."""

    id: str
    title: str
    description: str
    content: str


@dataclass
class PromptReport:
    """Assemble + count + lint in one pass — the island's live report."""

    assembled: str
    tokens: int
    warnings: List[str] = field(default_factory=list)


#: Blocks whose assembled size starts crowding the context on most models.
SYSTEM_PROMPT_SOFT_LIMIT_TOKENS: int = 2000

#: Ordered starter templates — the recommended skeleton of a system prompt.
SYSTEM_PROMPT_PRESETS: Tuple[PromptPreset, ...] = (
    PromptPreset(
        id="role",
        title="Role",
        description="Who the model is and what it optimizes for.",
        content=(
            "You are a senior software engineer. You give correct, concise "
            "answers and say so plainly when you are unsure."
        ),
    ),
    PromptPreset(
        id="context",
        title="Context",
        description="The situation the model is working in.",
        content=(
            "The user is a developer working in a TypeScript codebase. Prefer "
            "runnable examples over prose when both work."
        ),
    ),
    PromptPreset(
        id="constraints",
        title="Constraints",
        description="Hard rules the model must not break.",
        content=(
            "- Never invent library APIs; use only the ones in the provided code.\n"
            "- Keep answers under 300 words unless asked for more."
        ),
    ),
    PromptPreset(
        id="output-format",
        title="Output format",
        description="The exact shape of the answer.",
        content=(
            "Respond with: 1) a one-line summary, 2) a fenced code block, "
            "3) any caveats as bullet points."
        ),
    ),
    PromptPreset(
        id="examples",
        title="Examples",
        description="Few-shot demonstrations of the desired behavior.",
        content='Input: reverse "abc"\nOutput: "cba"',
    ),
    PromptPreset(
        id="tone",
        title="Tone",
        description="Voice and register.",
        content="Direct and friendly. No filler openers, no apologies.",
    ),
    PromptPreset(
        id="refusal",
        title="Refusal policy",
        description="How to handle out-of-scope requests.",
        content=(
            "If a request is outside your scope, say so in one sentence and "
            "suggest the closest thing you can do."
        ),
    ),
    PromptPreset(
        id="safety",
        title="Safety",
        description="Guardrails for sensitive content.",
        content=(
            "Refuse requests that could cause harm, and never echo secrets, "
            "keys, or credentials back in full."
        ),
    ),
)

# ---- token estimate (tokens figure only, from tokenEstimator.ts) ------------

#: Average characters per token, by content type.
CHARS_PER_TOKEN: Dict[str, float] = {"prose": 4.0, "code": 3.5, "json": 3.0, "cjk": 1.5}

ContentType = Union[str, None]  # 'prose' | 'code' | 'json' | 'cjk' | 'auto' | None


def _round_half_up(x: float) -> int:
    """JS-style Math.round (half away from zero for the positives used here)."""
    return int(math.floor(x + 0.5))


def _detect_line_type(line: str) -> str:
    """Classify a single line by its shape. Order: json, cjk, code, prose."""
    trimmed = line.strip()
    # JSON-ish: opens like a JSON fragment AND carries a separator.
    if trimmed[:1] in "{[}\"" and (":" in line or "," in line):
        return "json"
    # CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
    if any(
        0x4E00 <= ord(ch) <= 0x9FFF or 0x3040 <= ord(ch) <= 0x30FF or 0xAC00 <= ord(ch) <= 0xD82F
        for ch in line
    ):
        return "cjk"
    # Code: symbol-dense, or a statement terminator / block opener at EOL.
    density = sum(ch in "{}();=<>[]#" for ch in line) / len(line) if line else 0.0
    if density > 0.08 or trimmed.endswith((";", "{", "}")):
        return "code"
    return "prose"


def _estimate_tokens(text: str, content_type: ContentType = "prose") -> int:
    """Sum of per-line token estimates (excludes chat framing). 'auto'
    classifies per line, with a document that parses as JSON counted as json
    throughout; any other value forces one type on every line."""
    forced = content_type if content_type and content_type != "auto" else None
    whole_text_json = False
    if forced is None and text.strip():
        try:
            json.loads(text)
            whole_text_json = True
        except ValueError:
            pass  # not JSON — classify line by line
    tokens = 0
    for line in text.splitlines():
        if not line.strip():
            continue
        line_type = forced if forced is not None else ("json" if whole_text_json else _detect_line_type(line))
        tokens += max(1, _round_half_up(len(line) / CHARS_PER_TOKEN[line_type]))
    return tokens


def assemble_prompt(blocks: Sequence[PromptBlock], headers: bool = True) -> str:
    """Render enabled, non-empty blocks (in order) as one markdown-structured
    prompt (drop the '## Title' headers by passing headers=False)."""
    rendered = [
        (f"## {b.title.strip() or 'Untitled'}\n{b.content.strip()}" if headers else b.content.strip())
        for b in blocks
        if b.enabled and b.content.strip()
    ]
    return "\n\n".join(rendered).strip()


def add_block(
    blocks: Sequence[PromptBlock],
    id: str,
    title: str,
    content: str = "",
    enabled: bool = True,
) -> List[PromptBlock]:
    """Append a block (caller supplies the id so the lib stays pure)."""
    return [*blocks, PromptBlock(id=id, title=title, content=content, enabled=enabled)]


def update_block(
    blocks: Sequence[PromptBlock], id: str, patch: Optional[dict] = None
) -> List[PromptBlock]:
    """Patch one block by id (patch keys: title/content/enabled); unknown ids
    leave the list unchanged."""
    patch = patch or {}
    return [replace(b, **patch) if b.id == id else b for b in blocks]


def toggle_block(blocks: Sequence[PromptBlock], id: str) -> List[PromptBlock]:
    """Flip one block's enabled flag by id."""
    return [replace(b, enabled=not b.enabled) if b.id == id else b for b in blocks]


def remove_block(blocks: Sequence[PromptBlock], id: str) -> List[PromptBlock]:
    """Remove one block by id."""
    return [b for b in blocks if b.id != id]


def move_block(blocks: Sequence[PromptBlock], from_index: int, to_index: int) -> List[PromptBlock]:
    """Move a block (clamped; no-op when indexes are out of range or equal)."""
    n = len(blocks)
    if from_index < 0 or from_index >= n or to_index < 0 or to_index >= n or from_index == to_index:
        return list(blocks)
    nxt = list(blocks)
    nxt.insert(to_index, nxt.pop(from_index))
    return nxt


def build_report(blocks: Sequence[PromptBlock], content_type: str = "prose") -> PromptReport:
    """Assemble + count + lint in one pass — the island's live report."""
    assembled = assemble_prompt(blocks)
    tokens = _estimate_tokens(assembled, content_type) if assembled else 0
    warnings: List[str] = []
    if tokens > SYSTEM_PROMPT_SOFT_LIMIT_TOKENS:
        warnings.append(
            f"Assembled prompt is ~{tokens:,} tokens — beyond "
            f"{SYSTEM_PROMPT_SOFT_LIMIT_TOKENS:,} it starts crowding the "
            f"context window on most models."
        )
    if len(blocks) > 0 and not any(b.enabled and b.title.strip().lower() == "role" for b in blocks):
        warnings.append(
            'No enabled "Role" block — stating who the model is tends to '
            "anchor every following instruction."
        )
    if len(blocks) > 0 and assembled == "":
        warnings.append("Every block is disabled or empty — the assembled prompt is empty.")
    return PromptReport(assembled=assembled, tokens=tokens, warnings=warnings)


# ---- shareable state codec (URL-safe, compact) ------------------------------
# Triples of [enabled(0/1), title, content] keep URLs far smaller than the
# full object shape; ids are regenerated on decode (they are UI-local).

_MAX_ENCODED_LENGTH = 4000


def encode_blocks(blocks: Sequence[PromptBlock]) -> str:
    """Encode blocks to a compact base64url string; '' when blocks are empty."""
    if len(blocks) == 0:
        return ""
    compact = [[1 if b.enabled else 0, b.title, b.content] for b in blocks]
    payload = json.dumps(compact, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
    return base64.urlsafe_b64encode(payload).rstrip(b"=").decode("ascii")


def encoded_too_long(encoded: str) -> bool:
    """True when the encoded form would make an uncomfortably long URL."""
    return len(encoded) > _MAX_ENCODED_LENGTH


def decode_blocks(encoded: str) -> Optional[List[PromptBlock]]:
    """Decode `encode_blocks` output; regenerates ids (b1, b2, …). Returns
    None on malformed input — never raises; '' decodes to []."""
    if not encoded:
        return []
    try:
        padded = encoded + "=" * ((4 - len(encoded) % 4) % 4)
        raw = json.loads(base64.urlsafe_b64decode(padded).decode("utf-8"))
    except (ValueError, UnicodeDecodeError):
        return None
    if not isinstance(raw, list):
        return None
    blocks: List[PromptBlock] = []
    for i, entry in enumerate(raw):
        if not isinstance(entry, list) or len(entry) != 3:
            return None
        enabled, title, content = entry
        if not isinstance(enabled, int) or isinstance(enabled, bool):
            return None
        if not isinstance(title, str) or not isinstance(content, str):
            return None
        blocks.append(
            PromptBlock(id=f"b{i + 1}", title=title, content=content, enabled=enabled == 1)
        )
    return blocks

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →