Skip to content

System Prompt Builder — Rust source

Assemble a system prompt from ordered blocks — role, context, constraints, output format — with a live token count, soft-limit warnings, and a shareable URL. 100% client-side.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! system-prompt-builder — assemble an ordered list of prompt blocks into a
//! markdown-structured system prompt, with pure list operations, presets,
//! warnings, and a compact URL codec for shareable state.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source:   CosmoDev polyglot showcase port of the System Prompt Builder
//!           tool, ported from src/lib/systemPromptBuilder.ts (the canonical
//!           TypeScript implementation).
//! Tool page: https://dev.cosmolabs.org/tools/system-prompt-builder
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Token counting inlines the chars-per-token heuristic from
//! src/lib/tokenEstimator.ts (the original imports it). Line lengths are
//! counted in characters; the original counts UTF-16 code units, which only
//! differs for astral-plane text. The URL codec hand-rolls base64url and a
//! minimal JSON codec because the standard library ships neither.

/// One editable section of the system prompt.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct PromptBlock {
    pub id: String,
    pub title: String,
    pub content: String,
    pub enabled: bool,
}

/// A starter template from the recommended prompt skeleton.
#[derive(Debug, Clone, Copy)]
pub struct PromptPreset {
    pub id: &'static str,
    pub title: &'static str,
    pub description: &'static str,
    pub content: &'static str,
}

/// Assemble + count + lint in one pass — the island's live report.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct PromptReport {
    pub assembled: String,
    pub tokens: usize,
    pub warnings: Vec<String>,
}

/// Selects the chars-per-token rate for `build_report` ('auto' classifies
/// each line by its shape, as src/lib/tokenEstimator.ts does).
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
pub enum ContentType {
    #[default]
    Prose,
    Code,
    Json,
    Cjk,
    Auto,
}

/// Fields to patch on one block; `None` leaves that field untouched.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct PromptBlockPatch {
    pub title: Option<String>,
    pub content: Option<String>,
    pub enabled: Option<bool>,
}

/// Blocks whose assembled size starts crowding the context on most models.
pub const SYSTEM_PROMPT_SOFT_LIMIT_TOKENS: usize = 2000;

/// Ordered starter templates — the recommended skeleton of a system prompt.
pub const SYSTEM_PROMPT_PRESETS: [PromptPreset; 8] = [
    PromptPreset {
        id: "role",
        title: "Role",
        description: "Who the model is and what it optimizes for.",
        content: "You are a senior software engineer. You give correct, concise answers and say so plainly when you are unsure.",
    },
    PromptPreset {
        id: "context",
        title: "Context",
        description: "The situation the model is working in.",
        content: "The user is a developer working in a TypeScript codebase. Prefer runnable examples over prose when both work.",
    },
    PromptPreset {
        id: "constraints",
        title: "Constraints",
        description: "Hard rules the model must not break.",
        content: "- Never invent library APIs; use only the ones in the provided code.\n- Keep answers under 300 words unless asked for more.",
    },
    PromptPreset {
        id: "output-format",
        title: "Output format",
        description: "The exact shape of the answer.",
        content: "Respond with: 1) a one-line summary, 2) a fenced code block, 3) any caveats as bullet points.",
    },
    PromptPreset {
        id: "examples",
        title: "Examples",
        description: "Few-shot demonstrations of the desired behavior.",
        content: "Input: reverse \"abc\"\nOutput: \"cba\"",
    },
    PromptPreset {
        id: "tone",
        title: "Tone",
        description: "Voice and register.",
        content: "Direct and friendly. No filler openers, no apologies.",
    },
    PromptPreset {
        id: "refusal",
        title: "Refusal policy",
        description: "How to handle out-of-scope requests.",
        content: "If a request is outside your scope, say so in one sentence and suggest the closest thing you can do.",
    },
    PromptPreset {
        id: "safety",
        title: "Safety",
        description: "Guardrails for sensitive content.",
        content: "Refuse requests that could cause harm, and never echo secrets, keys, or credentials back in full.",
    },
];

// ---- token estimate (tokens figure only, from tokenEstimator.ts) ------------

#[derive(Clone, Copy)]
enum LineType {
    Prose,
    Code,
    Json,
    Cjk,
}

fn chars_per_token(t: LineType) -> f64 {
    match t {
        LineType::Prose => 4.0,
        LineType::Code => 3.5,
        LineType::Json => 3.0,
        LineType::Cjk => 1.5,
    }
}

/// CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
fn is_cjk(c: char) -> bool {
    let u = c as u32;
    (0x4E00..=0x9FFF).contains(&u) || (0x3040..=0x30FF).contains(&u) || (0xAC00..=0xD82F).contains(&u)
}

/// Classify a single line by its shape. Order: json, cjk, code, prose.
fn detect_line_type(line: &str) -> LineType {
    let trimmed = line.trim();
    // JSON-ish: opens like a JSON fragment AND carries a separator.
    let opens_like_json = matches!(trimmed.as_bytes().first(), Some(b'{') | Some(b'}') | Some(b'[') | Some(b'"'));
    if opens_like_json && (line.contains(':') || line.contains(',')) {
        return LineType::Json;
    }
    if line.chars().any(is_cjk) {
        return LineType::Cjk;
    }
    // Code: symbol-dense, or a statement terminator / block opener at EOL.
    let len = line.chars().count();
    let density = line.chars().filter(|c| "{}();=<>[]#".contains(*c)).count() as f64 / len as f64;
    if density > 0.08 || trimmed.ends_with(';') || trimmed.ends_with('{') || trimmed.ends_with('}') {
        return LineType::Code;
    }
    LineType::Prose
}

/// Sum of per-line token estimates (excludes chat framing).
fn estimate_tokens(text: &str, content_type: ContentType) -> usize {
    let forced = match content_type {
        ContentType::Auto => None,
        ContentType::Prose => Some(LineType::Prose),
        ContentType::Code => Some(LineType::Code),
        ContentType::Json => Some(LineType::Json),
        ContentType::Cjk => Some(LineType::Cjk),
    };
    // AUTO + whole-text JSON: a document that parses as JSON is json all the
    // way down.
    let whole_text_json = forced.is_none() && !text.trim().is_empty() && parse_json(text).is_some();
    let mut tokens = 0usize;
    for raw in text.split('\n') {
        let line = raw.strip_suffix('\r').unwrap_or(raw);
        if line.trim().is_empty() {
            continue;
        }
        let t = forced.unwrap_or(if whole_text_json {
            LineType::Json
        } else {
            detect_line_type(line)
        });
        // Math.max(1, Math.round(len / rate)) — round-half-up, JS style.
        let est = ((line.chars().count() as f64 / chars_per_token(t)) + 0.5).floor();
        tokens += est.max(1.0) as usize;
    }
    tokens
}

/// Group integer digits with commas the way toLocaleString('en-US') does.
fn thousands(n: usize) -> String {
    let digits = n.to_string();
    let mut out = String::with_capacity(digits.len() + digits.len() / 3);
    let len = digits.len();
    for (i, c) in digits.chars().enumerate() {
        if i > 0 && (len - i) % 3 == 0 {
            out.push(',');
        }
        out.push(c);
    }
    out
}

// ---- pure list operations ----------------------------------------------------

/// Render enabled, non-empty blocks (in order) as one markdown-structured
/// prompt (drop the "## Title" headers by passing `headers = false`).
pub fn assemble_prompt(blocks: &[PromptBlock], headers: bool) -> String {
    let rendered: Vec<String> = blocks
        .iter()
        .filter(|b| b.enabled && !b.content.trim().is_empty())
        .map(|b| {
            if headers {
                let title = {
                    let t = b.title.trim();
                    if t.is_empty() { "Untitled" } else { t }
                };
                format!("## {}\n{}", title, b.content.trim())
            } else {
                b.content.trim().to_string()
            }
        })
        .collect();
    rendered.join("\n\n").trim().to_string()
}

/// Append a block (caller supplies the id so the lib stays pure). The
/// original defaults `content` to "" and `enabled` to true.
pub fn add_block(blocks: &[PromptBlock], id: &str, title: &str, content: &str, enabled: bool) -> Vec<PromptBlock> {
    let mut next = blocks.to_vec();
    next.push(PromptBlock {
        id: id.to_string(),
        title: title.to_string(),
        content: content.to_string(),
        enabled,
    });
    next
}

/// Patch one block by id; unknown ids leave the list unchanged.
pub fn update_block(blocks: &[PromptBlock], id: &str, patch: &PromptBlockPatch) -> Vec<PromptBlock> {
    blocks
        .iter()
        .map(|b| {
            if b.id == id {
                PromptBlock {
                    id: b.id.clone(),
                    title: patch.title.clone().unwrap_or_else(|| b.title.clone()),
                    content: patch.content.clone().unwrap_or_else(|| b.content.clone()),
                    enabled: patch.enabled.unwrap_or(b.enabled),
                }
            } else {
                b.clone()
            }
        })
        .collect()
}

/// Flip one block's enabled flag by id.
pub fn toggle_block(blocks: &[PromptBlock], id: &str) -> Vec<PromptBlock> {
    blocks
        .iter()
        .map(|b| {
            if b.id == id {
                let mut next = b.clone();
                next.enabled = !next.enabled;
                next
            } else {
                b.clone()
            }
        })
        .collect()
}

/// Remove one block by id.
pub fn remove_block(blocks: &[PromptBlock], id: &str) -> Vec<PromptBlock> {
    blocks.iter().filter(|b| b.id != id).cloned().collect()
}

/// Move a block (no-op when the indexes are out of range or equal — the
/// original clamps negatives the same way because they are out of range).
pub fn move_block(blocks: &[PromptBlock], from: usize, to: usize) -> Vec<PromptBlock> {
    if from >= blocks.len() || to >= blocks.len() || from == to {
        return blocks.to_vec();
    }
    let mut next = blocks.to_vec();
    let moved = next.remove(from);
    next.insert(to, moved);
    next
}

/// Assemble + count + lint in one pass — the island's live report.
pub fn build_report(blocks: &[PromptBlock], content_type: ContentType) -> PromptReport {
    let assembled = assemble_prompt(blocks, true);
    let tokens = if assembled.is_empty() { 0 } else { estimate_tokens(&assembled, content_type) };
    let mut warnings: Vec<String> = Vec::new();
    if tokens > SYSTEM_PROMPT_SOFT_LIMIT_TOKENS {
        warnings.push(format!(
            "Assembled prompt is ~{} tokens — beyond {} it starts crowding the context window on most models.",
            thousands(tokens),
            thousands(SYSTEM_PROMPT_SOFT_LIMIT_TOKENS)
        ));
    }
    if !blocks.is_empty() && !blocks.iter().any(|b| b.enabled && b.title.trim().to_lowercase() == "role") {
        warnings.push(
            "No enabled \"Role\" block — stating who the model is tends to anchor every following instruction.".to_string(),
        );
    }
    if !blocks.is_empty() && assembled.is_empty() {
        warnings.push("Every block is disabled or empty — the assembled prompt is empty.".to_string());
    }
    PromptReport { assembled, tokens, warnings }
}

// ---- shareable state codec (URL-safe, compact) ------------------------------
// Triples of [enabled(0/1), title, content] keep URLs far smaller than the
// full object shape; ids are regenerated on decode (they are UI-local).

const MAX_ENCODED_LENGTH: usize = 4000;

const B64URL_ALPHABET: &[u8; 64] =
    b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_";

/// Standard base64 with the URL-safe alphabet and no trailing padding.
fn to_base64_url(data: &[u8]) -> String {
    let mut out = String::with_capacity(data.len().div_ceil(3) * 4);
    for chunk in data.chunks(3) {
        let b0 = chunk[0] as u32;
        let b1 = *chunk.get(1).unwrap_or(&0) as u32;
        let b2 = *chunk.get(2).unwrap_or(&0) as u32;
        let n = (b0 << 16) | (b1 << 8) | b2;
        out.push(B64URL_ALPHABET[(n >> 18 & 0x3F) as usize] as char);
        out.push(B64URL_ALPHABET[(n >> 12 & 0x3F) as usize] as char);
        if chunk.len() > 1 {
            out.push(B64URL_ALPHABET[(n >> 6 & 0x3F) as usize] as char);
        }
        if chunk.len() > 2 {
            out.push(B64URL_ALPHABET[(n & 0x3F) as usize] as char);
        }
    }
    out
}

/// Inverse of `to_base64_url`; None on characters outside the alphabet or an
/// impossible length.
fn from_base64_url(s: &str) -> Option<Vec<u8>> {
    fn val(c: u8) -> Option<u32> {
        match c {
            b'A'..=b'Z' => Some((c - b'A') as u32),
            b'a'..=b'z' => Some((c - b'a') as u32 + 26),
            b'0'..=b'9' => Some((c - b'0') as u32 + 52),
            b'-' => Some(62),
            b'_' => Some(63),
            _ => None,
        }
    }
    let bytes = s.as_bytes();
    if bytes.len() % 4 == 1 {
        return None;
    }
    let mut out = Vec::with_capacity(bytes.len() * 3 / 4);
    for chunk in bytes.chunks(4) {
        let mut n: u32 = 0;
        for (i, &c) in chunk.iter().enumerate() {
            n |= val(c)? << (18 - 6 * i);
        }
        out.push((n >> 16) as u8);
        if chunk.len() > 2 {
            out.push((n >> 8) as u8);
        }
        if chunk.len() > 3 {
            out.push(n as u8);
        }
    }
    Some(out)
}

// A minimal JSON value tree, plus the recursive-descent parser the codec and
// the whole-text-JSON probe share. It accepts exactly RFC 8259.

#[derive(Debug, Clone, PartialEq)]
enum Json {
    Null,
    Bool(bool),
    Num(f64),
    Str(String),
    Arr(Vec<Json>),
    Obj(Vec<(String, Json)>),
}

struct JsonParser<'a> {
    bytes: &'a [u8],
    pos: usize,
}

impl<'a> JsonParser<'a> {
    fn new(text: &'a str) -> Self {
        JsonParser { bytes: text.as_bytes(), pos: 0 }
    }

    fn skip_ws(&mut self) {
        while matches!(self.bytes.get(self.pos), Some(b' ' | b'\t' | b'\n' | b'\r')) {
            self.pos += 1;
        }
    }

    fn peek(&self) -> Option<u8> {
        self.bytes.get(self.pos).copied()
    }

    fn expect_lit(&mut self, lit: &str) -> Option<()> {
        if self.bytes[self.pos..].starts_with(lit.as_bytes()) {
            self.pos += lit.len();
            Some(())
        } else {
            None
        }
    }

    fn parse_value(&mut self) -> Option<Json> {
        self.skip_ws();
        match self.peek()? {
            b'n' => self.expect_lit("null").map(|_| Json::Null),
            b't' => self.expect_lit("true").map(|_| Json::Bool(true)),
            b'f' => self.expect_lit("false").map(|_| Json::Bool(false)),
            b'"' => self.parse_string().map(Json::Str),
            b'[' => self.parse_array(),
            b'{' => self.parse_object(),
            b'-' | b'0'..=b'9' => self.parse_number(),
            _ => None,
        }
    }

    fn parse_string(&mut self) -> Option<String> {
        self.pos += 1; // opening quote
        let mut out: Vec<u8> = Vec::new();
        loop {
            let c = *self.bytes.get(self.pos)?;
            self.pos += 1;
            match c {
                b'"' => return String::from_utf8(out).ok(),
                b'\\' => {
                    let esc = *self.bytes.get(self.pos)?;
                    self.pos += 1;
                    match esc {
                        b'"' => out.push(b'"'),
                        b'\\' => out.push(b'\\'),
                        b'/' => out.push(b'/'),
                        b'b' => out.push(0x08),
                        b'f' => out.push(0x0C),
                        b'n' => out.push(b'\n'),
                        b'r' => out.push(b'\r'),
                        b't' => out.push(b'\t'),
                        b'u' => {
                            let hi = self.parse_hex4()?;
                            // Combine a surrogate pair when present; a lone
                            // surrogate becomes U+FFFD (what a browser's
                            // TextDecoder would emit).
                            let cp = if (0xD800..=0xDBFF).contains(&hi) {
                                if self.bytes.get(self.pos) == Some(&b'\\')
                                    && self.bytes.get(self.pos + 1) == Some(&b'u')
                                {
                                    self.pos += 2;
                                    let lo = self.parse_hex4()?;
                                    if (0xDC00..=0xDFFF).contains(&lo) {
                                        0x10000 + ((hi - 0xD800) << 10) + (lo - 0xDC00)
                                    } else {
                                        0xFFFD
                                    }
                                } else {
                                    0xFFFD
                                }
                            } else {
                                hi
                            };
                            let c = char::from_u32(cp).unwrap_or('\u{FFFD}');
                            let mut buf = [0u8; 4];
                            out.extend_from_slice(c.encode_utf8(&mut buf).as_bytes());
                        }
                        _ => return None,
                    }
                }
                c if c < 0x20 => return None, // raw control characters are invalid
                c => out.push(c),             // multi-byte UTF-8 passes through
            }
        }
    }

    fn parse_hex4(&mut self) -> Option<u32> {
        let hex = self.bytes.get(self.pos..self.pos + 4)?;
        let s = std::str::from_utf8(hex).ok()?;
        let v = u32::from_str_radix(s, 16).ok()?;
        self.pos += 4;
        Some(v)
    }

    fn parse_number(&mut self) -> Option<Json> {
        let start = self.pos;
        if self.peek() == Some(b'-') {
            self.pos += 1;
        }
        while matches!(self.peek(), Some(b'0'..=b'9')) {
            self.pos += 1;
        }
        if self.peek() == Some(b'.') {
            self.pos += 1;
            while matches!(self.peek(), Some(b'0'..=b'9')) {
                self.pos += 1;
            }
        }
        if matches!(self.peek(), Some(b'e' | b'E')) {
            self.pos += 1;
            if matches!(self.peek(), Some(b'+' | b'-')) {
                self.pos += 1;
            }
            while matches!(self.peek(), Some(b'0'..=b'9')) {
                self.pos += 1;
            }
        }
        let text = std::str::from_utf8(&self.bytes[start..self.pos]).ok()?;
        text.parse::<f64>().ok().map(Json::Num)
    }

    fn parse_array(&mut self) -> Option<Json> {
        self.pos += 1; // '['
        let mut items = Vec::new();
        self.skip_ws();
        if self.peek() == Some(b']') {
            self.pos += 1;
            return Some(Json::Arr(items));
        }
        loop {
            items.push(self.parse_value()?);
            self.skip_ws();
            match self.peek()? {
                b',' => self.pos += 1,
                b']' => {
                    self.pos += 1;
                    return Some(Json::Arr(items));
                }
                _ => return None,
            }
        }
    }

    fn parse_object(&mut self) -> Option<Json> {
        self.pos += 1; // '{'
        let mut members = Vec::new();
        self.skip_ws();
        if self.peek() == Some(b'}') {
            self.pos += 1;
            return Some(Json::Obj(members));
        }
        loop {
            self.skip_ws();
            let key = self.parse_string()?;
            self.skip_ws();
            if self.peek()? != b':' {
                return None;
            }
            self.pos += 1;
            let value = self.parse_value()?;
            members.push((key, value));
            self.skip_ws();
            match self.peek()? {
                b',' => self.pos += 1,
                b'}' => {
                    self.pos += 1;
                    return Some(Json::Obj(members));
                }
                _ => return None,
            }
        }
    }
}

fn parse_json(text: &str) -> Option<Json> {
    let mut parser = JsonParser::new(text);
    let value = parser.parse_value()?;
    parser.skip_ws();
    if parser.pos == parser.bytes.len() {
        Some(value)
    } else {
        None
    }
}

/// Append `s` as a quoted JSON string, escaping exactly like JSON.stringify
/// (control characters below 0x20, quotes, and backslashes; everything else
/// passes through as UTF-8).
fn push_json_string(out: &mut String, s: &str) {
    out.push('"');
    for c in s.chars() {
        match c {
            '"' => out.push_str("\\\""),
            '\\' => out.push_str("\\\\"),
            '\u{08}' => out.push_str("\\b"),
            '\u{0C}' => out.push_str("\\f"),
            '\n' => out.push_str("\\n"),
            '\r' => out.push_str("\\r"),
            '\t' => out.push_str("\\t"),
            c if (c as u32) < 0x20 => out.push_str(&format!("\\u{:04x}", c as u32)),
            c => out.push(c),
        }
    }
    out.push('"');
}

/// Encode blocks to a compact base64url string; "" when blocks are empty.
pub fn encode_blocks(blocks: &[PromptBlock]) -> String {
    if blocks.is_empty() {
        return String::new();
    }
    let mut json = String::from("[");
    for (i, b) in blocks.iter().enumerate() {
        if i > 0 {
            json.push(',');
        }
        json.push('[');
        json.push(if b.enabled { '1' } else { '0' });
        json.push(',');
        push_json_string(&mut json, &b.title);
        json.push(',');
        push_json_string(&mut json, &b.content);
        json.push(']');
    }
    json.push(']');
    to_base64_url(json.as_bytes())
}

/// True when the encoded form would make an uncomfortably long URL.
pub fn encoded_too_long(encoded: &str) -> bool {
    encoded.len() > MAX_ENCODED_LENGTH
}

/// Decode `encode_blocks` output; regenerates ids (b1, b2, …). Returns None
/// on malformed input — never panics; "" decodes to an empty list.
pub fn decode_blocks(encoded: &str) -> Option<Vec<PromptBlock>> {
    if encoded.is_empty() {
        return Some(Vec::new());
    }
    let decoded = from_base64_url(encoded)?;
    let text = std::str::from_utf8(&decoded).ok()?;
    let items = match parse_json(text)? {
        Json::Arr(items) => items,
        _ => return None,
    };
    let mut blocks = Vec::with_capacity(items.len());
    for (i, entry) in items.into_iter().enumerate() {
        let triple = match entry {
            Json::Arr(t) if t.len() == 3 => t,
            _ => return None,
        };
        let (enabled, title, content) = match triple.as_slice() {
            [Json::Num(n), Json::Str(title), Json::Str(content)] => (*n == 1.0, title.clone(), content.clone()),
            _ => return None,
        };
        blocks.push(PromptBlock {
            id: format!("b{}", i + 1),
            title,
            content,
            enabled,
        });
    }
    Some(blocks)
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →