System Prompt Builder — Rust source
Assemble a system prompt from ordered blocks — role, context, constraints, output format — with a live token count, soft-limit warnings, and a shareable URL. 100% client-side.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! system-prompt-builder — assemble an ordered list of prompt blocks into a
//! markdown-structured system prompt, with pure list operations, presets,
//! warnings, and a compact URL codec for shareable state.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source: CosmoDev polyglot showcase port of the System Prompt Builder
//! tool, ported from src/lib/systemPromptBuilder.ts (the canonical
//! TypeScript implementation).
//! Tool page: https://dev.cosmolabs.org/tools/system-prompt-builder
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Token counting inlines the chars-per-token heuristic from
//! src/lib/tokenEstimator.ts (the original imports it). Line lengths are
//! counted in characters; the original counts UTF-16 code units, which only
//! differs for astral-plane text. The URL codec hand-rolls base64url and a
//! minimal JSON codec because the standard library ships neither.
/// One editable section of the system prompt.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct PromptBlock {
pub id: String,
pub title: String,
pub content: String,
pub enabled: bool,
}
/// A starter template from the recommended prompt skeleton.
#[derive(Debug, Clone, Copy)]
pub struct PromptPreset {
pub id: &'static str,
pub title: &'static str,
pub description: &'static str,
pub content: &'static str,
}
/// Assemble + count + lint in one pass — the island's live report.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct PromptReport {
pub assembled: String,
pub tokens: usize,
pub warnings: Vec<String>,
}
/// Selects the chars-per-token rate for `build_report` ('auto' classifies
/// each line by its shape, as src/lib/tokenEstimator.ts does).
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
pub enum ContentType {
#[default]
Prose,
Code,
Json,
Cjk,
Auto,
}
/// Fields to patch on one block; `None` leaves that field untouched.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct PromptBlockPatch {
pub title: Option<String>,
pub content: Option<String>,
pub enabled: Option<bool>,
}
/// Blocks whose assembled size starts crowding the context on most models.
pub const SYSTEM_PROMPT_SOFT_LIMIT_TOKENS: usize = 2000;
/// Ordered starter templates — the recommended skeleton of a system prompt.
pub const SYSTEM_PROMPT_PRESETS: [PromptPreset; 8] = [
PromptPreset {
id: "role",
title: "Role",
description: "Who the model is and what it optimizes for.",
content: "You are a senior software engineer. You give correct, concise answers and say so plainly when you are unsure.",
},
PromptPreset {
id: "context",
title: "Context",
description: "The situation the model is working in.",
content: "The user is a developer working in a TypeScript codebase. Prefer runnable examples over prose when both work.",
},
PromptPreset {
id: "constraints",
title: "Constraints",
description: "Hard rules the model must not break.",
content: "- Never invent library APIs; use only the ones in the provided code.\n- Keep answers under 300 words unless asked for more.",
},
PromptPreset {
id: "output-format",
title: "Output format",
description: "The exact shape of the answer.",
content: "Respond with: 1) a one-line summary, 2) a fenced code block, 3) any caveats as bullet points.",
},
PromptPreset {
id: "examples",
title: "Examples",
description: "Few-shot demonstrations of the desired behavior.",
content: "Input: reverse \"abc\"\nOutput: \"cba\"",
},
PromptPreset {
id: "tone",
title: "Tone",
description: "Voice and register.",
content: "Direct and friendly. No filler openers, no apologies.",
},
PromptPreset {
id: "refusal",
title: "Refusal policy",
description: "How to handle out-of-scope requests.",
content: "If a request is outside your scope, say so in one sentence and suggest the closest thing you can do.",
},
PromptPreset {
id: "safety",
title: "Safety",
description: "Guardrails for sensitive content.",
content: "Refuse requests that could cause harm, and never echo secrets, keys, or credentials back in full.",
},
];
// ---- token estimate (tokens figure only, from tokenEstimator.ts) ------------
#[derive(Clone, Copy)]
enum LineType {
Prose,
Code,
Json,
Cjk,
}
fn chars_per_token(t: LineType) -> f64 {
match t {
LineType::Prose => 4.0,
LineType::Code => 3.5,
LineType::Json => 3.0,
LineType::Cjk => 1.5,
}
}
/// CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
fn is_cjk(c: char) -> bool {
let u = c as u32;
(0x4E00..=0x9FFF).contains(&u) || (0x3040..=0x30FF).contains(&u) || (0xAC00..=0xD82F).contains(&u)
}
/// Classify a single line by its shape. Order: json, cjk, code, prose.
fn detect_line_type(line: &str) -> LineType {
let trimmed = line.trim();
// JSON-ish: opens like a JSON fragment AND carries a separator.
let opens_like_json = matches!(trimmed.as_bytes().first(), Some(b'{') | Some(b'}') | Some(b'[') | Some(b'"'));
if opens_like_json && (line.contains(':') || line.contains(',')) {
return LineType::Json;
}
if line.chars().any(is_cjk) {
return LineType::Cjk;
}
// Code: symbol-dense, or a statement terminator / block opener at EOL.
let len = line.chars().count();
let density = line.chars().filter(|c| "{}();=<>[]#".contains(*c)).count() as f64 / len as f64;
if density > 0.08 || trimmed.ends_with(';') || trimmed.ends_with('{') || trimmed.ends_with('}') {
return LineType::Code;
}
LineType::Prose
}
/// Sum of per-line token estimates (excludes chat framing).
fn estimate_tokens(text: &str, content_type: ContentType) -> usize {
let forced = match content_type {
ContentType::Auto => None,
ContentType::Prose => Some(LineType::Prose),
ContentType::Code => Some(LineType::Code),
ContentType::Json => Some(LineType::Json),
ContentType::Cjk => Some(LineType::Cjk),
};
// AUTO + whole-text JSON: a document that parses as JSON is json all the
// way down.
let whole_text_json = forced.is_none() && !text.trim().is_empty() && parse_json(text).is_some();
let mut tokens = 0usize;
for raw in text.split('\n') {
let line = raw.strip_suffix('\r').unwrap_or(raw);
if line.trim().is_empty() {
continue;
}
let t = forced.unwrap_or(if whole_text_json {
LineType::Json
} else {
detect_line_type(line)
});
// Math.max(1, Math.round(len / rate)) — round-half-up, JS style.
let est = ((line.chars().count() as f64 / chars_per_token(t)) + 0.5).floor();
tokens += est.max(1.0) as usize;
}
tokens
}
/// Group integer digits with commas the way toLocaleString('en-US') does.
fn thousands(n: usize) -> String {
let digits = n.to_string();
let mut out = String::with_capacity(digits.len() + digits.len() / 3);
let len = digits.len();
for (i, c) in digits.chars().enumerate() {
if i > 0 && (len - i) % 3 == 0 {
out.push(',');
}
out.push(c);
}
out
}
// ---- pure list operations ----------------------------------------------------
/// Render enabled, non-empty blocks (in order) as one markdown-structured
/// prompt (drop the "## Title" headers by passing `headers = false`).
pub fn assemble_prompt(blocks: &[PromptBlock], headers: bool) -> String {
let rendered: Vec<String> = blocks
.iter()
.filter(|b| b.enabled && !b.content.trim().is_empty())
.map(|b| {
if headers {
let title = {
let t = b.title.trim();
if t.is_empty() { "Untitled" } else { t }
};
format!("## {}\n{}", title, b.content.trim())
} else {
b.content.trim().to_string()
}
})
.collect();
rendered.join("\n\n").trim().to_string()
}
/// Append a block (caller supplies the id so the lib stays pure). The
/// original defaults `content` to "" and `enabled` to true.
pub fn add_block(blocks: &[PromptBlock], id: &str, title: &str, content: &str, enabled: bool) -> Vec<PromptBlock> {
let mut next = blocks.to_vec();
next.push(PromptBlock {
id: id.to_string(),
title: title.to_string(),
content: content.to_string(),
enabled,
});
next
}
/// Patch one block by id; unknown ids leave the list unchanged.
pub fn update_block(blocks: &[PromptBlock], id: &str, patch: &PromptBlockPatch) -> Vec<PromptBlock> {
blocks
.iter()
.map(|b| {
if b.id == id {
PromptBlock {
id: b.id.clone(),
title: patch.title.clone().unwrap_or_else(|| b.title.clone()),
content: patch.content.clone().unwrap_or_else(|| b.content.clone()),
enabled: patch.enabled.unwrap_or(b.enabled),
}
} else {
b.clone()
}
})
.collect()
}
/// Flip one block's enabled flag by id.
pub fn toggle_block(blocks: &[PromptBlock], id: &str) -> Vec<PromptBlock> {
blocks
.iter()
.map(|b| {
if b.id == id {
let mut next = b.clone();
next.enabled = !next.enabled;
next
} else {
b.clone()
}
})
.collect()
}
/// Remove one block by id.
pub fn remove_block(blocks: &[PromptBlock], id: &str) -> Vec<PromptBlock> {
blocks.iter().filter(|b| b.id != id).cloned().collect()
}
/// Move a block (no-op when the indexes are out of range or equal — the
/// original clamps negatives the same way because they are out of range).
pub fn move_block(blocks: &[PromptBlock], from: usize, to: usize) -> Vec<PromptBlock> {
if from >= blocks.len() || to >= blocks.len() || from == to {
return blocks.to_vec();
}
let mut next = blocks.to_vec();
let moved = next.remove(from);
next.insert(to, moved);
next
}
/// Assemble + count + lint in one pass — the island's live report.
pub fn build_report(blocks: &[PromptBlock], content_type: ContentType) -> PromptReport {
let assembled = assemble_prompt(blocks, true);
let tokens = if assembled.is_empty() { 0 } else { estimate_tokens(&assembled, content_type) };
let mut warnings: Vec<String> = Vec::new();
if tokens > SYSTEM_PROMPT_SOFT_LIMIT_TOKENS {
warnings.push(format!(
"Assembled prompt is ~{} tokens — beyond {} it starts crowding the context window on most models.",
thousands(tokens),
thousands(SYSTEM_PROMPT_SOFT_LIMIT_TOKENS)
));
}
if !blocks.is_empty() && !blocks.iter().any(|b| b.enabled && b.title.trim().to_lowercase() == "role") {
warnings.push(
"No enabled \"Role\" block — stating who the model is tends to anchor every following instruction.".to_string(),
);
}
if !blocks.is_empty() && assembled.is_empty() {
warnings.push("Every block is disabled or empty — the assembled prompt is empty.".to_string());
}
PromptReport { assembled, tokens, warnings }
}
// ---- shareable state codec (URL-safe, compact) ------------------------------
// Triples of [enabled(0/1), title, content] keep URLs far smaller than the
// full object shape; ids are regenerated on decode (they are UI-local).
const MAX_ENCODED_LENGTH: usize = 4000;
const B64URL_ALPHABET: &[u8; 64] =
b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_";
/// Standard base64 with the URL-safe alphabet and no trailing padding.
fn to_base64_url(data: &[u8]) -> String {
let mut out = String::with_capacity(data.len().div_ceil(3) * 4);
for chunk in data.chunks(3) {
let b0 = chunk[0] as u32;
let b1 = *chunk.get(1).unwrap_or(&0) as u32;
let b2 = *chunk.get(2).unwrap_or(&0) as u32;
let n = (b0 << 16) | (b1 << 8) | b2;
out.push(B64URL_ALPHABET[(n >> 18 & 0x3F) as usize] as char);
out.push(B64URL_ALPHABET[(n >> 12 & 0x3F) as usize] as char);
if chunk.len() > 1 {
out.push(B64URL_ALPHABET[(n >> 6 & 0x3F) as usize] as char);
}
if chunk.len() > 2 {
out.push(B64URL_ALPHABET[(n & 0x3F) as usize] as char);
}
}
out
}
/// Inverse of `to_base64_url`; None on characters outside the alphabet or an
/// impossible length.
fn from_base64_url(s: &str) -> Option<Vec<u8>> {
fn val(c: u8) -> Option<u32> {
match c {
b'A'..=b'Z' => Some((c - b'A') as u32),
b'a'..=b'z' => Some((c - b'a') as u32 + 26),
b'0'..=b'9' => Some((c - b'0') as u32 + 52),
b'-' => Some(62),
b'_' => Some(63),
_ => None,
}
}
let bytes = s.as_bytes();
if bytes.len() % 4 == 1 {
return None;
}
let mut out = Vec::with_capacity(bytes.len() * 3 / 4);
for chunk in bytes.chunks(4) {
let mut n: u32 = 0;
for (i, &c) in chunk.iter().enumerate() {
n |= val(c)? << (18 - 6 * i);
}
out.push((n >> 16) as u8);
if chunk.len() > 2 {
out.push((n >> 8) as u8);
}
if chunk.len() > 3 {
out.push(n as u8);
}
}
Some(out)
}
// A minimal JSON value tree, plus the recursive-descent parser the codec and
// the whole-text-JSON probe share. It accepts exactly RFC 8259.
#[derive(Debug, Clone, PartialEq)]
enum Json {
Null,
Bool(bool),
Num(f64),
Str(String),
Arr(Vec<Json>),
Obj(Vec<(String, Json)>),
}
struct JsonParser<'a> {
bytes: &'a [u8],
pos: usize,
}
impl<'a> JsonParser<'a> {
fn new(text: &'a str) -> Self {
JsonParser { bytes: text.as_bytes(), pos: 0 }
}
fn skip_ws(&mut self) {
while matches!(self.bytes.get(self.pos), Some(b' ' | b'\t' | b'\n' | b'\r')) {
self.pos += 1;
}
}
fn peek(&self) -> Option<u8> {
self.bytes.get(self.pos).copied()
}
fn expect_lit(&mut self, lit: &str) -> Option<()> {
if self.bytes[self.pos..].starts_with(lit.as_bytes()) {
self.pos += lit.len();
Some(())
} else {
None
}
}
fn parse_value(&mut self) -> Option<Json> {
self.skip_ws();
match self.peek()? {
b'n' => self.expect_lit("null").map(|_| Json::Null),
b't' => self.expect_lit("true").map(|_| Json::Bool(true)),
b'f' => self.expect_lit("false").map(|_| Json::Bool(false)),
b'"' => self.parse_string().map(Json::Str),
b'[' => self.parse_array(),
b'{' => self.parse_object(),
b'-' | b'0'..=b'9' => self.parse_number(),
_ => None,
}
}
fn parse_string(&mut self) -> Option<String> {
self.pos += 1; // opening quote
let mut out: Vec<u8> = Vec::new();
loop {
let c = *self.bytes.get(self.pos)?;
self.pos += 1;
match c {
b'"' => return String::from_utf8(out).ok(),
b'\\' => {
let esc = *self.bytes.get(self.pos)?;
self.pos += 1;
match esc {
b'"' => out.push(b'"'),
b'\\' => out.push(b'\\'),
b'/' => out.push(b'/'),
b'b' => out.push(0x08),
b'f' => out.push(0x0C),
b'n' => out.push(b'\n'),
b'r' => out.push(b'\r'),
b't' => out.push(b'\t'),
b'u' => {
let hi = self.parse_hex4()?;
// Combine a surrogate pair when present; a lone
// surrogate becomes U+FFFD (what a browser's
// TextDecoder would emit).
let cp = if (0xD800..=0xDBFF).contains(&hi) {
if self.bytes.get(self.pos) == Some(&b'\\')
&& self.bytes.get(self.pos + 1) == Some(&b'u')
{
self.pos += 2;
let lo = self.parse_hex4()?;
if (0xDC00..=0xDFFF).contains(&lo) {
0x10000 + ((hi - 0xD800) << 10) + (lo - 0xDC00)
} else {
0xFFFD
}
} else {
0xFFFD
}
} else {
hi
};
let c = char::from_u32(cp).unwrap_or('\u{FFFD}');
let mut buf = [0u8; 4];
out.extend_from_slice(c.encode_utf8(&mut buf).as_bytes());
}
_ => return None,
}
}
c if c < 0x20 => return None, // raw control characters are invalid
c => out.push(c), // multi-byte UTF-8 passes through
}
}
}
fn parse_hex4(&mut self) -> Option<u32> {
let hex = self.bytes.get(self.pos..self.pos + 4)?;
let s = std::str::from_utf8(hex).ok()?;
let v = u32::from_str_radix(s, 16).ok()?;
self.pos += 4;
Some(v)
}
fn parse_number(&mut self) -> Option<Json> {
let start = self.pos;
if self.peek() == Some(b'-') {
self.pos += 1;
}
while matches!(self.peek(), Some(b'0'..=b'9')) {
self.pos += 1;
}
if self.peek() == Some(b'.') {
self.pos += 1;
while matches!(self.peek(), Some(b'0'..=b'9')) {
self.pos += 1;
}
}
if matches!(self.peek(), Some(b'e' | b'E')) {
self.pos += 1;
if matches!(self.peek(), Some(b'+' | b'-')) {
self.pos += 1;
}
while matches!(self.peek(), Some(b'0'..=b'9')) {
self.pos += 1;
}
}
let text = std::str::from_utf8(&self.bytes[start..self.pos]).ok()?;
text.parse::<f64>().ok().map(Json::Num)
}
fn parse_array(&mut self) -> Option<Json> {
self.pos += 1; // '['
let mut items = Vec::new();
self.skip_ws();
if self.peek() == Some(b']') {
self.pos += 1;
return Some(Json::Arr(items));
}
loop {
items.push(self.parse_value()?);
self.skip_ws();
match self.peek()? {
b',' => self.pos += 1,
b']' => {
self.pos += 1;
return Some(Json::Arr(items));
}
_ => return None,
}
}
}
fn parse_object(&mut self) -> Option<Json> {
self.pos += 1; // '{'
let mut members = Vec::new();
self.skip_ws();
if self.peek() == Some(b'}') {
self.pos += 1;
return Some(Json::Obj(members));
}
loop {
self.skip_ws();
let key = self.parse_string()?;
self.skip_ws();
if self.peek()? != b':' {
return None;
}
self.pos += 1;
let value = self.parse_value()?;
members.push((key, value));
self.skip_ws();
match self.peek()? {
b',' => self.pos += 1,
b'}' => {
self.pos += 1;
return Some(Json::Obj(members));
}
_ => return None,
}
}
}
}
fn parse_json(text: &str) -> Option<Json> {
let mut parser = JsonParser::new(text);
let value = parser.parse_value()?;
parser.skip_ws();
if parser.pos == parser.bytes.len() {
Some(value)
} else {
None
}
}
/// Append `s` as a quoted JSON string, escaping exactly like JSON.stringify
/// (control characters below 0x20, quotes, and backslashes; everything else
/// passes through as UTF-8).
fn push_json_string(out: &mut String, s: &str) {
out.push('"');
for c in s.chars() {
match c {
'"' => out.push_str("\\\""),
'\\' => out.push_str("\\\\"),
'\u{08}' => out.push_str("\\b"),
'\u{0C}' => out.push_str("\\f"),
'\n' => out.push_str("\\n"),
'\r' => out.push_str("\\r"),
'\t' => out.push_str("\\t"),
c if (c as u32) < 0x20 => out.push_str(&format!("\\u{:04x}", c as u32)),
c => out.push(c),
}
}
out.push('"');
}
/// Encode blocks to a compact base64url string; "" when blocks are empty.
pub fn encode_blocks(blocks: &[PromptBlock]) -> String {
if blocks.is_empty() {
return String::new();
}
let mut json = String::from("[");
for (i, b) in blocks.iter().enumerate() {
if i > 0 {
json.push(',');
}
json.push('[');
json.push(if b.enabled { '1' } else { '0' });
json.push(',');
push_json_string(&mut json, &b.title);
json.push(',');
push_json_string(&mut json, &b.content);
json.push(']');
}
json.push(']');
to_base64_url(json.as_bytes())
}
/// True when the encoded form would make an uncomfortably long URL.
pub fn encoded_too_long(encoded: &str) -> bool {
encoded.len() > MAX_ENCODED_LENGTH
}
/// Decode `encode_blocks` output; regenerates ids (b1, b2, …). Returns None
/// on malformed input — never panics; "" decodes to an empty list.
pub fn decode_blocks(encoded: &str) -> Option<Vec<PromptBlock>> {
if encoded.is_empty() {
return Some(Vec::new());
}
let decoded = from_base64_url(encoded)?;
let text = std::str::from_utf8(&decoded).ok()?;
let items = match parse_json(text)? {
Json::Arr(items) => items,
_ => return None,
};
let mut blocks = Vec::with_capacity(items.len());
for (i, entry) in items.into_iter().enumerate() {
let triple = match entry {
Json::Arr(t) if t.len() == 3 => t,
_ => return None,
};
let (enabled, title, content) = match triple.as_slice() {
[Json::Num(n), Json::Str(title), Json::Str(content)] => (*n == 1.0, title.clone(), content.clone()),
_ => return None,
};
blocks.push(PromptBlock {
id: format!("b{}", i + 1),
title,
content,
enabled,
});
}
Some(blocks)
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →