System Prompt Builder — C source
Assemble a system prompt from ordered blocks — role, context, constraints, output format — with a live token count, soft-limit warnings, and a shareable URL. 100% client-side.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/* system-prompt-builder — C port: assemble an ordered list of prompt blocks
into a markdown-structured system prompt, with pure list operations,
presets, warnings, and a compact URL codec for shareable state. C11,
stdlib only. Port of src/lib/systemPromptBuilder.ts — same defaults and
edge-case behavior; allocation failure collapses to NULL/-1 since C has
no exceptions. Token counting inlines the chars-per-token heuristic from
src/lib/tokenEstimator.ts (the original imports it); line lengths are
counted in characters (the original counts UTF-16 code units).
Ownership: every spb_block returned by these functions owns its strings;
free with spb_free_blocks(). Strings returned alone (spb_assemble_prompt,
spb_encode_blocks) are malloc'd — free() them. spb_build_report fills an
spb_report the caller releases with spb_free_report(). */
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
/* Content-type rates for the token estimate; kSpbAuto classifies each line
by its shape, as src/lib/tokenEstimator.ts does. */
typedef enum { SPB_PROSE, SPB_CODE, SPB_JSON, SPB_CJK, SPB_AUTO } spb_content_type;
/* Blocks whose assembled size starts crowding the context on most models. */
#define SPB_SOFT_LIMIT_TOKENS 2000L
/* True when the encoded form would make an uncomfortably long URL. */
#define SPB_MAX_ENCODED_LENGTH 4000
/* Ordered starter templates — the recommended skeleton of a system prompt. */
#define SPB_PRESET_COUNT 8
typedef struct {
const char *id;
const char *title;
const char *description;
const char *content;
} spb_preset;
static const spb_preset kSpbPresets[] = {
{"role", "Role", "Who the model is and what it optimizes for.",
"You are a senior software engineer. You give correct, concise answers and say so "
"plainly when you are unsure."},
{"context", "Context", "The situation the model is working in.",
"The user is a developer working in a TypeScript codebase. Prefer runnable examples "
"over prose when both work."},
{"constraints", "Constraints", "Hard rules the model must not break.",
"- Never invent library APIs; use only the ones in the provided code.\n"
"- Keep answers under 300 words unless asked for more."},
{"output-format", "Output format", "The exact shape of the answer.",
"Respond with: 1) a one-line summary, 2) a fenced code block, 3) any caveats as "
"bullet points."},
{"examples", "Examples", "Few-shot demonstrations of the desired behavior.",
"Input: reverse \"abc\"\nOutput: \"cba\""},
{"tone", "Tone", "Voice and register.", "Direct and friendly. No filler openers, no apologies."},
{"refusal", "Refusal policy", "How to handle out-of-scope requests.",
"If a request is outside your scope, say so in one sentence and suggest the closest "
"thing you can do."},
{"safety", "Safety", "Guardrails for sensitive content.",
"Refuse requests that could cause harm, and never echo secrets, keys, or credentials "
"back in full."},
};
/* Ordered starter templates, in UI order; *count is always SPB_PRESET_COUNT. */
const spb_preset *spb_presets(size_t *count) {
*count = SPB_PRESET_COUNT;
return kSpbPresets;
}
/* One editable section of the system prompt; every field is owned. */
typedef struct {
char *id;
char *title;
char *content;
int enabled;
} spb_block;
/* Fields to patch on one block; a NULL string / -1 enabled leaves that
field untouched. */
typedef struct {
const char *title;
const char *content;
int enabled; /* -1 = leave untouched, 0/1 = set */
} spb_patch;
/* Assemble + count + lint in one pass — the island's live report. */
typedef struct {
char *assembled; /* owned */
long tokens;
char **warnings; /* owned array of owned strings */
size_t warning_count;
} spb_report;
/* decode_blocks error marker: *out_n = SPB_DECODE_ERROR means malformed. */
#define SPB_DECODE_ERROR ((size_t)-1)
/* ---- small string helpers -------------------------------------------------- */
static char *spb_strdup(const char *s) {
size_t n = strlen(s) + 1;
char *out = malloc(n);
if (out) memcpy(out, s, n);
return out;
}
/* Growable byte buffer used to build strings. */
typedef struct {
char *data;
size_t len, cap;
} spb_buf;
static int spb_buf_init(spb_buf *b) {
b->cap = 64;
b->len = 0;
b->data = malloc(b->cap);
if (!b->data) return -1;
b->data[0] = '\0';
return 0;
}
static int spb_buf_reserve(spb_buf *b, size_t extra) {
if (b->len + extra + 1 <= b->cap) return 0;
while (b->cap < b->len + extra + 1) b->cap *= 2;
char *grown = realloc(b->data, b->cap);
if (!grown) return -1;
b->data = grown;
return 0;
}
static int spb_buf_push(spb_buf *b, const char *s, size_t n) {
if (spb_buf_reserve(b, n) != 0) return -1;
memcpy(b->data + b->len, s, n);
b->len += n;
b->data[b->len] = '\0';
return 0;
}
static int spb_buf_str(spb_buf *b, const char *s) {
return spb_buf_push(b, s, strlen(s));
}
/* ASCII whitespace trim of [s, s+len), mirroring the original's .trim() for
the practical cases (JS additionally trims exotic Unicode spaces). Returns
the first non-space byte and writes the trimmed length. */
static const char *spb_trim(const char *s, size_t len, size_t *len_out) {
size_t start = 0;
while (start < len && (s[start] == ' ' || s[start] == '\t' || s[start] == '\n' ||
s[start] == '\r' || s[start] == '\v' || s[start] == '\f')) {
start++;
}
size_t end = len;
while (end > start && (s[end - 1] == ' ' || s[end - 1] == '\t' || s[end - 1] == '\n' ||
s[end - 1] == '\r' || s[end - 1] == '\v' || s[end - 1] == '\f')) {
end--;
}
*len_out = end - start;
return s + start;
}
/* Count characters (UTF-8 code points), the closest C analogue of the
original's string length. */
static size_t spb_utf8_len(const char *s, size_t n) {
size_t count = 0;
for (size_t i = 0; i < n; i++) {
if ((s[i] & 0xC0) != 0x80) count++;
}
return count;
}
/* Decode the code point at s[*i] and advance; returns -1 on invalid UTF-8. */
static long spb_utf8_next(const char *s, size_t len, size_t *i) {
unsigned char c = (unsigned char)s[*i];
if (c < 0x80) { (*i)++; return c; }
if (c >= 0xC2 && c <= 0xDF && *i + 1 < len && ((unsigned char)s[*i + 1] & 0xC0) == 0x80) {
long cp = ((c & 0x1F) << 6) | ((unsigned char)s[*i + 1] & 0x3F);
*i += 2;
return cp;
}
if (c >= 0xE0 && c <= 0xEF && *i + 2 < len && ((unsigned char)s[*i + 1] & 0xC0) == 0x80 &&
((unsigned char)s[*i + 2] & 0xC0) == 0x80) {
long cp = ((c & 0x0F) << 12) | (((unsigned char)s[*i + 1] & 0x3F) << 6) |
((unsigned char)s[*i + 2] & 0x3F);
*i += 3;
return cp;
}
if (c >= 0xF0 && c <= 0xF4 && *i + 3 < len && ((unsigned char)s[*i + 1] & 0xC0) == 0x80 &&
((unsigned char)s[*i + 2] & 0xC0) == 0x80 && ((unsigned char)s[*i + 3] & 0xC0) == 0x80) {
long cp = ((c & 0x07) << 18) | (((unsigned char)s[*i + 1] & 0x3F) << 12) |
(((unsigned char)s[*i + 2] & 0x3F) << 6) | ((unsigned char)s[*i + 3] & 0x3F);
*i += 4;
return cp;
}
return -1;
}
/* ---- token estimate (tokens figure only, from tokenEstimator.ts) ----------- */
static double spb_chars_per_token(int type) {
switch (type) {
case SPB_CODE: return 3.5;
case SPB_JSON: return 3.0;
case SPB_CJK: return 1.5;
default: return 4.0;
}
}
/* CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars. */
static int spb_has_cjk(const char *s, size_t len) {
size_t i = 0;
while (i < len) {
long cp = spb_utf8_next(s, len, &i);
if (cp < 0) return 0;
if ((cp >= 0x4E00 && cp <= 0x9FFF) || (cp >= 0x3040 && cp <= 0x30FF) ||
(cp >= 0xAC00 && cp <= 0xD82F)) {
return 1;
}
}
return 0;
}
/* Classify a single line by its shape. Order: json, cjk, code, prose. */
static int spb_detect_line_type(const char *line, size_t len) {
size_t tlen;
const char *trimmed = spb_trim(line, len, &tlen);
/* JSON-ish: opens like a JSON fragment AND carries a separator. */
if (tlen > 0 && (trimmed[0] == '{' || trimmed[0] == '}' || trimmed[0] == '[' ||
trimmed[0] == '"') &&
(memchr(line, ':', len) || memchr(line, ',', len))) {
return SPB_JSON;
}
if (spb_has_cjk(line, len)) return SPB_CJK;
/* Code: symbol-dense, or a statement terminator / block opener at EOL. */
size_t symbols = 0;
for (size_t i = 0; i < len; i++) {
if (strchr("{}();=<>[]#", line[i]) && line[i] != '\0') symbols++;
}
double density = len > 0 ? (double)symbols / (double)spb_utf8_len(line, len) : 0.0;
if (density > 0.08 || (tlen > 0 && (trimmed[tlen - 1] == ';' || trimmed[tlen - 1] == '{' ||
trimmed[tlen - 1] == '}'))) {
return SPB_CODE;
}
return SPB_PROSE;
}
/* ---- minimal JSON (validate + extract the codec's string triples) ---------- */
static void spb_json_skip_ws(const char *s, size_t len, size_t *i) {
while (*i < len && (s[*i] == ' ' || s[*i] == '\t' || s[*i] == '\n' || s[*i] == '\r')) (*i)++;
}
/* Advance past a JSON string (opening quote to closing quote); 0 invalid. */
static int spb_json_skip_string(const char *s, size_t len, size_t *i) {
if (*i >= len || s[*i] != '"') return 0;
(*i)++;
while (*i < len) {
char c = s[*i];
if (c == '"') { (*i)++; return 1; }
if (c == '\\') {
(*i)++;
if (*i >= len) return 0;
char esc = s[*i];
if (esc == 'u') {
for (int k = 1; k <= 4; k++) {
char h = (*i + (size_t)k < len) ? s[*i + (size_t)k] : '\0';
if (!((h >= '0' && h <= '9') || (h >= 'a' && h <= 'f') || (h >= 'A' && h <= 'F'))) {
return 0;
}
}
*i += 5;
} else if (strchr("\"\\/bfnrt", esc) && esc != '\0') {
(*i)++;
} else {
return 0;
}
} else if ((unsigned char)c < 0x20) {
return 0;
} else {
(*i)++;
}
}
return 0;
}
/* Recursive RFC 8259 validator used for the whole-text-JSON probe. */
static int spb_json_skip_value(const char *s, size_t len, size_t *i);
static int spb_json_skip_number(const char *s, size_t len, size_t *i) {
size_t start = *i;
if (*i < len && s[*i] == '-') (*i)++;
while (*i < len && s[*i] >= '0' && s[*i] <= '9') (*i)++;
if (*i < len && s[*i] == '.') {
(*i)++;
while (*i < len && s[*i] >= '0' && s[*i] <= '9') (*i)++;
}
if (*i < len && (s[*i] == 'e' || s[*i] == 'E')) {
(*i)++;
if (*i < len && (s[*i] == '+' || s[*i] == '-')) (*i)++;
while (*i < len && s[*i] >= '0' && s[*i] <= '9') (*i)++;
}
return *i > start;
}
static int spb_json_skip_value(const char *s, size_t len, size_t *i) {
if (*i >= len) return 0;
if (s[*i] == 'n') {
if (len - *i < 4 || strncmp(s + *i, "null", 4) != 0) return 0;
*i += 4;
return 1;
}
if (s[*i] == 't') {
if (len - *i < 4 || strncmp(s + *i, "true", 4) != 0) return 0;
*i += 4;
return 1;
}
if (s[*i] == 'f') {
if (len - *i < 5 || strncmp(s + *i, "false", 5) != 0) return 0;
*i += 5;
return 1;
}
if (s[*i] == '"') return spb_json_skip_string(s, len, i);
if (s[*i] == '[') {
(*i)++;
spb_json_skip_ws(s, len, i);
if (*i < len && s[*i] == ']') { (*i)++; return 1; }
for (;;) {
if (!spb_json_skip_value(s, len, i)) return 0;
spb_json_skip_ws(s, len, i);
if (*i >= len) return 0;
if (s[*i] == ',') { (*i)++; continue; }
if (s[*i] == ']') { (*i)++; return 1; }
return 0;
}
}
if (s[*i] == '{') {
(*i)++;
spb_json_skip_ws(s, len, i);
if (*i < len && s[*i] == '}') { (*i)++; return 1; }
for (;;) {
spb_json_skip_ws(s, len, i);
if (!spb_json_skip_string(s, len, i)) return 0;
spb_json_skip_ws(s, len, i);
if (*i >= len || s[*i] != ':') return 0;
(*i)++;
if (!spb_json_skip_value(s, len, i)) return 0;
spb_json_skip_ws(s, len, i);
if (*i >= len) return 0;
if (s[*i] == ',') { (*i)++; continue; }
if (s[*i] == '}') { (*i)++; return 1; }
return 0;
}
}
return spb_json_skip_number(s, len, i);
}
/* Whole-document JSON check (isValidJson in the original). */
static int spb_json_valid(const char *s) {
size_t i = 0, tlen;
const char *t = spb_trim(s, strlen(s), &tlen);
if (tlen == 0) return 0;
if (!spb_json_skip_value(t, tlen, &i)) return 0;
spb_json_skip_ws(t, tlen, &i);
return i == tlen;
}
/* Copy a validated JSON string body into a C string; *i sits on the opening
quote. Returns a malloc'd unescaped copy, or NULL on failure. */
static char *spb_json_read_string(const char *s, size_t len, size_t *i) {
if (*i >= len || s[*i] != '"') return NULL;
(*i)++;
spb_buf out;
if (spb_buf_init(&out) != 0) return NULL;
while (*i < len) {
char c = s[*i];
if (c == '"') {
(*i)++;
return out.data; /* NUL-terminated by spb_buf_push */
}
if (c == '\\') {
(*i)++;
if (*i >= len) goto fail;
char esc = s[*i];
if (esc == 'u') {
char hex[5] = {0};
if (*i + 4 >= len) goto fail;
memcpy(hex, s + *i + 1, 4);
long cp = strtol(hex, NULL, 16);
*i += 5;
/* Surrogate pairs combine; a lone surrogate is U+FFFD (what
a browser's TextDecoder would emit). */
if (cp >= 0xD800 && cp <= 0xDBFF && *i + 1 < len && s[*i] == '\\' && s[*i + 1] == 'u') {
char hex2[5] = {0};
memcpy(hex2, s + *i + 2, 4);
long lo = strtol(hex2, NULL, 16);
if (lo >= 0xDC00 && lo <= 0xDFFF) {
cp = 0x10000 + ((cp - 0xD800) << 10) + (lo - 0xDC00);
*i += 6;
} else {
cp = 0xFFFD;
}
} else if (cp >= 0xD800 && cp <= 0xDFFF) {
cp = 0xFFFD;
}
char utf8[4];
int n;
if (cp < 0x80) { utf8[0] = (char)cp; n = 1; }
else if (cp < 0x800) {
utf8[0] = (char)(0xC0 | (cp >> 6));
utf8[1] = (char)(0x80 | (cp & 0x3F));
n = 2;
} else if (cp < 0x10000) {
utf8[0] = (char)(0xE0 | (cp >> 12));
utf8[1] = (char)(0x80 | ((cp >> 6) & 0x3F));
utf8[2] = (char)(0x80 | (cp & 0x3F));
n = 3;
} else {
utf8[0] = (char)(0xF0 | (cp >> 18));
utf8[1] = (char)(0x80 | ((cp >> 12) & 0x3F));
utf8[2] = (char)(0x80 | ((cp >> 6) & 0x3F));
utf8[3] = (char)(0x80 | (cp & 0x3F));
n = 4;
}
if (spb_buf_push(&out, utf8, (size_t)n) != 0) goto fail;
} else {
switch (esc) {
case '"': case '\\': case '/': break; /* the char itself */
case 'b': esc = '\b'; break;
case 'f': esc = '\f'; break;
case 'n': esc = '\n'; break;
case 'r': esc = '\r'; break;
case 't': esc = '\t'; break;
default: goto fail;
}
if (spb_buf_push(&out, &esc, 1) != 0) goto fail;
(*i)++;
}
} else {
if (spb_buf_push(&out, s + *i, 1) != 0) goto fail;
(*i)++;
}
}
fail:
free(out.data);
return NULL;
}
/* Append s as a quoted JSON string, escaping exactly like JSON.stringify. */
static int spb_json_write_string(spb_buf *out, const char *s) {
if (spb_buf_push(out, "\"", 1) != 0) return -1;
for (const char *p = s; *p; p++) {
unsigned char c = (unsigned char)*p;
const char *esc = NULL;
char hex[8];
switch (c) {
case '"': esc = "\\\""; break;
case '\\': esc = "\\\\"; break;
case '\b': esc = "\\b"; break;
case '\f': esc = "\\f"; break;
case '\n': esc = "\\n"; break;
case '\r': esc = "\\r"; break;
case '\t': esc = "\\t"; break;
default:
if (c < 0x20) {
snprintf(hex, sizeof hex, "\\u%04x", c);
esc = hex;
}
}
if (esc) {
if (spb_buf_str(out, esc) != 0) return -1;
} else if (spb_buf_push(out, p, 1) != 0) {
return -1;
}
}
return spb_buf_push(out, "\"", 1);
}
/* Sum of per-line token estimates (excludes chat framing). SPB_AUTO
classifies per line, with a document that parses as JSON counted as json
throughout; any other type is forced on every line. */
static long spb_estimate_tokens(const char *text, int content_type) {
size_t tlen;
spb_trim(text, strlen(text), &tlen);
int forced = content_type == SPB_AUTO ? -1 : content_type;
int whole_text_json = forced < 0 && tlen > 0 && spb_json_valid(text);
long tokens = 0;
const char *p = text;
for (;;) {
const char *nl = strchr(p, '\n');
size_t len = nl ? (size_t)(nl - p) : strlen(p);
if (len > 0 && p[len - 1] == '\r') len--; /* split on \r?\n */
size_t trimmed_len;
spb_trim(p, len, &trimmed_len);
if (trimmed_len > 0) {
int type = forced >= 0 ? forced
: (whole_text_json ? SPB_JSON : spb_detect_line_type(p, len));
/* max(1, round(len / rate)) — floor(x + 0.5) is JS rounding. */
double est = (double)spb_utf8_len(p, len) / spb_chars_per_token(type) + 0.5;
long line_tokens = (long)est;
tokens += line_tokens > 1 ? line_tokens : 1;
}
if (!nl) break;
p = nl + 1;
}
return tokens;
}
/* ---- block ownership helpers ------------------------------------------------ */
void spb_free_blocks(spb_block *blocks, size_t n); /* defined below */
void spb_free_report(spb_report *report); /* defined below */
static int spb_block_copy(spb_block *dst, const char *id, const char *title,
const char *content, int enabled) {
dst->id = spb_strdup(id);
dst->title = spb_strdup(title);
dst->content = spb_strdup(content);
dst->enabled = enabled;
if (!dst->id || !dst->title || !dst->content) return -1;
return 0;
}
/* Deep-clone a block list; NULL on allocation failure. */
static spb_block *spb_blocks_clone(const spb_block *blocks, size_t n, size_t *out_n) {
spb_block *out = calloc(n ? n : 1, sizeof *out);
if (!out) return NULL;
for (size_t i = 0; i < n; i++) {
if (spb_block_copy(&out[i], blocks[i].id, blocks[i].title, blocks[i].content,
blocks[i].enabled) != 0) {
spb_free_blocks(out, i);
return NULL;
}
}
*out_n = n;
return out;
}
void spb_free_blocks(spb_block *blocks, size_t n) {
if (!blocks) return;
for (size_t i = 0; i < n; i++) {
free(blocks[i].id);
free(blocks[i].title);
free(blocks[i].content);
}
free(blocks);
}
/* ---- public API -------------------------------------------------------------- */
/* Render enabled, non-empty blocks (in order) as one markdown-structured
prompt; headers=0 drops the "## Title" lines. Malloc'd, or NULL on
allocation failure. */
char *spb_assemble_prompt(const spb_block *blocks, size_t n, int headers) {
spb_buf out;
if (spb_buf_init(&out) != 0) return NULL;
int first = 1;
for (size_t i = 0; i < n; i++) {
if (!blocks[i].enabled) continue;
size_t clen, tlen;
const char *content = spb_trim(blocks[i].content, strlen(blocks[i].content), &clen);
if (clen == 0) continue;
if (!first && spb_buf_str(&out, "\n\n") != 0) goto fail;
first = 0;
if (headers) {
const char *title = spb_trim(blocks[i].title, strlen(blocks[i].title), &tlen);
const char *heading = (tlen > 0) ? title : "Untitled";
size_t hlen = (tlen > 0) ? tlen : strlen("Untitled");
if (spb_buf_str(&out, "## ") != 0 || spb_buf_push(&out, heading, hlen) != 0 ||
spb_buf_str(&out, "\n") != 0) {
goto fail;
}
}
if (spb_buf_push(&out, content, clen) != 0) goto fail;
}
/* trim() the joined result: leading separators cannot occur, but a
trailing one would if a render ever produced only whitespace. */
size_t flen;
const char *final = spb_trim(out.data, out.len, &flen);
if (final != out.data || flen != out.len) {
char *trimmed = spb_strdup("");
if (!trimmed) goto fail;
trimmed = realloc(trimmed, flen + 1); /* fresh copy of the trimmed slice */
if (!trimmed) goto fail;
memcpy(trimmed, final, flen);
trimmed[flen] = '\0';
free(out.data);
return trimmed;
}
return out.data;
fail:
free(out.data);
return NULL;
}
/* Append a block (caller supplies the id so the lib stays pure). Returns a
new list of n+1 owned blocks, or NULL on allocation failure. */
spb_block *spb_add_block(const spb_block *blocks, size_t n, const char *id, const char *title,
const char *content, int enabled, size_t *out_n) {
spb_block *out = calloc(n + 1, sizeof *out);
if (!out) return NULL;
for (size_t i = 0; i < n; i++) {
if (spb_block_copy(&out[i], blocks[i].id, blocks[i].title, blocks[i].content,
blocks[i].enabled) != 0) {
spb_free_blocks(out, i);
return NULL;
}
}
if (spb_block_copy(&out[n], id, title, content, enabled) != 0) {
spb_free_blocks(out, n);
return NULL;
}
*out_n = n + 1;
return out;
}
/* Patch one block by id (NULL / -1 patch fields leave that field alone);
unknown ids leave the list unchanged. */
spb_block *spb_update_block(const spb_block *blocks, size_t n, const char *id,
const spb_patch *patch, size_t *out_n) {
spb_block *out = spb_blocks_clone(blocks, n, out_n);
if (!out) return NULL;
for (size_t i = 0; i < n; i++) {
if (strcmp(out[i].id, id) != 0) continue;
if (patch->title) {
char *t = spb_strdup(patch->title);
if (!t) { spb_free_blocks(out, n); return NULL; }
free(out[i].title);
out[i].title = t;
}
if (patch->content) {
char *c = spb_strdup(patch->content);
if (!c) { spb_free_blocks(out, n); return NULL; }
free(out[i].content);
out[i].content = c;
}
if (patch->enabled >= 0) out[i].enabled = patch->enabled;
}
return out;
}
/* Flip one block's enabled flag by id. */
spb_block *spb_toggle_block(const spb_block *blocks, size_t n, const char *id, size_t *out_n) {
spb_block *out = spb_blocks_clone(blocks, n, out_n);
if (!out) return NULL;
for (size_t i = 0; i < n; i++) {
if (strcmp(out[i].id, id) == 0) out[i].enabled = !out[i].enabled;
}
return out;
}
/* Remove one block by id (the result may be shorter than the input). */
spb_block *spb_remove_block(const spb_block *blocks, size_t n, const char *id, size_t *out_n) {
spb_block *out = calloc(n ? n : 1, sizeof *out);
if (!out) return NULL;
size_t k = 0;
for (size_t i = 0; i < n; i++) {
if (strcmp(blocks[i].id, id) == 0) continue;
if (spb_block_copy(&out[k], blocks[i].id, blocks[i].title, blocks[i].content,
blocks[i].enabled) != 0) {
spb_free_blocks(out, k);
return NULL;
}
k++;
}
*out_n = k;
return out;
}
/* Move a block (no-op when the indexes are out of range or equal — the
original clamps negatives the same way because they are out of range). */
spb_block *spb_move_block(const spb_block *blocks, size_t n, size_t from, size_t to,
size_t *out_n) {
spb_block *out = spb_blocks_clone(blocks, n, out_n);
if (!out) return NULL;
if (from >= n || to >= n || from == to) return out;
spb_block moved = out[from];
if (from < to) {
memmove(&out[from], &out[from + 1], (to - from) * sizeof *out);
} else {
memmove(&out[to + 1], &out[to], (from - to) * sizeof *out);
}
out[to] = moved;
return out;
}
/* Format n with thousands separators, like toLocaleString('en-US'). */
static void spb_thousands(long n, char *buf, size_t bufsz) {
char digits[24];
snprintf(digits, sizeof digits, "%ld", n);
size_t len = strlen(digits), o = 0;
for (size_t i = 0; i < len && o + 1 < bufsz; i++) {
if (i > 0 && (len - i) % 3 == 0 && digits[i - 1] != '-') buf[o++] = ',';
buf[o++] = digits[i];
}
buf[o] = '\0';
}
/* 1 when some enabled block's trimmed, lowercased title is "role". */
static int spb_has_role_block(const spb_block *blocks, size_t n) {
for (size_t i = 0; i < n; i++) {
if (!blocks[i].enabled) continue;
size_t tlen;
const char *t = spb_trim(blocks[i].title, strlen(blocks[i].title), &tlen);
if (tlen == 4) {
char lower[5];
for (size_t k = 0; k < 4; k++) lower[k] = (t[k] >= 'A' && t[k] <= 'Z') ? t[k] + 32 : t[k];
lower[4] = '\0';
if (strcmp(lower, "role") == 0) return 1;
}
}
return 0;
}
/* Assemble + count + lint in one pass — the island's live report. Returns 0
on success, -1 on allocation failure; release *out with spb_free_report. */
int spb_build_report(const spb_block *blocks, size_t n, int content_type, spb_report *out) {
memset(out, 0, sizeof *out);
out->assembled = spb_assemble_prompt(blocks, n, 1);
if (!out->assembled) return -1;
out->tokens = out->assembled[0] != '\0' ? spb_estimate_tokens(out->assembled, content_type) : 0;
out->warnings = NULL;
out->warning_count = 0;
char **warnings = NULL;
size_t count = 0;
if (out->tokens > SPB_SOFT_LIMIT_TOKENS) {
char tok[32], lim[32];
spb_thousands(out->tokens, tok, sizeof tok);
spb_thousands(SPB_SOFT_LIMIT_TOKENS, lim, sizeof lim);
int need = snprintf(NULL, 0,
"Assembled prompt is ~%s tokens — beyond %s it starts crowding "
"the context window on most models.",
tok, lim);
char *w = malloc((size_t)need + 1);
if (!w) goto fail;
snprintf(w, (size_t)need + 1,
"Assembled prompt is ~%s tokens — beyond %s it starts crowding "
"the context window on most models.",
tok, lim);
warnings = realloc(NULL, sizeof *warnings);
if (!warnings) { free(w); goto fail; }
warnings[count++] = w;
}
if (n > 0 && !spb_has_role_block(blocks, n)) {
const char *msg =
"No enabled \"Role\" block — stating who the model is tends to anchor every "
"following instruction.";
char *w = spb_strdup(msg);
char **grown = realloc(warnings, (count + 1) * sizeof *warnings);
if (!w || !grown) { free(w); goto fail; }
warnings = grown;
warnings[count++] = w;
}
if (n > 0 && out->assembled[0] == '\0') {
char *w = spb_strdup("Every block is disabled or empty — the assembled prompt is empty.");
char **grown = realloc(warnings, (count + 1) * sizeof *warnings);
if (!w || !grown) { free(w); goto fail; }
warnings = grown;
warnings[count++] = w;
}
out->warnings = warnings;
out->warning_count = count;
return 0;
fail:
for (size_t i = 0; i < count; i++) free(warnings[i]);
free(warnings);
spb_free_report(out);
return -1;
}
void spb_free_report(spb_report *report) {
if (!report) return;
free(report->assembled);
for (size_t i = 0; i < report->warning_count; i++) free(report->warnings[i]);
free(report->warnings);
memset(report, 0, sizeof *report);
}
/* ---- shareable state codec (URL-safe, compact) ------------------------------- */
/* Triples of [enabled(0/1), title, content] keep URLs far smaller than the
full object shape; ids are regenerated on decode (they are UI-local). */
static const char kSpbB64Url[] =
"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_";
/* Standard base64 with the URL-safe alphabet and no trailing padding. */
static char *spb_to_base64_url(const char *data, size_t n) {
char *out = malloc(n / 3 * 4 + 5);
if (!out) return NULL;
size_t o = 0;
for (size_t i = 0; i < n; i += 3) {
unsigned v = (unsigned char)data[i] << 16;
if (i + 1 < n) v |= (unsigned char)data[i + 1] << 8;
if (i + 2 < n) v |= (unsigned char)data[i + 2];
out[o++] = kSpbB64Url[(v >> 18) & 0x3F];
out[o++] = kSpbB64Url[(v >> 12) & 0x3F];
if (i + 1 < n) out[o++] = kSpbB64Url[(v >> 6) & 0x3F];
if (i + 2 < n) out[o++] = kSpbB64Url[v & 0x3F];
}
out[o] = '\0';
return out;
}
static int spb_b64_val(char c) {
if (c >= 'A' && c <= 'Z') return c - 'A';
if (c >= 'a' && c <= 'z') return c - 'a' + 26;
if (c >= '0' && c <= '9') return c - '0' + 52;
if (c == '-') return 62;
if (c == '_') return 63;
return -1;
}
/* Inverse of spb_to_base64_url; NULL on invalid characters or length. */
static unsigned char *spb_from_base64_url(const char *s, size_t *out_n) {
size_t n = strlen(s);
if (n % 4 == 1) return NULL;
unsigned char *out = malloc(n / 4 * 3 + 3);
if (!out) return NULL;
size_t o = 0;
for (size_t i = 0; i < n; i += 4) {
size_t m = n - i < 4 ? n - i : 4;
int v[4];
for (size_t k = 0; k < m; k++) {
v[k] = spb_b64_val(s[i + k]);
if (v[k] < 0) { free(out); return NULL; }
}
out[o++] = (unsigned char)((v[0] << 2) | ((m > 1 ? v[1] : 0) >> 4));
if (m > 2) out[o++] = (unsigned char)(((v[1] & 0xF) << 4) | (v[2] >> 2));
if (m > 3) out[o++] = (unsigned char)(((v[2] & 0x3) << 2) | v[3]);
}
*out_n = o;
return out;
}
/* 1 when every byte of [s, s+n) is valid UTF-8 (TextDecoder would reject the
decode otherwise and the original returns null). */
static int spb_utf8_valid(const unsigned char *s, size_t n) {
size_t i = 0;
while (i < n) {
long cp = spb_utf8_next((const char *)s, n, &i);
if (cp < 0) return 0;
}
return 1;
}
/* Encode blocks to a compact base64url JSON string; malloc'd "" when blocks
are empty; NULL on allocation failure. */
char *spb_encode_blocks(const spb_block *blocks, size_t n) {
if (n == 0) return spb_strdup("");
spb_buf json;
if (spb_buf_init(&json) != 0) return NULL;
int ok = spb_buf_push(&json, "[", 1) == 0;
for (size_t i = 0; ok && i < n; i++) {
ok = (i == 0 || spb_buf_push(&json, ",", 1) == 0) &&
spb_buf_push(&json, "[", 1) == 0 &&
spb_buf_push(&json, blocks[i].enabled ? "1" : "0", 1) == 0 &&
spb_buf_push(&json, ",", 1) == 0 &&
spb_json_write_string(&json, blocks[i].title) == 0 &&
spb_buf_push(&json, ",", 1) == 0 &&
spb_json_write_string(&json, blocks[i].content) == 0 &&
spb_buf_push(&json, "]", 1) == 0;
}
ok = ok && spb_buf_push(&json, "]", 1) == 0;
if (!ok) { free(json.data); return NULL; }
char *encoded = spb_to_base64_url(json.data, json.len);
free(json.data);
return encoded;
}
/* True when the encoded form would make an uncomfortably long URL. */
int spb_encoded_too_long(const char *encoded) {
return strlen(encoded) > SPB_MAX_ENCODED_LENGTH;
}
/* Decode spb_encode_blocks output; regenerates ids (b1, b2, …). Returns an
owned block list and sets *out_n to its length ('' input yields NULL with
*out_n = 0); malformed input yields NULL with *out_n = SPB_DECODE_ERROR. */
spb_block *spb_decode_blocks(const char *encoded, size_t *out_n) {
*out_n = 0;
if (encoded[0] == '\0') return NULL;
size_t byte_n;
unsigned char *bytes = spb_from_base64_url(encoded, &byte_n);
if (!bytes) { *out_n = SPB_DECODE_ERROR; return NULL; }
if (!spb_utf8_valid(bytes, byte_n)) { free(bytes); *out_n = SPB_DECODE_ERROR; return NULL; }
const char *s = (const char *)bytes;
size_t len = byte_n, i = 0;
spb_block *blocks = NULL;
size_t count = 0;
spb_json_skip_ws(s, len, &i);
if (i >= len || s[i] != '[') goto malformed;
i++;
spb_json_skip_ws(s, len, &i);
if (i < len && s[i] == ']') {
i++;
} else {
for (;;) {
if (i >= len || s[i] != '[') goto malformed;
i++;
spb_json_skip_ws(s, len, &i);
/* enabled: any JSON number, compared to 1 like the original's === 1 */
size_t num_start = i;
if (!spb_json_skip_number(s, len, &i)) goto malformed;
char num[64];
size_t num_len = i - num_start < sizeof num - 1 ? i - num_start : sizeof num - 1;
memcpy(num, s + num_start, num_len);
num[num_len] = '\0';
double enabled_val = strtod(num, NULL);
spb_json_skip_ws(s, len, &i);
if (i >= len || s[i] != ',') goto malformed;
i++;
spb_json_skip_ws(s, len, &i);
char *title = spb_json_read_string(s, len, &i);
if (!title) goto malformed;
spb_json_skip_ws(s, len, &i);
if (i >= len || s[i] != ',') { free(title); goto malformed; }
i++;
spb_json_skip_ws(s, len, &i);
char *content = spb_json_read_string(s, len, &i);
if (!content) { free(title); goto malformed; }
spb_json_skip_ws(s, len, &i);
if (i >= len || s[i] != ']') { free(title); free(content); goto malformed; }
i++;
spb_block *grown = realloc(blocks, (count + 1) * sizeof *blocks);
if (!grown) { free(title); free(content); goto oom; }
blocks = grown;
char id[24];
snprintf(id, sizeof id, "b%zu", count + 1);
blocks[count].id = spb_strdup(id);
blocks[count].title = title;
blocks[count].content = content;
blocks[count].enabled = enabled_val == 1.0;
if (!blocks[count].id) { free(title); free(content); goto oom; }
count++;
spb_json_skip_ws(s, len, &i);
if (i < len && s[i] == ',') { i++; continue; }
if (i < len && s[i] == ']') { i++; break; }
goto malformed;
}
}
spb_json_skip_ws(s, len, &i);
if (i != len) goto malformed;
free(bytes);
*out_n = count;
return blocks;
oom:
spb_free_blocks(blocks, count);
free(bytes);
*out_n = SPB_DECODE_ERROR;
return NULL;
malformed:
spb_free_blocks(blocks, count);
free(bytes);
*out_n = SPB_DECODE_ERROR;
return NULL;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →