Context Window Planner — C source
Paste your system prompt, docs, and history — see how they fill any model's context window, with overflow warnings and output headroom.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/*
* Context Window Planner — plan labeled prompt sections against a model's
* context window.
*
* Language: C (C11, standard library only)
* Source: CosmoDev polyglot showcase port of the Context Window Planner
* tool, ported from src/lib/contextPlanner.ts (the canonical
* TypeScript implementation).
* Live at: https://dev.cosmolabs.org/tools/context-window-planner
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; never crashes on any input (public API returns
* plain values, no allocation required beyond caller-provided storage).
* - Functionally equivalent to the TS reference: same inputs -> same outputs.
* - Self-contained: std only (no PCRE, no jansson — hand-rolled UTF-8
* decoding and a strict JSON validator instead).
*
* Port notes: the TS lib delegates to two siblings — estimateTokens from
* src/lib/tokenEstimator.ts and fitsWindow from src/lib/ai/models.ts (which
* defaults to the bundled pricing snapshot, src/data/ai-models.json). A
* dependency-free port cannot load that file, so the estimator is inlined
* below in the exact form the planner uses it (estimateTokens(text).tokens,
* auto content type — the full heuristic lives in the token-estimator port),
* window math is inlined from fitsWindow and models is an explicit parameter,
* never re-derived.
*
* Faithfulness notes (the places C's stdlib silently differs from JS):
* - Length: TS's String.length counts UTF-16 code units (an astral-plane
* character — emoji, rare CJK ext-B ideographs — counts as 2). C strings
* are bytes, so utf16_len() decodes UTF-8 and counts the same unit.
* - JSON: no stdlib JSON parser exists, so this port ships a small strict
* recursive-descent validator (is_valid_json) implementing exactly the
* grammar JSON.parse accepts. (Multibyte UTF-8 inside strings never
* contains an ASCII byte, so byte-level scanning is safe.)
* - Rounding: js_round() is floor(x + 0.5) — JS Math.round rounds halfway
* cases up, unlike C's round(), which rounds halfway away from zero
* (identical for the non-negative numbers used here, but stated anyway).
*/
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
#include <string.h>
/* ---------- core types ---------- */
/* One labeled block of the prompt (system / docs / history / ...).
* Mirrors the TS PlanSection interface. */
typedef struct {
const char *label; /* section label, e.g. "system" */
const char *text; /* the section's raw (NUL-terminated, UTF-8) text */
} cwp_section;
/* The subset of the TS AiModel record the planner reads. Production code
* passes the full snapshot entry; only these fields influence the plan. */
typedef struct {
const char *id; /* model id, e.g. "beta-pro" */
long long context_window; /* total context window in tokens */
long long max_output; /* the model's output cap (informational) */
} cwp_model;
/* Result of cwp_plan_window. Field-for-field twin of the TS WindowPlan
* interface (same keys, same meanings). */
typedef struct {
const char *id; /* the model id planned against */
long long input_tokens; /* sum of per-section token estimates */
long long context_window; /* the model's context window */
long long free; /* window - input; negative on overflow */
bool fits; /* raw fit: free >= 0 */
bool output_reserve_ok; /* room for the output reserve: free >= reserve */
long long max_output; /* the model's output cap (informational) */
} cwp_window_plan;
/* Content classification of a single line. The planner only needs each
* type's chars-per-token rate (mirrors CHARS_PER_TOKEN in
* src/lib/tokenEstimator.ts: prose 4, code 3.5, json 3, cjk 1.5). */
typedef enum {
CWP_PROSE = 0,
CWP_CODE,
CWP_JSON,
CWP_CJK
} cwp_content_type;
static double cwp_chars_per_token(cwp_content_type t) {
switch (t) {
case CWP_PROSE: return 4.0;
case CWP_CODE: return 3.5;
case CWP_JSON: return 3.0;
case CWP_CJK: return 1.5;
}
return 4.0; /* unreachable */
}
/* Sample table for standalone use (mirrors the shared test fixtures).
* Production code passes the model snapshot instead. */
static const cwp_model CWP_SAMPLE_MODELS[] = {
{ "alpha-mini", 200000, 10000 },
{ "beta-pro", 1000000, 10000 },
{ "gamma-open", 100000, 10000 },
};
static const size_t CWP_SAMPLE_MODELS_LEN =
sizeof CWP_SAMPLE_MODELS / sizeof CWP_SAMPLE_MODELS[0];
/* ---------- UTF-8 / UTF-16 helpers ---------- */
/* Decodes one UTF-8 code point at `p` (bounded by `end`), advances `*next`
* past it, and returns the code point. An invalid or truncated lead byte
* counts as one unit and advances by one byte. */
static uint32_t cwp_decode(const unsigned char *p, const unsigned char *end,
const unsigned char **next) {
uint32_t cp;
size_t step;
if (*p < 0x80) { cp = *p; step = 1; }
else if ((*p & 0xE0) == 0xC0) { cp = *p & 0x1F; step = 2; }
else if ((*p & 0xF0) == 0xE0) { cp = *p & 0x0F; step = 3; }
else if ((*p & 0xF8) == 0xF0) { cp = *p & 0x07; step = 4; }
else { cp = *p; step = 1; } /* invalid byte */
if ((size_t)(end - p) < step) { cp = *p; step = 1; }
for (size_t k = 1; k < step; k++)
cp = (cp << 6) | (p[k] & 0x3F);
*next = p + step;
return cp;
}
/* Length of the UTF-8 range [begin, end) in UTF-16 code units — the unit
* TS's String.length counts. BMP code points are one unit, astral-plane
* ones two. */
static long long cwp_utf16_len_range(const char *begin, const char *end) {
long long units = 0;
const unsigned char *p = (const unsigned char *)begin;
const unsigned char *stop = (const unsigned char *)end;
while (p < stop) {
const unsigned char *next;
uint32_t cp = cwp_decode(p, stop, &next);
units += cp > 0xFFFF ? 2 : 1;
p = next;
}
return units;
}
/* Reports whether the range [begin, end) contains a CJK ideograph
* (U+4E00–U+9FFF), kana (U+3040–U+30FF), or a Hangul syllable
* (U+AC00–U+D7AF). Mirrors CJK_RE in the TS lib. */
static bool cwp_has_cjk_range(const char *begin, const char *end) {
const unsigned char *p = (const unsigned char *)begin;
const unsigned char *stop = (const unsigned char *)end;
while (p < stop) {
const unsigned char *next;
uint32_t cp = cwp_decode(p, stop, &next);
if ((cp >= 0x4E00 && cp <= 0x9FFF) ||
(cp >= 0x3040 && cp <= 0x30FF) ||
(cp >= 0xAC00 && cp <= 0xD7AF))
return true;
p = next;
}
return false;
}
/* Reports whether `c` is one of the code-flavored symbols counted by
* CODE_SYMBOL_RE ({}();=<>[]#). */
static bool cwp_is_code_symbol(char c) {
return c == '{' || c == '}' || c == '(' || c == ')' || c == ';' ||
c == '=' || c == '<' || c == '>' || c == '[' || c == ']' || c == '#';
}
/* ---------- line helpers ---------- */
/* JS Math.round: halfway cases round up. floor(x + 0.5) over the
* non-negative inputs used here. */
static long long cwp_js_round(double x) {
return (long long)(x + 0.5);
}
static bool cwp_is_space(char c) {
return c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '\f' || c == '\v';
}
/* Classifies a single line (bounded by [begin, end)) by its shape.
* Order: json, cjk, code, prose. Inlined from detectLineType() in
* src/lib/tokenEstimator.ts. */
static cwp_content_type cwp_detect_line_type(const char *begin, const char *end) {
const char *t0 = begin, *t1 = end; /* trim both ends */
while (t0 < t1 && cwp_is_space(*t0)) t0++;
while (t1 > t0 && cwp_is_space(t1[-1])) t1--;
bool starts_jsonish = t0 < t1 &&
(*t0 == '{' || *t0 == '}' || *t0 == '[' || *t0 == '"');
if (starts_jsonish) {
for (const char *p = begin; p < end; p++)
if (*p == ':' || *p == ',') return CWP_JSON;
}
if (cwp_has_cjk_range(begin, end)) return CWP_CJK;
/* Code: symbol-dense, or a statement terminator / block opener at EOL. */
long long length = cwp_utf16_len_range(begin, end);
long long symbols = 0;
for (const char *p = begin; p < end; p++)
if (cwp_is_code_symbol(*p)) symbols++;
double density = length > 0 ? (double)symbols / (double)length : 0.0;
char last = t1 > t0 ? t1[-1] : '\0';
if (density > 0.08 || last == ';' || last == '{' || last == '}')
return CWP_CODE;
return CWP_PROSE;
}
/* ---------- strict JSON validator (the grammar JSON.parse accepts) ---------- */
typedef struct {
const unsigned char *bytes;
size_t pos, len;
} cwp_json_parser;
static void cwp_json_skip_ws(cwp_json_parser *p) {
while (p->pos < p->len) {
unsigned char b = p->bytes[p->pos];
if (b == ' ' || b == '\t' || b == '\n' || b == '\r') p->pos++;
else break;
}
}
static unsigned char cwp_json_peek(const cwp_json_parser *p) {
return p->pos < p->len ? p->bytes[p->pos] : 0;
}
static bool cwp_json_eat(cwp_json_parser *p, unsigned char b) {
if (cwp_json_peek(p) == b && p->pos < p->len) { p->pos++; return true; }
return false;
}
static bool cwp_json_literal(cwp_json_parser *p, const char *lit) {
size_t n = strlen(lit);
if (p->len - p->pos >= n && memcmp(p->bytes + p->pos, lit, n) == 0) {
p->pos += n;
return true;
}
return false;
}
static bool cwp_json_number(cwp_json_parser *p) {
cwp_json_eat(p, '-');
if (cwp_json_peek(p) == '0') {
p->pos++;
} else if (cwp_json_peek(p) >= '1' && cwp_json_peek(p) <= '9') {
while (cwp_json_peek(p) >= '0' && cwp_json_peek(p) <= '9') p->pos++;
} else {
return false;
}
if (cwp_json_peek(p) == '.') {
p->pos++;
size_t digits = 0;
while (cwp_json_peek(p) >= '0' && cwp_json_peek(p) <= '9') { p->pos++; digits++; }
if (digits == 0) return false;
}
if (cwp_json_peek(p) == 'e' || cwp_json_peek(p) == 'E') {
p->pos++;
if (cwp_json_peek(p) == '+' || cwp_json_peek(p) == '-') p->pos++;
size_t digits = 0;
while (cwp_json_peek(p) >= '0' && cwp_json_peek(p) <= '9') { p->pos++; digits++; }
if (digits == 0) return false;
}
return true;
}
/* string := '"' (escape | any byte >= 0x20)* '"'
* escape := '\' ("\"" | "/" | "\" | 'b' | 'f' | 'n' | 'r' | 't' | 'u' hex4) */
static bool cwp_json_string(cwp_json_parser *p) {
if (!cwp_json_eat(p, '"')) return false;
while (p->pos < p->len) {
unsigned char b = p->bytes[p->pos];
if (b == '"') { p->pos++; return true; }
if (b == '\\') {
p->pos++;
if (p->pos >= p->len) return false;
unsigned char esc = p->bytes[p->pos++];
switch (esc) {
case '"': case '/': case '\\': case 'b':
case 'f': case 'n': case 'r': case 't':
break;
case 'u':
for (int i = 0; i < 4; i++) {
unsigned char h = cwp_json_peek(p);
bool hex = (h >= '0' && h <= '9') || (h >= 'a' && h <= 'f') ||
(h >= 'A' && h <= 'F');
if (!hex) return false;
p->pos++;
}
break;
default:
return false;
}
} else if (b < 0x20) {
return false; /* raw control characters not allowed in strings */
} else {
p->pos++;
}
}
return false; /* unterminated string */
}
static bool cwp_json_value(cwp_json_parser *p);
static bool cwp_json_object(cwp_json_parser *p) {
if (!cwp_json_eat(p, '{')) return false;
cwp_json_skip_ws(p);
if (cwp_json_eat(p, '}')) return true;
for (;;) {
if (!cwp_json_string(p)) return false;
cwp_json_skip_ws(p);
if (!cwp_json_eat(p, ':')) return false;
if (!cwp_json_value(p)) return false;
cwp_json_skip_ws(p);
if (cwp_json_eat(p, ',')) cwp_json_skip_ws(p);
else return cwp_json_eat(p, '}');
}
}
static bool cwp_json_array(cwp_json_parser *p) {
if (!cwp_json_eat(p, '[')) return false;
cwp_json_skip_ws(p);
if (cwp_json_eat(p, ']')) return true;
for (;;) {
if (!cwp_json_value(p)) return false;
cwp_json_skip_ws(p);
if (cwp_json_eat(p, ',')) cwp_json_skip_ws(p);
else return cwp_json_eat(p, ']');
}
}
static bool cwp_json_value(cwp_json_parser *p) {
cwp_json_skip_ws(p);
switch (cwp_json_peek(p)) {
case '{': return cwp_json_object(p);
case '[': return cwp_json_array(p);
case '"': return cwp_json_string(p);
case '-': case '0': case '1': case '2': case '3': case '4':
case '5': case '6': case '7': case '8': case '9':
return cwp_json_number(p);
case 't': return cwp_json_literal(p, "true");
case 'f': return cwp_json_literal(p, "false");
case 'n': return cwp_json_literal(p, "null");
default: return false;
}
}
/* Whole-text JSON gate: a document that parses as JSON is json all the way
* down. Mirrors isValidJson() (JSON.parse in a try/catch); empty/whitespace
* text is not. */
static bool cwp_is_valid_json(const char *text) {
const unsigned char *p0 = (const unsigned char *)text;
size_t len = strlen(text), i = 0;
while (i < len && cwp_is_space((char)p0[i])) i++;
if (i == len) return false; /* whitespace-only */
cwp_json_parser jp = { p0, 0, len };
if (!cwp_json_value(&jp)) return false;
cwp_json_skip_ws(&jp);
return jp.pos == jp.len; /* reject trailing garbage */
}
/* ---------- estimator + planner ---------- */
/* Token count of `text` under auto content detection — exactly the slice of
* estimateTokens() the planner consumes (.tokens): per non-empty line,
* max(1, round(utf16_len / chars_per_token)). Framing tokens are the
* caller's job. */
static long long cwp_estimate_tokens(const char *text) {
bool whole_text_json = cwp_is_valid_json(text);
long long tokens = 0;
const char *line = text;
while (*line) {
const char *nl = strchr(line, '\n');
const char *end = nl ? nl : line + strlen(line);
const char *trimmed_end = end;
if (trimmed_end > line && trimmed_end[-1] == '\r') trimmed_end--; /* CRLF */
/* blank line? */
const char *p = line;
while (p < trimmed_end && cwp_is_space(*p)) p++;
if (p < trimmed_end) {
cwp_content_type t = whole_text_json
? CWP_JSON
: cwp_detect_line_type(line, trimmed_end);
long long line_tokens = cwp_js_round(
(double)cwp_utf16_len_range(line, trimmed_end) /
cwp_chars_per_token(t));
tokens += line_tokens >= 1 ? line_tokens : 1;
}
line = nl ? nl + 1 : end;
}
return tokens;
}
/* Sum of per-section token estimates (framing tokens are the caller's job).
* Mirrors inputTokenTotal() in the TS lib. */
long long cwp_input_token_total(const cwp_section *sections, size_t count) {
long long total = 0;
for (size_t i = 0; i < count; i++)
total += cwp_estimate_tokens(sections[i].text);
return total;
}
/* Plan one section set against one model's context window. Returns false
* for an unknown model id (window math is fitsWindow's, never re-derived).
* Mirrors planWindow() in the TS lib. */
bool cwp_plan_window(const cwp_section *sections, size_t count,
const char *model_id, long long output_reserve,
const cwp_model *models, size_t models_len,
cwp_window_plan *out) {
long long input_tokens = cwp_input_token_total(sections, count);
const cwp_model *m = NULL;
for (size_t i = 0; i < models_len; i++)
if (strcmp(models[i].id, model_id) == 0) { m = &models[i]; break; }
if (!m) return false;
long long free = m->context_window - input_tokens;
out->id = model_id;
out->input_tokens = input_tokens;
out->context_window = m->context_window;
out->free = free;
out->fits = free >= 0;
out->output_reserve_ok = free >= output_reserve;
out->max_output = m->max_output;
return true;
}
/* Plan against several models; unknown ids are dropped from the result.
* Returns how many plans were written to out (<= out_len). Mirrors planAll()
* in the TS lib. */
size_t cwp_plan_all(const cwp_section *sections, size_t count,
const char *const *model_ids, size_t ids_len,
long long output_reserve,
const cwp_model *models, size_t models_len,
cwp_window_plan *out, size_t out_len) {
size_t written = 0;
for (size_t i = 0; i < ids_len && written < out_len; i++)
if (cwp_plan_window(sections, count, model_ids[i], output_reserve,
models, models_len, &out[written]))
written++;
return written;
}
/* ---------- tests (showcase-only; the canonical suite lives in src/lib) ---------- */
#ifdef CWP_TEST
#include <assert.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
/* A 1600-char single line of 'a' is pure prose: 1600 / 4 = 400 tokens. */
static cwp_section g_sections[2];
static char g_line_a[1601];
static void build_sections(void) {
memset(g_line_a, 'a', 1600);
g_line_a[1600] = '\0';
g_sections[0].label = "sys";
g_sections[0].text = g_line_a;
g_sections[1].label = "docs";
g_sections[1].text = g_line_a;
}
int main(void) {
build_sections();
/* input totals */
assert(cwp_input_token_total(g_sections, 2) == 800);
assert(cwp_input_token_total(NULL, 0) == 0);
cwp_section empty = { "sys", "" };
assert(cwp_input_token_total(&empty, 1) == 0);
cwp_window_plan p;
/* plans two 400-token sections against beta-pro */
assert(cwp_plan_window(g_sections, 2, "beta-pro", 0,
CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
assert(strcmp(p.id, "beta-pro") == 0);
assert(p.input_tokens == 800);
assert(p.context_window == 1000000);
assert(p.free == 999200);
assert(p.fits);
assert(p.output_reserve_ok);
assert(p.max_output == 10000);
/* reserve larger than free leaves raw fit true */
assert(cwp_plan_window(g_sections, 2, "beta-pro", 1000000,
CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
assert(p.fits);
assert(!p.output_reserve_ok);
/* reserve exactly equal to free is ok */
assert(cwp_plan_window(g_sections, 2, "beta-pro", 999200,
CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
assert(p.output_reserve_ok);
/* smaller window leaves 199,200 free */
assert(cwp_plan_window(g_sections, 2, "alpha-mini", 0,
CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
assert(p.context_window == 200000);
assert(p.free == 199200);
assert(p.fits);
/* unknown model id returns failure */
assert(!cwp_plan_window(g_sections, 2, "ghost", 0,
CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
/* no sections: full window free */
assert(cwp_plan_window(NULL, 0, "beta-pro", 0,
CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
assert(p.input_tokens == 0);
assert(p.free == 1000000);
assert(p.fits);
/* overflow: fits false, reserve false */
char *big = malloc(4400001);
memset(big, 'z', 4400000);
big[4400000] = '\0';
cwp_section big_sec = { "big", big };
assert(cwp_plan_window(&big_sec, 1, "beta-pro", 0,
CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
assert(p.input_tokens == 1100000);
assert(p.free == -100000);
assert(!p.fits);
assert(!p.output_reserve_ok);
free(big);
/* plan_all drops unknown ids and keeps order */
const char *ids[] = { "beta-pro", "alpha-mini", "ghost" };
cwp_window_plan plans[3];
size_t n = cwp_plan_all(g_sections, 2, ids, 3, 0,
CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN,
plans, 3);
assert(n == 2);
assert(strcmp(plans[0].id, "beta-pro") == 0);
assert(strcmp(plans[1].id, "alpha-mini") == 0);
assert(plans[1].free == 199200);
assert(cwp_plan_all(g_sections, 2, NULL, 0, 0,
CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, plans, 3) == 0);
printf("context-window-planner (C): all tests passed\n");
return 0;
}
#endif /* CWP_TEST */
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →