Skip to content

Context Window Planner — C source

Paste your system prompt, docs, and history — see how they fill any model's context window, with overflow warnings and output headroom.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * Context Window Planner — plan labeled prompt sections against a model's
 * context window.
 *
 * Language: C (C11, standard library only)
 * Source:   CosmoDev polyglot showcase port of the Context Window Planner
 *           tool, ported from src/lib/contextPlanner.ts (the canonical
 *           TypeScript implementation).
 * Live at:  https://dev.cosmolabs.org/tools/context-window-planner
 * License:  display source — part of CosmoDev's polyglot tool pages.
 *
 * Design goals:
 *   - Pure + deterministic; never crashes on any input (public API returns
 *     plain values, no allocation required beyond caller-provided storage).
 *   - Functionally equivalent to the TS reference: same inputs -> same outputs.
 *   - Self-contained: std only (no PCRE, no jansson — hand-rolled UTF-8
 *     decoding and a strict JSON validator instead).
 *
 * Port notes: the TS lib delegates to two siblings — estimateTokens from
 * src/lib/tokenEstimator.ts and fitsWindow from src/lib/ai/models.ts (which
 * defaults to the bundled pricing snapshot, src/data/ai-models.json). A
 * dependency-free port cannot load that file, so the estimator is inlined
 * below in the exact form the planner uses it (estimateTokens(text).tokens,
 * auto content type — the full heuristic lives in the token-estimator port),
 * window math is inlined from fitsWindow and models is an explicit parameter,
 * never re-derived.
 *
 * Faithfulness notes (the places C's stdlib silently differs from JS):
 *   - Length: TS's String.length counts UTF-16 code units (an astral-plane
 *     character — emoji, rare CJK ext-B ideographs — counts as 2). C strings
 *     are bytes, so utf16_len() decodes UTF-8 and counts the same unit.
 *   - JSON: no stdlib JSON parser exists, so this port ships a small strict
 *     recursive-descent validator (is_valid_json) implementing exactly the
 *     grammar JSON.parse accepts. (Multibyte UTF-8 inside strings never
 *     contains an ASCII byte, so byte-level scanning is safe.)
 *   - Rounding: js_round() is floor(x + 0.5) — JS Math.round rounds halfway
 *     cases up, unlike C's round(), which rounds halfway away from zero
 *     (identical for the non-negative numbers used here, but stated anyway).
 */

#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
#include <string.h>

/* ---------- core types ---------- */

/* One labeled block of the prompt (system / docs / history / ...).
 * Mirrors the TS PlanSection interface. */
typedef struct {
    const char *label; /* section label, e.g. "system" */
    const char *text;  /* the section's raw (NUL-terminated, UTF-8) text */
} cwp_section;

/* The subset of the TS AiModel record the planner reads. Production code
 * passes the full snapshot entry; only these fields influence the plan. */
typedef struct {
    const char *id;       /* model id, e.g. "beta-pro" */
    long long context_window; /* total context window in tokens */
    long long max_output;     /* the model's output cap (informational) */
} cwp_model;

/* Result of cwp_plan_window. Field-for-field twin of the TS WindowPlan
 * interface (same keys, same meanings). */
typedef struct {
    const char *id;       /* the model id planned against */
    long long input_tokens;   /* sum of per-section token estimates */
    long long context_window; /* the model's context window */
    long long free;           /* window - input; negative on overflow */
    bool fits;                /* raw fit: free >= 0 */
    bool output_reserve_ok;   /* room for the output reserve: free >= reserve */
    long long max_output;     /* the model's output cap (informational) */
} cwp_window_plan;

/* Content classification of a single line. The planner only needs each
 * type's chars-per-token rate (mirrors CHARS_PER_TOKEN in
 * src/lib/tokenEstimator.ts: prose 4, code 3.5, json 3, cjk 1.5). */
typedef enum {
    CWP_PROSE = 0,
    CWP_CODE,
    CWP_JSON,
    CWP_CJK
} cwp_content_type;

static double cwp_chars_per_token(cwp_content_type t) {
    switch (t) {
    case CWP_PROSE: return 4.0;
    case CWP_CODE:  return 3.5;
    case CWP_JSON:  return 3.0;
    case CWP_CJK:   return 1.5;
    }
    return 4.0; /* unreachable */
}

/* Sample table for standalone use (mirrors the shared test fixtures).
 * Production code passes the model snapshot instead. */
static const cwp_model CWP_SAMPLE_MODELS[] = {
    { "alpha-mini", 200000,   10000 },
    { "beta-pro",   1000000,  10000 },
    { "gamma-open", 100000,   10000 },
};
static const size_t CWP_SAMPLE_MODELS_LEN =
    sizeof CWP_SAMPLE_MODELS / sizeof CWP_SAMPLE_MODELS[0];

/* ---------- UTF-8 / UTF-16 helpers ---------- */

/* Decodes one UTF-8 code point at `p` (bounded by `end`), advances `*next`
 * past it, and returns the code point. An invalid or truncated lead byte
 * counts as one unit and advances by one byte. */
static uint32_t cwp_decode(const unsigned char *p, const unsigned char *end,
                           const unsigned char **next) {
    uint32_t cp;
    size_t step;
    if (*p < 0x80) {                 cp = *p;        step = 1; }
    else if ((*p & 0xE0) == 0xC0) {  cp = *p & 0x1F; step = 2; }
    else if ((*p & 0xF0) == 0xE0) {  cp = *p & 0x0F; step = 3; }
    else if ((*p & 0xF8) == 0xF0) {  cp = *p & 0x07; step = 4; }
    else {                           cp = *p;        step = 1; } /* invalid byte */
    if ((size_t)(end - p) < step) { cp = *p; step = 1; }
    for (size_t k = 1; k < step; k++)
        cp = (cp << 6) | (p[k] & 0x3F);
    *next = p + step;
    return cp;
}

/* Length of the UTF-8 range [begin, end) in UTF-16 code units — the unit
 * TS's String.length counts. BMP code points are one unit, astral-plane
 * ones two. */
static long long cwp_utf16_len_range(const char *begin, const char *end) {
    long long units = 0;
    const unsigned char *p = (const unsigned char *)begin;
    const unsigned char *stop = (const unsigned char *)end;
    while (p < stop) {
        const unsigned char *next;
        uint32_t cp = cwp_decode(p, stop, &next);
        units += cp > 0xFFFF ? 2 : 1;
        p = next;
    }
    return units;
}

/* Reports whether the range [begin, end) contains a CJK ideograph
 * (U+4E00–U+9FFF), kana (U+3040–U+30FF), or a Hangul syllable
 * (U+AC00–U+D7AF). Mirrors CJK_RE in the TS lib. */
static bool cwp_has_cjk_range(const char *begin, const char *end) {
    const unsigned char *p = (const unsigned char *)begin;
    const unsigned char *stop = (const unsigned char *)end;
    while (p < stop) {
        const unsigned char *next;
        uint32_t cp = cwp_decode(p, stop, &next);
        if ((cp >= 0x4E00 && cp <= 0x9FFF) ||
            (cp >= 0x3040 && cp <= 0x30FF) ||
            (cp >= 0xAC00 && cp <= 0xD7AF))
            return true;
        p = next;
    }
    return false;
}

/* Reports whether `c` is one of the code-flavored symbols counted by
 * CODE_SYMBOL_RE ({}();=<>[]#). */
static bool cwp_is_code_symbol(char c) {
    return c == '{' || c == '}' || c == '(' || c == ')' || c == ';' ||
           c == '=' || c == '<' || c == '>' || c == '[' || c == ']' || c == '#';
}

/* ---------- line helpers ---------- */

/* JS Math.round: halfway cases round up. floor(x + 0.5) over the
 * non-negative inputs used here. */
static long long cwp_js_round(double x) {
    return (long long)(x + 0.5);
}

static bool cwp_is_space(char c) {
    return c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '\f' || c == '\v';
}

/* Classifies a single line (bounded by [begin, end)) by its shape.
 * Order: json, cjk, code, prose. Inlined from detectLineType() in
 * src/lib/tokenEstimator.ts. */
static cwp_content_type cwp_detect_line_type(const char *begin, const char *end) {
    const char *t0 = begin, *t1 = end; /* trim both ends */
    while (t0 < t1 && cwp_is_space(*t0)) t0++;
    while (t1 > t0 && cwp_is_space(t1[-1])) t1--;

    bool starts_jsonish = t0 < t1 &&
        (*t0 == '{' || *t0 == '}' || *t0 == '[' || *t0 == '"');
    if (starts_jsonish) {
        for (const char *p = begin; p < end; p++)
            if (*p == ':' || *p == ',') return CWP_JSON;
    }
    if (cwp_has_cjk_range(begin, end)) return CWP_CJK;
    /* Code: symbol-dense, or a statement terminator / block opener at EOL. */
    long long length = cwp_utf16_len_range(begin, end);
    long long symbols = 0;
    for (const char *p = begin; p < end; p++)
        if (cwp_is_code_symbol(*p)) symbols++;
    double density = length > 0 ? (double)symbols / (double)length : 0.0;
    char last = t1 > t0 ? t1[-1] : '\0';
    if (density > 0.08 || last == ';' || last == '{' || last == '}')
        return CWP_CODE;
    return CWP_PROSE;
}

/* ---------- strict JSON validator (the grammar JSON.parse accepts) ---------- */

typedef struct {
    const unsigned char *bytes;
    size_t pos, len;
} cwp_json_parser;

static void cwp_json_skip_ws(cwp_json_parser *p) {
    while (p->pos < p->len) {
        unsigned char b = p->bytes[p->pos];
        if (b == ' ' || b == '\t' || b == '\n' || b == '\r') p->pos++;
        else break;
    }
}

static unsigned char cwp_json_peek(const cwp_json_parser *p) {
    return p->pos < p->len ? p->bytes[p->pos] : 0;
}

static bool cwp_json_eat(cwp_json_parser *p, unsigned char b) {
    if (cwp_json_peek(p) == b && p->pos < p->len) { p->pos++; return true; }
    return false;
}

static bool cwp_json_literal(cwp_json_parser *p, const char *lit) {
    size_t n = strlen(lit);
    if (p->len - p->pos >= n && memcmp(p->bytes + p->pos, lit, n) == 0) {
        p->pos += n;
        return true;
    }
    return false;
}

static bool cwp_json_number(cwp_json_parser *p) {
    cwp_json_eat(p, '-');
    if (cwp_json_peek(p) == '0') {
        p->pos++;
    } else if (cwp_json_peek(p) >= '1' && cwp_json_peek(p) <= '9') {
        while (cwp_json_peek(p) >= '0' && cwp_json_peek(p) <= '9') p->pos++;
    } else {
        return false;
    }
    if (cwp_json_peek(p) == '.') {
        p->pos++;
        size_t digits = 0;
        while (cwp_json_peek(p) >= '0' && cwp_json_peek(p) <= '9') { p->pos++; digits++; }
        if (digits == 0) return false;
    }
    if (cwp_json_peek(p) == 'e' || cwp_json_peek(p) == 'E') {
        p->pos++;
        if (cwp_json_peek(p) == '+' || cwp_json_peek(p) == '-') p->pos++;
        size_t digits = 0;
        while (cwp_json_peek(p) >= '0' && cwp_json_peek(p) <= '9') { p->pos++; digits++; }
        if (digits == 0) return false;
    }
    return true;
}

/* string := '"' (escape | any byte >= 0x20)* '"'
 * escape := '\' ("\"" | "/" | "\" | 'b' | 'f' | 'n' | 'r' | 't' | 'u' hex4) */
static bool cwp_json_string(cwp_json_parser *p) {
    if (!cwp_json_eat(p, '"')) return false;
    while (p->pos < p->len) {
        unsigned char b = p->bytes[p->pos];
        if (b == '"') { p->pos++; return true; }
        if (b == '\\') {
            p->pos++;
            if (p->pos >= p->len) return false;
            unsigned char esc = p->bytes[p->pos++];
            switch (esc) {
            case '"': case '/': case '\\': case 'b':
            case 'f': case 'n': case 'r': case 't':
                break;
            case 'u':
                for (int i = 0; i < 4; i++) {
                    unsigned char h = cwp_json_peek(p);
                    bool hex = (h >= '0' && h <= '9') || (h >= 'a' && h <= 'f') ||
                               (h >= 'A' && h <= 'F');
                    if (!hex) return false;
                    p->pos++;
                }
                break;
            default:
                return false;
            }
        } else if (b < 0x20) {
            return false; /* raw control characters not allowed in strings */
        } else {
            p->pos++;
        }
    }
    return false; /* unterminated string */
}

static bool cwp_json_value(cwp_json_parser *p);

static bool cwp_json_object(cwp_json_parser *p) {
    if (!cwp_json_eat(p, '{')) return false;
    cwp_json_skip_ws(p);
    if (cwp_json_eat(p, '}')) return true;
    for (;;) {
        if (!cwp_json_string(p)) return false;
        cwp_json_skip_ws(p);
        if (!cwp_json_eat(p, ':')) return false;
        if (!cwp_json_value(p)) return false;
        cwp_json_skip_ws(p);
        if (cwp_json_eat(p, ',')) cwp_json_skip_ws(p);
        else return cwp_json_eat(p, '}');
    }
}

static bool cwp_json_array(cwp_json_parser *p) {
    if (!cwp_json_eat(p, '[')) return false;
    cwp_json_skip_ws(p);
    if (cwp_json_eat(p, ']')) return true;
    for (;;) {
        if (!cwp_json_value(p)) return false;
        cwp_json_skip_ws(p);
        if (cwp_json_eat(p, ',')) cwp_json_skip_ws(p);
        else return cwp_json_eat(p, ']');
    }
}

static bool cwp_json_value(cwp_json_parser *p) {
    cwp_json_skip_ws(p);
    switch (cwp_json_peek(p)) {
    case '{': return cwp_json_object(p);
    case '[': return cwp_json_array(p);
    case '"': return cwp_json_string(p);
    case '-': case '0': case '1': case '2': case '3': case '4':
    case '5': case '6': case '7': case '8': case '9':
        return cwp_json_number(p);
    case 't': return cwp_json_literal(p, "true");
    case 'f': return cwp_json_literal(p, "false");
    case 'n': return cwp_json_literal(p, "null");
    default:  return false;
    }
}

/* Whole-text JSON gate: a document that parses as JSON is json all the way
 * down. Mirrors isValidJson() (JSON.parse in a try/catch); empty/whitespace
 * text is not. */
static bool cwp_is_valid_json(const char *text) {
    const unsigned char *p0 = (const unsigned char *)text;
    size_t len = strlen(text), i = 0;
    while (i < len && cwp_is_space((char)p0[i])) i++;
    if (i == len) return false; /* whitespace-only */
    cwp_json_parser jp = { p0, 0, len };
    if (!cwp_json_value(&jp)) return false;
    cwp_json_skip_ws(&jp);
    return jp.pos == jp.len; /* reject trailing garbage */
}

/* ---------- estimator + planner ---------- */

/* Token count of `text` under auto content detection — exactly the slice of
 * estimateTokens() the planner consumes (.tokens): per non-empty line,
 * max(1, round(utf16_len / chars_per_token)). Framing tokens are the
 * caller's job. */
static long long cwp_estimate_tokens(const char *text) {
    bool whole_text_json = cwp_is_valid_json(text);
    long long tokens = 0;
    const char *line = text;
    while (*line) {
        const char *nl = strchr(line, '\n');
        const char *end = nl ? nl : line + strlen(line);
        const char *trimmed_end = end;
        if (trimmed_end > line && trimmed_end[-1] == '\r') trimmed_end--; /* CRLF */
        /* blank line? */
        const char *p = line;
        while (p < trimmed_end && cwp_is_space(*p)) p++;
        if (p < trimmed_end) {
            cwp_content_type t = whole_text_json
                ? CWP_JSON
                : cwp_detect_line_type(line, trimmed_end);
            long long line_tokens = cwp_js_round(
                (double)cwp_utf16_len_range(line, trimmed_end) /
                cwp_chars_per_token(t));
            tokens += line_tokens >= 1 ? line_tokens : 1;
        }
        line = nl ? nl + 1 : end;
    }
    return tokens;
}

/* Sum of per-section token estimates (framing tokens are the caller's job).
 * Mirrors inputTokenTotal() in the TS lib. */
long long cwp_input_token_total(const cwp_section *sections, size_t count) {
    long long total = 0;
    for (size_t i = 0; i < count; i++)
        total += cwp_estimate_tokens(sections[i].text);
    return total;
}

/* Plan one section set against one model's context window. Returns false
 * for an unknown model id (window math is fitsWindow's, never re-derived).
 * Mirrors planWindow() in the TS lib. */
bool cwp_plan_window(const cwp_section *sections, size_t count,
                     const char *model_id, long long output_reserve,
                     const cwp_model *models, size_t models_len,
                     cwp_window_plan *out) {
    long long input_tokens = cwp_input_token_total(sections, count);
    const cwp_model *m = NULL;
    for (size_t i = 0; i < models_len; i++)
        if (strcmp(models[i].id, model_id) == 0) { m = &models[i]; break; }
    if (!m) return false;
    long long free = m->context_window - input_tokens;
    out->id = model_id;
    out->input_tokens = input_tokens;
    out->context_window = m->context_window;
    out->free = free;
    out->fits = free >= 0;
    out->output_reserve_ok = free >= output_reserve;
    out->max_output = m->max_output;
    return true;
}

/* Plan against several models; unknown ids are dropped from the result.
 * Returns how many plans were written to out (<= out_len). Mirrors planAll()
 * in the TS lib. */
size_t cwp_plan_all(const cwp_section *sections, size_t count,
                    const char *const *model_ids, size_t ids_len,
                    long long output_reserve,
                    const cwp_model *models, size_t models_len,
                    cwp_window_plan *out, size_t out_len) {
    size_t written = 0;
    for (size_t i = 0; i < ids_len && written < out_len; i++)
        if (cwp_plan_window(sections, count, model_ids[i], output_reserve,
                            models, models_len, &out[written]))
            written++;
    return written;
}

/* ---------- tests (showcase-only; the canonical suite lives in src/lib) ---------- */

#ifdef CWP_TEST

#include <assert.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

/* A 1600-char single line of 'a' is pure prose: 1600 / 4 = 400 tokens. */
static cwp_section g_sections[2];
static char g_line_a[1601];

static void build_sections(void) {
    memset(g_line_a, 'a', 1600);
    g_line_a[1600] = '\0';
    g_sections[0].label = "sys";
    g_sections[0].text = g_line_a;
    g_sections[1].label = "docs";
    g_sections[1].text = g_line_a;
}

int main(void) {
    build_sections();

    /* input totals */
    assert(cwp_input_token_total(g_sections, 2) == 800);
    assert(cwp_input_token_total(NULL, 0) == 0);
    cwp_section empty = { "sys", "" };
    assert(cwp_input_token_total(&empty, 1) == 0);

    cwp_window_plan p;

    /* plans two 400-token sections against beta-pro */
    assert(cwp_plan_window(g_sections, 2, "beta-pro", 0,
                           CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
    assert(strcmp(p.id, "beta-pro") == 0);
    assert(p.input_tokens == 800);
    assert(p.context_window == 1000000);
    assert(p.free == 999200);
    assert(p.fits);
    assert(p.output_reserve_ok);
    assert(p.max_output == 10000);

    /* reserve larger than free leaves raw fit true */
    assert(cwp_plan_window(g_sections, 2, "beta-pro", 1000000,
                           CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
    assert(p.fits);
    assert(!p.output_reserve_ok);

    /* reserve exactly equal to free is ok */
    assert(cwp_plan_window(g_sections, 2, "beta-pro", 999200,
                           CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
    assert(p.output_reserve_ok);

    /* smaller window leaves 199,200 free */
    assert(cwp_plan_window(g_sections, 2, "alpha-mini", 0,
                           CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
    assert(p.context_window == 200000);
    assert(p.free == 199200);
    assert(p.fits);

    /* unknown model id returns failure */
    assert(!cwp_plan_window(g_sections, 2, "ghost", 0,
                            CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));

    /* no sections: full window free */
    assert(cwp_plan_window(NULL, 0, "beta-pro", 0,
                           CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
    assert(p.input_tokens == 0);
    assert(p.free == 1000000);
    assert(p.fits);

    /* overflow: fits false, reserve false */
    char *big = malloc(4400001);
    memset(big, 'z', 4400000);
    big[4400000] = '\0';
    cwp_section big_sec = { "big", big };
    assert(cwp_plan_window(&big_sec, 1, "beta-pro", 0,
                           CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, &p));
    assert(p.input_tokens == 1100000);
    assert(p.free == -100000);
    assert(!p.fits);
    assert(!p.output_reserve_ok);
    free(big);

    /* plan_all drops unknown ids and keeps order */
    const char *ids[] = { "beta-pro", "alpha-mini", "ghost" };
    cwp_window_plan plans[3];
    size_t n = cwp_plan_all(g_sections, 2, ids, 3, 0,
                            CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN,
                            plans, 3);
    assert(n == 2);
    assert(strcmp(plans[0].id, "beta-pro") == 0);
    assert(strcmp(plans[1].id, "alpha-mini") == 0);
    assert(plans[1].free == 199200);
    assert(cwp_plan_all(g_sections, 2, NULL, 0, 0,
                        CWP_SAMPLE_MODELS, CWP_SAMPLE_MODELS_LEN, plans, 3) == 0);

    printf("context-window-planner (C): all tests passed\n");
    return 0;
}

#endif /* CWP_TEST */

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →