Context Window Planner — PHP source
Paste your system prompt, docs, and history — see how they fill any model's context window, with overflow warnings and output headroom.
This is the PHP implementation — the same logic the interactive tool runs, in a shareable, citable form.
<?php
/**
* Context Window Planner — plan labeled prompt sections against a model's
* context window.
*
* Language: PHP (8.1+, standard library only — mbstring for multibyte string
* arithmetic, which is effectively universal in modern PHP)
* Source: CosmoDev polyglot showcase port of the Context Window Planner
* tool, ported from src/lib/contextPlanner.ts (the canonical
* TypeScript implementation).
* Live at: https://dev.cosmolabs.org/tools/context-window-planner
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; never throws.
* - Functionally equivalent to the TS reference: same inputs -> same outputs.
* - Self-contained: stdlib only (no Composer packages).
*
* Port notes: the TS lib delegates to two siblings — `estimateTokens` from
* src/lib/tokenEstimator.ts and `fitsWindow` from src/lib/ai/models.ts (which
* defaults to the bundled pricing snapshot, src/data/ai-models.json). A
* dependency-free port cannot load that file, so the estimator is inlined
* below in the exact form the planner uses it (`estimateTokens(text).tokens`,
* auto content type — the full heuristic lives in the token-estimator port),
* window math is inlined from fitsWindow() and `$models` is an explicit
* parameter (defaulting to an empty list), never re-derived.
*
* Faithfulness notes (the places PHP defaults silently differ from JS):
* - Length: TS's String.length counts UTF-16 code units (an astral-plane
* character — emoji, rare CJK ext-B ideographs — counts as 2). PHP
* strings are byte strings, so line arithmetic goes through
* cwp_utf16_len(), which counts UTF-8 lead bytes and adjusts per
* sequence length — same unit as TS, in O(1) memory (no per-character
* array allocation), assuming well-formed UTF-8.
* - Trim: PHP's trim() only strips ASCII whitespace, while JS trim()
* strips Unicode whitespace (NBSP, ideographic space, ...). cwp_trim()
* strips the ASCII set plus Unicode separator characters (\pZ) so
* whitespace-only-line handling matches the TS reference.
* - Rounding: PHP round() rounds halfway cases away from zero, which for
* the non-negative numbers used here is exactly JS Math.round.
* - JSON validity: json_decode with JSON_THROW_ON_ERROR gives the same
* strict grammar as JSON.parse (no trailing commas, no NaN/Infinity,
* no comments).
*/
declare(strict_types=1);
/**
* Average characters per token, by content type.
* Mirrors CHARS_PER_TOKEN in src/lib/tokenEstimator.ts.
*/
const CWP_CHARS_PER_TOKEN = [
'prose' => 4.0,
'code' => 3.5,
'json' => 3.0,
'cjk' => 1.5,
];
/**
* Sample table for standalone use (mirrors the shared test fixtures).
* Production code passes the model snapshot instead. Each entry carries the
* subset of the TS `AiModel` record the planner reads.
*/
const CWP_SAMPLE_MODELS = [
['id' => 'alpha-mini', 'contextWindow' => 200_000, 'maxOutput' => 10_000],
['id' => 'beta-pro', 'contextWindow' => 1_000_000, 'maxOutput' => 10_000],
['id' => 'gamma-open', 'contextWindow' => 100_000, 'maxOutput' => 10_000],
];
/**
* Length of a string in UTF-16 code units — the unit TS's String.length
* counts. BMP code points are one unit, astral-plane ones two.
*
* Counted from UTF-8 lead bytes instead of mb_str_split so a multi-megabyte
* line costs O(1) memory: every byte is one unit except that a 2-byte
* sequence contributes 1 unit (1 overcounted), and 3- and 4-byte sequences
* contribute 1 and 2 units respectively (2 overcounted each). Assumes
* well-formed UTF-8.
*/
function cwp_utf16_len(string $s): int
{
$bytes = strlen($s);
if ($bytes === 0) {
return 0;
}
$lead2 = (int) preg_match_all('/[\xC0-\xDF]/', $s);
$lead3 = (int) preg_match_all('/[\xE0-\xEF]/', $s);
$lead4 = (int) preg_match_all('/[\xF0-\xF7]/', $s);
return $bytes - $lead2 - 2 * $lead3 - 2 * $lead4;
}
/**
* Unicode-aware trim: strips ASCII whitespace plus Unicode separators
* (mirrors JS String.prototype.trim more closely than plain trim()).
*/
function cwp_trim(string $s): string
{
$trimmed = preg_replace('/^[\s\pZ]+|[\s\pZ]+$/u', '', $s);
return $trimmed ?? '';
}
/**
* Classify a single line by its shape. Order: json, cjk, code, prose.
* Inlined from detectLineType() in src/lib/tokenEstimator.ts.
*/
function cwp_detect_line_type(string $line): string
{
$trimmed = cwp_trim($line);
// JSON-ish: opens like a JSON fragment AND carries a separator.
$first = $trimmed !== '' ? mb_substr($trimmed, 0, 1, 'UTF-8') : '';
if (($first === '{' || $first === '}' || $first === '[' || $first === '"')
&& (str_contains($line, ':') || str_contains($line, ','))) {
return 'json';
}
// CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul
// syllables (U+AC00..U+D7AF). Mirrors CJK_RE = /[一-鿿-ヿ가-]/.
if (preg_match('/[\x{4E00}-\x{9FFF}\x{3040}-\x{30FF}\x{AC00}-\x{D7AF}]/u', $line) === 1) {
return 'cjk';
}
// Code: symbol-dense, or a statement terminator / block opener at EOL.
// Symbols are ASCII, so they never occur inside a UTF-8 multibyte
// sequence — a byte-oriented match count is exact.
$length = cwp_utf16_len($line);
$symbols = (int) preg_match_all('/[{}();=<>\[\]#]/', $line);
$density = $length > 0 ? $symbols / $length : 0.0;
if ($density > 0.08
|| str_ends_with($trimmed, ';')
|| str_ends_with($trimmed, '{')
|| str_ends_with($trimmed, '}')) {
return 'code';
}
return 'prose';
}
/**
* Whole-text JSON gate: a document that parses as JSON is json all the way
* down. Mirrors isValidJson() (JSON.parse in a try/catch); empty/whitespace
* text is not.
*/
function cwp_is_valid_json(string $text): bool
{
if (cwp_trim($text) === '') {
return false;
}
try {
json_decode($text, null, 512, JSON_THROW_ON_ERROR);
return true;
} catch (JsonException) {
return false;
}
}
/**
* Token count of text under auto content detection — exactly the slice of
* estimateTokens() the planner consumes (`.tokens`): per non-empty line,
* max(1, round(utf16_len / CHARS_PER_TOKEN[type])). Framing tokens are the
* caller's job.
*/
function cwp_estimate_tokens(string $text): int
{
// AUTO + whole-text JSON: json's 3 chars/token rate applies to every
// line, not just the reported contentType.
$whole_text_json = cwp_is_valid_json($text);
$tokens = 0;
foreach (preg_split('/\r?\n/', $text) ?: [] as $line) {
if (cwp_trim($line) === '') {
continue;
}
$type = $whole_text_json ? 'json' : cwp_detect_line_type($line);
$tokens += max(1, (int) round(cwp_utf16_len($line) / CWP_CHARS_PER_TOKEN[$type]));
}
return $tokens;
}
/**
* Sum of per-section token estimates (framing tokens are the caller's job).
* Mirrors inputTokenTotal() in the TS lib.
*
* @param array<int, array{label: string, text: string}> $sections
*/
function input_token_total(array $sections): int
{
$total = 0;
foreach ($sections as $section) {
$total += cwp_estimate_tokens($section['text']);
}
return $total;
}
/**
* Plan one section set against one model's context window. Returns null for
* an unknown model id (window math is fitsWindow's, never re-derived).
* Mirrors planWindow() in the TS lib.
*
* @param array<int, array{label: string, text: string}> $sections
* @param array<int, array{id: string, contextWindow: int, maxOutput: int}> $models
* @return array{id: string, inputTokens: int, contextWindow: int, free: int,
* fits: bool, outputReserveOk: bool, maxOutput: int}|null
*/
function plan_window(array $sections, string $model_id, int $output_reserve = 0, array $models = []): ?array
{
$input_tokens = input_token_total($sections);
// Fit check inlined from fitsWindow() in src/lib/ai/models.ts.
$model = null;
foreach ($models as $m) {
if ($m['id'] === $model_id) {
$model = $m;
break;
}
}
if ($model === null) {
return null;
}
$free = $model['contextWindow'] - $input_tokens;
return [
'id' => $model_id,
'inputTokens' => $input_tokens,
'contextWindow' => $model['contextWindow'],
'free' => $free,
'fits' => $free >= 0,
'outputReserveOk' => $free >= $output_reserve,
'maxOutput' => $model['maxOutput'],
];
}
/**
* Plan against several models; unknown ids are dropped from the result.
* Mirrors planAll() in the TS lib.
*
* @param array<int, array{label: string, text: string}> $sections
* @param array<int, string> $model_ids
* @param array<int, array{id: string, contextWindow: int, maxOutput: int}> $models
* @return array<int, array{id: string, inputTokens: int, contextWindow: int, free: int,
* fits: bool, outputReserveOk: bool, maxOutput: int}>
*/
function plan_all(array $sections, array $model_ids, int $output_reserve = 0, array $models = []): array
{
$plans = [];
foreach ($model_ids as $model_id) {
$plan = plan_window($sections, $model_id, $output_reserve, $models);
if ($plan !== null) {
$plans[] = $plan;
}
}
return $plans;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →