Context Window Planner — Python source
Paste your system prompt, docs, and history — see how they fill any model's context window, with overflow warnings and output headroom.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""Context Window Planner — plan labeled prompt sections against a model's
context window.
Language: Python (3.9+, standard library only)
Source: CosmoDev polyglot showcase port of the Context Window Planner
tool, ported from src/lib/contextPlanner.ts (the canonical
TypeScript implementation).
Tool page: https://dev.cosmolabs.org/tools/context-window-planner
License: display source — part of CosmoDev's polyglot tool pages.
Design goals:
- Pure + deterministic; never raises.
- Functionally equivalent to the TS reference: same inputs -> same outputs.
- Self-contained: stdlib only (no pip packages).
Port notes: the TS lib delegates to two siblings — ``estimateTokens`` from
src/lib/tokenEstimator.ts and ``fitsWindow`` from src/lib/ai/models.ts (which
defaults to the bundled pricing snapshot, src/data/ai-models.json). A
dependency-free port cannot load that file, so the estimator is inlined below
in the exact form the planner uses it (``estimateTokens(text).tokens``, auto
content type — the full heuristic lives in the token-estimator port), window
math is inlined from ``fitsWindow`` and ``models`` is an explicit parameter
(defaulting to an empty tuple), never re-derived.
Faithfulness notes (the places Python defaults silently differ from JS):
- Rounding: JS ``Math.round`` rounds halfway cases UP, Python's built-in
``round`` rounds halfway cases to the nearest EVEN integer (banker's
rounding — ``round(2.5) == 2``). Every token count goes through
``_js_round`` (``floor(x + 0.5)``) so a 10-char prose line estimates 3
tokens here exactly as it does in TS.
- Length: TS's ``String.length`` counts UTF-16 code units (an astral-plane
character — emoji, rare CJK ext-B ideographs — counts as 2). Python's
``len(str)`` counts code points. Line arithmetic goes through
``_utf16_len`` so multi-byte text estimates identically on both sides.
- JSON validity: ``json.loads`` accepts ``NaN`` / ``Infinity`` literals that
``JSON.parse`` rejects; ``parse_constant=_reject_constant`` restores the
strict grammar.
"""
from __future__ import annotations
import json
import math
import re
from dataclasses import dataclass
from typing import List, Optional, Sequence, Tuple
__all__ = [
"PlanSection",
"Model",
"WindowPlan",
"SAMPLE_MODELS",
"input_token_total",
"plan_window",
"plan_all",
]
#: Average characters per token, by content type. Mirrors CHARS_PER_TOKEN
#: in src/lib/tokenEstimator.ts.
CHARS_PER_TOKEN = {
"prose": 4,
"code": 3.5,
"json": 3,
"cjk": 1.5,
}
# CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul syllables
# (U+AC00..U+D7AF). Mirrors CJK_RE = /[一-鿿-ヿ가-]/ in the TS lib.
_CJK_RE = re.compile(r"[一-鿿-ヿ가-]")
# Code-flavored symbols, counted over the raw line. Mirrors CODE_SYMBOL_RE.
_CODE_SYMBOL_RE = re.compile(r"[{}();=<>\[\]#]")
# Line split on LF or CRLF. Mirrors text.split(/\r?\n/).
_NEWLINE_RE = re.compile(r"\r?\n")
@dataclass(frozen=True)
class PlanSection:
"""One labeled block of the prompt (system / docs / history / ...)."""
label: str
text: str
@dataclass(frozen=True)
class Model:
"""The subset of the TS ``AiModel`` record the planner reads.
Production code passes the full snapshot entry; only these fields
influence the plan.
"""
id: str
context_window: int
"""Total context window in tokens."""
max_output: int
"""The model's output cap (informational)."""
#: Sample table for standalone use (mirrors the shared test fixtures).
#: Production code passes the model snapshot instead.
SAMPLE_MODELS: Tuple[Model, ...] = (
Model("alpha-mini", 200_000, 10_000),
Model("beta-pro", 1_000_000, 10_000),
Model("gamma-open", 100_000, 10_000),
)
@dataclass(frozen=True)
class WindowPlan:
"""Result of ``plan_window``. Field-for-field twin of the TS
``WindowPlan`` interface (same keys, same meanings)."""
id: str
"""The model id planned against."""
input_tokens: int
"""Sum of per-section token estimates."""
context_window: int
"""The model's context window."""
free: int
"""Context tokens left after the request; negative on overflow."""
fits: bool
"""Raw fit: free >= 0."""
output_reserve_ok: bool
"""Room for the output reserve: free >= output_reserve."""
max_output: int
"""The model's output cap (informational)."""
def _utf16_len(s: str) -> int:
"""Length of ``s`` in UTF-16 code units — the unit TS's ``String.length``
counts. BMP code points are one unit, astral-plane ones two."""
return sum(2 if ord(ch) > 0xFFFF else 1 for ch in s)
def _js_round(x: float) -> int:
"""Round halfway cases up, like JS ``Math.round`` (Python's built-in
``round`` would round 2.5 to 2, not 3)."""
return math.floor(x + 0.5)
def _reject_constant(name: str) -> None:
"""``json.loads`` parse_constant hook: raise on NaN / Infinity so the
validity check matches ``JSON.parse``'s strict grammar."""
raise ValueError(f"non-standard JSON constant: {name}")
def _detect_line_type(line: str) -> str:
"""Classify a single line by its shape. Order: json, cjk, code, prose.
Inlined from detectLineType() in src/lib/tokenEstimator.ts."""
trimmed = line.strip()
# JSON-ish: opens like a JSON fragment AND carries a separator.
if trimmed[:1] in ("{", "}", "[", '"') and (":" in line or "," in line):
return "json"
# CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
if _CJK_RE.search(line):
return "cjk"
# Code: symbol-dense, or a statement terminator / block opener at EOL.
length = _utf16_len(line)
density = len(_CODE_SYMBOL_RE.findall(line)) / length if length else 0.0
if density > 0.08 or trimmed.endswith((";", "{", "}")):
return "code"
return "prose"
def _is_valid_json(text: str) -> bool:
"""Whole-text JSON gate: mirrors isValidJson() (JSON.parse in a
try/except); empty/whitespace text is not."""
if not text.strip():
return False
try:
json.loads(text, parse_constant=_reject_constant)
except ValueError:
return False
return True
def _estimate_tokens(text: str) -> int:
"""Token count of ``text`` under auto content detection — exactly the
slice of estimateTokens() the planner consumes (``.tokens``).
Per non-empty line: ``max(1, round(utf16_len / CHARS_PER_TOKEN[type]))``.
Framing tokens are the caller's job.
"""
# AUTO + whole-text JSON: a document that parses as JSON is json all the
# way down — json's 3 chars/token rate applies to every line.
whole_text_json = _is_valid_json(text)
tokens = 0
for line in _NEWLINE_RE.split(text):
if line.strip() == "":
continue
line_type = "json" if whole_text_json else _detect_line_type(line)
tokens += max(1, _js_round(_utf16_len(line) / CHARS_PER_TOKEN[line_type]))
return tokens
def input_token_total(sections: Sequence[PlanSection]) -> int:
"""Sum of per-section token estimates (framing tokens are the caller's
job). Mirrors inputTokenTotal() in the TS lib."""
return sum(_estimate_tokens(s.text) for s in sections)
def _fits_window(model_id: str, tokens: int, models: Sequence[Model]) -> Optional[Tuple[Model, int, int, bool]]:
"""Fit check for one request. Inlined from fitsWindow() in
src/lib/ai/models.ts — ``(model, context_window, free, fits)`` or None
for an unknown id."""
for m in models:
if m.id == model_id:
free = m.context_window - tokens
return (m, m.context_window, free, free >= 0)
return None
def plan_window(
sections: Sequence[PlanSection],
model_id: str,
output_reserve: int = 0,
models: Sequence[Model] = (),
) -> Optional[WindowPlan]:
"""Plan one section set against one model's context window.
Returns ``None`` for an unknown model id (window math is fitsWindow's,
never re-derived). Mirrors planWindow() in the TS lib.
"""
input_tokens = input_token_total(sections)
fit = _fits_window(model_id, input_tokens, models)
if fit is None:
return None
model, context_window, free, fits = fit
return WindowPlan(
id=model_id,
input_tokens=input_tokens,
context_window=context_window,
free=free,
fits=fits,
output_reserve_ok=free >= output_reserve,
max_output=model.max_output,
)
def plan_all(
sections: Sequence[PlanSection],
model_ids: Sequence[str],
output_reserve: int = 0,
models: Sequence[Model] = (),
) -> List[WindowPlan]:
"""Plan against several models; unknown ids are dropped from the result.
Mirrors planAll() in the TS lib."""
plans: List[WindowPlan] = []
for model_id in model_ids:
plan = plan_window(sections, model_id, output_reserve, models)
if plan is not None:
plans.append(plan)
return plans
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →