Skip to content

Context Window Planner — Python source

Paste your system prompt, docs, and history — see how they fill any model's context window, with overflow warnings and output headroom.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""Context Window Planner — plan labeled prompt sections against a model's
context window.

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the Context Window Planner
          tool, ported from src/lib/contextPlanner.ts (the canonical
          TypeScript implementation).
Tool page: https://dev.cosmolabs.org/tools/context-window-planner
License:  display source — part of CosmoDev's polyglot tool pages.

Design goals:
  - Pure + deterministic; never raises.
  - Functionally equivalent to the TS reference: same inputs -> same outputs.
  - Self-contained: stdlib only (no pip packages).

Port notes: the TS lib delegates to two siblings — ``estimateTokens`` from
src/lib/tokenEstimator.ts and ``fitsWindow`` from src/lib/ai/models.ts (which
defaults to the bundled pricing snapshot, src/data/ai-models.json). A
dependency-free port cannot load that file, so the estimator is inlined below
in the exact form the planner uses it (``estimateTokens(text).tokens``, auto
content type — the full heuristic lives in the token-estimator port), window
math is inlined from ``fitsWindow`` and ``models`` is an explicit parameter
(defaulting to an empty tuple), never re-derived.

Faithfulness notes (the places Python defaults silently differ from JS):
  - Rounding: JS ``Math.round`` rounds halfway cases UP, Python's built-in
    ``round`` rounds halfway cases to the nearest EVEN integer (banker's
    rounding — ``round(2.5) == 2``). Every token count goes through
    ``_js_round`` (``floor(x + 0.5)``) so a 10-char prose line estimates 3
    tokens here exactly as it does in TS.
  - Length: TS's ``String.length`` counts UTF-16 code units (an astral-plane
    character — emoji, rare CJK ext-B ideographs — counts as 2). Python's
    ``len(str)`` counts code points. Line arithmetic goes through
    ``_utf16_len`` so multi-byte text estimates identically on both sides.
  - JSON validity: ``json.loads`` accepts ``NaN`` / ``Infinity`` literals that
    ``JSON.parse`` rejects; ``parse_constant=_reject_constant`` restores the
    strict grammar.
"""

from __future__ import annotations

import json
import math
import re
from dataclasses import dataclass
from typing import List, Optional, Sequence, Tuple

__all__ = [
    "PlanSection",
    "Model",
    "WindowPlan",
    "SAMPLE_MODELS",
    "input_token_total",
    "plan_window",
    "plan_all",
]

#: Average characters per token, by content type. Mirrors CHARS_PER_TOKEN
#: in src/lib/tokenEstimator.ts.
CHARS_PER_TOKEN = {
    "prose": 4,
    "code": 3.5,
    "json": 3,
    "cjk": 1.5,
}

# CJK ideographs (U+4E00..U+9FFF), kana (U+3040..U+30FF), Hangul syllables
# (U+AC00..U+D7AF). Mirrors CJK_RE = /[一-鿿぀-ヿ가-힯]/ in the TS lib.
_CJK_RE = re.compile(r"[一-鿿぀-ヿ가-힯]")

# Code-flavored symbols, counted over the raw line. Mirrors CODE_SYMBOL_RE.
_CODE_SYMBOL_RE = re.compile(r"[{}();=<>\[\]#]")

# Line split on LF or CRLF. Mirrors text.split(/\r?\n/).
_NEWLINE_RE = re.compile(r"\r?\n")


@dataclass(frozen=True)
class PlanSection:
    """One labeled block of the prompt (system / docs / history / ...)."""

    label: str
    text: str


@dataclass(frozen=True)
class Model:
    """The subset of the TS ``AiModel`` record the planner reads.

    Production code passes the full snapshot entry; only these fields
    influence the plan.
    """

    id: str
    context_window: int
    """Total context window in tokens."""

    max_output: int
    """The model's output cap (informational)."""


#: Sample table for standalone use (mirrors the shared test fixtures).
#: Production code passes the model snapshot instead.
SAMPLE_MODELS: Tuple[Model, ...] = (
    Model("alpha-mini", 200_000, 10_000),
    Model("beta-pro", 1_000_000, 10_000),
    Model("gamma-open", 100_000, 10_000),
)


@dataclass(frozen=True)
class WindowPlan:
    """Result of ``plan_window``. Field-for-field twin of the TS
    ``WindowPlan`` interface (same keys, same meanings)."""

    id: str
    """The model id planned against."""

    input_tokens: int
    """Sum of per-section token estimates."""

    context_window: int
    """The model's context window."""

    free: int
    """Context tokens left after the request; negative on overflow."""

    fits: bool
    """Raw fit: free >= 0."""

    output_reserve_ok: bool
    """Room for the output reserve: free >= output_reserve."""

    max_output: int
    """The model's output cap (informational)."""


def _utf16_len(s: str) -> int:
    """Length of ``s`` in UTF-16 code units — the unit TS's ``String.length``
    counts. BMP code points are one unit, astral-plane ones two."""
    return sum(2 if ord(ch) > 0xFFFF else 1 for ch in s)


def _js_round(x: float) -> int:
    """Round halfway cases up, like JS ``Math.round`` (Python's built-in
    ``round`` would round 2.5 to 2, not 3)."""
    return math.floor(x + 0.5)


def _reject_constant(name: str) -> None:
    """``json.loads`` parse_constant hook: raise on NaN / Infinity so the
    validity check matches ``JSON.parse``'s strict grammar."""
    raise ValueError(f"non-standard JSON constant: {name}")


def _detect_line_type(line: str) -> str:
    """Classify a single line by its shape. Order: json, cjk, code, prose.
    Inlined from detectLineType() in src/lib/tokenEstimator.ts."""
    trimmed = line.strip()
    # JSON-ish: opens like a JSON fragment AND carries a separator.
    if trimmed[:1] in ("{", "}", "[", '"') and (":" in line or "," in line):
        return "json"
    # CJK ideographs / kana / Hangul pack roughly one token per 1.5 chars.
    if _CJK_RE.search(line):
        return "cjk"
    # Code: symbol-dense, or a statement terminator / block opener at EOL.
    length = _utf16_len(line)
    density = len(_CODE_SYMBOL_RE.findall(line)) / length if length else 0.0
    if density > 0.08 or trimmed.endswith((";", "{", "}")):
        return "code"
    return "prose"


def _is_valid_json(text: str) -> bool:
    """Whole-text JSON gate: mirrors isValidJson() (JSON.parse in a
    try/except); empty/whitespace text is not."""
    if not text.strip():
        return False
    try:
        json.loads(text, parse_constant=_reject_constant)
    except ValueError:
        return False
    return True


def _estimate_tokens(text: str) -> int:
    """Token count of ``text`` under auto content detection — exactly the
    slice of estimateTokens() the planner consumes (``.tokens``).

    Per non-empty line: ``max(1, round(utf16_len / CHARS_PER_TOKEN[type]))``.
    Framing tokens are the caller's job.
    """
    # AUTO + whole-text JSON: a document that parses as JSON is json all the
    # way down — json's 3 chars/token rate applies to every line.
    whole_text_json = _is_valid_json(text)
    tokens = 0
    for line in _NEWLINE_RE.split(text):
        if line.strip() == "":
            continue
        line_type = "json" if whole_text_json else _detect_line_type(line)
        tokens += max(1, _js_round(_utf16_len(line) / CHARS_PER_TOKEN[line_type]))
    return tokens


def input_token_total(sections: Sequence[PlanSection]) -> int:
    """Sum of per-section token estimates (framing tokens are the caller's
    job). Mirrors inputTokenTotal() in the TS lib."""
    return sum(_estimate_tokens(s.text) for s in sections)


def _fits_window(model_id: str, tokens: int, models: Sequence[Model]) -> Optional[Tuple[Model, int, int, bool]]:
    """Fit check for one request. Inlined from fitsWindow() in
    src/lib/ai/models.ts — ``(model, context_window, free, fits)`` or None
    for an unknown id."""
    for m in models:
        if m.id == model_id:
            free = m.context_window - tokens
            return (m, m.context_window, free, free >= 0)
    return None


def plan_window(
    sections: Sequence[PlanSection],
    model_id: str,
    output_reserve: int = 0,
    models: Sequence[Model] = (),
) -> Optional[WindowPlan]:
    """Plan one section set against one model's context window.

    Returns ``None`` for an unknown model id (window math is fitsWindow's,
    never re-derived). Mirrors planWindow() in the TS lib.
    """
    input_tokens = input_token_total(sections)
    fit = _fits_window(model_id, input_tokens, models)
    if fit is None:
        return None
    model, context_window, free, fits = fit
    return WindowPlan(
        id=model_id,
        input_tokens=input_tokens,
        context_window=context_window,
        free=free,
        fits=fits,
        output_reserve_ok=free >= output_reserve,
        max_output=model.max_output,
    )


def plan_all(
    sections: Sequence[PlanSection],
    model_ids: Sequence[str],
    output_reserve: int = 0,
    models: Sequence[Model] = (),
) -> List[WindowPlan]:
    """Plan against several models; unknown ids are dropped from the result.
    Mirrors planAll() in the TS lib."""
    plans: List[WindowPlan] = []
    for model_id in model_ids:
        plan = plan_window(sections, model_id, output_reserve, models)
        if plan is not None:
            plans.append(plan)
    return plans

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →