Skip to content

Conversation Pruner — Python source

Plan how to fit a long chat history into a context budget — which turns to keep, fold into a summary, or drop, protecting system messages and the current request. 100% client-side.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""Conversation Pruner — compute a deterministic pruning plan for a
token-budgeted chat history.

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the Conversation Pruner
          tool (slug: conversation-pruner).
Port of src/lib/conversationPruner.ts (the canonical TypeScript
          implementation).
Tool page: https://dev.cosmolabs.org/tools/conversation-pruner
License:  display source — part of CosmoDev's polyglot tool pages.

Given per-message token counts and a context budget, decide which messages
to keep verbatim, which to fold into one running summary, and which to
drop outright — protecting system messages, pinned turns, the first turn
(the opening user request), and the current (last user) request. Token
counts are plain integers; the only float math is the summary cost: fixed
framing plus 10% of the folded content, rounded up.

Field names use snake_case where the TypeScript uses camelCase
(keptTokens -> kept_tokens, planPrune -> plan_prune).
"""

from __future__ import annotations

import math
from dataclasses import dataclass, field
from typing import List, Sequence

#: A chat participant. Mirrors the TypeScript ``ChatRole`` union.
ChatRole = str

#: What the plan does with one message.
PruneAction = str  # one of "keep", "summarize", "drop"


@dataclass
class ConversationMessage:
    """One message plus its prompt-side token count.

    ``pinned`` messages are never dropped or summarized (the TypeScript's
    optional flag defaults to false).
    """

    role: ChatRole  # "system" | "user" | "assistant" | "tool"
    content: str
    tokens: int
    pinned: bool = False


@dataclass
class PruneDecision:
    index: int
    role: ChatRole
    action: PruneAction
    tokens: int


@dataclass
class PrunePlan:
    decisions: List[PruneDecision] = field(default_factory=list)
    kept_tokens: int = 0
    summarized_tokens: int = 0
    dropped_tokens: int = 0
    #: Tokens the summary placeholder itself will cost in the prompt.
    summary_cost_tokens: int = 0
    projected_tokens: int = 0
    fits_budget: bool = False
    warnings: List[str] = field(default_factory=list)


#: Summary compression model: fixed framing tokens.
SUMMARY_FIXED_TOKENS = 60

#: Summary compression model: share of the folded content's tokens.
SUMMARY_RATIO = 0.1


def plan_prune(messages: Sequence[ConversationMessage], budget_tokens: int) -> PrunePlan:
    """Compute the pruning plan.

    Raises ValueError (the TypeScript's RangeError) on a negative budget or
    any negative per-message token count.
    """
    warnings: List[str] = []
    if budget_tokens < 0:
        raise ValueError("budgetTokens must be >= 0")
    if any(m.tokens < 0 for m in messages):
        raise ValueError("message tokens must be >= 0")

    n = len(messages)
    last_user = -1
    for i in range(n - 1, -1, -1):
        if messages[i].role == "user":
            last_user = i
            break

    # Untouchable: every system message, pinned messages, the first turn (the
    # opening user request that anchors the conversation), and the current
    # request (the last user message and everything after it).
    protected: set = set()
    for i, m in enumerate(messages):
        if m.role == "system" or m.pinned:
            protected.add(i)
    if n > 0:
        protected.add(0)
    first_turn = -1
    for i, m in enumerate(messages):
        if m.role != "system":
            first_turn = i
            break
    if first_turn != -1:
        protected.add(first_turn)
    tail_start = n - 1 if last_user == -1 else last_user
    for i in range(max(tail_start, 0), n):
        protected.add(i)

    protected_tokens = sum(messages[i].tokens for i in protected)
    if protected_tokens > budget_tokens:
        warnings.append(
            f"Protected messages alone are {protected_tokens:,} tokens against a "
            f"{budget_tokens:,} budget — raise the budget (or reserve less for the "
            f"reply) before pruning anything else."
        )

    # Fill the remaining budget newest-to-oldest through the middle.
    actions: List[PruneAction] = ["drop"] * n
    for i in protected:
        actions[i] = "keep"
    used = protected_tokens
    for i in range(n - 1, -1, -1):
        if actions[i] != "drop":
            continue
        if used + messages[i].tokens <= budget_tokens:
            actions[i] = "keep"
            used += messages[i].tokens
        else:
            break  # oldest-unfilled remain drop/summarize candidates, newest first stopped

    # Everything still 'drop' in the middle folds into ONE running summary when
    # the compressed form fits where the raw turns did not.
    summarize_idx = [i for i in range(n) if actions[i] == "drop" and i not in protected]
    summarize_tokens = sum(messages[i].tokens for i in summarize_idx)
    attempted_summary_cost = (
        SUMMARY_FIXED_TOKENS + math.ceil(summarize_tokens * SUMMARY_RATIO)
        if summarize_idx
        else 0
    )

    # The summary only costs anything when it is actually applied — otherwise
    # those turns drop and cost zero.
    summary_cost = 0
    if attempted_summary_cost > 0 and used + attempted_summary_cost <= budget_tokens:
        for i in summarize_idx:
            actions[i] = "summarize"
        summary_cost = attempted_summary_cost
        used += summary_cost
    elif attempted_summary_cost > 0:
        warnings.append(
            f"Even the compressed summary ({attempted_summary_cost:,} tokens) does not "
            f"fit the remaining budget — the oldest turns are dropped instead."
        )

    decisions = [
        PruneDecision(index=i, role=messages[i].role, action=actions[i], tokens=messages[i].tokens)
        for i in range(n)
    ]

    kept_tokens = sum(d.tokens for d in decisions if d.action == "keep")
    dropped_tokens = sum(d.tokens for d in decisions if d.action == "drop")
    folded_tokens = sum(d.tokens for d in decisions if d.action == "summarize")

    return PrunePlan(
        decisions=decisions,
        kept_tokens=kept_tokens,
        summarized_tokens=folded_tokens,
        dropped_tokens=dropped_tokens,
        summary_cost_tokens=summary_cost,
        projected_tokens=kept_tokens + summary_cost,
        fits_budget=kept_tokens + summary_cost <= budget_tokens,
        warnings=warnings,
    )


def describe_prune(plan: PrunePlan) -> str:
    """Human-readable one-line summary of a plan."""
    if not plan.fits_budget:
        return f"Does not fit: {plan.projected_tokens:,} tokens projected against the budget."
    parts = [f"{plan.kept_tokens:,} kept"]
    if plan.summarized_tokens > 0:
        parts.append(
            f"{plan.summarized_tokens:,} folded into a {plan.summary_cost_tokens:,}-token summary"
        )
    if plan.dropped_tokens > 0:
        parts.append(f"{plan.dropped_tokens:,} dropped")
    return " · ".join(parts) + " — fits the budget."

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →