Skip to content

Conversation Pruner — Kotlin source

Plan how to fit a long chat history into a context budget — which turns to keep, fold into a summary, or drop, protecting system messages and the current request. 100% client-side.

This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Conversation Pruner — compute a deterministic pruning plan for a
// token-budgeted chat history.
//
// Language: Kotlin (JVM 17+, zero dependencies)
// Port of src/lib/conversationPruner.ts (the canonical TypeScript
//           implementation) for the CosmoDev polyglot showcase
//           (slug: conversation-pruner).
//
// Given per-message token counts and a context budget, decide which messages
// to keep verbatim, which to fold into one running summary, and which to drop
// outright — protecting system messages, pinned turns, the first turn, and
// the current (last user) request.

import kotlin.math.ceil

enum class ChatRole { SYSTEM, USER, ASSISTANT, TOOL }
enum class PruneAction { KEEP, SUMMARIZE, DROP }

/** One chat message with its caller-supplied token count. */
data class ConversationMessage(
    val role: ChatRole,
    val content: String,
    val tokens: Int,
    val pinned: Boolean = false,
)

/** One per-message decision. */
data class PruneDecision(
    val index: Int,
    val role: ChatRole,
    val action: PruneAction,
    val tokens: Int,
)

/** The full plan: decisions, tallies, projection, warnings. */
data class PrunePlan(
    val decisions: List<PruneDecision>,
    val keptTokens: Int,
    val summarizedTokens: Int,
    val droppedTokens: Int,
    val summaryCostTokens: Int,
    val projectedTokens: Int,
    val fitsBudget: Boolean,
    val warnings: List<String>,
)

/** Summary compression model: fixed framing tokens. */
const val SUMMARY_FIXED_TOKENS = 60

/** Summary compression model: share of the folded content. */
const val SUMMARY_RATIO = 0.1

private fun fmt(v: Int): String = String.format("%,d", v)

/**
 * Compute the pruning plan for [messages] under [budgetTokens].
 * Throws [IllegalArgumentException] on a negative budget or any negative
 * per-message token count.
 */
fun planPrune(messages: List<ConversationMessage>, budgetTokens: Int): PrunePlan {
    val warnings = mutableListOf<String>()
    require(budgetTokens >= 0) { "budgetTokens must be >= 0" }
    require(messages.none { it.tokens < 0 }) { "message tokens must be >= 0" }

    val n = messages.size
    val lastUser = messages.indexOfLast { it.role == ChatRole.USER }

    // Untouchable: every system message, pinned messages, the first turn
    // (the opening user request), and the current request (the last user
    // message and everything after it).
    val protectedIdx = buildSet {
        messages.forEachIndexed { i, m -> if (m.role == ChatRole.SYSTEM || m.pinned) add(i) }
        if (n > 0) add(0)
        val firstTurn = messages.indexOfFirst { it.role != ChatRole.SYSTEM }
        if (firstTurn != -1) add(firstTurn)
        val from = if (lastUser == -1) n - 1 else lastUser
        for (i in maxOf(from, 0) until n) add(i)
    }

    val protectedTokens = protectedIdx.sumOf { messages[it].tokens }
    if (protectedTokens > budgetTokens) {
        warnings.add(
            "Protected messages alone are ${fmt(protectedTokens)} tokens against a " +
                "${fmt(budgetTokens)} budget — raise the budget (or reserve less for the reply) " +
                "before pruning anything else.",
        )
    }

    // Fill the remaining budget newest-to-oldest through the middle.
    val actions = Array(n) { PruneAction.DROP }
    protectedIdx.forEach { actions[it] = PruneAction.KEEP }
    var used = protectedTokens
    for (i in n - 1 downTo 0) {
        if (actions[i] != PruneAction.DROP) continue
        val t = messages[i].tokens
        if (used + t <= budgetTokens) {
            actions[i] = PruneAction.KEEP
            used += t
        } else break // oldest-unfilled remain drop/summarize candidates
    }

    // Everything still 'drop' in the middle folds into ONE running summary
    // when the compressed form fits where the raw turns did not.
    val summarizeIdx = (0 until n).filter { actions[it] == PruneAction.DROP && it !in protectedIdx }
    val summarizeTokens = summarizeIdx.sumOf { messages[it].tokens }
    val attempted =
        if (summarizeIdx.isEmpty()) 0
        else SUMMARY_FIXED_TOKENS + ceil(summarizeTokens * SUMMARY_RATIO).toInt()

    // The summary only costs anything when it is actually applied.
    var summaryCost = 0
    if (attempted > 0 && used + attempted <= budgetTokens) {
        summarizeIdx.forEach { actions[it] = PruneAction.SUMMARIZE }
        summaryCost = attempted
        used += summaryCost
    } else if (attempted > 0) {
        warnings.add(
            "Even the compressed summary (${fmt(attempted)} tokens) does not fit the " +
                "remaining budget — the oldest turns are dropped instead.",
        )
    }

    val decisions = messages.mapIndexed { i, m -> PruneDecision(i, m.role, actions[i], m.tokens) }
    var kept = 0; var dropped = 0; var folded = 0
    for (d in decisions) when (d.action) {
        PruneAction.KEEP -> kept += d.tokens
        PruneAction.DROP -> dropped += d.tokens
        PruneAction.SUMMARIZE -> folded += d.tokens
    }

    return PrunePlan(
        decisions, kept, folded, dropped, summaryCost,
        kept + summaryCost, kept + summaryCost <= budgetTokens, warnings,
    )
}

/** Human-readable one-line summary of a plan. */
fun describePrune(plan: PrunePlan): String {
    if (!plan.fitsBudget) {
        return "Does not fit: ${fmt(plan.projectedTokens)} tokens projected against the budget."
    }
    val parts = mutableListOf("${fmt(plan.keptTokens)} kept")
    if (plan.summarizedTokens > 0) {
        parts.add("${fmt(plan.summarizedTokens)} folded into a ${fmt(plan.summaryCostTokens)}-token summary")
    }
    if (plan.droppedTokens > 0) parts.add("${fmt(plan.droppedTokens)} dropped")
    return parts.joinToString(" · ") + " — fits the budget."
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →