Conversation Pruner — Kotlin source
Plan how to fit a long chat history into a context budget — which turns to keep, fold into a summary, or drop, protecting system messages and the current request. 100% client-side.
This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Conversation Pruner — compute a deterministic pruning plan for a
// token-budgeted chat history.
//
// Language: Kotlin (JVM 17+, zero dependencies)
// Port of src/lib/conversationPruner.ts (the canonical TypeScript
// implementation) for the CosmoDev polyglot showcase
// (slug: conversation-pruner).
//
// Given per-message token counts and a context budget, decide which messages
// to keep verbatim, which to fold into one running summary, and which to drop
// outright — protecting system messages, pinned turns, the first turn, and
// the current (last user) request.
import kotlin.math.ceil
enum class ChatRole { SYSTEM, USER, ASSISTANT, TOOL }
enum class PruneAction { KEEP, SUMMARIZE, DROP }
/** One chat message with its caller-supplied token count. */
data class ConversationMessage(
val role: ChatRole,
val content: String,
val tokens: Int,
val pinned: Boolean = false,
)
/** One per-message decision. */
data class PruneDecision(
val index: Int,
val role: ChatRole,
val action: PruneAction,
val tokens: Int,
)
/** The full plan: decisions, tallies, projection, warnings. */
data class PrunePlan(
val decisions: List<PruneDecision>,
val keptTokens: Int,
val summarizedTokens: Int,
val droppedTokens: Int,
val summaryCostTokens: Int,
val projectedTokens: Int,
val fitsBudget: Boolean,
val warnings: List<String>,
)
/** Summary compression model: fixed framing tokens. */
const val SUMMARY_FIXED_TOKENS = 60
/** Summary compression model: share of the folded content. */
const val SUMMARY_RATIO = 0.1
private fun fmt(v: Int): String = String.format("%,d", v)
/**
* Compute the pruning plan for [messages] under [budgetTokens].
* Throws [IllegalArgumentException] on a negative budget or any negative
* per-message token count.
*/
fun planPrune(messages: List<ConversationMessage>, budgetTokens: Int): PrunePlan {
val warnings = mutableListOf<String>()
require(budgetTokens >= 0) { "budgetTokens must be >= 0" }
require(messages.none { it.tokens < 0 }) { "message tokens must be >= 0" }
val n = messages.size
val lastUser = messages.indexOfLast { it.role == ChatRole.USER }
// Untouchable: every system message, pinned messages, the first turn
// (the opening user request), and the current request (the last user
// message and everything after it).
val protectedIdx = buildSet {
messages.forEachIndexed { i, m -> if (m.role == ChatRole.SYSTEM || m.pinned) add(i) }
if (n > 0) add(0)
val firstTurn = messages.indexOfFirst { it.role != ChatRole.SYSTEM }
if (firstTurn != -1) add(firstTurn)
val from = if (lastUser == -1) n - 1 else lastUser
for (i in maxOf(from, 0) until n) add(i)
}
val protectedTokens = protectedIdx.sumOf { messages[it].tokens }
if (protectedTokens > budgetTokens) {
warnings.add(
"Protected messages alone are ${fmt(protectedTokens)} tokens against a " +
"${fmt(budgetTokens)} budget — raise the budget (or reserve less for the reply) " +
"before pruning anything else.",
)
}
// Fill the remaining budget newest-to-oldest through the middle.
val actions = Array(n) { PruneAction.DROP }
protectedIdx.forEach { actions[it] = PruneAction.KEEP }
var used = protectedTokens
for (i in n - 1 downTo 0) {
if (actions[i] != PruneAction.DROP) continue
val t = messages[i].tokens
if (used + t <= budgetTokens) {
actions[i] = PruneAction.KEEP
used += t
} else break // oldest-unfilled remain drop/summarize candidates
}
// Everything still 'drop' in the middle folds into ONE running summary
// when the compressed form fits where the raw turns did not.
val summarizeIdx = (0 until n).filter { actions[it] == PruneAction.DROP && it !in protectedIdx }
val summarizeTokens = summarizeIdx.sumOf { messages[it].tokens }
val attempted =
if (summarizeIdx.isEmpty()) 0
else SUMMARY_FIXED_TOKENS + ceil(summarizeTokens * SUMMARY_RATIO).toInt()
// The summary only costs anything when it is actually applied.
var summaryCost = 0
if (attempted > 0 && used + attempted <= budgetTokens) {
summarizeIdx.forEach { actions[it] = PruneAction.SUMMARIZE }
summaryCost = attempted
used += summaryCost
} else if (attempted > 0) {
warnings.add(
"Even the compressed summary (${fmt(attempted)} tokens) does not fit the " +
"remaining budget — the oldest turns are dropped instead.",
)
}
val decisions = messages.mapIndexed { i, m -> PruneDecision(i, m.role, actions[i], m.tokens) }
var kept = 0; var dropped = 0; var folded = 0
for (d in decisions) when (d.action) {
PruneAction.KEEP -> kept += d.tokens
PruneAction.DROP -> dropped += d.tokens
PruneAction.SUMMARIZE -> folded += d.tokens
}
return PrunePlan(
decisions, kept, folded, dropped, summaryCost,
kept + summaryCost, kept + summaryCost <= budgetTokens, warnings,
)
}
/** Human-readable one-line summary of a plan. */
fun describePrune(plan: PrunePlan): String {
if (!plan.fitsBudget) {
return "Does not fit: ${fmt(plan.projectedTokens)} tokens projected against the budget."
}
val parts = mutableListOf("${fmt(plan.keptTokens)} kept")
if (plan.summarizedTokens > 0) {
parts.add("${fmt(plan.summarizedTokens)} folded into a ${fmt(plan.summaryCostTokens)}-token summary")
}
if (plan.droppedTokens > 0) parts.add("${fmt(plan.droppedTokens)} dropped")
return parts.joinToString(" · ") + " — fits the budget."
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →