Skip to content

Cache Breakpoint Planner — Go source

Find what your prompts share — common prefix and suffix blocks — and place prompt-cache breakpoints where they pay, with an estimated cost saving. 100% client-side.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package cachebreakpointplanner is the Go twin of CosmoDev's
// src/lib/cacheBreakpointPlanner.ts (dual source: the web lib is TypeScript,
// the CLI lib is Go — kept in lock-step). Pure + deterministic, never panics.
// The table-driven tests in cachebreakpointplanner_test.go share vectors with
// src/lib/cacheBreakpointPlanner.test.ts so the two implementations are held
// to the same contract.
//
// The package finds the structure shared across a set of prompts (common
// prefix/suffix blocks by position) and places cache breakpoints where they
// pay: everything stable goes before the breakpoint, everything per-request
// after it. Token figures reuse the token-estimator twin's prose heuristic
// (cosmodev/token-estimator, mirroring the TS lib's estimateTokens import);
// savings use the 10× cached-read discount providers like Anthropic document.
//
// The TS surface takes no options — planBreakpoints(sessions) only — so, unlike
// the slugify twin, there is no Options struct to default.
package cachebreakpointplanner

import (
	"fmt"
	"strings"

	tokenestimator "cosmodev/token-estimator"
)

// PromptSession is one pasted prompt with its ordered blocks (system, docs,
// history turns, user ask…). Field-for-field twin of the TS PromptSession
// interface (camelCase JSON tags, module convention).
type PromptSession struct {
	ID     string   `json:"id"`
	Blocks []string `json:"blocks"`
}

// Breakpoint is one recommended cache breakpoint. Mirrors the TS Breakpoint
// interface.
type Breakpoint struct {
	// Place the cache breakpoint AFTER this block index (0-based).
	AfterBlock   int    `json:"afterBlock"`
	Label        string `json:"label"`
	Reason       string `json:"reason"`
	CachedTokens int    `json:"cachedTokens"`
}

// PerSessionRow is the per-prompt token accounting. Mirrors the anonymous
// perSession entry type of the TS BreakpointPlan.
type PerSessionRow struct {
	ID           string  `json:"id"`
	TotalTokens  int     `json:"totalTokens"`
	UniqueTokens int     `json:"uniqueTokens"`
	CachedRatio  float64 `json:"cachedRatio"`
}

// BreakpointPlan is the full result of planning cache breakpoints across a
// set of sessions. Mirrors the TS BreakpointPlan interface.
type BreakpointPlan struct {
	// Blocks shared by every session, in order, at the front.
	PrefixBlocks []string `json:"prefixBlocks"`
	PrefixTokens int      `json:"prefixTokens"`
	// Blocks shared by every session at the END.
	SuffixBlocks []string        `json:"suffixBlocks"`
	SuffixTokens int             `json:"suffixTokens"`
	Breakpoints  []Breakpoint    `json:"breakpoints"`
	PerSession   []PerSessionRow `json:"perSession"`
	// EstimatedSavings is the estimated cost saving across the sessions vs no
	// caching (0–1).
	EstimatedSavings float64  `json:"estimatedSavings"`
	Warnings         []string `json:"warnings"`
}

// CacheReadDiscount is the rate cached reads bill at (~0.1×) — the saving on
// the cached share is ~90%. Mirrors CACHE_READ_DISCOUNT.
const CacheReadDiscount = 0.1

// PlanBreakpoints finds the shared prefix/suffix structure of sessions and
// returns where to place cache breakpoints. It is the Go twin of
// planBreakpoints() in src/lib/cacheBreakpointPlanner.ts and must agree with
// it on every shared vector.
func PlanBreakpoints(sessions []PromptSession) BreakpointPlan {
	warnings := []string{}
	// TS filters sessions to those whose blocks is an array; a Go nil slice is
	// a valid empty block list, so every session is valid here.
	valid := sessions

	if len(valid) == 0 {
		return BreakpointPlan{
			PrefixBlocks:     []string{},
			PrefixTokens:     0,
			SuffixBlocks:     []string{},
			SuffixTokens:     0,
			Breakpoints:      []Breakpoint{},
			PerSession:       []PerSessionRow{},
			EstimatedSavings: 0,
			Warnings:         []string{"No sessions given — paste at least two prompts to compare."},
		}
	}
	if len(valid) == 1 {
		warnings = append(warnings, "Only one session — a prefix needs at least two prompts to detect.")
	}

	// Common leading blocks by position.
	shortest := len(valid[0].Blocks)
	for _, s := range valid[1:] {
		if len(s.Blocks) < shortest {
			shortest = len(s.Blocks)
		}
	}
	prefixEnd := 0
	for prefixEnd < shortest && allMatchPrefix(valid, prefixEnd) {
		prefixEnd++
	}

	// Common trailing blocks, matched from each session's own tail, never
	// overlapping the prefix.
	suffixLen := 0
	for suffixLen < shortest-prefixEnd && allMatchSuffix(valid, suffixLen) {
		suffixLen++
	}

	prefixBlocks := valid[0].Blocks[:prefixEnd]
	suffixBlocks := []string{}
	if suffixLen > 0 {
		suffixBlocks = valid[0].Blocks[len(valid[0].Blocks)-suffixLen:]
	}
	prefixTokens := tok(strings.Join(prefixBlocks, "\n"))
	suffixTokens := tok(strings.Join(suffixBlocks, "\n"))

	breakpoints := []Breakpoint{}
	if len(prefixBlocks) > 0 {
		breakpoints = append(breakpoints, Breakpoint{
			AfterBlock:   prefixEnd - 1,
			Label:        "after the shared prefix",
			Reason:       fmt.Sprintf("%d block(s) identical across every session — cache once, hit on every request.", len(prefixBlocks)),
			CachedTokens: prefixTokens,
		})
	}
	if suffixLen > 0 {
		breakpoints = append(breakpoints, Breakpoint{
			AfterBlock:   -1, // terminal: the shared tail sits at the end of each request
			Label:        "shared tail",
			Reason:       fmt.Sprintf("%d trailing block(s) also identical — extend the cache segment or accept the re-read.", suffixLen),
			CachedTokens: suffixTokens,
		})
	}
	if len(breakpoints) == 0 {
		warnings = append(warnings, "No shared leading or trailing blocks — nothing to cache across these sessions.")
	}

	perSession := make([]PerSessionRow, 0, len(valid))
	totalSum := 0
	for _, s := range valid {
		totalTokens := tok(strings.Join(s.Blocks, "\n"))
		totalSum += totalTokens
		uniqueTokens := totalTokens - prefixTokens - suffixTokens
		if uniqueTokens < 0 {
			uniqueTokens = 0
		}
		cachedRatio := 0.0
		if totalTokens > 0 {
			cachedRatio = min(float64(prefixTokens+suffixTokens)/float64(totalTokens), 1)
		}
		perSession = append(perSession, PerSessionRow{
			ID:           s.ID,
			TotalTokens:  totalTokens,
			UniqueTokens: uniqueTokens,
			CachedRatio:  cachedRatio,
		})
	}

	avgTotal := float64(totalSum) / float64(len(perSession))
	cachedShare := 0.0
	if avgTotal > 0 {
		cachedShare = min(float64(prefixTokens+suffixTokens)/avgTotal, 1)
	}
	estimatedSavings := cachedShare * (1 - CacheReadDiscount)

	return BreakpointPlan{
		PrefixBlocks:     prefixBlocks,
		PrefixTokens:     prefixTokens,
		SuffixBlocks:     suffixBlocks,
		SuffixTokens:     suffixTokens,
		Breakpoints:      breakpoints,
		PerSession:       perSession,
		EstimatedSavings: estimatedSavings,
		Warnings:         warnings,
	}
}

// allMatchPrefix reports whether every session's block at index i equals the
// first session's block at i. Callers guarantee i < len(s.Blocks).
func allMatchPrefix(valid []PromptSession, i int) bool {
	for _, s := range valid {
		if s.Blocks[i] != valid[0].Blocks[i] {
			return false
		}
	}
	return true
}

// allMatchSuffix reports whether every session's block at offset n from its
// own tail equals the first session's block at offset n from its tail.
// Callers guarantee n < shortest-prefixEnd, so both indexes stay >= prefixEnd.
func allMatchSuffix(valid []PromptSession, n int) bool {
	first := len(valid[0].Blocks) - 1 - n
	for _, s := range valid {
		if s.Blocks[len(s.Blocks)-1-n] != valid[0].Blocks[first] {
			return false
		}
	}
	return true
}

// tok estimates the token count of text with the prose heuristic. It mirrors
// the tok() helper in the TS lib (empty text is 0 tokens, everything else is
// estimated as prose).
func tok(text string) int {
	if text == "" {
		return 0
	}
	return tokenestimator.EstimateTokens(text, &tokenestimator.EstimateOptions{
		ContentType: tokenestimator.TypeProse,
	}).Tokens
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →