Cache Breakpoint Planner — Go source
Find what your prompts share — common prefix and suffix blocks — and place prompt-cache breakpoints where they pay, with an estimated cost saving. 100% client-side.
This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Package cachebreakpointplanner is the Go twin of CosmoDev's
// src/lib/cacheBreakpointPlanner.ts (dual source: the web lib is TypeScript,
// the CLI lib is Go — kept in lock-step). Pure + deterministic, never panics.
// The table-driven tests in cachebreakpointplanner_test.go share vectors with
// src/lib/cacheBreakpointPlanner.test.ts so the two implementations are held
// to the same contract.
//
// The package finds the structure shared across a set of prompts (common
// prefix/suffix blocks by position) and places cache breakpoints where they
// pay: everything stable goes before the breakpoint, everything per-request
// after it. Token figures reuse the token-estimator twin's prose heuristic
// (cosmodev/token-estimator, mirroring the TS lib's estimateTokens import);
// savings use the 10× cached-read discount providers like Anthropic document.
//
// The TS surface takes no options — planBreakpoints(sessions) only — so, unlike
// the slugify twin, there is no Options struct to default.
package cachebreakpointplanner
import (
"fmt"
"strings"
tokenestimator "cosmodev/token-estimator"
)
// PromptSession is one pasted prompt with its ordered blocks (system, docs,
// history turns, user ask…). Field-for-field twin of the TS PromptSession
// interface (camelCase JSON tags, module convention).
type PromptSession struct {
ID string `json:"id"`
Blocks []string `json:"blocks"`
}
// Breakpoint is one recommended cache breakpoint. Mirrors the TS Breakpoint
// interface.
type Breakpoint struct {
// Place the cache breakpoint AFTER this block index (0-based).
AfterBlock int `json:"afterBlock"`
Label string `json:"label"`
Reason string `json:"reason"`
CachedTokens int `json:"cachedTokens"`
}
// PerSessionRow is the per-prompt token accounting. Mirrors the anonymous
// perSession entry type of the TS BreakpointPlan.
type PerSessionRow struct {
ID string `json:"id"`
TotalTokens int `json:"totalTokens"`
UniqueTokens int `json:"uniqueTokens"`
CachedRatio float64 `json:"cachedRatio"`
}
// BreakpointPlan is the full result of planning cache breakpoints across a
// set of sessions. Mirrors the TS BreakpointPlan interface.
type BreakpointPlan struct {
// Blocks shared by every session, in order, at the front.
PrefixBlocks []string `json:"prefixBlocks"`
PrefixTokens int `json:"prefixTokens"`
// Blocks shared by every session at the END.
SuffixBlocks []string `json:"suffixBlocks"`
SuffixTokens int `json:"suffixTokens"`
Breakpoints []Breakpoint `json:"breakpoints"`
PerSession []PerSessionRow `json:"perSession"`
// EstimatedSavings is the estimated cost saving across the sessions vs no
// caching (0–1).
EstimatedSavings float64 `json:"estimatedSavings"`
Warnings []string `json:"warnings"`
}
// CacheReadDiscount is the rate cached reads bill at (~0.1×) — the saving on
// the cached share is ~90%. Mirrors CACHE_READ_DISCOUNT.
const CacheReadDiscount = 0.1
// PlanBreakpoints finds the shared prefix/suffix structure of sessions and
// returns where to place cache breakpoints. It is the Go twin of
// planBreakpoints() in src/lib/cacheBreakpointPlanner.ts and must agree with
// it on every shared vector.
func PlanBreakpoints(sessions []PromptSession) BreakpointPlan {
warnings := []string{}
// TS filters sessions to those whose blocks is an array; a Go nil slice is
// a valid empty block list, so every session is valid here.
valid := sessions
if len(valid) == 0 {
return BreakpointPlan{
PrefixBlocks: []string{},
PrefixTokens: 0,
SuffixBlocks: []string{},
SuffixTokens: 0,
Breakpoints: []Breakpoint{},
PerSession: []PerSessionRow{},
EstimatedSavings: 0,
Warnings: []string{"No sessions given — paste at least two prompts to compare."},
}
}
if len(valid) == 1 {
warnings = append(warnings, "Only one session — a prefix needs at least two prompts to detect.")
}
// Common leading blocks by position.
shortest := len(valid[0].Blocks)
for _, s := range valid[1:] {
if len(s.Blocks) < shortest {
shortest = len(s.Blocks)
}
}
prefixEnd := 0
for prefixEnd < shortest && allMatchPrefix(valid, prefixEnd) {
prefixEnd++
}
// Common trailing blocks, matched from each session's own tail, never
// overlapping the prefix.
suffixLen := 0
for suffixLen < shortest-prefixEnd && allMatchSuffix(valid, suffixLen) {
suffixLen++
}
prefixBlocks := valid[0].Blocks[:prefixEnd]
suffixBlocks := []string{}
if suffixLen > 0 {
suffixBlocks = valid[0].Blocks[len(valid[0].Blocks)-suffixLen:]
}
prefixTokens := tok(strings.Join(prefixBlocks, "\n"))
suffixTokens := tok(strings.Join(suffixBlocks, "\n"))
breakpoints := []Breakpoint{}
if len(prefixBlocks) > 0 {
breakpoints = append(breakpoints, Breakpoint{
AfterBlock: prefixEnd - 1,
Label: "after the shared prefix",
Reason: fmt.Sprintf("%d block(s) identical across every session — cache once, hit on every request.", len(prefixBlocks)),
CachedTokens: prefixTokens,
})
}
if suffixLen > 0 {
breakpoints = append(breakpoints, Breakpoint{
AfterBlock: -1, // terminal: the shared tail sits at the end of each request
Label: "shared tail",
Reason: fmt.Sprintf("%d trailing block(s) also identical — extend the cache segment or accept the re-read.", suffixLen),
CachedTokens: suffixTokens,
})
}
if len(breakpoints) == 0 {
warnings = append(warnings, "No shared leading or trailing blocks — nothing to cache across these sessions.")
}
perSession := make([]PerSessionRow, 0, len(valid))
totalSum := 0
for _, s := range valid {
totalTokens := tok(strings.Join(s.Blocks, "\n"))
totalSum += totalTokens
uniqueTokens := totalTokens - prefixTokens - suffixTokens
if uniqueTokens < 0 {
uniqueTokens = 0
}
cachedRatio := 0.0
if totalTokens > 0 {
cachedRatio = min(float64(prefixTokens+suffixTokens)/float64(totalTokens), 1)
}
perSession = append(perSession, PerSessionRow{
ID: s.ID,
TotalTokens: totalTokens,
UniqueTokens: uniqueTokens,
CachedRatio: cachedRatio,
})
}
avgTotal := float64(totalSum) / float64(len(perSession))
cachedShare := 0.0
if avgTotal > 0 {
cachedShare = min(float64(prefixTokens+suffixTokens)/avgTotal, 1)
}
estimatedSavings := cachedShare * (1 - CacheReadDiscount)
return BreakpointPlan{
PrefixBlocks: prefixBlocks,
PrefixTokens: prefixTokens,
SuffixBlocks: suffixBlocks,
SuffixTokens: suffixTokens,
Breakpoints: breakpoints,
PerSession: perSession,
EstimatedSavings: estimatedSavings,
Warnings: warnings,
}
}
// allMatchPrefix reports whether every session's block at index i equals the
// first session's block at i. Callers guarantee i < len(s.Blocks).
func allMatchPrefix(valid []PromptSession, i int) bool {
for _, s := range valid {
if s.Blocks[i] != valid[0].Blocks[i] {
return false
}
}
return true
}
// allMatchSuffix reports whether every session's block at offset n from its
// own tail equals the first session's block at offset n from its tail.
// Callers guarantee n < shortest-prefixEnd, so both indexes stay >= prefixEnd.
func allMatchSuffix(valid []PromptSession, n int) bool {
first := len(valid[0].Blocks) - 1 - n
for _, s := range valid {
if s.Blocks[len(s.Blocks)-1-n] != valid[0].Blocks[first] {
return false
}
}
return true
}
// tok estimates the token count of text with the prose heuristic. It mirrors
// the tok() helper in the TS lib (empty text is 0 tokens, everything else is
// estimated as prose).
func tok(text string) int {
if text == "" {
return 0
}
return tokenestimator.EstimateTokens(text, &tokenestimator.EstimateOptions{
ContentType: tokenestimator.TypeProse,
}).Tokens
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →