Cache Breakpoint Planner — TypeScript source
Find what your prompts share — common prefix and suffix blocks — and place prompt-cache breakpoints where they pay, with an estimated cost saving. 100% client-side.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Pure logic for the Cache Breakpoint Planner tool (slug: cache-breakpoint-planner).
// Finds the structure shared across a set of prompts (common prefix/suffix
// blocks by position) and places cache breakpoints where they pay: everything
// stable goes before the breakpoint, everything per-request after it. Token
// figures reuse the tokenEstimator prose heuristic; savings use the 10×
// cached-read discount providers like Anthropic document.
import { estimateTokens } from './tokenEstimator';
export interface PromptSession {
id: string;
/** Ordered prompt blocks (system, docs, history turns, user ask…). */
blocks: string[];
}
export interface Breakpoint {
/** Place the cache breakpoint AFTER this block index (0-based). */
afterBlock: number;
label: string;
reason: string;
cachedTokens: number;
}
export interface BreakpointPlan {
/** Blocks shared by every session, in order, at the front. */
prefixBlocks: string[];
prefixTokens: number;
/** Blocks shared by every session at the END. */
suffixBlocks: string[];
suffixTokens: number;
breakpoints: Breakpoint[];
perSession: { id: string; totalTokens: number; uniqueTokens: number; cachedRatio: number }[];
/** Estimated cost saving across the sessions vs no caching (0–1). */
estimatedSavings: number;
warnings: string[];
}
/** Cached reads bill at ~0.1× — the saving on the cached share is ~90%. */
export const CACHE_READ_DISCOUNT = 0.1;
export function planBreakpoints(sessions: readonly PromptSession[]): BreakpointPlan {
const warnings: string[] = [];
const valid = sessions.filter((s) => Array.isArray(s.blocks));
if (valid.length === 0) {
return {
prefixBlocks: [],
prefixTokens: 0,
suffixBlocks: [],
suffixTokens: 0,
breakpoints: [],
perSession: [],
estimatedSavings: 0,
warnings: ['No sessions given — paste at least two prompts to compare.'],
};
}
if (valid.length === 1) {
warnings.push('Only one session — a prefix needs at least two prompts to detect.');
}
// Common leading blocks by position.
const shortest = Math.min(...valid.map((s) => s.blocks.length));
let prefixEnd = 0;
while (
prefixEnd < shortest &&
valid.every((s) => s.blocks[prefixEnd] === valid[0].blocks[prefixEnd])
) {
prefixEnd++;
}
// Common trailing blocks, matched from each session's own tail, never
// overlapping the prefix.
let suffixLen = 0;
while (
suffixLen < shortest - prefixEnd &&
valid.every((s) => s.blocks[s.blocks.length - 1 - suffixLen] === valid[0].blocks[valid[0].blocks.length - 1 - suffixLen])
) {
suffixLen++;
}
const prefixBlocks = valid[0].blocks.slice(0, prefixEnd);
const suffixBlocks = suffixLen > 0 ? valid[0].blocks.slice(valid[0].blocks.length - suffixLen) : [];
const prefixTokens = tok(prefixBlocks.join('\n'));
const suffixTokens = tok(suffixBlocks.join('\n'));
const breakpoints: Breakpoint[] = [];
if (prefixBlocks.length > 0) {
breakpoints.push({
afterBlock: prefixEnd - 1,
label: 'after the shared prefix',
reason: `${prefixBlocks.length} block(s) identical across every session — cache once, hit on every request.`,
cachedTokens: prefixTokens,
});
}
if (suffixLen > 0) {
breakpoints.push({
afterBlock: -1, // terminal: the shared tail sits at the end of each request
label: 'shared tail',
reason: `${suffixLen} trailing block(s) also identical — extend the cache segment or accept the re-read.`,
cachedTokens: suffixTokens,
});
}
if (breakpoints.length === 0) {
warnings.push('No shared leading or trailing blocks — nothing to cache across these sessions.');
}
const perSession = valid.map((s) => {
const totalTokens = tok(s.blocks.join('\n'));
const uniqueTokens = totalTokens - prefixTokens - suffixTokens;
return {
id: s.id,
totalTokens,
uniqueTokens: Math.max(uniqueTokens, 0),
cachedRatio: totalTokens > 0 ? Math.min((prefixTokens + suffixTokens) / totalTokens, 1) : 0,
};
});
const avgTotal = perSession.reduce((sum, p) => sum + p.totalTokens, 0) / perSession.length;
const cachedShare = avgTotal > 0 ? Math.min((prefixTokens + suffixTokens) / avgTotal, 1) : 0;
const estimatedSavings = cachedShare * (1 - CACHE_READ_DISCOUNT);
return {
prefixBlocks,
prefixTokens,
suffixBlocks,
suffixTokens,
breakpoints,
perSession,
estimatedSavings,
warnings,
};
}
function tok(text: string): number {
return text ? estimateTokens(text, { type: 'prose' }).tokens : 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →