Skip to content

Cache Breakpoint Planner — TypeScript source

Find what your prompts share — common prefix and suffix blocks — and place prompt-cache breakpoints where they pay, with an estimated cost saving. 100% client-side.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Pure logic for the Cache Breakpoint Planner tool (slug: cache-breakpoint-planner).
// Finds the structure shared across a set of prompts (common prefix/suffix
// blocks by position) and places cache breakpoints where they pay: everything
// stable goes before the breakpoint, everything per-request after it. Token
// figures reuse the tokenEstimator prose heuristic; savings use the 10×
// cached-read discount providers like Anthropic document.
import { estimateTokens } from './tokenEstimator';

export interface PromptSession {
  id: string;
  /** Ordered prompt blocks (system, docs, history turns, user ask…). */
  blocks: string[];
}

export interface Breakpoint {
  /** Place the cache breakpoint AFTER this block index (0-based). */
  afterBlock: number;
  label: string;
  reason: string;
  cachedTokens: number;
}

export interface BreakpointPlan {
  /** Blocks shared by every session, in order, at the front. */
  prefixBlocks: string[];
  prefixTokens: number;
  /** Blocks shared by every session at the END. */
  suffixBlocks: string[];
  suffixTokens: number;
  breakpoints: Breakpoint[];
  perSession: { id: string; totalTokens: number; uniqueTokens: number; cachedRatio: number }[];
  /** Estimated cost saving across the sessions vs no caching (0–1). */
  estimatedSavings: number;
  warnings: string[];
}

/** Cached reads bill at ~0.1× — the saving on the cached share is ~90%. */
export const CACHE_READ_DISCOUNT = 0.1;

export function planBreakpoints(sessions: readonly PromptSession[]): BreakpointPlan {
  const warnings: string[] = [];
  const valid = sessions.filter((s) => Array.isArray(s.blocks));

  if (valid.length === 0) {
    return {
      prefixBlocks: [],
      prefixTokens: 0,
      suffixBlocks: [],
      suffixTokens: 0,
      breakpoints: [],
      perSession: [],
      estimatedSavings: 0,
      warnings: ['No sessions given — paste at least two prompts to compare.'],
    };
  }
  if (valid.length === 1) {
    warnings.push('Only one session — a prefix needs at least two prompts to detect.');
  }

  // Common leading blocks by position.
  const shortest = Math.min(...valid.map((s) => s.blocks.length));
  let prefixEnd = 0;
  while (
    prefixEnd < shortest &&
    valid.every((s) => s.blocks[prefixEnd] === valid[0].blocks[prefixEnd])
  ) {
    prefixEnd++;
  }

  // Common trailing blocks, matched from each session's own tail, never
  // overlapping the prefix.
  let suffixLen = 0;
  while (
    suffixLen < shortest - prefixEnd &&
    valid.every((s) => s.blocks[s.blocks.length - 1 - suffixLen] === valid[0].blocks[valid[0].blocks.length - 1 - suffixLen])
  ) {
    suffixLen++;
  }

  const prefixBlocks = valid[0].blocks.slice(0, prefixEnd);
  const suffixBlocks = suffixLen > 0 ? valid[0].blocks.slice(valid[0].blocks.length - suffixLen) : [];
  const prefixTokens = tok(prefixBlocks.join('\n'));
  const suffixTokens = tok(suffixBlocks.join('\n'));

  const breakpoints: Breakpoint[] = [];
  if (prefixBlocks.length > 0) {
    breakpoints.push({
      afterBlock: prefixEnd - 1,
      label: 'after the shared prefix',
      reason: `${prefixBlocks.length} block(s) identical across every session — cache once, hit on every request.`,
      cachedTokens: prefixTokens,
    });
  }
  if (suffixLen > 0) {
    breakpoints.push({
      afterBlock: -1, // terminal: the shared tail sits at the end of each request
      label: 'shared tail',
      reason: `${suffixLen} trailing block(s) also identical — extend the cache segment or accept the re-read.`,
      cachedTokens: suffixTokens,
    });
  }
  if (breakpoints.length === 0) {
    warnings.push('No shared leading or trailing blocks — nothing to cache across these sessions.');
  }

  const perSession = valid.map((s) => {
    const totalTokens = tok(s.blocks.join('\n'));
    const uniqueTokens = totalTokens - prefixTokens - suffixTokens;
    return {
      id: s.id,
      totalTokens,
      uniqueTokens: Math.max(uniqueTokens, 0),
      cachedRatio: totalTokens > 0 ? Math.min((prefixTokens + suffixTokens) / totalTokens, 1) : 0,
    };
  });

  const avgTotal = perSession.reduce((sum, p) => sum + p.totalTokens, 0) / perSession.length;
  const cachedShare = avgTotal > 0 ? Math.min((prefixTokens + suffixTokens) / avgTotal, 1) : 0;
  const estimatedSavings = cachedShare * (1 - CACHE_READ_DISCOUNT);

  return {
    prefixBlocks,
    prefixTokens,
    suffixBlocks,
    suffixTokens,
    breakpoints,
    perSession,
    estimatedSavings,
    warnings,
  };
}

function tok(text: string): number {
  return text ? estimateTokens(text, { type: 'prose' }).tokens : 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →