Skip to content

Cache Breakpoint Planner — Swift source

Find what your prompts share — common prefix and suffix blocks — and place prompt-cache breakpoints where they pay, with an estimated cost saving. 100% client-side.

This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Cache Breakpoint Planner — find the blocks a set of prompts share and
// place cache breakpoints where they pay.
//
// Language: Swift (5.9+, zero dependencies)
// Port of src/lib/cacheBreakpointPlanner.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/cache-breakpoint-planner

/// Cached reads bill at ~0.1x — the saving on the cached share is ~90%.
let CACHE_READ_DISCOUNT = 0.1

/// One prompt session: an id plus its ordered blocks.
struct PromptSession {
    var id: String
    var blocks: [String]
}

/// Place the cache breakpoint AFTER this block index (0-based); -1 = terminal.
struct Breakpoint {
    var afterBlock: Int
    var label: String
    var reason: String
    var cachedTokens: Int
}

struct PerSessionRow {
    var id: String
    var totalTokens: Int
    var uniqueTokens: Int
    var cachedRatio: Double
}

/// The full plan: shared blocks, breakpoints, rows, savings.
struct BreakpointPlan {
    var prefixBlocks: [String]
    var prefixTokens: Int
    var suffixBlocks: [String]
    var suffixTokens: Int
    var breakpoints: [Breakpoint]
    var perSession: [PerSessionRow]
    /// Estimated cost saving across the sessions vs no caching (0...1).
    var estimatedSavings: Double
    var warnings: [String]
}

/// The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
/// line costs max(1, round(length / 4)) tokens; empty text is 0.
private func tok(_ text: String) -> Int {
    if text.isEmpty { return 0 }
    var tokens = 0
    for line in text.split(separator: "\n", omittingEmptySubsequences: false) {
        if !line.isEmpty {
            tokens += max(1, Int((Double(line.count) / 4.0).rounded()))
        }
    }
    return tokens
}

/// Plan cache breakpoints for a set of prompt sessions: find the common
/// leading/trailing blocks across every session and place breakpoints where
/// the cache pays.
func planBreakpoints(_ sessions: [PromptSession]) -> BreakpointPlan {
    var warnings: [String] = []
    let valid = sessions // Swift typing stands in for TS's Array.isArray filter

    if valid.isEmpty {
        return BreakpointPlan(
            prefixBlocks: [], prefixTokens: 0, suffixBlocks: [], suffixTokens: 0,
            breakpoints: [], perSession: [], estimatedSavings: 0.0,
            warnings: ["No sessions given — paste at least two prompts to compare."])
    }
    if valid.count == 1 {
        warnings.append("Only one session — a prefix needs at least two prompts to detect.")
    }

    // Common leading blocks by position.
    let shortest = valid.map { $0.blocks.count }.min() ?? 0
    var prefixEnd = 0
    while prefixEnd < shortest
        && valid.allSatisfy({ $0.blocks[prefixEnd] == valid[0].blocks[prefixEnd] })
    {
        prefixEnd += 1
    }

    // Common trailing blocks, matched from each session's own tail, never
    // overlapping the prefix.
    var suffixLen = 0
    while suffixLen < shortest - prefixEnd
        && valid.allSatisfy({
            $0.blocks[$0.blocks.count - 1 - suffixLen]
                == valid[0].blocks[valid[0].blocks.count - 1 - suffixLen]
        })
    {
        suffixLen += 1
    }

    let prefixBlocks = Array(valid[0].blocks[..<prefixEnd])
    let suffixBlocks: [String] = suffixLen > 0
        ? Array(valid[0].blocks[(valid[0].blocks.count - suffixLen)...])
        : []
    let prefixTokens = tok(prefixBlocks.joined(separator: "\n"))
    let suffixTokens = tok(suffixBlocks.joined(separator: "\n"))

    var breakpoints: [Breakpoint] = []
    if !prefixBlocks.isEmpty {
        breakpoints.append(Breakpoint(
            afterBlock: prefixEnd - 1,
            label: "after the shared prefix",
            reason: "\(prefixBlocks.count) block(s) identical across every session — "
                + "cache once, hit on every request.",
            cachedTokens: prefixTokens))
    }
    if suffixLen > 0 {
        breakpoints.append(Breakpoint(
            afterBlock: -1, // terminal: the shared tail sits at the end
            label: "shared tail",
            reason: "\(suffixLen) trailing block(s) also identical — extend the cache "
                + "segment or accept the re-read.",
            cachedTokens: suffixTokens))
    }
    if breakpoints.isEmpty {
        warnings.append(
            "No shared leading or trailing blocks — nothing to cache across these sessions.")
    }

    let perSession: [PerSessionRow] = valid.map { s in
        let totalTokens = tok(s.blocks.joined(separator: "\n"))
        let unique = max(totalTokens - prefixTokens - suffixTokens, 0)
        let cachedRatio = totalTokens > 0
            ? min(Double(prefixTokens + suffixTokens) / Double(totalTokens), 1.0)
            : 0.0
        return PerSessionRow(id: s.id, totalTokens: totalTokens, uniqueTokens: unique,
                             cachedRatio: cachedRatio)
    }

    let avgTotal = perSession.map { Double($0.totalTokens) }.reduce(0, +)
        / Double(perSession.count)
    let cachedShare = avgTotal > 0
        ? min(Double(prefixTokens + suffixTokens) / avgTotal, 1.0)
        : 0.0
    let estimatedSavings = cachedShare * (1 - CACHE_READ_DISCOUNT)

    return BreakpointPlan(
        prefixBlocks: prefixBlocks,
        prefixTokens: prefixTokens,
        suffixBlocks: suffixBlocks,
        suffixTokens: suffixTokens,
        breakpoints: breakpoints,
        perSession: perSession,
        estimatedSavings: estimatedSavings,
        warnings: warnings)
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →