Skip to content

RAG Chunk Comparator — Swift source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.

// RAG Chunk Comparator — chunk one document three ways and report the stats
// that matter for retrieval.
//
// Language: Swift (5.9+, zero dependencies)
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator

import Foundation

enum ChunkStrategy: String, CaseIterable {
    case fixed, sentence, markdown
}

/// Errors mirroring the TS RangeError contract.
enum ChunkError: Error, CustomStringConvertible {
    case sizeTokens
    case overlapTokens

    var description: String {
        switch self {
        case .sizeTokens: return "sizeTokens must be > 0"
        case .overlapTokens: return "overlapTokens must be in [0, sizeTokens)"
        }
    }
}

/// Target chunk size in tokens + fixed-strategy overlap.
struct ChunkOptions {
    var sizeTokens: Int
    var overlapTokens: Int = 0
}

/// One chunk: index, text, tokens, and (markdown only) the section heading.
struct Chunk {
    var index: Int
    var text: String
    var tokens: Int
    var heading: String?
}

struct StrategyStats {
    var count: Int
    var minTokens: Int
    var maxTokens: Int
    var avgTokens: Int
    /// Share of chunk boundaries that fall on a sentence end (0...1).
    var sentenceBoundaryShare: Double
}

struct StrategyResult {
    var strategy: ChunkStrategy
    var chunks: [Chunk]
    var stats: StrategyStats
}

/// The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
/// line costs max(1, round(length / 4)) tokens; empty text is 0.
private func tok(_ s: String) -> Int {
    if s.isEmpty { return 0 }
    var tokens = 0
    for line in s.split(separator: "\n", omittingEmptySubsequences: false) {
        if !line.isEmpty {
            tokens += max(1, Int((Double(line.count) / 4.0).rounded()))
        }
    }
    return tokens
}

/// Split on sentence enders followed by whitespace or end of text.
func splitSentences(_ text: String) -> [String] {
    let collapsed = text
        .split(whereSeparator: { $0.isWhitespace })
        .joined(separator: " ")
    var out: [String] = []
    var current = ""
    for ch in collapsed {
        current.append(ch)
        if ch == "." || ch == "!" || ch == "?" {
            out.append(current)
            current = ""
        }
    }
    if !current.isEmpty { out.append(current) }
    return out.map { $0.trimmingCharacters(in: .whitespaces) }.filter { !$0.isEmpty }
}

private func endsSentence(_ s: String) -> Bool {
    let t = String(s.reversed())
    var i = t.startIndex
    if i < t.endIndex, "\"')]".contains(t[i]) {
        i = t.index(after: i)
    }
    guard i < t.endIndex else { return false }
    let c = t[i]
    return c == "." || c == "!" || c == "?"
}

/// Greedy character accumulation to a token target (overlapping allowed).
func chunkFixed(_ text: String, _ opts: ChunkOptions) throws -> [Chunk] {
    if opts.sizeTokens <= 0 { throw ChunkError.sizeTokens }
    if opts.overlapTokens < 0 || opts.overlapTokens >= opts.sizeTokens {
        throw ChunkError.overlapTokens
    }
    let clean = text.trimmingCharacters(in: .whitespacesAndNewlines)
    if clean.isEmpty { return [] }
    let chars = Array(clean)
    // ~4 chars per prose token: step by tokens, verify with the estimator.
    let charStep = max(1, Int((Double(opts.sizeTokens) * 4.0).rounded()))
    let overlapChars = Int((Double(opts.overlapTokens) * 4.0).rounded())
    var chunks: [Chunk] = []
    var start = 0
    while start < chars.count {
        var end = min(start + charStep, chars.count)
        // Prefer cutting at whitespace near the target.
        if end < chars.count {
            var cut = -1
            var i = end
            while i > start {
                i -= 1
                if chars[i] == " " { cut = i; break }
            }
            if cut > start { end = cut }
        }
        let piece = String(chars[start..<end]).trimmingCharacters(in: .whitespaces)
        if !piece.isEmpty {
            chunks.append(Chunk(index: chunks.count, text: piece, tokens: tok(piece), heading: nil))
        }
        if end >= chars.count { break }
        start = max(end - overlapChars, start + 1)
    }
    return chunks
}

/// Group whole sentences up to the token target; boundaries never split a
/// sentence. A single sentence larger than the target becomes its own chunk.
func chunkBySentences(_ text: String, _ opts: ChunkOptions) throws -> [Chunk] {
    if opts.sizeTokens <= 0 { throw ChunkError.sizeTokens }
    let sentences = splitSentences(text)
    if sentences.isEmpty { return [] }
    var chunks: [Chunk] = []
    var current: [String] = []
    var currentTokens = 0
    func flush() {
        guard !current.isEmpty else { return }
        let piece = current.joined(separator: " ")
        chunks.append(Chunk(index: chunks.count, text: piece, tokens: tok(piece), heading: nil))
        current = []
        currentTokens = 0
    }
    for sentence in sentences {
        let t = tok(sentence)
        if currentTokens > 0 && currentTokens + t > opts.sizeTokens { flush() }
        current.append(sentence)
        currentTokens += t
    }
    flush()
    return chunks
}

/// One markdown line's heading text, when the line is 1-6 '#' + whitespace.
private func headingText(_ line: String) -> String? {
    var hashes = 0
    var idx = line.startIndex
    while idx < line.endIndex, line[idx] == "#" {
        hashes += 1
        idx = line.index(after: idx)
    }
    guard hashes >= 1, hashes <= 6, idx < line.endIndex,
          line[idx] == " " || line[idx] == "\t" else { return nil }
    let rest = line[idx...].drop(while: { $0 == " " || $0 == "\t" })
    let text = String(rest).trimmingCharacters(in: .whitespaces)
    return text.isEmpty ? nil : text
}

/// Split on markdown headings; oversized sections fall back to sentence
/// grouping, and every chunk carries the section heading.
func chunkMarkdown(_ text: String, _ opts: ChunkOptions) throws -> [Chunk] {
    if opts.sizeTokens <= 0 { throw ChunkError.sizeTokens }
    var sections: [(heading: String?, body: [String])] = []
    var pendingHeading: String? = nil
    var pendingBody: [String] = []
    for line in text.split(separator: "\n", omittingEmptySubsequences: false) {
        if let h = headingText(String(line)) {
            if !pendingBody.isEmpty {
                sections.append((pendingHeading, pendingBody))
            }
            pendingHeading = h
            pendingBody = []
        } else {
            pendingBody.append(String(line))
        }
    }
    if !pendingBody.isEmpty { sections.append((pendingHeading, pendingBody)) }

    var chunks: [Chunk] = []
    for section in sections {
        let clean = section.body.joined(separator: "\n")
            .trimmingCharacters(in: .whitespacesAndNewlines)
        if clean.isEmpty { continue }
        let whole = section.heading != nil ? "# \(section.heading!)\n\(clean)" : clean
        if tok(whole) <= opts.sizeTokens {
            chunks.append(Chunk(index: chunks.count, text: whole, tokens: tok(whole),
                                heading: section.heading))
            continue
        }
        for c in try chunkBySentences(clean, opts) {
            chunks.append(Chunk(index: chunks.count, text: c.text, tokens: c.tokens,
                                heading: section.heading))
        }
    }
    return chunks
}

private func statsFor(_ strategy: ChunkStrategy, _ chunks: [Chunk]) -> StrategyResult {
    let count = chunks.count
    let sizes = chunks.map { $0.tokens }
    let boundaries = chunks.dropLast().map { endsSentence($0.text) }
    return StrategyResult(
        strategy: strategy,
        chunks: chunks,
        stats: StrategyStats(
            count: count,
            minTokens: count > 0 ? sizes.min()! : 0,
            maxTokens: count > 0 ? sizes.max()! : 0,
            avgTokens: count > 0 ? Int((Double(sizes.reduce(0, +)) / Double(count)).rounded()) : 0,
            sentenceBoundaryShare: boundaries.isEmpty
                ? 1.0 // a single chunk has no internal boundaries to botch
                : Double(boundaries.filter { $0 }.count) / Double(boundaries.count)))
}

/// Run all three strategies over one document and report comparable stats.
func compareStrategies(_ text: String, _ opts: ChunkOptions) throws
    -> [ChunkStrategy: StrategyResult]
{
    [
        .fixed: statsFor(.fixed, try chunkFixed(text, opts)),
        .sentence: statsFor(.sentence, try chunkBySentences(text, opts)),
        .markdown: statsFor(.markdown, try chunkMarkdown(text, opts)),
    ]
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →