RAG Chunk Comparator — Swift source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.
// RAG Chunk Comparator — chunk one document three ways and report the stats
// that matter for retrieval.
//
// Language: Swift (5.9+, zero dependencies)
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
import Foundation
enum ChunkStrategy: String, CaseIterable {
case fixed, sentence, markdown
}
/// Errors mirroring the TS RangeError contract.
enum ChunkError: Error, CustomStringConvertible {
case sizeTokens
case overlapTokens
var description: String {
switch self {
case .sizeTokens: return "sizeTokens must be > 0"
case .overlapTokens: return "overlapTokens must be in [0, sizeTokens)"
}
}
}
/// Target chunk size in tokens + fixed-strategy overlap.
struct ChunkOptions {
var sizeTokens: Int
var overlapTokens: Int = 0
}
/// One chunk: index, text, tokens, and (markdown only) the section heading.
struct Chunk {
var index: Int
var text: String
var tokens: Int
var heading: String?
}
struct StrategyStats {
var count: Int
var minTokens: Int
var maxTokens: Int
var avgTokens: Int
/// Share of chunk boundaries that fall on a sentence end (0...1).
var sentenceBoundaryShare: Double
}
struct StrategyResult {
var strategy: ChunkStrategy
var chunks: [Chunk]
var stats: StrategyStats
}
/// The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
/// line costs max(1, round(length / 4)) tokens; empty text is 0.
private func tok(_ s: String) -> Int {
if s.isEmpty { return 0 }
var tokens = 0
for line in s.split(separator: "\n", omittingEmptySubsequences: false) {
if !line.isEmpty {
tokens += max(1, Int((Double(line.count) / 4.0).rounded()))
}
}
return tokens
}
/// Split on sentence enders followed by whitespace or end of text.
func splitSentences(_ text: String) -> [String] {
let collapsed = text
.split(whereSeparator: { $0.isWhitespace })
.joined(separator: " ")
var out: [String] = []
var current = ""
for ch in collapsed {
current.append(ch)
if ch == "." || ch == "!" || ch == "?" {
out.append(current)
current = ""
}
}
if !current.isEmpty { out.append(current) }
return out.map { $0.trimmingCharacters(in: .whitespaces) }.filter { !$0.isEmpty }
}
private func endsSentence(_ s: String) -> Bool {
let t = String(s.reversed())
var i = t.startIndex
if i < t.endIndex, "\"')]".contains(t[i]) {
i = t.index(after: i)
}
guard i < t.endIndex else { return false }
let c = t[i]
return c == "." || c == "!" || c == "?"
}
/// Greedy character accumulation to a token target (overlapping allowed).
func chunkFixed(_ text: String, _ opts: ChunkOptions) throws -> [Chunk] {
if opts.sizeTokens <= 0 { throw ChunkError.sizeTokens }
if opts.overlapTokens < 0 || opts.overlapTokens >= opts.sizeTokens {
throw ChunkError.overlapTokens
}
let clean = text.trimmingCharacters(in: .whitespacesAndNewlines)
if clean.isEmpty { return [] }
let chars = Array(clean)
// ~4 chars per prose token: step by tokens, verify with the estimator.
let charStep = max(1, Int((Double(opts.sizeTokens) * 4.0).rounded()))
let overlapChars = Int((Double(opts.overlapTokens) * 4.0).rounded())
var chunks: [Chunk] = []
var start = 0
while start < chars.count {
var end = min(start + charStep, chars.count)
// Prefer cutting at whitespace near the target.
if end < chars.count {
var cut = -1
var i = end
while i > start {
i -= 1
if chars[i] == " " { cut = i; break }
}
if cut > start { end = cut }
}
let piece = String(chars[start..<end]).trimmingCharacters(in: .whitespaces)
if !piece.isEmpty {
chunks.append(Chunk(index: chunks.count, text: piece, tokens: tok(piece), heading: nil))
}
if end >= chars.count { break }
start = max(end - overlapChars, start + 1)
}
return chunks
}
/// Group whole sentences up to the token target; boundaries never split a
/// sentence. A single sentence larger than the target becomes its own chunk.
func chunkBySentences(_ text: String, _ opts: ChunkOptions) throws -> [Chunk] {
if opts.sizeTokens <= 0 { throw ChunkError.sizeTokens }
let sentences = splitSentences(text)
if sentences.isEmpty { return [] }
var chunks: [Chunk] = []
var current: [String] = []
var currentTokens = 0
func flush() {
guard !current.isEmpty else { return }
let piece = current.joined(separator: " ")
chunks.append(Chunk(index: chunks.count, text: piece, tokens: tok(piece), heading: nil))
current = []
currentTokens = 0
}
for sentence in sentences {
let t = tok(sentence)
if currentTokens > 0 && currentTokens + t > opts.sizeTokens { flush() }
current.append(sentence)
currentTokens += t
}
flush()
return chunks
}
/// One markdown line's heading text, when the line is 1-6 '#' + whitespace.
private func headingText(_ line: String) -> String? {
var hashes = 0
var idx = line.startIndex
while idx < line.endIndex, line[idx] == "#" {
hashes += 1
idx = line.index(after: idx)
}
guard hashes >= 1, hashes <= 6, idx < line.endIndex,
line[idx] == " " || line[idx] == "\t" else { return nil }
let rest = line[idx...].drop(while: { $0 == " " || $0 == "\t" })
let text = String(rest).trimmingCharacters(in: .whitespaces)
return text.isEmpty ? nil : text
}
/// Split on markdown headings; oversized sections fall back to sentence
/// grouping, and every chunk carries the section heading.
func chunkMarkdown(_ text: String, _ opts: ChunkOptions) throws -> [Chunk] {
if opts.sizeTokens <= 0 { throw ChunkError.sizeTokens }
var sections: [(heading: String?, body: [String])] = []
var pendingHeading: String? = nil
var pendingBody: [String] = []
for line in text.split(separator: "\n", omittingEmptySubsequences: false) {
if let h = headingText(String(line)) {
if !pendingBody.isEmpty {
sections.append((pendingHeading, pendingBody))
}
pendingHeading = h
pendingBody = []
} else {
pendingBody.append(String(line))
}
}
if !pendingBody.isEmpty { sections.append((pendingHeading, pendingBody)) }
var chunks: [Chunk] = []
for section in sections {
let clean = section.body.joined(separator: "\n")
.trimmingCharacters(in: .whitespacesAndNewlines)
if clean.isEmpty { continue }
let whole = section.heading != nil ? "# \(section.heading!)\n\(clean)" : clean
if tok(whole) <= opts.sizeTokens {
chunks.append(Chunk(index: chunks.count, text: whole, tokens: tok(whole),
heading: section.heading))
continue
}
for c in try chunkBySentences(clean, opts) {
chunks.append(Chunk(index: chunks.count, text: c.text, tokens: c.tokens,
heading: section.heading))
}
}
return chunks
}
private func statsFor(_ strategy: ChunkStrategy, _ chunks: [Chunk]) -> StrategyResult {
let count = chunks.count
let sizes = chunks.map { $0.tokens }
let boundaries = chunks.dropLast().map { endsSentence($0.text) }
return StrategyResult(
strategy: strategy,
chunks: chunks,
stats: StrategyStats(
count: count,
minTokens: count > 0 ? sizes.min()! : 0,
maxTokens: count > 0 ? sizes.max()! : 0,
avgTokens: count > 0 ? Int((Double(sizes.reduce(0, +)) / Double(count)).rounded()) : 0,
sentenceBoundaryShare: boundaries.isEmpty
? 1.0 // a single chunk has no internal boundaries to botch
: Double(boundaries.filter { $0 }.count) / Double(boundaries.count)))
}
/// Run all three strategies over one document and report comparable stats.
func compareStrategies(_ text: String, _ opts: ChunkOptions) throws
-> [ChunkStrategy: StrategyResult]
{
[
.fixed: statsFor(.fixed, try chunkFixed(text, opts)),
.sentence: statsFor(.sentence, try chunkBySentences(text, opts)),
.markdown: statsFor(.markdown, try chunkMarkdown(text, opts)),
]
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →