RAG Chunk Comparator — Kotlin source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.
// RAG Chunk Comparator — chunk one document three ways and report the stats
// that matter for retrieval.
//
// Language: Kotlin (JVM 17+, zero dependencies)
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
import kotlin.math.max
import kotlin.math.min
import kotlin.math.roundToInt
enum class ChunkStrategy { FIXED, SENTENCE, MARKDOWN }
/** Target chunk size in tokens + fixed-strategy overlap. */
data class ChunkOptions(val sizeTokens: Int, val overlapTokens: Int = 0)
data class Chunk(
val index: Int,
val text: String,
val tokens: Int,
/** Nearest markdown heading for markdown chunks; null otherwise. */
val heading: String? = null,
)
data class StrategyStats(
val count: Int,
val minTokens: Int,
val maxTokens: Int,
val avgTokens: Int,
/** Share of chunk boundaries that fall on a sentence end (0..1). */
val sentenceBoundaryShare: Double,
)
data class StrategyResult(
val strategy: ChunkStrategy,
val chunks: List<Chunk>,
val stats: StrategyStats,
)
private val SENTENCE_SPLIT = Regex("(?<=[.!?]) +")
private val ENDS_SENTENCE = Regex("[.!?][\"'\\)\\]]?$")
private val HEADING = Regex("^(#{1,6})\\s+(.*)$")
/**
* The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
* line costs max(1, round(length / 4)) tokens; empty text is 0.
*/
private fun tok(s: String): Int {
if (s.isEmpty()) return 0
var tokens = 0
for (line in s.split('\n')) {
if (line.isNotEmpty()) tokens += max(1, (line.length / 4.0).roundToInt())
}
return tokens
}
/** Split on sentence enders followed by whitespace or end of text. */
fun splitSentences(text: String): List<String> =
SENTENCE_SPLIT.split(text.replace(Regex("\\s+"), " ").trim()).filter { it.isNotEmpty() }
private fun endsSentence(s: String): Boolean = ENDS_SENTENCE.containsMatchIn(s.trim())
/**
* Greedy character accumulation to a token target (overlapping allowed).
* @throws IllegalArgumentException on impossible options (the TS RangeError contract).
*/
fun chunkFixed(text: String, opts: ChunkOptions): List<Chunk> {
require(opts.sizeTokens > 0) { "sizeTokens must be > 0" }
require(opts.overlapTokens in 0 until opts.sizeTokens) {
"overlapTokens must be in [0, sizeTokens)"
}
val clean = text.trim()
if (clean.isEmpty()) return emptyList()
// ~4 chars per prose token: step by tokens, verify with the estimator.
val charStep = max(1, (opts.sizeTokens * 4.0).roundToInt())
val overlapChars = (opts.overlapTokens * 4.0).roundToInt()
val chunks = mutableListOf<Chunk>()
var start = 0
while (start < clean.length) {
var end = min(start + charStep, clean.length)
// Prefer cutting at whitespace near the target.
if (end < clean.length) {
val cut = clean.lastIndexOf(' ', end)
if (cut > start) end = cut
}
val piece = clean.substring(start, end).trim()
if (piece.isNotEmpty()) chunks += Chunk(chunks.size, piece, tok(piece))
if (end >= clean.length) break
start = max(end - overlapChars, start + 1)
}
return chunks
}
/**
* Group whole sentences up to the token target; boundaries never split a
* sentence. A single sentence larger than the target becomes its own chunk.
*/
fun chunkBySentences(text: String, opts: ChunkOptions): List<Chunk> {
require(opts.sizeTokens > 0) { "sizeTokens must be > 0" }
val sentences = splitSentences(text)
if (sentences.isEmpty()) return emptyList()
val chunks = mutableListOf<Chunk>()
var current = mutableListOf<String>()
var currentTokens = 0
fun flush() {
if (current.isEmpty()) return
val piece = current.joinToString(" ")
chunks += Chunk(chunks.size, piece, tok(piece))
current = mutableListOf()
currentTokens = 0
}
for (sentence in sentences) {
val t = tok(sentence)
if (currentTokens > 0 && currentTokens + t > opts.sizeTokens) flush()
current.add(sentence)
currentTokens += t
}
flush()
return chunks
}
/**
* Split on markdown headings; oversized sections fall back to sentence
* grouping, and every chunk carries the section heading.
*/
fun chunkMarkdown(text: String, opts: ChunkOptions): List<Chunk> {
require(opts.sizeTokens > 0) { "sizeTokens must be > 0" }
data class Section(val heading: String?, val body: List<String>)
val sections = mutableListOf<Section>()
var pendingHeading: String? = null
var pendingBody = mutableListOf<String>()
for (line in text.split('\n')) {
val m = HEADING.matchEntire(line)
if (m != null) {
if (pendingBody.isNotEmpty()) sections += Section(pendingHeading, pendingBody)
pendingHeading = m.groupValues[2].trim()
pendingBody = mutableListOf()
} else {
pendingBody.add(line)
}
}
if (pendingBody.isNotEmpty()) sections += Section(pendingHeading, pendingBody)
val chunks = mutableListOf<Chunk>()
for ((heading, body) in sections) {
val clean = body.joinToString("\n").trim()
if (clean.isEmpty()) continue
val whole = if (heading != null) "# $heading\n$clean" else clean
if (tok(whole) <= opts.sizeTokens) {
chunks += Chunk(chunks.size, whole, tok(whole), heading)
continue
}
for (c in chunkBySentences(clean, opts)) {
chunks += Chunk(chunks.size, c.text, c.tokens, heading)
}
}
return chunks
}
private fun statsFor(strategy: ChunkStrategy, chunks: List<Chunk>): StrategyResult {
val count = chunks.size
val sizes = chunks.map { it.tokens }
val minTokens = if (count > 0) sizes.min() else 0
val maxTokens = if (count > 0) sizes.max() else 0
val avgTokens = if (count > 0) (sizes.sum().toDouble() / count).roundToInt() else 0
val boundaries = chunks.dropLast(1).map { endsSentence(it.text) }
val share = if (boundaries.isNotEmpty()) {
boundaries.count { it }.toDouble() / boundaries.size
} else 1.0 // a single chunk has no internal boundaries to botch
return StrategyResult(strategy, chunks,
StrategyStats(count, minTokens, maxTokens, avgTokens, share))
}
/** Run all three strategies over one document and report comparable stats. */
fun compareStrategies(text: String, opts: ChunkOptions): Map<ChunkStrategy, StrategyResult> =
linkedMapOf(
ChunkStrategy.FIXED to statsFor(ChunkStrategy.FIXED, chunkFixed(text, opts)),
ChunkStrategy.SENTENCE to statsFor(ChunkStrategy.SENTENCE, chunkBySentences(text, opts)),
ChunkStrategy.MARKDOWN to statsFor(ChunkStrategy.MARKDOWN, chunkMarkdown(text, opts)),
)
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →