Skip to content

RAG Chunk Comparator — Kotlin source

Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.

This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.

// RAG Chunk Comparator — chunk one document three ways and report the stats
// that matter for retrieval.
//
// Language: Kotlin (JVM 17+, zero dependencies)
// Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
// implementation). Field names stay camelCase to match the TS surface.
// Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator

import kotlin.math.max
import kotlin.math.min
import kotlin.math.roundToInt

enum class ChunkStrategy { FIXED, SENTENCE, MARKDOWN }

/** Target chunk size in tokens + fixed-strategy overlap. */
data class ChunkOptions(val sizeTokens: Int, val overlapTokens: Int = 0)

data class Chunk(
    val index: Int,
    val text: String,
    val tokens: Int,
    /** Nearest markdown heading for markdown chunks; null otherwise. */
    val heading: String? = null,
)

data class StrategyStats(
    val count: Int,
    val minTokens: Int,
    val maxTokens: Int,
    val avgTokens: Int,
    /** Share of chunk boundaries that fall on a sentence end (0..1). */
    val sentenceBoundaryShare: Double,
)

data class StrategyResult(
    val strategy: ChunkStrategy,
    val chunks: List<Chunk>,
    val stats: StrategyStats,
)

private val SENTENCE_SPLIT = Regex("(?<=[.!?]) +")
private val ENDS_SENTENCE = Regex("[.!?][\"'\\)\\]]?$")
private val HEADING = Regex("^(#{1,6})\\s+(.*)$")

/**
 * The `type: 'prose'` path of the tokenEstimator, inlined: every non-empty
 * line costs max(1, round(length / 4)) tokens; empty text is 0.
 */
private fun tok(s: String): Int {
    if (s.isEmpty()) return 0
    var tokens = 0
    for (line in s.split('\n')) {
        if (line.isNotEmpty()) tokens += max(1, (line.length / 4.0).roundToInt())
    }
    return tokens
}

/** Split on sentence enders followed by whitespace or end of text. */
fun splitSentences(text: String): List<String> =
    SENTENCE_SPLIT.split(text.replace(Regex("\\s+"), " ").trim()).filter { it.isNotEmpty() }

private fun endsSentence(s: String): Boolean = ENDS_SENTENCE.containsMatchIn(s.trim())

/**
 * Greedy character accumulation to a token target (overlapping allowed).
 * @throws IllegalArgumentException on impossible options (the TS RangeError contract).
 */
fun chunkFixed(text: String, opts: ChunkOptions): List<Chunk> {
    require(opts.sizeTokens > 0) { "sizeTokens must be > 0" }
    require(opts.overlapTokens in 0 until opts.sizeTokens) {
        "overlapTokens must be in [0, sizeTokens)"
    }
    val clean = text.trim()
    if (clean.isEmpty()) return emptyList()
    // ~4 chars per prose token: step by tokens, verify with the estimator.
    val charStep = max(1, (opts.sizeTokens * 4.0).roundToInt())
    val overlapChars = (opts.overlapTokens * 4.0).roundToInt()
    val chunks = mutableListOf<Chunk>()
    var start = 0
    while (start < clean.length) {
        var end = min(start + charStep, clean.length)
        // Prefer cutting at whitespace near the target.
        if (end < clean.length) {
            val cut = clean.lastIndexOf(' ', end)
            if (cut > start) end = cut
        }
        val piece = clean.substring(start, end).trim()
        if (piece.isNotEmpty()) chunks += Chunk(chunks.size, piece, tok(piece))
        if (end >= clean.length) break
        start = max(end - overlapChars, start + 1)
    }
    return chunks
}

/**
 * Group whole sentences up to the token target; boundaries never split a
 * sentence. A single sentence larger than the target becomes its own chunk.
 */
fun chunkBySentences(text: String, opts: ChunkOptions): List<Chunk> {
    require(opts.sizeTokens > 0) { "sizeTokens must be > 0" }
    val sentences = splitSentences(text)
    if (sentences.isEmpty()) return emptyList()
    val chunks = mutableListOf<Chunk>()
    var current = mutableListOf<String>()
    var currentTokens = 0
    fun flush() {
        if (current.isEmpty()) return
        val piece = current.joinToString(" ")
        chunks += Chunk(chunks.size, piece, tok(piece))
        current = mutableListOf()
        currentTokens = 0
    }
    for (sentence in sentences) {
        val t = tok(sentence)
        if (currentTokens > 0 && currentTokens + t > opts.sizeTokens) flush()
        current.add(sentence)
        currentTokens += t
    }
    flush()
    return chunks
}

/**
 * Split on markdown headings; oversized sections fall back to sentence
 * grouping, and every chunk carries the section heading.
 */
fun chunkMarkdown(text: String, opts: ChunkOptions): List<Chunk> {
    require(opts.sizeTokens > 0) { "sizeTokens must be > 0" }
    data class Section(val heading: String?, val body: List<String>)
    val sections = mutableListOf<Section>()
    var pendingHeading: String? = null
    var pendingBody = mutableListOf<String>()
    for (line in text.split('\n')) {
        val m = HEADING.matchEntire(line)
        if (m != null) {
            if (pendingBody.isNotEmpty()) sections += Section(pendingHeading, pendingBody)
            pendingHeading = m.groupValues[2].trim()
            pendingBody = mutableListOf()
        } else {
            pendingBody.add(line)
        }
    }
    if (pendingBody.isNotEmpty()) sections += Section(pendingHeading, pendingBody)

    val chunks = mutableListOf<Chunk>()
    for ((heading, body) in sections) {
        val clean = body.joinToString("\n").trim()
        if (clean.isEmpty()) continue
        val whole = if (heading != null) "# $heading\n$clean" else clean
        if (tok(whole) <= opts.sizeTokens) {
            chunks += Chunk(chunks.size, whole, tok(whole), heading)
            continue
        }
        for (c in chunkBySentences(clean, opts)) {
            chunks += Chunk(chunks.size, c.text, c.tokens, heading)
        }
    }
    return chunks
}

private fun statsFor(strategy: ChunkStrategy, chunks: List<Chunk>): StrategyResult {
    val count = chunks.size
    val sizes = chunks.map { it.tokens }
    val minTokens = if (count > 0) sizes.min() else 0
    val maxTokens = if (count > 0) sizes.max() else 0
    val avgTokens = if (count > 0) (sizes.sum().toDouble() / count).roundToInt() else 0
    val boundaries = chunks.dropLast(1).map { endsSentence(it.text) }
    val share = if (boundaries.isNotEmpty()) {
        boundaries.count { it }.toDouble() / boundaries.size
    } else 1.0 // a single chunk has no internal boundaries to botch
    return StrategyResult(strategy, chunks,
        StrategyStats(count, minTokens, maxTokens, avgTokens, share))
}

/** Run all three strategies over one document and report comparable stats. */
fun compareStrategies(text: String, opts: ChunkOptions): Map<ChunkStrategy, StrategyResult> =
    linkedMapOf(
        ChunkStrategy.FIXED to statsFor(ChunkStrategy.FIXED, chunkFixed(text, opts)),
        ChunkStrategy.SENTENCE to statsFor(ChunkStrategy.SENTENCE, chunkBySentences(text, opts)),
        ChunkStrategy.MARKDOWN to statsFor(ChunkStrategy.MARKDOWN, chunkMarkdown(text, opts)),
    )

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →