RAG Chunk Comparator — PHP source
Chunk one document three ways — fixed-size, sentence-aware, markdown-heading-aware — and compare counts, size spread, and how often boundaries cut sentences mid-thought. 100% client-side.
This is the PHP implementation — the same logic the interactive tool runs, in a shareable, citable form.
<?php
/**
* RAG Chunk Comparator — chunk one document three ways (fixed-size,
* sentence-aware, markdown-heading-aware) and compare retrieval stats.
*
* Language: PHP (8.1+, standard library only)
* Source: CosmoDev polyglot showcase port of the RAG Chunk Comparator
* tool (slug: rag-chunk-comparator).
* Port of src/lib/ragChunkComparator.ts (the canonical TypeScript
* implementation).
* Tool page: https://dev.cosmolabs.org/tools/rag-chunk-comparator
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Token sizes inline the tokenEstimator prose heuristic (~4 chars per
* token, per non-empty line, minimum one token per line) so this file is
* self-contained; lengths are byte counts, exact for ASCII text. Array
* keys keep the TypeScript field names (sizeTokens, sentenceBoundaryShare,
* …) so results serialize to the same shape. The RangeError becomes
* RangeException.
*/
declare(strict_types=1);
/** Prose token estimate: chars/4 per non-empty line, min 1 per line. */
function prose_tokens(string $s): int
{
$total = 0;
foreach (preg_split('/\r?\n/', $s) ?: [] as $line) {
if (trim($line) === '') {
continue;
}
$total += max(1, (int) round(strlen($line) / 4));
}
return $total;
}
/** Split on sentence enders followed by whitespace or end of text. */
function split_sentences(string $text): array
{
$normalized = trim((string) preg_replace('/\s+/', ' ', $text));
if ($normalized === '') {
return [];
}
$parts = preg_split('/(?<=[.!?]) +/', $normalized, -1, PREG_SPLIT_NO_EMPTY);
return $parts ?: [];
}
function ends_sentence(string $s): bool
{
return (bool) preg_match('/[.!?]["\')\]]?$/', trim($s));
}
/** Greedy character accumulation to a token target (overlapping allowed). */
function chunk_fixed(string $text, array $opts): array
{
$sizeTokens = $opts['sizeTokens'];
$overlapTokens = $opts['overlapTokens'] ?? 0;
if ($sizeTokens <= 0) {
throw new RangeException('sizeTokens must be > 0');
}
if ($overlapTokens < 0 || $overlapTokens >= $sizeTokens) {
throw new RangeException('overlapTokens must be in [0, sizeTokens)');
}
$clean = trim($text);
if ($clean === '') {
return [];
}
// ~4 chars per prose token: step by tokens, verify with the estimator.
$charStep = max(1, (int) round($sizeTokens * 4));
$overlapChars = (int) round($overlapTokens * 4);
$len = strlen($clean);
$chunks = [];
$start = 0;
while ($start < $len) {
$end = min($start + $charStep, $len);
// Prefer cutting at whitespace near the target — but never trim the
// document's final piece back to a word when it already fits.
if ($end < $len) {
$cut = strrpos(substr($clean, 0, $end + 1), ' ');
if ($cut !== false && $cut > $start) {
$end = $cut;
}
}
$piece = trim(substr($clean, $start, $end - $start));
if ($piece !== '') {
$chunks[] = ['index' => count($chunks), 'text' => $piece, 'tokens' => prose_tokens($piece)];
}
if ($end >= $len) {
break;
}
$start = max($end - $overlapChars, $start + 1);
}
return $chunks;
}
/** Group whole sentences up to the token target; boundaries never split a sentence. */
function chunk_by_sentences(string $text, array $opts): array
{
$sizeTokens = $opts['sizeTokens'];
if ($sizeTokens <= 0) {
throw new RangeException('sizeTokens must be > 0');
}
$sentences = split_sentences($text);
if (count($sentences) === 0) {
return [];
}
$chunks = [];
$current = [];
$currentTokens = 0;
foreach ($sentences as $sentence) {
$t = prose_tokens($sentence);
if ($currentTokens > 0 && $currentTokens + $t > $sizeTokens) {
$piece = implode(' ', $current);
$chunks[] = ['index' => count($chunks), 'text' => $piece, 'tokens' => prose_tokens($piece)];
$current = [];
$currentTokens = 0;
}
$current[] = $sentence;
$currentTokens += $t;
// A single sentence larger than the target becomes its own chunk.
}
if (count($current) > 0) {
$piece = implode(' ', $current);
$chunks[] = ['index' => count($chunks), 'text' => $piece, 'tokens' => prose_tokens($piece)];
}
return $chunks;
}
/** Split on markdown headings; oversized sections fall back to sentence grouping. */
function chunk_markdown(string $text, array $opts): array
{
$sizeTokens = $opts['sizeTokens'];
if ($sizeTokens <= 0) {
throw new RangeException('sizeTokens must be > 0');
}
$sections = [];
$current = ['heading' => null, 'body' => []];
foreach (explode("\n", $text) as $line) {
if (preg_match('/^(#{1,6})\s+(.*)$/', $line, $m)) {
if (count($current['body']) > 0) {
$sections[] = $current;
}
$current = ['heading' => trim($m[2]), 'body' => []];
} else {
$current['body'][] = $line;
}
}
if (count($current['body']) > 0) {
$sections[] = $current;
}
$chunks = [];
foreach ($sections as $section) {
$body = trim(implode("\n", $section['body']));
if ($body === '') {
continue;
}
$whole = $section['heading'] !== null ? "# {$section['heading']}\n{$body}" : $body;
if (prose_tokens($whole) <= $sizeTokens) {
$chunks[] = [
'index' => count($chunks),
'text' => $whole,
'tokens' => prose_tokens($whole),
'heading' => $section['heading'],
];
continue;
}
// Oversized section: sentence-group the body, stamp every chunk with
// the heading.
foreach (chunk_by_sentences($body, $opts) as $c) {
$chunks[] = [
'index' => count($chunks),
'text' => $c['text'],
'tokens' => $c['tokens'],
'heading' => $section['heading'],
];
}
}
return $chunks;
}
function stats_for(string $strategy, array $chunks): array
{
$sizes = array_column($chunks, 'tokens');
$count = count($chunks);
$boundaries = array_map(
static fn(array $c): bool => ends_sentence($c['text']),
array_slice($chunks, 0, -1)
);
return [
'strategy' => $strategy,
'chunks' => $chunks,
'stats' => [
'count' => $count,
'minTokens' => $count ? min($sizes) : 0,
'maxTokens' => $count ? max($sizes) : 0,
'avgTokens' => $count ? (int) round(array_sum($sizes) / $count) : 0,
// A single chunk has no internal boundaries to botch.
'sentenceBoundaryShare' => count($boundaries)
? count(array_filter($boundaries)) / count($boundaries)
: 1,
],
];
}
/** Run all three strategies over one document and report comparable stats. */
function compare_strategies(string $text, array $opts): array
{
return [
'fixed' => stats_for('fixed', chunk_fixed($text, $opts)),
'sentence' => stats_for('sentence', chunk_by_sentences($text, $opts)),
'markdown' => stats_for('markdown', chunk_markdown($text, $opts)),
];
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →