Text Diff Viewer — PHP source
Compare two pieces of text and see exactly what changed. Highlights added and removed lines, words, or characters, shows a per-side summary, and exports a unified diff you can paste into a PR or commit. Runs 100% in your browser.
This is the PHP implementation — the same logic the interactive tool runs, in a shareable, citable form.
<?php
/**
* text-diff — PHP port (CosmoDev polyglot showcase).
*
* Computes a diff between two strings at line / word / char granularity using a
* classic LCS (longest-common-subsequence) dynamic-programming table. No
* third-party dependencies, fully deterministic.
*
* Ported from src/lib/text-diff.ts — display source, part of CosmoDev's
* polyglot tool pages (dev.cosmolabs.org). Behavior is functionally equivalent
* to the canonical TypeScript implementation.
*
* Token invariant: every tokenizer splits a string into tokens whose exact
* concatenation reconstructs the original, so concatenating every part's `text`
* in order reconstructs the changed string `$b` (under default options).
*/
declare(strict_types=1);
namespace CosmoDev\TextDiff;
/**
* Granularity constants — the token size the diff operates on.
*/
const GRAN_LINE = 'line';
const GRAN_WORD = 'word';
const GRAN_CHAR = 'char';
/**
* Diff type constants — the category of a diff part.
*/
const TYPE_EQUAL = 'equal';
const TYPE_ADDED = 'added';
const TYPE_REMOVED = 'removed';
/** Lines of context kept around each change in unified output. */
const CONTEXT_LINES = 3;
/**
* Normalize a token for COMPARISON only; the original is always emitted.
*
* Order matters: ignoreWhitespace first (collapse + trim), then an explicit
* trim, then case folding. `mb_strtolower` is the Unicode-aware lowercase map.
*
* @param string $token
* @param array{ignoreCase?: bool, trim?: bool, ignoreWhitespace?: bool} $opts
*/
function normalize_key(string $token, array $opts): string
{
$s = $token;
if (!empty($opts['ignoreWhitespace'])) {
// Collapse every whitespace run to a single space, then trim.
$s = preg_replace('/\s+/', ' ', $s) ?? $s;
$s = trim($s);
}
if (!empty($opts['trim'])) {
$s = trim($s);
}
if (!empty($opts['ignoreCase'])) {
$s = mb_strtolower($s, 'UTF-8');
}
return $s;
}
/**
* Split text into reconstructable tokens.
*
* - char: code-point split (UTF-8 aware); concatenation === text.
* - word: alternating maximal whitespace runs and non-whitespace runs;
* concatenation === text.
* - line: content-only lines (explode on "\n"); a trailing empty string marks
* that the text ends with a newline. Reconstruction joins with "\n".
*
* @return list<string>
*/
function tokenize(string $text, string $granularity): array
{
if ($text === '') {
return [];
}
switch ($granularity) {
case GRAN_CHAR:
// preg_split with the /u flag splits on UTF-8 code-point boundaries;
// PREG_SPLIT_NO_EMPTY drops the empty ends the zero-width pattern yields.
/** @var list<string> $parts */
$parts = preg_split('//u', $text, -1, PREG_SPLIT_NO_EMPTY);
return $parts;
case GRAN_WORD:
// The capture group keeps whitespace runs as their own tokens so
// concatenation reconstructs the input exactly.
/** @var list<string> $parts */
$parts = preg_split('/(\s+)/', $text, -1, PREG_SPLIT_DELIM_CAPTURE);
return array_values(array_filter($parts, fn ($t) => $t !== ''));
case GRAN_LINE:
default:
return explode("\n", $text);
}
}
/**
* Compute a diff between $a (original) and $b (changed) at the requested
* granularity. Returns merged runs of ['type' => ..., 'text' => ...]. Default
* granularity is 'line', default options is [] (no normalization). Opt-in
* normalization compares on a normalized key but emits the ORIGINAL token.
*
* @param string $a
* @param string $b
* @param string $granularity
* @param array{ignoreCase?: bool, trim?: bool, ignoreWhitespace?: bool} $opts
* @return list<array{type: string, text: string}>
*/
function diff(string $a, string $b, string $granularity = GRAN_LINE, array $opts = []): array
{
$A = tokenize($a, $granularity);
$B = tokenize($b, $granularity);
// Compare on a normalized key; emit the original token.
$aKey = array_map(fn ($t) => normalize_key($t, $opts), $A);
$bKey = array_map(fn ($t) => normalize_key($t, $opts), $B);
$n = count($aKey);
$m = count($bKey);
// dp[i][j] = length of the LCS of aKey[i..] and bKey[j..], built backwards so
// each cell only depends on already-computed cells (i+1, j+1).
$dp = array_fill(0, $n + 1, array_fill(0, $m + 1, 0));
for ($i = $n - 1; $i >= 0; $i--) {
for ($j = $m - 1; $j >= 0; $j--) {
if ($aKey[$i] === $bKey[$j]) {
$dp[$i][$j] = $dp[$i + 1][$j + 1] + 1;
} elseif ($dp[$i + 1][$j] >= $dp[$i][$j + 1]) {
$dp[$i][$j] = $dp[$i + 1][$j];
} else {
$dp[$i][$j] = $dp[$i][$j + 1];
}
}
}
// Greedy walk: equal on a key match; otherwise drop the side whose remaining
// LCS is larger. The >= tie favors 'removed', matching the canonical walk.
$raw = [];
$i = 0;
$j = 0;
while ($i < $n && $j < $m) {
if ($aKey[$i] === $bKey[$j]) {
$raw[] = ['type' => TYPE_EQUAL, 'text' => $A[$i]];
$i++;
$j++;
} elseif ($dp[$i + 1][$j] >= $dp[$i][$j + 1]) {
$raw[] = ['type' => TYPE_REMOVED, 'text' => $A[$i]];
$i++;
} else {
$raw[] = ['type' => TYPE_ADDED, 'text' => $B[$j]];
$j++;
}
}
while ($i < $n) {
$raw[] = ['type' => TYPE_REMOVED, 'text' => $A[$i]];
$i++;
}
while ($j < $m) {
$raw[] = ['type' => TYPE_ADDED, 'text' => $B[$j]];
$j++;
}
// Merge consecutive runs of the same type. Lines rejoin with "\n"; char/word
// tokens already carry their separators and concatenate with "".
$sep = $granularity === GRAN_LINE ? "\n" : '';
$merged = [];
foreach ($raw as $r) {
$last = $merged[count($merged) - 1] ?? null;
if ($last !== null && $last['type'] === $r['type']) {
$merged[count($merged) - 1]['text'] .= $sep . $r['text'];
} else {
$merged[] = ['type' => $r['type'], 'text' => $r['text']];
}
}
return $merged;
}
/**
* Count characters per diff category (granularity-agnostic). Counts UTF-8
* characters via mb_strlen. Under opt-in normalization an equal part carries
* $a's text, so 'unchanged' reflects $a's length, not $b's.
*
* @param list<array{type: string, text: string}> $parts
* @return array{added: int, removed: int, unchanged: int}
*/
function summary(array $parts): array
{
$out = ['added' => 0, 'removed' => 0, 'unchanged' => 0];
foreach ($parts as $p) {
$len = mb_strlen($p['text'], 'UTF-8');
if ($p['type'] === TYPE_ADDED) {
$out['added'] += $len;
} elseif ($p['type'] === TYPE_REMOVED) {
$out['removed'] += $len;
} else {
$out['unchanged'] += $len;
}
}
return $out;
}
/**
* Prefix character for each line type in unified output.
*/
function prefix_for(string $type): string
{
if ($type === TYPE_ADDED) {
return '+';
}
if ($type === TYPE_REMOVED) {
return '-';
}
return ' ';
}
/**
* Expand diff parts into one entry per output line. A trailing newline produces
* no phantom empty line — it terminates the preceding line.
*
* @param list<array{type: string, text: string}> $parts
* @return list<array{type: string, text: string}>
*/
function expand_lines(array $parts): array
{
$entries = [];
foreach ($parts as $p) {
$segs = explode("\n", $p['text']);
if (str_ends_with($p['text'], "\n")) {
array_pop($segs); // drop the phantom empty segment
}
foreach ($segs as $s) {
$entries[] = ['type' => $p['type'], 'text' => $s];
}
}
return $entries;
}
/**
* Render a unified-diff string. Emits `--- `/`+++ ` header lines when $headers
* is provided. For 'line' granularity, groups changes into hunks with 3 lines
* of context and a `@@ -oldStart,oldLen +newStart,newLen @@` header per hunk.
* For 'word'/'char', emits one prefixed line per output line with no hunks.
*
* @param list<array{type: string, text: string}> $parts
* @param array{old: string, new: string}|null $headers
* @param string $granularity
*/
function to_unified_diff(array $parts, ?array $headers = null, string $granularity = GRAN_LINE): string
{
$out = [];
if ($headers !== null) {
$out[] = '--- ' . $headers['old'];
$out[] = '+++ ' . $headers['new'];
}
$entries = expand_lines($parts);
if ($granularity !== GRAN_LINE) {
foreach ($entries as $e) {
$out[] = prefix_for($e['type']) . $e['text'];
}
return implode("\n", $out);
}
// Line granularity: group changes into hunks bounded by CONTEXT_LINES.
$changedIdx = [];
foreach ($entries as $k => $e) {
if ($e['type'] !== TYPE_EQUAL) {
$changedIdx[] = $k;
}
}
if (count($changedIdx) === 0) {
return implode("\n", $out);
}
$last = count($entries) - 1;
// Merge changes within 2*CONTEXT_LINES of each other into one hunk [start..end].
$ranges = [];
$cur = [
'start' => max(0, $changedIdx[0] - CONTEXT_LINES),
'end' => min($last, $changedIdx[0] + CONTEXT_LINES),
];
for ($k = 1; $k < count($changedIdx); $k++) {
$idx = $changedIdx[$k];
$s = max(0, $idx - CONTEXT_LINES);
if ($s <= $cur['end'] + 1) {
$cur['end'] = min($last, $idx + CONTEXT_LINES);
} else {
$ranges[] = $cur;
$cur = ['start' => $s, 'end' => min($last, $idx + CONTEXT_LINES)];
}
}
$ranges[] = $cur;
foreach ($ranges as $r) {
// Old side = entries that are not 'added'; new side = not 'removed'.
$oldBefore = 0;
$newBefore = 0;
for ($k = 0; $k < $r['start']; $k++) {
if ($entries[$k]['type'] !== TYPE_ADDED) {
$oldBefore++;
}
if ($entries[$k]['type'] !== TYPE_REMOVED) {
$newBefore++;
}
}
$oldLen = 0;
$newLen = 0;
for ($k = $r['start']; $k <= $r['end']; $k++) {
if ($entries[$k]['type'] !== TYPE_ADDED) {
$oldLen++;
}
if ($entries[$k]['type'] !== TYPE_REMOVED) {
$newLen++;
}
}
// Empty-side convention: a length-0 side reports the preceding line number
// (or 0 at the very start of an empty file).
$oldStart = $oldLen === 0 ? $oldBefore : $oldBefore + 1;
$newStart = $newLen === 0 ? $newBefore : $newBefore + 1;
$out[] = "@@ -{$oldStart},{$oldLen} +{$newStart},{$newLen} @@";
for ($k = $r['start']; $k <= $r['end']; $k++) {
$out[] = prefix_for($entries[$k]['type']) . $entries[$k]['text'];
}
}
return implode("\n", $out);
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →