Text Diff Viewer — JavaScript source
Compare two pieces of text and see exactly what changed. Highlights added and removed lines, words, or characters, shows a per-side summary, and exports a unified diff you can paste into a PR or commit. Runs 100% in your browser.
This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// text-diff - JavaScript port (CosmoDev polyglot showcase).
//
// Computes a diff between two strings at line / word / char granularity using a
// classic LCS (longest-common-subsequence) dynamic-programming table. No
// dependencies, fully deterministic, safe to run in the browser or in Node.
//
// Ported from src/lib/text-diff.ts - display source, part of CosmoDev's
// polyglot tool pages (dev.cosmolabs.org). Behavior is functionally equivalent
// to the canonical TypeScript implementation.
//
// Token invariant: every tokenizer splits a string into tokens whose exact
// concatenation reconstructs the original, so concatenating every DiffPart.text
// in order reconstructs the *changed* string `b` (under default options).
/* @typedef {'equal'|'added'|'removed'} DiffType */
/* @typedef {{ type: DiffType, text: string }} DiffPart */
/* @typedef {'line'|'word'|'char'} Granularity */
/* @typedef {{ ignoreCase?: boolean, trim?: boolean, ignoreWhitespace?: boolean }} DiffOptions */
/* @typedef {{ added: number, removed: number, unchanged: number }} DiffSummary */
/* @typedef {{ old: string, new: string }} UnifiedHeaders */
/**
* Normalize a token for COMPARISON only - the original token is always what is
* emitted in the diff output. With empty opts this is the identity function, so
* the reconstruction invariant holds exactly; opt-in normalization may relax it
* for tokens that match only after normalization (an equal part then carries
* `a`'s original text).
*/
function normalizeKey(token, opts) {
let s = token;
if (opts.ignoreWhitespace) s = s.replace(/\s+/g, ' ').trim();
if (opts.trim) s = s.trim();
if (opts.ignoreCase) s = s.toLowerCase();
return s;
}
/**
* Split `text` into tokens.
* - char: code-point split (unicode-safe via Array.from); concatenation === text.
* - word: alternating maximal whitespace runs and non-whitespace runs;
* concatenation === text.
* - line: content-only lines (text.split('\n')); a trailing '' element marks
* that the text ends with a newline. Reconstruction joins tokens with '\n'.
*/
export function tokenize(text, granularity) {
if (text === '') return [];
switch (granularity) {
case 'char':
return Array.from(text);
case 'word':
// The capture group keeps the separators, so whitespace runs survive as
// their own tokens and concatenation reconstructs the input exactly.
return text.split(/(\s+)/).filter((t) => t.length > 0);
case 'line':
return text.split('\n');
}
}
/**
* Compute a diff between `a` (original) and `b` (changed) at the requested
* granularity. Default granularity is 'line', default options is {} (no
* normalization). Returns merged runs of { type, text }. Opt-in normalization
* compares on a normalized key but emits the ORIGINAL token, so an equal part
* may carry `a`'s text. Uses an LCS dynamic-programming table.
*/
export function diff(a, b, granularity = 'line', opts = {}) {
const A = tokenize(a, granularity);
const B = tokenize(b, granularity);
// Compare on a normalized key so opt-in ignoreCase/trim/ignoreWhitespace can
// match tokens that differ superficially; the ORIGINAL token is always emitted.
const aKey = A.map((t) => normalizeKey(t, opts));
const bKey = B.map((t) => normalizeKey(t, opts));
const n = aKey.length;
const m = bKey.length;
// dp[i][j] = length of the longest common subsequence of aKey[i..] and bKey[j..].
// Built backwards so each cell only depends on cells already computed (i+1, j+1).
const dp = Array.from({ length: n + 1 }, () => new Array(m + 1).fill(0));
for (let i = n - 1; i >= 0; i--) {
for (let j = m - 1; j >= 0; j--) {
if (aKey[i] === bKey[j]) dp[i][j] = dp[i + 1][j + 1] + 1;
else dp[i][j] = Math.max(dp[i + 1][j], dp[i][j + 1]);
}
}
// Greedy walk through the table. When keys match we emit an equal part;
// otherwise we drop the side whose remaining LCS is larger. The `>=` tie
// favors 'removed' to match the canonical implementation exactly.
const raw = [];
let i = 0;
let j = 0;
while (i < n && j < m) {
if (aKey[i] === bKey[j]) {
raw.push({ type: 'equal', text: A[i] });
i++;
j++;
} else if (dp[i + 1][j] >= dp[i][j + 1]) {
raw.push({ type: 'removed', text: A[i] });
i++;
} else {
raw.push({ type: 'added', text: B[j] });
j++;
}
}
while (i < n) {
raw.push({ type: 'removed', text: A[i] });
i++;
}
while (j < m) {
raw.push({ type: 'added', text: B[j] });
j++;
}
// Merge consecutive runs of the same type into a single part. Line tokens are
// content-only, so they rejoin with '\n'; char/word tokens already contain
// their separators and concatenate directly with ''.
const sep = granularity === 'line' ? '\n' : '';
const merged = [];
for (const r of raw) {
const last = merged[merged.length - 1];
if (last && last.type === r.type) last.text += sep + r.text;
else merged.push({ type: r.type, text: r.text });
}
return merged;
}
/**
* Count characters per diff category (granularity-agnostic). Under opt-in
* normalization, an equal part carries `a`'s text, so `unchanged` reflects
* `a`'s length, not `b`'s (and `added`/`removed` can read 0 for a≠b).
*/
export function summary(parts) {
const out = { added: 0, removed: 0, unchanged: 0 };
for (const p of parts) {
const len = p.text.length;
if (p.type === 'added') out.added += len;
else if (p.type === 'removed') out.removed += len;
else out.unchanged += len;
}
return out;
}
/** Unified-diff context: unchanged lines kept around each change. */
const CONTEXT = 3;
/** Prefix character for each line type in unified output. */
function prefixFor(t) {
return t === 'added' ? '+' : t === 'removed' ? '-' : ' ';
}
/**
* Expand diff parts into one entry per output line. A trailing newline produces
* no phantom empty line - it is the terminator of the preceding line.
*/
function expandLines(parts) {
const entries = [];
for (const p of parts) {
const segs = p.text.split('\n');
if (p.text.endsWith('\n')) segs.pop();
for (const s of segs) entries.push({ type: p.type, text: s });
}
return entries;
}
/**
* Render a unified-diff string. Emits `--- `/`+++ ` header lines when `headers`
* is provided. For `line` granularity, groups changes into hunks with 3 lines
* of context and a `@@ -oldStart,oldLen +newStart,newLen @@` header per hunk.
* For `word`/`char`, emits one prefixed line per output line with no hunk headers.
*/
export function toUnifiedDiff(parts, headers, granularity = 'line') {
const out = [];
if (headers) {
out.push(`--- ${headers.old}`);
out.push(`+++ ${headers.new}`);
}
const entries = expandLines(parts);
if (granularity !== 'line') {
for (const e of entries) out.push(prefixFor(e.type) + e.text);
return out.join('\n');
}
// Line granularity: group changes into hunks bounded by CONTEXT lines.
const changedIdx = [];
for (let k = 0; k < entries.length; k++) {
if (entries[k].type !== 'equal') changedIdx.push(k);
}
if (changedIdx.length === 0) return out.join('\n');
// Merge changes within 2*CONTEXT of each other into one hunk range [start..end].
const ranges = [];
let cur = {
start: Math.max(0, changedIdx[0] - CONTEXT),
end: Math.min(entries.length - 1, changedIdx[0] + CONTEXT),
};
for (let k = 1; k < changedIdx.length; k++) {
const s = Math.max(0, changedIdx[k] - CONTEXT);
if (s <= cur.end + 1) {
cur.end = Math.min(entries.length - 1, changedIdx[k] + CONTEXT);
} else {
ranges.push(cur);
cur = { start: s, end: Math.min(entries.length - 1, changedIdx[k] + CONTEXT) };
}
}
ranges.push(cur);
for (const r of ranges) {
// Count old/new line numbers: every entry that is not 'added' exists in the
// old file; every entry that is not 'removed' exists in the new file.
let oldBefore = 0;
let newBefore = 0;
for (let k = 0; k < r.start; k++) {
if (entries[k].type !== 'added') oldBefore++;
if (entries[k].type !== 'removed') newBefore++;
}
let oldLen = 0;
let newLen = 0;
for (let k = r.start; k <= r.end; k++) {
if (entries[k].type !== 'added') oldLen++;
if (entries[k].type !== 'removed') newLen++;
}
// Empty-side convention: a length-0 side reports the line number of the
// preceding line (or 0 when at the very start of an empty file).
const oldStart = oldLen === 0 ? oldBefore : oldBefore + 1;
const newStart = newLen === 0 ? newBefore : newBefore + 1;
out.push(`@@ -${oldStart},${oldLen} +${newStart},${newLen} @@`);
for (let k = r.start; k <= r.end; k++) {
out.push(prefixFor(entries[k].type) + entries[k].text);
}
}
return out.join('\n');
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →