Text Diff Viewer — TypeScript source
Compare two pieces of text and see exactly what changed. Highlights added and removed lines, words, or characters, shows a per-side summary, and exports a unified diff you can paste into a PR or commit. Runs 100% in your browser.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Text Diff Viewer - pure logic.
//
// Computes a diff between two strings at line / word / char granularity using a
// classic LCS (longest-common-subsequence) dynamic-programming table. No
// dependencies, fully deterministic, safe to run in the browser.
//
// Token invariant: every tokenizer splits a string into tokens whose exact
// concatenation reconstructs the original. That means concatenating every
// `DiffPart.text` in order reconstructs the *changed* string `b`.
export type DiffType = 'equal' | 'added' | 'removed';
export interface DiffPart {
type: DiffType;
text: string;
}
export type Granularity = 'line' | 'word' | 'char';
export interface DiffOptions {
/**
* Normalization applies to the comparison key only - the original token is
* always emitted. Designed for `line` granularity; at `word`/`char` the
* whitespace options collapse every whitespace token/char to an empty key,
* which is deterministic but can be surprising.
*/
/** Case-insensitive comparison (the original text is still displayed). */
ignoreCase?: boolean;
/** Ignore leading/trailing whitespace per token when comparing. */
trim?: boolean;
/** Collapse internal whitespace runs to a single space (and trim) when comparing. */
ignoreWhitespace?: boolean;
}
/**
* Normalize a token for COMPARISON only - the original token is always what's
* emitted in the diff output. With default `{}` opts this is the identity fn,
* so the reconstruction invariant (concatenating equal+added `.text` === `b`)
* holds exactly; opt-in normalization may relax it for tokens that match only
* after normalization (an equal part then carries `a`'s original text).
*/
function normalizeKey(token: string, opts: DiffOptions): string {
let s = token;
if (opts.ignoreWhitespace) s = s.replace(/\s+/g, ' ').trim();
if (opts.trim) s = s.trim();
if (opts.ignoreCase) s = s.toLowerCase();
return s;
}
/**
* Split `text` into tokens.
* - `char`: code-point split (unicode-safe); concatenation === `text`.
* - `word`: alternating maximal whitespace runs and non-whitespace runs;
* concatenation === `text`.
* - `line`: content-only lines (`text.split('\n')`); a trailing `''` element
* marks that the text ends with a newline. Lines are compared by content, so
* `"b"` (last line, no newline) still matches `"b"` mid-file - no spurious
* remove/add when appending. Reconstruction joins line tokens with `'\n'`.
*/
export function tokenize(text: string, granularity: Granularity): string[] {
if (text === '') return [];
switch (granularity) {
case 'char':
return Array.from(text);
case 'word':
return text.split(/(\s+)/).filter((t) => t.length > 0);
case 'line':
return text.split('\n');
}
}
/**
* Compute a diff between `a` (original) and `b` (changed) at the requested
* granularity. Returns merged runs of `{ type, text }`; concatenating every
* `.text` reconstructs `b` exactly with default opts. Opt-in normalization
* (`ignoreCase`/`trim`/`ignoreWhitespace`) compares on a normalized key but
* emits the ORIGINAL token, so a normalized-equal part carries `a`'s text and
* may relax this invariant. Uses an LCS dynamic-programming table.
*/
export function diff(
a: string,
b: string,
granularity: Granularity = 'line',
opts: DiffOptions = {},
): DiffPart[] {
const A = tokenize(a, granularity);
const B = tokenize(b, granularity);
// Compare on a normalized key so opt-in ignoreCase/trim/ignoreWhitespace can
// match tokens that differ superficially; the ORIGINAL token is always emitted.
const aKey = A.map((t) => normalizeKey(t, opts));
const bKey = B.map((t) => normalizeKey(t, opts));
const n = aKey.length;
const m = bKey.length;
// dp[i][j] = length of the longest common subsequence of aKey[i..] and bKey[j..].
const dp: number[][] = Array.from({ length: n + 1 }, () => new Array<number>(m + 1).fill(0));
for (let i = n - 1; i >= 0; i--) {
for (let j = m - 1; j >= 0; j--) {
if (aKey[i] === bKey[j]) dp[i][j] = dp[i + 1][j + 1] + 1;
else dp[i][j] = Math.max(dp[i + 1][j], dp[i][j + 1]);
}
}
const raw: { type: DiffType; text: string }[] = [];
let i = 0;
let j = 0;
while (i < n && j < m) {
if (aKey[i] === bKey[j]) {
raw.push({ type: 'equal', text: A[i] });
i++;
j++;
} else if (dp[i + 1][j] >= dp[i][j + 1]) {
raw.push({ type: 'removed', text: A[i] });
i++;
} else {
raw.push({ type: 'added', text: B[j] });
j++;
}
}
while (i < n) {
raw.push({ type: 'removed', text: A[i] });
i++;
}
while (j < m) {
raw.push({ type: 'added', text: B[j] });
j++;
}
// Merge consecutive runs of the same type into a single part. Line tokens are
// content-only, so they are rejoined with '\n'; char/word tokens concatenate
// directly (their separators are already inside the tokens).
const sep = granularity === 'line' ? '\n' : '';
const merged: DiffPart[] = [];
for (const r of raw) {
const last = merged[merged.length - 1];
if (last && last.type === r.type) last.text += sep + r.text;
else merged.push({ type: r.type, text: r.text });
}
return merged;
}
export interface DiffSummary {
added: number;
removed: number;
unchanged: number;
}
/**
* Count characters per diff category (granularity-agnostic). Under opt-in
* normalization, an equal part carries `a`'s text, so `unchanged` reflects
* `a`'s length, not `b`'s (and `added`/`removed` can read 0 for a≠b).
*/
export function summary(parts: DiffPart[]): DiffSummary {
const out: DiffSummary = { added: 0, removed: 0, unchanged: 0 };
for (const p of parts) {
const len = p.text.length;
if (p.type === 'added') out.added += len;
else if (p.type === 'removed') out.removed += len;
else out.unchanged += len;
}
return out;
}
export interface UnifiedHeaders {
old: string;
new: string;
}
/** Unified-diff context: unchanged lines kept around each change. */
const CONTEXT = 3;
interface LineEntry {
type: DiffType;
text: string;
}
/**
* Expand diff parts into one entry per output line. A trailing newline produces
* no phantom empty line (it is the terminator of the preceding line).
*/
function expandLines(parts: DiffPart[]): LineEntry[] {
const entries: LineEntry[] = [];
for (const p of parts) {
const segs = p.text.split('\n');
if (p.text.endsWith('\n')) segs.pop();
for (const s of segs) entries.push({ type: p.type, text: s });
}
return entries;
}
function prefixFor(t: DiffType): string {
return t === 'added' ? '+' : t === 'removed' ? '-' : ' ';
}
/**
* Render a unified-diff string. Emits `--- `/`+++ ` header lines when `headers`
* is provided. For `line` granularity, groups changes into hunks with 3 lines of
* context and a `@@ -oldStart,oldLen +newStart,newLen @@` header per hunk. For
* `word`/`char`, emits one prefixed line per output line with no hunk headers.
*/
export function toUnifiedDiff(
parts: DiffPart[],
headers?: UnifiedHeaders,
granularity: Granularity = 'line',
): string {
const out: string[] = [];
if (headers) {
out.push(`--- ${headers.old}`);
out.push(`+++ ${headers.new}`);
}
const entries = expandLines(parts);
if (granularity !== 'line') {
for (const e of entries) out.push(prefixFor(e.type) + e.text);
return out.join('\n');
}
// Line granularity: group changes into hunks bounded by CONTEXT lines.
const changedIdx: number[] = [];
for (let k = 0; k < entries.length; k++) {
if (entries[k].type !== 'equal') changedIdx.push(k);
}
if (changedIdx.length === 0) {
return out.join('\n');
}
// Merge changes within 2*CONTEXT of each other into one hunk range [start..end].
const ranges: { start: number; end: number }[] = [];
let cur = {
start: Math.max(0, changedIdx[0] - CONTEXT),
end: Math.min(entries.length - 1, changedIdx[0] + CONTEXT),
};
for (let k = 1; k < changedIdx.length; k++) {
const s = Math.max(0, changedIdx[k] - CONTEXT);
if (s <= cur.end + 1) {
cur.end = Math.min(entries.length - 1, changedIdx[k] + CONTEXT);
} else {
ranges.push(cur);
cur = { start: s, end: Math.min(entries.length - 1, changedIdx[k] + CONTEXT) };
}
}
ranges.push(cur);
for (const r of ranges) {
let oldBefore = 0;
let newBefore = 0;
for (let k = 0; k < r.start; k++) {
if (entries[k].type !== 'added') oldBefore++;
if (entries[k].type !== 'removed') newBefore++;
}
let oldLen = 0;
let newLen = 0;
for (let k = r.start; k <= r.end; k++) {
if (entries[k].type !== 'added') oldLen++;
if (entries[k].type !== 'removed') newLen++;
}
// Empty side convention: a length-0 side reports the line number of the
// preceding line (or 0 when at the very start of an empty file).
const oldStart = oldLen === 0 ? oldBefore : oldBefore + 1;
const newStart = newLen === 0 ? newBefore : newBefore + 1;
out.push(`@@ -${oldStart},${oldLen} +${newStart},${newLen} @@`);
for (let k = r.start; k <= r.end; k++) {
out.push(prefixFor(entries[k].type) + entries[k].text);
}
}
return out.join('\n');
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →