Skip to content

Text Diff Viewer — TypeScript source

Compare two pieces of text and see exactly what changed. Highlights added and removed lines, words, or characters, shows a per-side summary, and exports a unified diff you can paste into a PR or commit. Runs 100% in your browser.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Text Diff Viewer - pure logic.
//
// Computes a diff between two strings at line / word / char granularity using a
// classic LCS (longest-common-subsequence) dynamic-programming table. No
// dependencies, fully deterministic, safe to run in the browser.
//
// Token invariant: every tokenizer splits a string into tokens whose exact
// concatenation reconstructs the original. That means concatenating every
// `DiffPart.text` in order reconstructs the *changed* string `b`.

export type DiffType = 'equal' | 'added' | 'removed';

export interface DiffPart {
  type: DiffType;
  text: string;
}

export type Granularity = 'line' | 'word' | 'char';

export interface DiffOptions {
  /**
   * Normalization applies to the comparison key only - the original token is
   * always emitted. Designed for `line` granularity; at `word`/`char` the
   * whitespace options collapse every whitespace token/char to an empty key,
   * which is deterministic but can be surprising.
   */
  /** Case-insensitive comparison (the original text is still displayed). */
  ignoreCase?: boolean;
  /** Ignore leading/trailing whitespace per token when comparing. */
  trim?: boolean;
  /** Collapse internal whitespace runs to a single space (and trim) when comparing. */
  ignoreWhitespace?: boolean;
}

/**
 * Normalize a token for COMPARISON only - the original token is always what's
 * emitted in the diff output. With default `{}` opts this is the identity fn,
 * so the reconstruction invariant (concatenating equal+added `.text` === `b`)
 * holds exactly; opt-in normalization may relax it for tokens that match only
 * after normalization (an equal part then carries `a`'s original text).
 */
function normalizeKey(token: string, opts: DiffOptions): string {
  let s = token;
  if (opts.ignoreWhitespace) s = s.replace(/\s+/g, ' ').trim();
  if (opts.trim) s = s.trim();
  if (opts.ignoreCase) s = s.toLowerCase();
  return s;
}

/**
 * Split `text` into tokens.
 * - `char`: code-point split (unicode-safe); concatenation === `text`.
 * - `word`: alternating maximal whitespace runs and non-whitespace runs;
 *   concatenation === `text`.
 * - `line`: content-only lines (`text.split('\n')`); a trailing `''` element
 *   marks that the text ends with a newline. Lines are compared by content, so
 *   `"b"` (last line, no newline) still matches `"b"` mid-file - no spurious
 *   remove/add when appending. Reconstruction joins line tokens with `'\n'`.
 */
export function tokenize(text: string, granularity: Granularity): string[] {
  if (text === '') return [];
  switch (granularity) {
    case 'char':
      return Array.from(text);
    case 'word':
      return text.split(/(\s+)/).filter((t) => t.length > 0);
    case 'line':
      return text.split('\n');
  }
}

/**
 * Compute a diff between `a` (original) and `b` (changed) at the requested
 * granularity. Returns merged runs of `{ type, text }`; concatenating every
 * `.text` reconstructs `b` exactly with default opts. Opt-in normalization
 * (`ignoreCase`/`trim`/`ignoreWhitespace`) compares on a normalized key but
 * emits the ORIGINAL token, so a normalized-equal part carries `a`'s text and
 * may relax this invariant. Uses an LCS dynamic-programming table.
 */
export function diff(
  a: string,
  b: string,
  granularity: Granularity = 'line',
  opts: DiffOptions = {},
): DiffPart[] {
  const A = tokenize(a, granularity);
  const B = tokenize(b, granularity);
  // Compare on a normalized key so opt-in ignoreCase/trim/ignoreWhitespace can
  // match tokens that differ superficially; the ORIGINAL token is always emitted.
  const aKey = A.map((t) => normalizeKey(t, opts));
  const bKey = B.map((t) => normalizeKey(t, opts));
  const n = aKey.length;
  const m = bKey.length;

  // dp[i][j] = length of the longest common subsequence of aKey[i..] and bKey[j..].
  const dp: number[][] = Array.from({ length: n + 1 }, () => new Array<number>(m + 1).fill(0));
  for (let i = n - 1; i >= 0; i--) {
    for (let j = m - 1; j >= 0; j--) {
      if (aKey[i] === bKey[j]) dp[i][j] = dp[i + 1][j + 1] + 1;
      else dp[i][j] = Math.max(dp[i + 1][j], dp[i][j + 1]);
    }
  }

  const raw: { type: DiffType; text: string }[] = [];
  let i = 0;
  let j = 0;
  while (i < n && j < m) {
    if (aKey[i] === bKey[j]) {
      raw.push({ type: 'equal', text: A[i] });
      i++;
      j++;
    } else if (dp[i + 1][j] >= dp[i][j + 1]) {
      raw.push({ type: 'removed', text: A[i] });
      i++;
    } else {
      raw.push({ type: 'added', text: B[j] });
      j++;
    }
  }
  while (i < n) {
    raw.push({ type: 'removed', text: A[i] });
    i++;
  }
  while (j < m) {
    raw.push({ type: 'added', text: B[j] });
    j++;
  }

  // Merge consecutive runs of the same type into a single part. Line tokens are
  // content-only, so they are rejoined with '\n'; char/word tokens concatenate
  // directly (their separators are already inside the tokens).
  const sep = granularity === 'line' ? '\n' : '';
  const merged: DiffPart[] = [];
  for (const r of raw) {
    const last = merged[merged.length - 1];
    if (last && last.type === r.type) last.text += sep + r.text;
    else merged.push({ type: r.type, text: r.text });
  }
  return merged;
}

export interface DiffSummary {
  added: number;
  removed: number;
  unchanged: number;
}

/**
 * Count characters per diff category (granularity-agnostic). Under opt-in
 * normalization, an equal part carries `a`'s text, so `unchanged` reflects
 * `a`'s length, not `b`'s (and `added`/`removed` can read 0 for a≠b).
 */
export function summary(parts: DiffPart[]): DiffSummary {
  const out: DiffSummary = { added: 0, removed: 0, unchanged: 0 };
  for (const p of parts) {
    const len = p.text.length;
    if (p.type === 'added') out.added += len;
    else if (p.type === 'removed') out.removed += len;
    else out.unchanged += len;
  }
  return out;
}

export interface UnifiedHeaders {
  old: string;
  new: string;
}

/** Unified-diff context: unchanged lines kept around each change. */
const CONTEXT = 3;

interface LineEntry {
  type: DiffType;
  text: string;
}

/**
 * Expand diff parts into one entry per output line. A trailing newline produces
 * no phantom empty line (it is the terminator of the preceding line).
 */
function expandLines(parts: DiffPart[]): LineEntry[] {
  const entries: LineEntry[] = [];
  for (const p of parts) {
    const segs = p.text.split('\n');
    if (p.text.endsWith('\n')) segs.pop();
    for (const s of segs) entries.push({ type: p.type, text: s });
  }
  return entries;
}

function prefixFor(t: DiffType): string {
  return t === 'added' ? '+' : t === 'removed' ? '-' : ' ';
}

/**
 * Render a unified-diff string. Emits `--- `/`+++ ` header lines when `headers`
 * is provided. For `line` granularity, groups changes into hunks with 3 lines of
 * context and a `@@ -oldStart,oldLen +newStart,newLen @@` header per hunk. For
 * `word`/`char`, emits one prefixed line per output line with no hunk headers.
 */
export function toUnifiedDiff(
  parts: DiffPart[],
  headers?: UnifiedHeaders,
  granularity: Granularity = 'line',
): string {
  const out: string[] = [];
  if (headers) {
    out.push(`--- ${headers.old}`);
    out.push(`+++ ${headers.new}`);
  }

  const entries = expandLines(parts);

  if (granularity !== 'line') {
    for (const e of entries) out.push(prefixFor(e.type) + e.text);
    return out.join('\n');
  }

  // Line granularity: group changes into hunks bounded by CONTEXT lines.
  const changedIdx: number[] = [];
  for (let k = 0; k < entries.length; k++) {
    if (entries[k].type !== 'equal') changedIdx.push(k);
  }
  if (changedIdx.length === 0) {
    return out.join('\n');
  }

  // Merge changes within 2*CONTEXT of each other into one hunk range [start..end].
  const ranges: { start: number; end: number }[] = [];
  let cur = {
    start: Math.max(0, changedIdx[0] - CONTEXT),
    end: Math.min(entries.length - 1, changedIdx[0] + CONTEXT),
  };
  for (let k = 1; k < changedIdx.length; k++) {
    const s = Math.max(0, changedIdx[k] - CONTEXT);
    if (s <= cur.end + 1) {
      cur.end = Math.min(entries.length - 1, changedIdx[k] + CONTEXT);
    } else {
      ranges.push(cur);
      cur = { start: s, end: Math.min(entries.length - 1, changedIdx[k] + CONTEXT) };
    }
  }
  ranges.push(cur);

  for (const r of ranges) {
    let oldBefore = 0;
    let newBefore = 0;
    for (let k = 0; k < r.start; k++) {
      if (entries[k].type !== 'added') oldBefore++;
      if (entries[k].type !== 'removed') newBefore++;
    }
    let oldLen = 0;
    let newLen = 0;
    for (let k = r.start; k <= r.end; k++) {
      if (entries[k].type !== 'added') oldLen++;
      if (entries[k].type !== 'removed') newLen++;
    }
    // Empty side convention: a length-0 side reports the line number of the
    // preceding line (or 0 when at the very start of an empty file).
    const oldStart = oldLen === 0 ? oldBefore : oldBefore + 1;
    const newStart = newLen === 0 ? newBefore : newBefore + 1;
    out.push(`@@ -${oldStart},${oldLen} +${newStart},${newLen} @@`);
    for (let k = r.start; k <= r.end; k++) {
      out.push(prefixFor(entries[k].type) + entries[k].text);
    }
  }

  return out.join('\n');
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →