Skip to content

Text Diff Viewer — Go source

Compare two pieces of text and see exactly what changed. Highlights added and removed lines, words, or characters, shows a per-side summary, and exports a unified diff you can paste into a PR or commit. Runs 100% in your browser.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package textdiff is the Go twin of CosmoDev's src/lib/text-diff.ts (dual
// source: the web lib is TypeScript, the CLI lib is Go — kept in lock-step).
// It computes a diff between two strings at line / word / char granularity
// using a classic LCS (longest-common-subsequence) dynamic-programming table.
// No dependencies, fully deterministic, never panics. The table-driven tests
// in text-diff_test.go share vectors with src/lib/text-diff.test.ts so the two
// implementations are held to the same contract.
//
// Token invariant (mirrors the TS lib): every tokenizer splits a string into
// tokens whose exact concatenation reconstructs the original, so concatenating
// every DiffPart.Text in order (with default opts) reconstructs the changed
// string b.
package textdiff

import (
	"fmt"
	"regexp"
	"strings"
	"unicode"
)

// DiffType labels a diff segment.
type DiffType string

const (
	Equal   DiffType = "equal"
	Added   DiffType = "added"
	Removed DiffType = "removed"
)

// DiffPart is one merged run of the diff: a type and its original text.
type DiffPart struct {
	Type DiffType
	Text string
}

// Granularity selects the diff/tokenize unit.
type Granularity string

const (
	Line Granularity = "line"
	Word Granularity = "word"
	Char Granularity = "char"
)

// DiffOptions mirrors DiffOptions in the TS lib. The zero value
// (DiffOptions{}) matches the TS default ({}): no normalization, so the
// reconstruction invariant holds exactly. DiffOptions is compared by value and
// is never mutated by Diff.
type DiffOptions struct {
	IgnoreCase       bool
	Trim             bool
	IgnoreWhitespace bool
}

// UnifiedHeaders carries the optional --- / +++ file labels for ToUnifiedDiff.
type UnifiedHeaders struct {
	Old string
	New string
}

// contextLines is the unified-diff context window (mirrors CONTEXT in the TS).
const contextLines = 3

var whitespaceRun = regexp.MustCompile(`\s+`)

// normalizeKey normalizes a token for COMPARISON only — the original token is
// always emitted in the diff output. With the zero-value DiffOptions this is
// the identity function, so the reconstruction invariant holds exactly;
// opt-in normalization may relax it for tokens that match only after
// normalization (an equal part then carries a's original text).
func normalizeKey(token string, opts DiffOptions) string {
	s := token
	if opts.IgnoreWhitespace {
		s = whitespaceRun.ReplaceAllString(s, " ")
		s = strings.TrimSpace(s)
	}
	if opts.Trim {
		s = strings.TrimSpace(s)
	}
	if opts.IgnoreCase {
		s = strings.ToLower(s)
	}
	return s
}

// Tokenize splits text into tokens, mirroring tokenize() in the TS lib:
//   - Char: code-point split (unicode-safe); concatenation === text.
//   - Word: alternating maximal whitespace runs and non-whitespace runs;
//     concatenation === text.
//   - Line: content-only lines (strings.Split(text, "\n")); a trailing ""
//     element marks that the text ends with a newline. Lines are compared by
//     content, so a final no-newline line still matches a mid-file line.
//
// The empty string tokenizes to a nil slice at every granularity.
func Tokenize(text string, g Granularity) []string {
	if text == "" {
		return nil
	}
	switch g {
	case Char:
		out := make([]string, 0, len(text))
		for _, r := range text {
			out = append(out, string(r))
		}
		return out
	case Word:
		return tokenizeWord(text)
	case Line:
		return strings.Split(text, "\n")
	}
	return nil
}

// tokenizeWord splits text into alternating whitespace / non-whitespace runs,
// mirroring text.split(/(\s+)/).filter(t => t.length > 0). Each run is built
// maximal and non-empty, so the empty-string filter is implicit. unicode.IsSpace
// matches JS \s closely (ASCII spaces plus Unicode space characters).
func tokenizeWord(text string) []string {
	runes := []rune(text)
	var out []string
	for i := 0; i < len(runes); {
		space := unicode.IsSpace(runes[i])
		j := i + 1
		for j < len(runes) && unicode.IsSpace(runes[j]) == space {
			j++
		}
		out = append(out, string(runes[i:j]))
		i = j
	}
	return out
}

// Diff computes a diff between a (original) and b (changed) at the requested
// granularity. It returns merged runs of {Type, Text}; concatenating every
// Text (with default opts) reconstructs b exactly. Opt-in normalization
// compares on a normalized key but emits the ORIGINAL token, so a
// normalized-equal part carries a's text and may relax that invariant. Uses an
// LCS dynamic-programming table.
//
// granularity and opts are required in Go (no default args); pass Line and
// DiffOptions{} to match the TS defaults diff(a, b) and diff(a, b, g).
func Diff(a, b string, g Granularity, opts DiffOptions) []DiffPart {
	A := Tokenize(a, g)
	B := Tokenize(b, g)
	// Compare on a normalized key so opt-in ignoreCase/trim/ignoreWhitespace can
	// match tokens that differ superficially; the ORIGINAL token is emitted.
	aKey := make([]string, len(A))
	for i, t := range A {
		aKey[i] = normalizeKey(t, opts)
	}
	bKey := make([]string, len(B))
	for i, t := range B {
		bKey[i] = normalizeKey(t, opts)
	}
	n, m := len(aKey), len(bKey)

	// dp[i][j] = length of the longest common subsequence of aKey[i..] and bKey[j..].
	dp := make([][]int, n+1)
	for i := range dp {
		dp[i] = make([]int, m+1)
	}
	for i := n - 1; i >= 0; i-- {
		for j := m - 1; j >= 0; j-- {
			if aKey[i] == bKey[j] {
				dp[i][j] = dp[i+1][j+1] + 1
			} else {
				dp[i][j] = max(dp[i+1][j], dp[i][j+1])
			}
		}
	}

	var raw []DiffPart
	i, j := 0, 0
	for i < n && j < m {
		if aKey[i] == bKey[j] {
			raw = append(raw, DiffPart{Equal, A[i]})
			i++
			j++
		} else if dp[i+1][j] >= dp[i][j+1] {
			raw = append(raw, DiffPart{Removed, A[i]})
			i++
		} else {
			raw = append(raw, DiffPart{Added, B[j]})
			j++
		}
	}
	for i < n {
		raw = append(raw, DiffPart{Removed, A[i]})
		i++
	}
	for j < m {
		raw = append(raw, DiffPart{Added, B[j]})
		j++
	}

	// Merge consecutive runs of the same type into a single part. Line tokens
	// are content-only, so they rejoin with "\n"; char/word tokens concatenate
	// directly (their separators are already inside the tokens).
	sep := ""
	if g == Line {
		sep = "\n"
	}
	var merged []DiffPart
	for _, r := range raw {
		if len(merged) > 0 && merged[len(merged)-1].Type == r.Type {
			merged[len(merged)-1].Text += sep + r.Text
		} else {
			merged = append(merged, r)
		}
	}
	return merged
}

// DiffSummary counts characters per diff category (granularity-agnostic).
type DiffSummary struct {
	Added     int
	Removed   int
	Unchanged int
}

// Summary counts characters (Unicode code points) per diff category, mirroring
// summary() in the TS lib. Under opt-in normalization an equal part carries
// a's text, so Unchanged reflects a's length, not b's (and Added/Removed can
// read 0 for a≠b).
func Summary(parts []DiffPart) DiffSummary {
	var out DiffSummary
	for _, p := range parts {
		n := len([]rune(p.Text))
		switch p.Type {
		case Added:
			out.Added += n
		case Removed:
			out.Removed += n
		default:
			out.Unchanged += n
		}
	}
	return out
}

// lineEntry is one expanded output line (mirrors the unexported LineEntry in TS).
type lineEntry struct {
	typ  DiffType
	text string
}

// expandLines expands diff parts into one entry per output line. A trailing
// newline produces no phantom empty line — it is the terminator of the
// preceding line.
func expandLines(parts []DiffPart) []lineEntry {
	var entries []lineEntry
	for _, p := range parts {
		segs := strings.Split(p.Text, "\n")
		if strings.HasSuffix(p.Text, "\n") {
			segs = segs[:len(segs)-1]
		}
		for _, s := range segs {
			entries = append(entries, lineEntry{p.Type, s})
		}
	}
	return entries
}

func prefixFor(t DiffType) string {
	switch t {
	case Added:
		return "+"
	case Removed:
		return "-"
	default:
		return " "
	}
}

// ToUnifiedDiff renders a unified-diff string. It emits "--- "/"+++ " header
// lines when headers is non-nil. For Line granularity it groups changes into
// hunks with contextLines of context and a "@@ -oldStart,oldLen +newStart,newLen @@"
// header per hunk. For Word/Char granularity it emits one prefixed line per
// output line with no hunk headers.
//
// granularity is required in Go; pass Line to match the TS default.
func ToUnifiedDiff(parts []DiffPart, headers *UnifiedHeaders, g Granularity) string {
	var out []string
	if headers != nil {
		out = append(out, "--- "+headers.Old, "+++ "+headers.New)
	}

	entries := expandLines(parts)

	if g != Line {
		for _, e := range entries {
			out = append(out, prefixFor(e.typ)+e.text)
		}
		return strings.Join(out, "\n")
	}

	// Line granularity: group changes into hunks bounded by contextLines.
	var changedIdx []int
	for k, e := range entries {
		if e.typ != Equal {
			changedIdx = append(changedIdx, k)
		}
	}
	if len(changedIdx) == 0 {
		return strings.Join(out, "\n")
	}

	// Merge changes within 2*contextLines of each other into one hunk range.
	type rng struct{ start, end int }
	var ranges []rng
	cur := rng{
		start: max(0, changedIdx[0]-contextLines),
		end:   min(len(entries)-1, changedIdx[0]+contextLines),
	}
	for k := 1; k < len(changedIdx); k++ {
		s := max(0, changedIdx[k]-contextLines)
		if s <= cur.end+1 {
			cur.end = min(len(entries)-1, changedIdx[k]+contextLines)
		} else {
			ranges = append(ranges, cur)
			cur = rng{start: s, end: min(len(entries)-1, changedIdx[k]+contextLines)}
		}
	}
	ranges = append(ranges, cur)

	for _, r := range ranges {
		oldBefore, newBefore := 0, 0
		for k := 0; k < r.start; k++ {
			if entries[k].typ != Added {
				oldBefore++
			}
			if entries[k].typ != Removed {
				newBefore++
			}
		}
		oldLen, newLen := 0, 0
		for k := r.start; k <= r.end; k++ {
			if entries[k].typ != Added {
				oldLen++
			}
			if entries[k].typ != Removed {
				newLen++
			}
		}
		// Empty side convention: a length-0 side reports the line number of the
		// preceding line (or 0 when at the very start of an empty file).
		oldStart := oldBefore
		if oldLen != 0 {
			oldStart = oldBefore + 1
		}
		newStart := newBefore
		if newLen != 0 {
			newStart = newBefore + 1
		}
		out = append(out, fmt.Sprintf("@@ -%d,%d +%d,%d @@", oldStart, oldLen, newStart, newLen))
		for k := r.start; k <= r.end; k++ {
			out = append(out, prefixFor(entries[k].typ)+entries[k].text)
		}
	}

	return strings.Join(out, "\n")
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →