Skip to content

Text Diff Viewer — Rust source

Compare two pieces of text and see exactly what changed. Highlights added and removed lines, words, or characters, shows a per-side summary, and exports a unified diff you can paste into a PR or commit. Runs 100% in your browser.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! text-diff — Rust port (CosmoDev polyglot showcase).
//!
//! Computes a diff between two strings at line / word / char granularity using a
//! classic LCS (longest-common-subsequence) dynamic-programming table. No
//! external crates, fully deterministic.
//!
//! Ported from `src/lib/text-diff.ts` — display source, part of CosmoDev's
//! polyglot tool pages (dev.cosmolabs.org). Behavior is functionally equivalent
//! to the canonical TypeScript implementation.
//!
//! Token invariant: every tokenizer splits a string into tokens whose exact
//! concatenation reconstructs the original, so concatenating every `DiffPart.text`
//! in order reconstructs the changed string `b` (under default options).

/// The category of a diff part.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum DiffType {
    Equal,
    Added,
    Removed,
}

/// Selects the token size the diff operates on.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Granularity {
    Line,
    Word,
    Char,
}

/// A single merged run of the diff: a category and its text.
#[derive(Clone, Debug)]
pub struct DiffPart {
    pub kind: DiffType,
    pub text: String,
}

/// Comparison-key normalization. The original token is always emitted. The
/// default (all `false`) is the identity — reconstruction holds exactly; opting
/// in may relax it.
#[derive(Clone, Debug, Default)]
pub struct DiffOptions {
    pub ignore_case: bool,
    pub trim: bool,
    pub ignore_whitespace: bool,
}

/// Per-category character counts (granularity-agnostic).
#[derive(Clone, Debug, Default)]
pub struct DiffSummary {
    pub added: usize,
    pub removed: usize,
    pub unchanged: usize,
}

/// Filename labels for the `---`/`+++` header lines.
#[derive(Clone, Debug, Default)]
pub struct UnifiedHeaders {
    pub old: String,
    pub new: String,
}

/// Lines of context kept around each change in unified output.
const CONTEXT: usize = 3;

/// Normalize a token for COMPARISON only; the original is always emitted.
///
/// Order matters: ignoreWhitespace first (collapse + trim), then an explicit
/// trim, then case folding. `to_lowercase` is Rust's Unicode-aware lowercase
/// mapping.
fn normalize_key(token: &str, opts: &DiffOptions) -> String {
    let mut s = String::from(token);
    if opts.ignore_whitespace {
        // split_whitespace is unicode-aware and skips runs, so joining with a
        // single space collapses every internal run and trims both ends.
        s = s.split_whitespace().collect::<Vec<_>>().join(" ");
    }
    if opts.trim {
        s = s.trim().to_string();
    }
    if opts.ignore_case {
        s = s.to_lowercase();
    }
    s
}

/// Split `text` into reconstructable tokens.
///
/// - `Char` — Unicode scalar value split (`chars()`); concatenation === text.
/// - `Word` — alternating maximal whitespace runs and non-whitespace runs;
///   concatenation === text.
/// - `Line` — content-only lines (split on `'\n'`); a trailing empty string
///   marks that the text ends with a newline. Reconstruction joins with `'\n'`.
pub fn tokenize(text: &str, granularity: Granularity) -> Vec<String> {
    if text.is_empty() {
        return Vec::new();
    }
    match granularity {
        Granularity::Char => text.chars().map(|c| c.to_string()).collect(),
        Granularity::Word => split_words(text),
        Granularity::Line => text.split('\n').map(|s| s.to_string()).collect(),
    }
}

/// Group consecutive characters by their whitespace class so whitespace runs and
/// non-whitespace runs survive as separate tokens. `char::is_whitespace` is
/// unicode-aware, so this matches the spirit of `\s+` without needing the regex
/// crate (kept dependency-free for the showcase).
fn split_words(text: &str) -> Vec<String> {
    let mut tokens = Vec::new();
    let mut cur = String::new();
    let mut cur_ws: Option<bool> = None;
    for c in text.chars() {
        let is_ws = c.is_whitespace();
        match cur_ws {
            None => {
                cur.push(c);
                cur_ws = Some(is_ws);
            }
            Some(w) if w == is_ws => cur.push(c),
            Some(_) => {
                tokens.push(std::mem::take(&mut cur));
                cur.push(c);
                cur_ws = Some(is_ws);
            }
        }
    }
    if !cur.is_empty() {
        tokens.push(cur);
    }
    tokens
}

/// Compute a diff between `a` (original) and `b` (changed) at the requested
/// granularity. Pass `Granularity::Line` and `&DiffOptions::default()` for the
/// canonical defaults. Returns merged runs of `DiffPart`. Uses an LCS
/// dynamic-programming table; opt-in normalization compares on a normalized key
/// but emits the ORIGINAL token.
pub fn diff(a: &str, b: &str, granularity: Granularity, opts: &DiffOptions) -> Vec<DiffPart> {
    let a_tokens = tokenize(a, granularity);
    let b_tokens = tokenize(b, granularity);

    // Compare on a normalized key; emit the original token.
    let a_key: Vec<String> = a_tokens.iter().map(|t| normalize_key(t, opts)).collect();
    let b_key: Vec<String> = b_tokens.iter().map(|t| normalize_key(t, opts)).collect();
    let n = a_key.len();
    let m = b_key.len();

    // dp[i][j] = length of the LCS of a_key[i..] and b_key[j..], built backwards
    // so each cell only depends on already-computed cells (i+1, j+1).
    let mut dp = vec![vec![0usize; m + 1]; n + 1];
    for i in (0..n).rev() {
        for j in (0..m).rev() {
            if a_key[i] == b_key[j] {
                dp[i][j] = dp[i + 1][j + 1] + 1;
            } else if dp[i + 1][j] >= dp[i][j + 1] {
                dp[i][j] = dp[i + 1][j];
            } else {
                dp[i][j] = dp[i][j + 1];
            }
        }
    }

    // Greedy walk: equal on a key match; otherwise drop the side whose remaining
    // LCS is larger. The `>=` tie favors 'removed', matching the canonical walk.
    let mut raw: Vec<DiffPart> = Vec::new();
    let mut i = 0;
    let mut j = 0;
    while i < n && j < m {
        if a_key[i] == b_key[j] {
            raw.push(DiffPart { kind: DiffType::Equal, text: a_tokens[i].clone() });
            i += 1;
            j += 1;
        } else if dp[i + 1][j] >= dp[i][j + 1] {
            raw.push(DiffPart { kind: DiffType::Removed, text: a_tokens[i].clone() });
            i += 1;
        } else {
            raw.push(DiffPart { kind: DiffType::Added, text: b_tokens[j].clone() });
            j += 1;
        }
    }
    while i < n {
        raw.push(DiffPart { kind: DiffType::Removed, text: a_tokens[i].clone() });
        i += 1;
    }
    while j < m {
        raw.push(DiffPart { kind: DiffType::Added, text: b_tokens[j].clone() });
        j += 1;
    }

    // Merge consecutive runs of the same type. Lines rejoin with '\n'; char/word
    // tokens already carry their separators and concatenate with "".
    let sep = if granularity == Granularity::Line { "\n" } else { "" };
    let mut merged: Vec<DiffPart> = Vec::new();
    for r in raw {
        if let Some(last) = merged.last_mut() {
            if last.kind == r.kind {
                last.text.push_str(sep);
                last.text.push_str(&r.text);
                continue;
            }
        }
        merged.push(r);
    }
    merged
}

/// Count characters per diff category. Counts Unicode scalar values
/// (`chars().count()`). Under opt-in normalization an equal part carries `a`'s
/// text, so `unchanged` reflects `a`'s length, not `b`'s.
pub fn summary(parts: &[DiffPart]) -> DiffSummary {
    let mut out = DiffSummary::default();
    for p in parts {
        let len = p.text.chars().count();
        match p.kind {
            DiffType::Added => out.added += len,
            DiffType::Removed => out.removed += len,
            DiffType::Equal => out.unchanged += len,
        }
    }
    out
}

/// Return the unified-diff prefix character for a line type.
fn prefix_for(t: DiffType) -> &'static str {
    match t {
        DiffType::Added => "+",
        DiffType::Removed => "-",
        DiffType::Equal => " ",
    }
}

/// One output line within unified rendering.
struct LineEntry {
    kind: DiffType,
    text: String,
}

/// Expand diff parts into one entry per output line. A trailing newline produces
/// no phantom empty line — it terminates the preceding line.
fn expand_lines(parts: &[DiffPart]) -> Vec<LineEntry> {
    let mut entries = Vec::new();
    for p in parts {
        let mut segs: Vec<&str> = p.text.split('\n').collect();
        if p.text.ends_with('\n') {
            segs.pop(); // drop the phantom empty segment from the trailing separator
        }
        for s in segs {
            entries.push(LineEntry { kind: p.kind, text: s.to_string() });
        }
    }
    entries
}

/// Render a unified-diff string. Pass `Some(headers)` to emit `---`/`+++` header
/// lines. For `Granularity::Line`, groups changes into hunks with `CONTEXT` (3)
/// lines of context and a `@@ -oldStart,oldLen +newStart,newLen @@` header per
/// hunk. For `Word`/`Char` emits one prefixed line per output line, no hunks.
pub fn to_unified_diff(
    parts: &[DiffPart],
    headers: Option<&UnifiedHeaders>,
    granularity: Granularity,
) -> String {
    let mut out: Vec<String> = Vec::new();
    if let Some(h) = headers {
        out.push(format!("--- {}", h.old));
        out.push(format!("+++ {}", h.new));
    }

    let entries = expand_lines(parts);

    if granularity != Granularity::Line {
        for e in &entries {
            out.push(format!("{}{}", prefix_for(e.kind), e.text));
        }
        return out.join("\n");
    }

    // Line granularity: group changes into hunks bounded by CONTEXT lines.
    let changed_idx: Vec<usize> = entries
        .iter()
        .enumerate()
        .filter(|(_, e)| e.kind != DiffType::Equal)
        .map(|(k, _)| k)
        .collect();
    if changed_idx.is_empty() {
        return out.join("\n");
    }
    let last = entries.len() - 1;

    // Merge changes within 2*CONTEXT of each other into one hunk [start..end].
    let mut ranges: Vec<(usize, usize)> = Vec::new();
    let mut cur = (
        changed_idx[0].saturating_sub(CONTEXT),
        (changed_idx[0] + CONTEXT).min(last),
    );
    for &k in &changed_idx[1..] {
        let s = k.saturating_sub(CONTEXT);
        if s <= cur.1 + 1 {
            cur.1 = (k + CONTEXT).min(last);
        } else {
            ranges.push(cur);
            cur = (s, (k + CONTEXT).min(last));
        }
    }
    ranges.push(cur);

    for &(r_start, r_end) in &ranges {
        // Old side = entries that are not 'added'; new side = not 'removed'.
        let mut old_before = 0;
        let mut new_before = 0;
        for k in 0..r_start {
            if entries[k].kind != DiffType::Added {
                old_before += 1;
            }
            if entries[k].kind != DiffType::Removed {
                new_before += 1;
            }
        }
        let mut old_len = 0;
        let mut new_len = 0;
        for k in r_start..=r_end {
            if entries[k].kind != DiffType::Added {
                old_len += 1;
            }
            if entries[k].kind != DiffType::Removed {
                new_len += 1;
            }
        }
        // Empty-side convention: a length-0 side reports the preceding line number
        // (or 0 at the very start of an empty file).
        let old_start = if old_len == 0 { old_before } else { old_before + 1 };
        let new_start = if new_len == 0 { new_before } else { new_before + 1 };
        out.push(format!(
            "@@ -{},{} +{},{} @@",
            old_start, old_len, new_start, new_len
        ));
        for k in r_start..=r_end {
            out.push(format!("{}{}", prefix_for(entries[k].kind), entries[k].text));
        }
    }

    out.join("\n")
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →