Text Diff Viewer — Rust source
Compare two pieces of text and see exactly what changed. Highlights added and removed lines, words, or characters, shows a per-side summary, and exports a unified diff you can paste into a PR or commit. Runs 100% in your browser.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! text-diff — Rust port (CosmoDev polyglot showcase).
//!
//! Computes a diff between two strings at line / word / char granularity using a
//! classic LCS (longest-common-subsequence) dynamic-programming table. No
//! external crates, fully deterministic.
//!
//! Ported from `src/lib/text-diff.ts` — display source, part of CosmoDev's
//! polyglot tool pages (dev.cosmolabs.org). Behavior is functionally equivalent
//! to the canonical TypeScript implementation.
//!
//! Token invariant: every tokenizer splits a string into tokens whose exact
//! concatenation reconstructs the original, so concatenating every `DiffPart.text`
//! in order reconstructs the changed string `b` (under default options).
/// The category of a diff part.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum DiffType {
Equal,
Added,
Removed,
}
/// Selects the token size the diff operates on.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Granularity {
Line,
Word,
Char,
}
/// A single merged run of the diff: a category and its text.
#[derive(Clone, Debug)]
pub struct DiffPart {
pub kind: DiffType,
pub text: String,
}
/// Comparison-key normalization. The original token is always emitted. The
/// default (all `false`) is the identity — reconstruction holds exactly; opting
/// in may relax it.
#[derive(Clone, Debug, Default)]
pub struct DiffOptions {
pub ignore_case: bool,
pub trim: bool,
pub ignore_whitespace: bool,
}
/// Per-category character counts (granularity-agnostic).
#[derive(Clone, Debug, Default)]
pub struct DiffSummary {
pub added: usize,
pub removed: usize,
pub unchanged: usize,
}
/// Filename labels for the `---`/`+++` header lines.
#[derive(Clone, Debug, Default)]
pub struct UnifiedHeaders {
pub old: String,
pub new: String,
}
/// Lines of context kept around each change in unified output.
const CONTEXT: usize = 3;
/// Normalize a token for COMPARISON only; the original is always emitted.
///
/// Order matters: ignoreWhitespace first (collapse + trim), then an explicit
/// trim, then case folding. `to_lowercase` is Rust's Unicode-aware lowercase
/// mapping.
fn normalize_key(token: &str, opts: &DiffOptions) -> String {
let mut s = String::from(token);
if opts.ignore_whitespace {
// split_whitespace is unicode-aware and skips runs, so joining with a
// single space collapses every internal run and trims both ends.
s = s.split_whitespace().collect::<Vec<_>>().join(" ");
}
if opts.trim {
s = s.trim().to_string();
}
if opts.ignore_case {
s = s.to_lowercase();
}
s
}
/// Split `text` into reconstructable tokens.
///
/// - `Char` — Unicode scalar value split (`chars()`); concatenation === text.
/// - `Word` — alternating maximal whitespace runs and non-whitespace runs;
/// concatenation === text.
/// - `Line` — content-only lines (split on `'\n'`); a trailing empty string
/// marks that the text ends with a newline. Reconstruction joins with `'\n'`.
pub fn tokenize(text: &str, granularity: Granularity) -> Vec<String> {
if text.is_empty() {
return Vec::new();
}
match granularity {
Granularity::Char => text.chars().map(|c| c.to_string()).collect(),
Granularity::Word => split_words(text),
Granularity::Line => text.split('\n').map(|s| s.to_string()).collect(),
}
}
/// Group consecutive characters by their whitespace class so whitespace runs and
/// non-whitespace runs survive as separate tokens. `char::is_whitespace` is
/// unicode-aware, so this matches the spirit of `\s+` without needing the regex
/// crate (kept dependency-free for the showcase).
fn split_words(text: &str) -> Vec<String> {
let mut tokens = Vec::new();
let mut cur = String::new();
let mut cur_ws: Option<bool> = None;
for c in text.chars() {
let is_ws = c.is_whitespace();
match cur_ws {
None => {
cur.push(c);
cur_ws = Some(is_ws);
}
Some(w) if w == is_ws => cur.push(c),
Some(_) => {
tokens.push(std::mem::take(&mut cur));
cur.push(c);
cur_ws = Some(is_ws);
}
}
}
if !cur.is_empty() {
tokens.push(cur);
}
tokens
}
/// Compute a diff between `a` (original) and `b` (changed) at the requested
/// granularity. Pass `Granularity::Line` and `&DiffOptions::default()` for the
/// canonical defaults. Returns merged runs of `DiffPart`. Uses an LCS
/// dynamic-programming table; opt-in normalization compares on a normalized key
/// but emits the ORIGINAL token.
pub fn diff(a: &str, b: &str, granularity: Granularity, opts: &DiffOptions) -> Vec<DiffPart> {
let a_tokens = tokenize(a, granularity);
let b_tokens = tokenize(b, granularity);
// Compare on a normalized key; emit the original token.
let a_key: Vec<String> = a_tokens.iter().map(|t| normalize_key(t, opts)).collect();
let b_key: Vec<String> = b_tokens.iter().map(|t| normalize_key(t, opts)).collect();
let n = a_key.len();
let m = b_key.len();
// dp[i][j] = length of the LCS of a_key[i..] and b_key[j..], built backwards
// so each cell only depends on already-computed cells (i+1, j+1).
let mut dp = vec![vec![0usize; m + 1]; n + 1];
for i in (0..n).rev() {
for j in (0..m).rev() {
if a_key[i] == b_key[j] {
dp[i][j] = dp[i + 1][j + 1] + 1;
} else if dp[i + 1][j] >= dp[i][j + 1] {
dp[i][j] = dp[i + 1][j];
} else {
dp[i][j] = dp[i][j + 1];
}
}
}
// Greedy walk: equal on a key match; otherwise drop the side whose remaining
// LCS is larger. The `>=` tie favors 'removed', matching the canonical walk.
let mut raw: Vec<DiffPart> = Vec::new();
let mut i = 0;
let mut j = 0;
while i < n && j < m {
if a_key[i] == b_key[j] {
raw.push(DiffPart { kind: DiffType::Equal, text: a_tokens[i].clone() });
i += 1;
j += 1;
} else if dp[i + 1][j] >= dp[i][j + 1] {
raw.push(DiffPart { kind: DiffType::Removed, text: a_tokens[i].clone() });
i += 1;
} else {
raw.push(DiffPart { kind: DiffType::Added, text: b_tokens[j].clone() });
j += 1;
}
}
while i < n {
raw.push(DiffPart { kind: DiffType::Removed, text: a_tokens[i].clone() });
i += 1;
}
while j < m {
raw.push(DiffPart { kind: DiffType::Added, text: b_tokens[j].clone() });
j += 1;
}
// Merge consecutive runs of the same type. Lines rejoin with '\n'; char/word
// tokens already carry their separators and concatenate with "".
let sep = if granularity == Granularity::Line { "\n" } else { "" };
let mut merged: Vec<DiffPart> = Vec::new();
for r in raw {
if let Some(last) = merged.last_mut() {
if last.kind == r.kind {
last.text.push_str(sep);
last.text.push_str(&r.text);
continue;
}
}
merged.push(r);
}
merged
}
/// Count characters per diff category. Counts Unicode scalar values
/// (`chars().count()`). Under opt-in normalization an equal part carries `a`'s
/// text, so `unchanged` reflects `a`'s length, not `b`'s.
pub fn summary(parts: &[DiffPart]) -> DiffSummary {
let mut out = DiffSummary::default();
for p in parts {
let len = p.text.chars().count();
match p.kind {
DiffType::Added => out.added += len,
DiffType::Removed => out.removed += len,
DiffType::Equal => out.unchanged += len,
}
}
out
}
/// Return the unified-diff prefix character for a line type.
fn prefix_for(t: DiffType) -> &'static str {
match t {
DiffType::Added => "+",
DiffType::Removed => "-",
DiffType::Equal => " ",
}
}
/// One output line within unified rendering.
struct LineEntry {
kind: DiffType,
text: String,
}
/// Expand diff parts into one entry per output line. A trailing newline produces
/// no phantom empty line — it terminates the preceding line.
fn expand_lines(parts: &[DiffPart]) -> Vec<LineEntry> {
let mut entries = Vec::new();
for p in parts {
let mut segs: Vec<&str> = p.text.split('\n').collect();
if p.text.ends_with('\n') {
segs.pop(); // drop the phantom empty segment from the trailing separator
}
for s in segs {
entries.push(LineEntry { kind: p.kind, text: s.to_string() });
}
}
entries
}
/// Render a unified-diff string. Pass `Some(headers)` to emit `---`/`+++` header
/// lines. For `Granularity::Line`, groups changes into hunks with `CONTEXT` (3)
/// lines of context and a `@@ -oldStart,oldLen +newStart,newLen @@` header per
/// hunk. For `Word`/`Char` emits one prefixed line per output line, no hunks.
pub fn to_unified_diff(
parts: &[DiffPart],
headers: Option<&UnifiedHeaders>,
granularity: Granularity,
) -> String {
let mut out: Vec<String> = Vec::new();
if let Some(h) = headers {
out.push(format!("--- {}", h.old));
out.push(format!("+++ {}", h.new));
}
let entries = expand_lines(parts);
if granularity != Granularity::Line {
for e in &entries {
out.push(format!("{}{}", prefix_for(e.kind), e.text));
}
return out.join("\n");
}
// Line granularity: group changes into hunks bounded by CONTEXT lines.
let changed_idx: Vec<usize> = entries
.iter()
.enumerate()
.filter(|(_, e)| e.kind != DiffType::Equal)
.map(|(k, _)| k)
.collect();
if changed_idx.is_empty() {
return out.join("\n");
}
let last = entries.len() - 1;
// Merge changes within 2*CONTEXT of each other into one hunk [start..end].
let mut ranges: Vec<(usize, usize)> = Vec::new();
let mut cur = (
changed_idx[0].saturating_sub(CONTEXT),
(changed_idx[0] + CONTEXT).min(last),
);
for &k in &changed_idx[1..] {
let s = k.saturating_sub(CONTEXT);
if s <= cur.1 + 1 {
cur.1 = (k + CONTEXT).min(last);
} else {
ranges.push(cur);
cur = (s, (k + CONTEXT).min(last));
}
}
ranges.push(cur);
for &(r_start, r_end) in &ranges {
// Old side = entries that are not 'added'; new side = not 'removed'.
let mut old_before = 0;
let mut new_before = 0;
for k in 0..r_start {
if entries[k].kind != DiffType::Added {
old_before += 1;
}
if entries[k].kind != DiffType::Removed {
new_before += 1;
}
}
let mut old_len = 0;
let mut new_len = 0;
for k in r_start..=r_end {
if entries[k].kind != DiffType::Added {
old_len += 1;
}
if entries[k].kind != DiffType::Removed {
new_len += 1;
}
}
// Empty-side convention: a length-0 side reports the preceding line number
// (or 0 at the very start of an empty file).
let old_start = if old_len == 0 { old_before } else { old_before + 1 };
let new_start = if new_len == 0 { new_before } else { new_before + 1 };
out.push(format!(
"@@ -{},{} +{},{} @@",
old_start, old_len, new_start, new_len
));
for k in r_start..=r_end {
out.push(format!("{}{}", prefix_for(entries[k].kind), entries[k].text));
}
}
out.join("\n")
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →