Skip to content

Text Diff Viewer — C source

Compare two pieces of text and see exactly what changed. Highlights added and removed lines, words, or characters, shows a per-side summary, and exports a unified diff you can paste into a PR or commit. Runs 100% in your browser.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/* text-diff — line-granularity diff via an LCS dynamic-programming table. Language: C (C11). Port of src/lib/text-diff.ts — core tokenizer/backwards-DP/greedy-walk/run-merge; word/char granularity, normalization options and unified hunk headers live in this dir's javascript.js (80-line budget). */
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

typedef enum { EQUAL = 0, REMOVED, ADDED } Type;
typedef struct { Type type; char *text; } Part; /* merged run of same-type lines */

/* Content-only lines — the TS tokenizer's 'line' case: tokens joined with '\n' reconstruct the text. */
static char **split_lines(const char *text, size_t *n)
{
    size_t cap = 8, cnt = 0;
    char **lines = malloc(cap * sizeof *lines);
    const char *p = text;
    for (;;) {
        const char *nl = strchr(p, '\n');
        size_t len = nl ? (size_t)(nl - p) : strlen(p);
        if (cnt == cap) lines = realloc(lines, (cap *= 2) * sizeof *lines);
        char *line = memcpy(malloc(len + 1), p, len);
        line[len] = '\0';
        lines[cnt++] = line;
        if (!nl) break;
        p = nl + 1;
    }
    *n = cnt;
    return lines;
}

/* dp[i][j] = LCS length of a[i..] and b[j..], built backwards. The greedy walk
 * emits equal on match, else drops the larger-remaining-LCS side ('>=' tie favors 'removed'). */
static Part *diff_lines(char **a, size_t n, char **b, size_t m, size_t *np)
{
    size_t w = m + 1;
    size_t *dp = calloc((n + 1) * w, sizeof *dp);
    for (size_t k = n; k-- > 0;)
        for (size_t l = m; l-- > 0;)
            dp[k * w + l] = strcmp(a[k], b[l]) == 0 ? dp[(k + 1) * w + l + 1] + 1
                : dp[(k + 1) * w + l] >= dp[k * w + l + 1] ? dp[(k + 1) * w + l] : dp[k * w + l + 1];
    Part *parts = NULL; /* merge step: consecutive same-type tokens rejoin '\n' */
    size_t cnt = 0, i = 0, j = 0;
    while (i < n || j < m) {
        Type t;
        const char *tok;
        if (i < n && j < m && strcmp(a[i], b[j]) == 0) { t = EQUAL; tok = a[i]; i++; j++; }
        else if (j == m || (i < n && dp[(i + 1) * w + j] >= dp[i * w + j + 1])) { t = REMOVED; tok = a[i++]; }
        else { t = ADDED; tok = b[j++]; }
        if (cnt && parts[cnt - 1].type == t) {
            Part *p = &parts[cnt - 1];
            p->text = realloc(p->text, strlen(p->text) + strlen(tok) + 2), strcat(strcat(p->text, "\n"), tok);
        } else {
            parts = realloc(parts, (cnt + 1) * sizeof *parts);
            parts[cnt++] = (Part){ t, strcpy(malloc(strlen(tok) + 1), tok) };
        }
    }
    free(dp);
    *np = cnt;
    return parts;
}

int main(void)
{
    const char *a = "const x = 1;\nfunction greet(name) {\n  return 'hi ' + name;\n}\nconsole.log(greet('dev'));";
    const char *b = "const x = 2;\nfunction greet(name) {\n  return 'hello, ' + name + '!';\n}\nconsole.log(greet('dev'));";
    size_t n, m, np, add = 0, rem = 0, same = 0;
    char **al = split_lines(a, &n), **bl = split_lines(b, &m);
    Part *parts = diff_lines(al, n, bl, m, &np);
    for (size_t k = 0; k < np; k++) { /* one prefix per line inside each part */
        static const char pre[3] = { ' ', '-', '+' };
        for (char *line = parts[k].text, *nl; ; line = nl + 1) {
            nl = strchr(line, '\n');
            printf("%c %.*s\n", pre[parts[k].type], nl ? (int)(nl - line) : (int)strlen(line), line);
            if (!nl) break;
        }
        if (parts[k].type == ADDED) add += strlen(parts[k].text);
        else if (parts[k].type == REMOVED) rem += strlen(parts[k].text);
        else same += strlen(parts[k].text);
    }
    printf("summary: +%zu added, -%zu removed, =%zu unchanged chars\n", add, rem, same);
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →