Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion .github/workflows/deploy-site.yml
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,8 @@ on:
branches: [main]
paths:
- 'apps/web/**'
- 'packages/**'
- 'scripts/**'

env:
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true
Expand Down Expand Up @@ -62,7 +64,7 @@ jobs:
run: npx tsx scripts/generate-diffs.ts --repo apps/web/content-data --output apps/web/public/diffs/

- name: Build site
run: pnpm --filter @civic-source/web build
run: pnpm --filter @civic-source/web... build

- name: Upload artifact
uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5.0.0
Expand Down
88 changes: 88 additions & 0 deletions apps/web/src/__tests__/word-diff.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,88 @@
import { describe, it, expect } from 'vitest';
import { computeWordDiff, tokenizeWords } from '../lib/word-diff';

describe('word-diff (Myers algorithm)', () => {
it('tokenizes words, whitespace, and punctuation', () => {
const tokens = tokenizeWords('Sec. 101(a) -- Definitions;');
expect(tokens).toEqual(['Sec', '.', ' ', '101', '(', 'a', ')', ' ', '-', '-', ' ', 'Definitions', ';']);
});

it('handles identical strings', () => {
const diff = computeWordDiff('No changes made here.', 'No changes made here.');
expect(diff).toEqual([{ type: 'context', text: 'No changes made here.' }]);
});

it('handles empty strings', () => {
expect(computeWordDiff('', '')).toEqual([]);
expect(computeWordDiff('', 'new text')).toEqual([
{ type: 'add', text: 'new' },
{ type: 'add', text: ' ' },
{ type: 'add', text: 'text' },
]);
expect(computeWordDiff('old text', '')).toEqual([
{ type: 'del', text: 'old' },
{ type: 'del', text: ' ' },
{ type: 'del', text: 'text' },
]);
});

it('accurately identifies word substitutions', () => {
const oldText = 'The Attorney General shall report annually to Congress.';
const newText = 'The Attorney General shall report quarterly to Congress.';

const diff = computeWordDiff(oldText, newText);

const oldReconstructed = diff.filter((t) => t.type !== 'add').map((t) => t.text).join('');
const newReconstructed = diff.filter((t) => t.type !== 'del').map((t) => t.text).join('');

expect(oldReconstructed).toBe(oldText);
expect(newReconstructed).toBe(newText);

expect(diff.some((t) => t.type === 'del' && t.text === 'annually')).toBe(true);
expect(diff.some((t) => t.type === 'add' && t.text === 'quarterly')).toBe(true);
});

it('handles legislative amendments accurately', () => {
const oldText =
'Section 101. The Commission consists of 5 members appointed by the President by and with the advice and consent of the Senate.';
const newText =
'Section 101. The Commission consists of 7 members appointed by the President, with the advice and consent of the Senate, for terms of six years.';

const diff = computeWordDiff(oldText, newText);

const oldReconstructed = diff.filter((t) => t.type !== 'add').map((t) => t.text).join('');
const newReconstructed = diff.filter((t) => t.type !== 'del').map((t) => t.text).join('');

expect(oldReconstructed).toBe(oldText);
expect(newReconstructed).toBe(newText);
});

it('handles large inputs without UI lag', () => {
const base = 'The Secretary of the Treasury shall issue regulations governing financial instruments under this title. '.repeat(40);
const modified = 'The Secretary of the Treasury, in consultation with the Board of Governors, shall issue regulations governing designated financial instruments under this title. '.repeat(40);

const start = performance.now();
const diff = computeWordDiff(base, modified);
const duration = performance.now() - start;

expect(duration).toBeLessThan(100); // Myers SES with prefix/suffix strip is extremely fast (<100ms)

const oldReconstructed = diff.filter((t) => t.type !== 'add').map((t) => t.text).join('');
const newReconstructed = diff.filter((t) => t.type !== 'del').map((t) => t.text).join('');

expect(oldReconstructed).toBe(base);
expect(newReconstructed).toBe(modified);
});

it('gracefully triggers safety fallback cap on gigantic differences', () => {
const oldText = 'a '.repeat(1000);
const newText = 'b '.repeat(1000);

const diff = computeWordDiff(oldText, newText);

// Should complete cleanly without crashing or hanging
expect(diff.length).toBeGreaterThan(0);
expect(diff.some((t) => t.type === 'del')).toBe(true);
expect(diff.some((t) => t.type === 'add')).toBe(true);
});
});
48 changes: 1 addition & 47 deletions apps/web/src/components/DiffViewer.svelte
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@
isRateLimited, formatTagName, extractYear,
type CommitInfo, type DiffLine, type ReleaseTag,
} from "../lib/github";
import { computeWordDiff, type WordToken } from "../lib/word-diff";

interface SectionDiff {
from: string;
Expand All @@ -24,10 +25,6 @@
changedCount: number;
}

interface WordToken {
type: "context" | "add" | "del";
text: string;
}

interface RedlineRow {
type: "paired" | "del" | "add" | "context";
Expand Down Expand Up @@ -71,49 +68,6 @@
let diffMode = $state<"redline" | "split" | "raw">("redline");
const HISTORY_PREVIEW_COUNT = 10;

/** Simple word/punctuation tokenizer */
function tokenizeWords(text: string): string[] {
return text.match(/\w+|\s+|[^\w\s]/g) || [text];
}

/** Longest Common Subsequence word-level diff */
function computeWordDiff(oldText: string, newText: string): WordToken[] {
const a = tokenizeWords(oldText);
const b = tokenizeWords(newText);
const m = a.length;
const n = b.length;

// LCS table
const dp: number[][] = Array.from({ length: m + 1 }, () => new Array(n + 1).fill(0));
for (let i = 0; i < m; i++) {
for (let j = 0; j < n; j++) {
if (a[i] === b[j]) {
dp[i + 1]![j + 1] = dp[i]![j]! + 1;
} else {
dp[i + 1]![j + 1] = Math.max(dp[i + 1]![j]!, dp[i]![j + 1]!);
}
}
}

// Backtrack to build tokens
const tokens: WordToken[] = [];
let i = m;
let j = n;
while (i > 0 || j > 0) {
if (i > 0 && j > 0 && a[i - 1] === b[j - 1]) {
tokens.unshift({ type: "context", text: a[i - 1]! });
i--;
j--;
} else if (j > 0 && (i === 0 || dp[i]![j - 1]! >= dp[i - 1]![j]!)) {
tokens.unshift({ type: "add", text: b[j - 1]! });
j--;
} else if (i > 0 && (j === 0 || dp[i]![j - 1]! < dp[i - 1]![j]!)) {
tokens.unshift({ type: "del", text: a[i - 1]! });
i--;
}
}
return tokens;
}

/** Build redline rows by pairing consecutive del and add lines */
let redlineRows = $derived.by<RedlineRow[]>(() => {
Expand Down
185 changes: 185 additions & 0 deletions apps/web/src/lib/word-diff.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,185 @@
/**
* Myers' diff algorithm for word-level intraline redline comparison.
*
* Implements Eugene W. Myers' $O(ND)$ Shortest Edit Script (SES) algorithm
* with $O(N)$ common prefix/suffix optimization and safety threshold fallback.
*/

export interface WordToken {
type: 'context' | 'add' | 'del';
text: string;
}

/**
* Tokenize text into words, whitespace sequences, and individual punctuation marks.
*/
export function tokenizeWords(text: string): string[] {
if (!text) return [];
return text.match(/\w+|\s+|[^\w\s]/g) || [];
}

/**
* Maximum product of tokens (N * M) or token count before falling back to
* block-level diffing to guarantee the UI thread is never locked.
*/
const MAX_MYERS_CELLS = 250_000;
const MAX_TOTAL_TOKENS = 1_500;

/**
* Compute word-level diff between two text strings using Myers' algorithm.
*/
export function computeWordDiff(oldText: string, newText: string): WordToken[] {
if (oldText === newText) {
return oldText.length > 0 ? [{ type: 'context', text: oldText }] : [];
}

const a = tokenizeWords(oldText);
const b = tokenizeWords(newText);
const n = a.length;
const m = b.length;

if (n === 0) {
return b.map((text) => ({ type: 'add' as const, text }));
}
if (m === 0) {
return a.map((text) => ({ type: 'del' as const, text }));
}

// 1. Fast common prefix strip: O(min(N, M))
let prefix = 0;
while (prefix < n && prefix < m && a[prefix] === b[prefix]) {
prefix++;
}

// 2. Fast common suffix strip: O(min(N, M))
let suffix = 0;
while (
suffix < n - prefix &&
suffix < m - prefix &&
a[n - 1 - suffix] === b[m - 1 - suffix]
) {
suffix++;
}

const result: WordToken[] = [];
for (let i = 0; i < prefix; i++) {
result.push({ type: 'context', text: a[i]! });
}

const midA = a.slice(prefix, n - suffix);
const midB = b.slice(prefix, m - suffix);
const midN = midA.length;
const midM = midB.length;

if (midN === 0 && midM === 0) {
// Exact match apart from prefix/suffix
} else if (midN === 0) {
for (const text of midB) {
result.push({ type: 'add', text });
}
} else if (midM === 0) {
for (const text of midA) {
result.push({ type: 'del', text });
}
} else if (midN * midM > MAX_MYERS_CELLS || midN + midM > MAX_TOTAL_TOKENS) {
// Safety cap fallback for extremely massive structural diffs
for (const text of midA) result.push({ type: 'del', text });
for (const text of midB) result.push({ type: 'add', text });
} else {
// 3. Myers' SES on the remaining divergent middle slice
const max = midN + midM;
const vOffset = max;
const v = new Int32Array(2 * max + 1);
v[vOffset + 1] = 0;
const trace: Int32Array[] = [];

let reached = false;
for (let d = 0; d <= max; d++) {
trace.push(new Int32Array(v));
for (let k = -d; k <= d; k += 2) {
let x: number;
if (k === -d || (k !== d && v[vOffset + k - 1]! < v[vOffset + k + 1]!)) {
x = v[vOffset + k + 1]!; // insertion (down)
} else {
x = v[vOffset + k - 1]! + 1; // deletion (right)
}
let y = x - k;

while (x < midN && y < midM && midA[x] === midB[y]) {
x++;
y++;
}
v[vOffset + k] = x;

if (x >= midN && y >= midM) {
reached = true;
break;
}
}
if (reached) break;
}

// 4. Backtrack through trace history to reconstruct edit operations
let currX = midN;
let currY = midM;
const midTokens: WordToken[] = [];

for (let d = trace.length - 1; d > 0; d--) {
const prevV = trace[d]!;
const k = currX - currY;

let prevK: number;
if (k === -d || (k !== d && prevV[vOffset + k - 1]! < prevV[vOffset + k + 1]!)) {
prevK = k + 1;
} else {
prevK = k - 1;
}

const prevX = prevV[vOffset + prevK]!;
const prevY = prevX - prevK;

// Diagonal snake items (context matches)
while (currX > prevX && currY > prevY) {
currX--;
currY--;
midTokens.unshift({ type: 'context', text: midA[currX]! });
}

if (d > 0) {
if (currX === prevX) {
// Insertion
currY--;
midTokens.unshift({ type: 'add', text: midB[currY]! });
} else {
// Deletion
currX--;
midTokens.unshift({ type: 'del', text: midA[currX]! });
}
}
}

// Any remaining leading items
while (currX > 0 && currY > 0) {
currX--;
currY--;
midTokens.unshift({ type: 'context', text: midA[currX]! });
}
while (currX > 0) {
currX--;
midTokens.unshift({ type: 'del', text: midA[currX]! });
}
while (currY > 0) {
currY--;
midTokens.unshift({ type: 'add', text: midB[currY]! });
}

result.push(...midTokens);
}

// 5. Append common suffix
for (let i = n - suffix; i < n; i++) {
result.push({ type: 'context', text: a[i]! });
}

return result;
}
Loading