From 407097490cd8f6018e4aac906049c1640847aa17 Mon Sep 17 00:00:00 2001 From: William Zujkowski Date: Tue, 6 Oct 2026 10:01:38 -0400 Subject: [PATCH] perf(web): replace quadratic LCS with Myers diff and fix deploy workflow - Replace O(M*N) 2D matrix LCS in DiffViewer with Myers' O(ND) algorithm in apps/web/src/lib/word-diff.ts - Add prefix/suffix trimming and max-cells fallback cap to prevent UI locking on large statutory diffs - Add comprehensive vitest suite in apps/web/src/__tests__/word-diff.test.ts - Build workspace package dependencies in deploy-site workflow before web build - Resolves #280 --- .github/workflows/deploy-site.yml | 4 +- apps/web/src/__tests__/word-diff.test.ts | 88 ++++++++++ apps/web/src/components/DiffViewer.svelte | 48 +----- apps/web/src/lib/word-diff.ts | 185 ++++++++++++++++++++++ 4 files changed, 277 insertions(+), 48 deletions(-) create mode 100644 apps/web/src/__tests__/word-diff.test.ts create mode 100644 apps/web/src/lib/word-diff.ts diff --git a/.github/workflows/deploy-site.yml b/.github/workflows/deploy-site.yml index c56aad2..d8678b1 100644 --- a/.github/workflows/deploy-site.yml +++ b/.github/workflows/deploy-site.yml @@ -7,6 +7,8 @@ on: branches: [main] paths: - 'apps/web/**' + - 'packages/**' + - 'scripts/**' env: FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true @@ -62,7 +64,7 @@ jobs: run: npx tsx scripts/generate-diffs.ts --repo apps/web/content-data --output apps/web/public/diffs/ - name: Build site - run: pnpm --filter @civic-source/web build + run: pnpm --filter @civic-source/web... build - name: Upload artifact uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5.0.0 diff --git a/apps/web/src/__tests__/word-diff.test.ts b/apps/web/src/__tests__/word-diff.test.ts new file mode 100644 index 0000000..117f562 --- /dev/null +++ b/apps/web/src/__tests__/word-diff.test.ts @@ -0,0 +1,88 @@ +import { describe, it, expect } from 'vitest'; +import { computeWordDiff, tokenizeWords } from '../lib/word-diff'; + +describe('word-diff (Myers algorithm)', () => { + it('tokenizes words, whitespace, and punctuation', () => { + const tokens = tokenizeWords('Sec. 101(a) -- Definitions;'); + expect(tokens).toEqual(['Sec', '.', ' ', '101', '(', 'a', ')', ' ', '-', '-', ' ', 'Definitions', ';']); + }); + + it('handles identical strings', () => { + const diff = computeWordDiff('No changes made here.', 'No changes made here.'); + expect(diff).toEqual([{ type: 'context', text: 'No changes made here.' }]); + }); + + it('handles empty strings', () => { + expect(computeWordDiff('', '')).toEqual([]); + expect(computeWordDiff('', 'new text')).toEqual([ + { type: 'add', text: 'new' }, + { type: 'add', text: ' ' }, + { type: 'add', text: 'text' }, + ]); + expect(computeWordDiff('old text', '')).toEqual([ + { type: 'del', text: 'old' }, + { type: 'del', text: ' ' }, + { type: 'del', text: 'text' }, + ]); + }); + + it('accurately identifies word substitutions', () => { + const oldText = 'The Attorney General shall report annually to Congress.'; + const newText = 'The Attorney General shall report quarterly to Congress.'; + + const diff = computeWordDiff(oldText, newText); + + const oldReconstructed = diff.filter((t) => t.type !== 'add').map((t) => t.text).join(''); + const newReconstructed = diff.filter((t) => t.type !== 'del').map((t) => t.text).join(''); + + expect(oldReconstructed).toBe(oldText); + expect(newReconstructed).toBe(newText); + + expect(diff.some((t) => t.type === 'del' && t.text === 'annually')).toBe(true); + expect(diff.some((t) => t.type === 'add' && t.text === 'quarterly')).toBe(true); + }); + + it('handles legislative amendments accurately', () => { + const oldText = + 'Section 101. The Commission consists of 5 members appointed by the President by and with the advice and consent of the Senate.'; + const newText = + 'Section 101. The Commission consists of 7 members appointed by the President, with the advice and consent of the Senate, for terms of six years.'; + + const diff = computeWordDiff(oldText, newText); + + const oldReconstructed = diff.filter((t) => t.type !== 'add').map((t) => t.text).join(''); + const newReconstructed = diff.filter((t) => t.type !== 'del').map((t) => t.text).join(''); + + expect(oldReconstructed).toBe(oldText); + expect(newReconstructed).toBe(newText); + }); + + it('handles large inputs without UI lag', () => { + const base = 'The Secretary of the Treasury shall issue regulations governing financial instruments under this title. '.repeat(40); + const modified = 'The Secretary of the Treasury, in consultation with the Board of Governors, shall issue regulations governing designated financial instruments under this title. '.repeat(40); + + const start = performance.now(); + const diff = computeWordDiff(base, modified); + const duration = performance.now() - start; + + expect(duration).toBeLessThan(100); // Myers SES with prefix/suffix strip is extremely fast (<100ms) + + const oldReconstructed = diff.filter((t) => t.type !== 'add').map((t) => t.text).join(''); + const newReconstructed = diff.filter((t) => t.type !== 'del').map((t) => t.text).join(''); + + expect(oldReconstructed).toBe(base); + expect(newReconstructed).toBe(modified); + }); + + it('gracefully triggers safety fallback cap on gigantic differences', () => { + const oldText = 'a '.repeat(1000); + const newText = 'b '.repeat(1000); + + const diff = computeWordDiff(oldText, newText); + + // Should complete cleanly without crashing or hanging + expect(diff.length).toBeGreaterThan(0); + expect(diff.some((t) => t.type === 'del')).toBe(true); + expect(diff.some((t) => t.type === 'add')).toBe(true); + }); +}); diff --git a/apps/web/src/components/DiffViewer.svelte b/apps/web/src/components/DiffViewer.svelte index c1577fc..c2f01b5 100644 --- a/apps/web/src/components/DiffViewer.svelte +++ b/apps/web/src/components/DiffViewer.svelte @@ -4,6 +4,7 @@ isRateLimited, formatTagName, extractYear, type CommitInfo, type DiffLine, type ReleaseTag, } from "../lib/github"; + import { computeWordDiff, type WordToken } from "../lib/word-diff"; interface SectionDiff { from: string; @@ -24,10 +25,6 @@ changedCount: number; } - interface WordToken { - type: "context" | "add" | "del"; - text: string; - } interface RedlineRow { type: "paired" | "del" | "add" | "context"; @@ -71,49 +68,6 @@ let diffMode = $state<"redline" | "split" | "raw">("redline"); const HISTORY_PREVIEW_COUNT = 10; - /** Simple word/punctuation tokenizer */ - function tokenizeWords(text: string): string[] { - return text.match(/\w+|\s+|[^\w\s]/g) || [text]; - } - - /** Longest Common Subsequence word-level diff */ - function computeWordDiff(oldText: string, newText: string): WordToken[] { - const a = tokenizeWords(oldText); - const b = tokenizeWords(newText); - const m = a.length; - const n = b.length; - - // LCS table - const dp: number[][] = Array.from({ length: m + 1 }, () => new Array(n + 1).fill(0)); - for (let i = 0; i < m; i++) { - for (let j = 0; j < n; j++) { - if (a[i] === b[j]) { - dp[i + 1]![j + 1] = dp[i]![j]! + 1; - } else { - dp[i + 1]![j + 1] = Math.max(dp[i + 1]![j]!, dp[i]![j + 1]!); - } - } - } - - // Backtrack to build tokens - const tokens: WordToken[] = []; - let i = m; - let j = n; - while (i > 0 || j > 0) { - if (i > 0 && j > 0 && a[i - 1] === b[j - 1]) { - tokens.unshift({ type: "context", text: a[i - 1]! }); - i--; - j--; - } else if (j > 0 && (i === 0 || dp[i]![j - 1]! >= dp[i - 1]![j]!)) { - tokens.unshift({ type: "add", text: b[j - 1]! }); - j--; - } else if (i > 0 && (j === 0 || dp[i]![j - 1]! < dp[i - 1]![j]!)) { - tokens.unshift({ type: "del", text: a[i - 1]! }); - i--; - } - } - return tokens; - } /** Build redline rows by pairing consecutive del and add lines */ let redlineRows = $derived.by(() => { diff --git a/apps/web/src/lib/word-diff.ts b/apps/web/src/lib/word-diff.ts new file mode 100644 index 0000000..26ce935 --- /dev/null +++ b/apps/web/src/lib/word-diff.ts @@ -0,0 +1,185 @@ +/** + * Myers' diff algorithm for word-level intraline redline comparison. + * + * Implements Eugene W. Myers' $O(ND)$ Shortest Edit Script (SES) algorithm + * with $O(N)$ common prefix/suffix optimization and safety threshold fallback. + */ + +export interface WordToken { + type: 'context' | 'add' | 'del'; + text: string; +} + +/** + * Tokenize text into words, whitespace sequences, and individual punctuation marks. + */ +export function tokenizeWords(text: string): string[] { + if (!text) return []; + return text.match(/\w+|\s+|[^\w\s]/g) || []; +} + +/** + * Maximum product of tokens (N * M) or token count before falling back to + * block-level diffing to guarantee the UI thread is never locked. + */ +const MAX_MYERS_CELLS = 250_000; +const MAX_TOTAL_TOKENS = 1_500; + +/** + * Compute word-level diff between two text strings using Myers' algorithm. + */ +export function computeWordDiff(oldText: string, newText: string): WordToken[] { + if (oldText === newText) { + return oldText.length > 0 ? [{ type: 'context', text: oldText }] : []; + } + + const a = tokenizeWords(oldText); + const b = tokenizeWords(newText); + const n = a.length; + const m = b.length; + + if (n === 0) { + return b.map((text) => ({ type: 'add' as const, text })); + } + if (m === 0) { + return a.map((text) => ({ type: 'del' as const, text })); + } + + // 1. Fast common prefix strip: O(min(N, M)) + let prefix = 0; + while (prefix < n && prefix < m && a[prefix] === b[prefix]) { + prefix++; + } + + // 2. Fast common suffix strip: O(min(N, M)) + let suffix = 0; + while ( + suffix < n - prefix && + suffix < m - prefix && + a[n - 1 - suffix] === b[m - 1 - suffix] + ) { + suffix++; + } + + const result: WordToken[] = []; + for (let i = 0; i < prefix; i++) { + result.push({ type: 'context', text: a[i]! }); + } + + const midA = a.slice(prefix, n - suffix); + const midB = b.slice(prefix, m - suffix); + const midN = midA.length; + const midM = midB.length; + + if (midN === 0 && midM === 0) { + // Exact match apart from prefix/suffix + } else if (midN === 0) { + for (const text of midB) { + result.push({ type: 'add', text }); + } + } else if (midM === 0) { + for (const text of midA) { + result.push({ type: 'del', text }); + } + } else if (midN * midM > MAX_MYERS_CELLS || midN + midM > MAX_TOTAL_TOKENS) { + // Safety cap fallback for extremely massive structural diffs + for (const text of midA) result.push({ type: 'del', text }); + for (const text of midB) result.push({ type: 'add', text }); + } else { + // 3. Myers' SES on the remaining divergent middle slice + const max = midN + midM; + const vOffset = max; + const v = new Int32Array(2 * max + 1); + v[vOffset + 1] = 0; + const trace: Int32Array[] = []; + + let reached = false; + for (let d = 0; d <= max; d++) { + trace.push(new Int32Array(v)); + for (let k = -d; k <= d; k += 2) { + let x: number; + if (k === -d || (k !== d && v[vOffset + k - 1]! < v[vOffset + k + 1]!)) { + x = v[vOffset + k + 1]!; // insertion (down) + } else { + x = v[vOffset + k - 1]! + 1; // deletion (right) + } + let y = x - k; + + while (x < midN && y < midM && midA[x] === midB[y]) { + x++; + y++; + } + v[vOffset + k] = x; + + if (x >= midN && y >= midM) { + reached = true; + break; + } + } + if (reached) break; + } + + // 4. Backtrack through trace history to reconstruct edit operations + let currX = midN; + let currY = midM; + const midTokens: WordToken[] = []; + + for (let d = trace.length - 1; d > 0; d--) { + const prevV = trace[d]!; + const k = currX - currY; + + let prevK: number; + if (k === -d || (k !== d && prevV[vOffset + k - 1]! < prevV[vOffset + k + 1]!)) { + prevK = k + 1; + } else { + prevK = k - 1; + } + + const prevX = prevV[vOffset + prevK]!; + const prevY = prevX - prevK; + + // Diagonal snake items (context matches) + while (currX > prevX && currY > prevY) { + currX--; + currY--; + midTokens.unshift({ type: 'context', text: midA[currX]! }); + } + + if (d > 0) { + if (currX === prevX) { + // Insertion + currY--; + midTokens.unshift({ type: 'add', text: midB[currY]! }); + } else { + // Deletion + currX--; + midTokens.unshift({ type: 'del', text: midA[currX]! }); + } + } + } + + // Any remaining leading items + while (currX > 0 && currY > 0) { + currX--; + currY--; + midTokens.unshift({ type: 'context', text: midA[currX]! }); + } + while (currX > 0) { + currX--; + midTokens.unshift({ type: 'del', text: midA[currX]! }); + } + while (currY > 0) { + currY--; + midTokens.unshift({ type: 'add', text: midB[currY]! }); + } + + result.push(...midTokens); + } + + // 5. Append common suffix + for (let i = n - suffix; i < n; i++) { + result.push({ type: 'context', text: a[i]! }); + } + + return result; +}