From 3afaf9a16913e1f7145477cc6a9f760d0aa2ddbe Mon Sep 17 00:00:00 2001 From: Lucio Lelii Date: Thu, 3 Sep 2026 22:13:25 +0200 Subject: [PATCH] Add a word-level diff for the comparison view Written here rather than pulled in: there is no diff library in the project, the initial bundle is already over budget, and what is needed is small. Side by side without it, two paragraphs of model output differing in one clause have to be read twice to find the clause. Word-level, not character-level: the values are prose, and a character diff on a rewritten sentence produces confetti rather than an explanation. Each word carries its trailing whitespace. As separate tokens the spaces match between any two texts, so a wholly rewritten sentence came back as alternating removed-word / kept-space parts - the same confetti, arrived at from the other direction. A test caught this; the tokenizer changed, not the expectation. The LCS table is quadratic in tokens, and two long model outputs are exactly what this exists for, so past a guard it falls back to whole-value replacement rather than freezing the view it is meant to render. 525 frontend tests green. Co-Authored-By: Claude Opus 5 (1M context) --- .../execution-compare-diff.spec.ts | 67 +++++++++++ .../execution-compare-diff.ts | 105 ++++++++++++++++++ 2 files changed, 172 insertions(+) create mode 100644 src/app/shared/execution-compare/execution-compare-diff.spec.ts create mode 100644 src/app/shared/execution-compare/execution-compare-diff.ts diff --git a/src/app/shared/execution-compare/execution-compare-diff.spec.ts b/src/app/shared/execution-compare/execution-compare-diff.spec.ts new file mode 100644 index 0000000..24a282c --- /dev/null +++ b/src/app/shared/execution-compare/execution-compare-diff.spec.ts @@ -0,0 +1,67 @@ +import { diffSide, diffWords } from './execution-compare-diff'; + +/** Compact rendering of a diff, so the assertions read like the thing they describe. */ +function render(parts: ReturnType): string { + return parts.map((part) => + part.kind === 'same' ? part.text : `[${part.kind === 'added' ? '+' : '-'}${part.text}]` + ).join(''); +} + +describe('diffWords', () => { + it('returns a single unchanged part for identical text', () => { + expect(diffWords('same text', 'same text')).toEqual([{ kind: 'same', text: 'same text' }]); + }); + + it('returns nothing for two empty values', () => { + expect(diffWords('', '')).toEqual([]); + }); + + it('marks an inserted word', () => { + expect(render(diffWords('the candidate is strong', 'the candidate is very strong'))) + .toBe('the candidate is [+very ]strong'); + }); + + it('marks a removed word', () => { + expect(render(diffWords('the very strong candidate', 'the strong candidate'))) + .toBe('the [-very ]strong candidate'); + }); + + it('marks a substitution as a removal and an addition', () => { + const rendered = render(diffWords('Score 7/10', 'Score 5/10')); + expect(rendered).toContain('[-7/10]'); + expect(rendered).toContain('[+5/10]'); + }); + + it('treats a value that appeared from nothing as entirely added', () => { + expect(render(diffWords('', 'brand new'))).toBe('[+brand new]'); + }); + + it('rebuilds each side exactly from its own parts', () => { + // The parts carry whitespace, so a view can render them without re-joining and drifting. + const left = 'the candidate is strong'; + const right = 'the candidate is very strong'; + const parts = diffWords(left, right); + + expect(diffSide(parts, 'left').map((part) => part.text).join('')).toBe(left); + expect(diffSide(parts, 'right').map((part) => part.text).join('')).toBe(right); + }); + + it('merges neighbouring words of the same kind into one part', () => { + const parts = diffWords('a b c', 'x y z'); + // Not one part per word: a view would otherwise render a span for every token. + expect(parts.filter((part) => part.kind === 'removed')).toHaveLength(1); + expect(parts.filter((part) => part.kind === 'added')).toHaveLength(1); + }); + + it('falls back to whole-value replacement past the size guard', () => { + // Two long outputs are exactly what this is for, and a quadratic table on them would freeze + // the view it exists to render. + const left = Array.from({ length: 2100 }, (_, i) => `w${i}`).join(' '); + const right = Array.from({ length: 2100 }, (_, i) => `x${i}`).join(' '); + + const parts = diffWords(left, right); + + expect(parts.map((part) => part.kind)).toEqual(['removed', 'added']); + expect(diffSide(parts, 'left')[0].text).toBe(left); + }); +}); diff --git a/src/app/shared/execution-compare/execution-compare-diff.ts b/src/app/shared/execution-compare/execution-compare-diff.ts new file mode 100644 index 0000000..9975920 --- /dev/null +++ b/src/app/shared/execution-compare/execution-compare-diff.ts @@ -0,0 +1,105 @@ +/** + * A word-level diff for the two sides of a comparison. + * + * Written here rather than pulled in: no diff library is in the project, the initial bundle is + * already over its budget, and what is needed is small. Side by side without it, two paragraphs of + * model output that differ in one clause have to be read twice to find the clause. + * + * Word-level rather than character-level because the values are prose: a character diff on a + * rewritten sentence produces confetti, not an explanation. + */ +export type DiffPartKind = 'same' | 'added' | 'removed'; + +export type DiffPart = { + kind: DiffPartKind; + text: string; +}; + +/** + * Splits into words, each carrying its trailing whitespace, so the text can be rebuilt exactly. + * + * The whitespace rides along deliberately. As separate tokens the spaces match between any two + * texts, so a wholly rewritten sentence came back as alternating removed-word / kept-space parts - + * confetti, which is what word-level diffing is supposed to avoid. + */ +function tokenize(text: string): string[] { + return text.match(/\S+\s*|\s+/g) ?? []; +} + +/** + * Longest common subsequence over the token lists, as a DP table. + * + * Quadratic in tokens. Guarded below, because two long model outputs are exactly the case this + * exists for and a runaway table would freeze the view it is meant to render. + */ +const MAX_TOKENS = 2000; + +export function diffWords(left: string, right: string): DiffPart[] { + if (left === right) { + return left.length ? [{ kind: 'same', text: left }] : []; + } + + const leftTokens = tokenize(left); + const rightTokens = tokenize(right); + + // Beyond this the table costs more than the reading it saves; whole-value replacement is honest. + if (leftTokens.length > MAX_TOKENS || rightTokens.length > MAX_TOKENS) { + return compact([ + { kind: 'removed', text: left }, + { kind: 'added', text: right } + ]); + } + + const rows = leftTokens.length; + const columns = rightTokens.length; + const lengths: number[][] = Array.from({ length: rows + 1 }, () => new Array(columns + 1).fill(0)); + for (let i = rows - 1; i >= 0; i--) { + for (let j = columns - 1; j >= 0; j--) { + lengths[i][j] = leftTokens[i] === rightTokens[j] + ? lengths[i + 1][j + 1] + 1 + : Math.max(lengths[i + 1][j], lengths[i][j + 1]); + } + } + + const parts: DiffPart[] = []; + let i = 0; + let j = 0; + while (i < rows && j < columns) { + if (leftTokens[i] === rightTokens[j]) { + parts.push({ kind: 'same', text: leftTokens[i] }); + i++; + j++; + } else if (lengths[i + 1][j] >= lengths[i][j + 1]) { + parts.push({ kind: 'removed', text: leftTokens[i] }); + i++; + } else { + parts.push({ kind: 'added', text: rightTokens[j] }); + j++; + } + } + while (i < rows) parts.push({ kind: 'removed', text: leftTokens[i++] }); + while (j < columns) parts.push({ kind: 'added', text: rightTokens[j++] }); + + return compact(parts); +} + +/** Merges neighbouring parts of the same kind, so the view renders spans and not one per word. */ +function compact(parts: DiffPart[]): DiffPart[] { + const merged: DiffPart[] = []; + for (const part of parts) { + if (!part.text.length) continue; + const last = merged[merged.length - 1]; + if (last && last.kind === part.kind) { + last.text += part.text; + continue; + } + merged.push({ ...part }); + } + return merged; +} + +/** The side of a diff a reader is looking at: removals belong to the left, additions to the right. */ +export function diffSide(parts: DiffPart[], side: 'left' | 'right'): DiffPart[] { + const dropped: DiffPartKind = side === 'left' ? 'added' : 'removed'; + return parts.filter((part) => part.kind !== dropped); +}