diff --git a/src/app/shared/execution-compare/execution-compare-diff.spec.ts b/src/app/shared/execution-compare/execution-compare-diff.spec.ts new file mode 100644 index 0000000..24a282c --- /dev/null +++ b/src/app/shared/execution-compare/execution-compare-diff.spec.ts @@ -0,0 +1,67 @@ +import { diffSide, diffWords } from './execution-compare-diff'; + +/** Compact rendering of a diff, so the assertions read like the thing they describe. */ +function render(parts: ReturnType): string { + return parts.map((part) => + part.kind === 'same' ? part.text : `[${part.kind === 'added' ? '+' : '-'}${part.text}]` + ).join(''); +} + +describe('diffWords', () => { + it('returns a single unchanged part for identical text', () => { + expect(diffWords('same text', 'same text')).toEqual([{ kind: 'same', text: 'same text' }]); + }); + + it('returns nothing for two empty values', () => { + expect(diffWords('', '')).toEqual([]); + }); + + it('marks an inserted word', () => { + expect(render(diffWords('the candidate is strong', 'the candidate is very strong'))) + .toBe('the candidate is [+very ]strong'); + }); + + it('marks a removed word', () => { + expect(render(diffWords('the very strong candidate', 'the strong candidate'))) + .toBe('the [-very ]strong candidate'); + }); + + it('marks a substitution as a removal and an addition', () => { + const rendered = render(diffWords('Score 7/10', 'Score 5/10')); + expect(rendered).toContain('[-7/10]'); + expect(rendered).toContain('[+5/10]'); + }); + + it('treats a value that appeared from nothing as entirely added', () => { + expect(render(diffWords('', 'brand new'))).toBe('[+brand new]'); + }); + + it('rebuilds each side exactly from its own parts', () => { + // The parts carry whitespace, so a view can render them without re-joining and drifting. + const left = 'the candidate is strong'; + const right = 'the candidate is very strong'; + const parts = diffWords(left, right); + + expect(diffSide(parts, 'left').map((part) => part.text).join('')).toBe(left); + expect(diffSide(parts, 'right').map((part) => part.text).join('')).toBe(right); + }); + + it('merges neighbouring words of the same kind into one part', () => { + const parts = diffWords('a b c', 'x y z'); + // Not one part per word: a view would otherwise render a span for every token. + expect(parts.filter((part) => part.kind === 'removed')).toHaveLength(1); + expect(parts.filter((part) => part.kind === 'added')).toHaveLength(1); + }); + + it('falls back to whole-value replacement past the size guard', () => { + // Two long outputs are exactly what this is for, and a quadratic table on them would freeze + // the view it exists to render. + const left = Array.from({ length: 2100 }, (_, i) => `w${i}`).join(' '); + const right = Array.from({ length: 2100 }, (_, i) => `x${i}`).join(' '); + + const parts = diffWords(left, right); + + expect(parts.map((part) => part.kind)).toEqual(['removed', 'added']); + expect(diffSide(parts, 'left')[0].text).toBe(left); + }); +}); diff --git a/src/app/shared/execution-compare/execution-compare-diff.ts b/src/app/shared/execution-compare/execution-compare-diff.ts new file mode 100644 index 0000000..9975920 --- /dev/null +++ b/src/app/shared/execution-compare/execution-compare-diff.ts @@ -0,0 +1,105 @@ +/** + * A word-level diff for the two sides of a comparison. + * + * Written here rather than pulled in: no diff library is in the project, the initial bundle is + * already over its budget, and what is needed is small. Side by side without it, two paragraphs of + * model output that differ in one clause have to be read twice to find the clause. + * + * Word-level rather than character-level because the values are prose: a character diff on a + * rewritten sentence produces confetti, not an explanation. + */ +export type DiffPartKind = 'same' | 'added' | 'removed'; + +export type DiffPart = { + kind: DiffPartKind; + text: string; +}; + +/** + * Splits into words, each carrying its trailing whitespace, so the text can be rebuilt exactly. + * + * The whitespace rides along deliberately. As separate tokens the spaces match between any two + * texts, so a wholly rewritten sentence came back as alternating removed-word / kept-space parts - + * confetti, which is what word-level diffing is supposed to avoid. + */ +function tokenize(text: string): string[] { + return text.match(/\S+\s*|\s+/g) ?? []; +} + +/** + * Longest common subsequence over the token lists, as a DP table. + * + * Quadratic in tokens. Guarded below, because two long model outputs are exactly the case this + * exists for and a runaway table would freeze the view it is meant to render. + */ +const MAX_TOKENS = 2000; + +export function diffWords(left: string, right: string): DiffPart[] { + if (left === right) { + return left.length ? [{ kind: 'same', text: left }] : []; + } + + const leftTokens = tokenize(left); + const rightTokens = tokenize(right); + + // Beyond this the table costs more than the reading it saves; whole-value replacement is honest. + if (leftTokens.length > MAX_TOKENS || rightTokens.length > MAX_TOKENS) { + return compact([ + { kind: 'removed', text: left }, + { kind: 'added', text: right } + ]); + } + + const rows = leftTokens.length; + const columns = rightTokens.length; + const lengths: number[][] = Array.from({ length: rows + 1 }, () => new Array(columns + 1).fill(0)); + for (let i = rows - 1; i >= 0; i--) { + for (let j = columns - 1; j >= 0; j--) { + lengths[i][j] = leftTokens[i] === rightTokens[j] + ? lengths[i + 1][j + 1] + 1 + : Math.max(lengths[i + 1][j], lengths[i][j + 1]); + } + } + + const parts: DiffPart[] = []; + let i = 0; + let j = 0; + while (i < rows && j < columns) { + if (leftTokens[i] === rightTokens[j]) { + parts.push({ kind: 'same', text: leftTokens[i] }); + i++; + j++; + } else if (lengths[i + 1][j] >= lengths[i][j + 1]) { + parts.push({ kind: 'removed', text: leftTokens[i] }); + i++; + } else { + parts.push({ kind: 'added', text: rightTokens[j] }); + j++; + } + } + while (i < rows) parts.push({ kind: 'removed', text: leftTokens[i++] }); + while (j < columns) parts.push({ kind: 'added', text: rightTokens[j++] }); + + return compact(parts); +} + +/** Merges neighbouring parts of the same kind, so the view renders spans and not one per word. */ +function compact(parts: DiffPart[]): DiffPart[] { + const merged: DiffPart[] = []; + for (const part of parts) { + if (!part.text.length) continue; + const last = merged[merged.length - 1]; + if (last && last.kind === part.kind) { + last.text += part.text; + continue; + } + merged.push({ ...part }); + } + return merged; +} + +/** The side of a diff a reader is looking at: removals belong to the left, additions to the right. */ +export function diffSide(parts: DiffPart[], side: 'left' | 'right'): DiffPart[] { + const dropped: DiffPartKind = side === 'left' ? 'added' : 'removed'; + return parts.filter((part) => part.kind !== dropped); +}