From 381e3d381705f4034f6b8d0e446d9305f4d30328 Mon Sep 17 00:00:00 2001 From: Lucio Lelii Date: Thu, 3 Sep 2026 22:12:11 +0200 Subject: [PATCH] Join two runs of a flow node by node The first half of comparing two executions: a pure function, so the semantics are testable without a TestBed - the same shape as flow-grouping and planInputSaves. It is not the bias comparison. That one scopes itself to the nodes a probe was activated on and refuses a run that is not an experiment, so for two ordinary runs its node set is empty and it compares nothing. Generalising it would mean rewriting its scoping, not relaxing a condition. The join key is the step id, stable across runs of one flow. Not the node name: names are not unique and can be edited, and a rename would report every node as replaced. Two runs sharing no step at all are reported as disjoint - almost always the flow was edited between them, which makes the join meaningless rather than merely empty. resolveStepOutputs is the part that makes this work on a real flow. context.result holds only the *unconnected* outputs, so on a flow whose nodes feed one another it is nearly empty, and on one ending in an End node it is empty outright - the run this was built for has result {} and one outcome. The value of a connected output is still observable as what arrived at the input on the other end of the wire, which is how the backend reconstructs node outputs for a bias report. Outcomes are compared too, for the same reason. 516 frontend tests green. Co-Authored-By: Claude Opus 5 (1M context) --- .../execution-compare.spec.ts | 194 ++++++++++++++++++ .../execution-compare/execution-compare.ts | 175 ++++++++++++++++ .../execution-viewer.utils.ts | 35 ++++ 3 files changed, 404 insertions(+) create mode 100644 src/app/shared/execution-compare/execution-compare.spec.ts create mode 100644 src/app/shared/execution-compare/execution-compare.ts diff --git a/src/app/shared/execution-compare/execution-compare.spec.ts b/src/app/shared/execution-compare/execution-compare.spec.ts new file mode 100644 index 0000000..e128e41 --- /dev/null +++ b/src/app/shared/execution-compare/execution-compare.spec.ts @@ -0,0 +1,194 @@ +import { TaskExecution } from '@models/task-execution'; +import { compareExecutions } from './execution-compare'; + +type StepSpec = { + id: string; + name?: string; + status?: string; + inputs?: Record; + outputs?: string[]; +}; + +function execution(id: string, steps: StepSpec[], overrides: Partial = {}): TaskExecution { + return { + id, + name: id, + creationTime: 1, + context: { + inputs: {}, + result: {}, + errors: {}, + warnings: {}, + waitingSteps: [], + status: 'SUCCESS', + steps: Object.fromEntries(steps.map((step) => [step.id, { + id: step.id, + status: step.status ?? 'COMPLETED', + simulated: false, + node: { + id: step.id, + name: step.name ?? step.id, + nodeFamily: 'block' as const, + typeName: 'LLMBlock', + inputs: Object.keys(step.inputs ?? {}).map((name) => ({ name, type: 'TEXT' })), + outputs: (step.outputs ?? []).map((name) => ({ name, type: 'TEXT' })), + specificConfiguration: {} + }, + inputs: Object.entries(step.inputs ?? {}).map(([name, value]) => ({ + descriptor: { name, type: 'TEXT' }, value, set: true, registered: false + })), + outputs: (step.outputs ?? []).map((name) => ({ descriptor: { name, type: 'TEXT' }, connected: false })) + }])), + ...overrides + } + } as unknown as TaskExecution; +} + +describe('compareExecutions', () => { + it('marks a node whose output differs, and leaves an identical one alone', () => { + const left = execution('run-1', [{ id: 's1', name: 'Evaluate', outputs: ['response'] }], + { result: { 's1:response': 'Score 7/10' } }); + const right = execution('run-2', [{ id: 's1', name: 'Evaluate', outputs: ['response'] }], + { result: { 's1:response': 'Score 5/10' } }); + + const { nodes, changedNodeCount } = compareExecutions(left, right); + + expect(changedNodeCount).toBe(1); + const value = nodes[0].values.find((one) => one.key === 'outputs.response'); + expect(value).toMatchObject({ left: 'Score 7/10', right: 'Score 5/10', state: 'changed' }); + }); + + it('reports an unchanged node as equal', () => { + const step = { id: 's1', outputs: ['response'] }; + const result = { 's1:response': 'same' }; + const comparison = compareExecutions( + execution('run-1', [step], { result }), + execution('run-2', [step], { result }) + ); + + expect(comparison.changedNodeCount).toBe(0); + expect(comparison.nodes[0].state).toBe('equal'); + }); + + it('reads a connected output from the input it fed, since context.result never holds it', () => { + // This is the case that matters on a real flow: every wired output is absent from + // context.result, so comparing that map alone would report almost nothing. + const wiring = { + stepConnections: [ + { id: 'c1', sourceId: 's1', sourceName: 'response', targetId: 's2', targetName: 'summary' } + ] + }; + const left = { + ...execution('run-1', [ + { id: 's1', name: 'Interview', outputs: ['response'] }, + { id: 's2', name: 'Evaluate', inputs: { summary: 'interview went well' } } + ], { inputs: { 's2:summary': 'interview went well' } }), + ...wiring + }; + const right = { + ...execution('run-2', [ + { id: 's1', name: 'Interview', outputs: ['response'] }, + { id: 's2', name: 'Evaluate', inputs: { summary: 'interview went badly' } } + ], { inputs: { 's2:summary': 'interview went badly' } }), + ...wiring + }; + + const { nodes } = compareExecutions(left as TaskExecution, right as TaskExecution); + + const interview = nodes.find((node) => node.title === 'Interview')!; + expect(interview.values.find((one) => one.key === 'outputs.response')).toMatchObject({ + left: 'interview went well', + right: 'interview went badly', + state: 'changed' + }); + }); + + it('compares the outcomes, which is where the answer lives when the flow ends on an End node', () => { + // The real run this was built for has result {} and a single outcome: omitting outcomes would + // make the comparison useless in exactly the case it was asked for. + const outcome = (payload: string) => ({ + outcomes: [{ stepId: 's9', code: 'MULTI_CV_RANKING_ACCEPTED', label: 'Final ranking', payload, timestamp: 1 }] + }); + const comparison = compareExecutions( + execution('run-1', [{ id: 's1' }], outcome('1. Alice — 7/10')), + execution('run-2', [{ id: 's1' }], outcome('1. Bob — 6/10')) + ); + + expect(comparison.outcomes).toHaveLength(1); + expect(comparison.outcomes[0]).toMatchObject({ + code: 'MULTI_CV_RANKING_ACCEPTED', + left: '1. Alice — 7/10', + right: '1. Bob — 6/10', + state: 'changed' + }); + }); + + it('flags an outcome only one run reached, even with no payload to compare', () => { + const comparison = compareExecutions( + execution('run-1', [{ id: 's1' }], { + outcomes: [{ stepId: 's9', code: 'ACCEPTED', label: 'Accepted', payload: null, timestamp: 1 }] + }), + execution('run-2', [{ id: 's1' }], { + outcomes: [{ stepId: 's9', code: 'REVIEW_REQUESTED', label: 'Review', payload: null, timestamp: 1 }] + }) + ); + + expect(comparison.outcomes.map((one) => [one.code, one.state])) + .toEqual([['ACCEPTED', 'only-left'], ['REVIEW_REQUESTED', 'only-right']]); + }); + + it('marks a node that only one run has', () => { + const comparison = compareExecutions( + execution('run-1', [{ id: 's1', name: 'Shared' }, { id: 's2', name: 'Dropped' }]), + execution('run-2', [{ id: 's1', name: 'Shared' }]) + ); + + expect(comparison.nodes.find((node) => node.title === 'Dropped')?.state).toBe('only-left'); + expect(comparison.disjoint).toBe(false); + }); + + it('reports two runs with no node in common as disjoint', () => { + // Almost always the flow was edited between the runs, which makes a per-node join meaningless + // rather than merely empty - the view has to say so instead of showing everything as replaced. + const comparison = compareExecutions( + execution('run-1', [{ id: 'old-1' }]), + execution('run-2', [{ id: 'new-1' }]) + ); + + expect(comparison.disjoint).toBe(true); + }); + + it('treats a differing step status as a change even when the values match', () => { + const comparison = compareExecutions( + execution('run-1', [{ id: 's1', status: 'COMPLETED' }]), + execution('run-2', [{ id: 's1', status: 'SKIPPED' }]) + ); + + expect(comparison.nodes[0].statusChanged).toBe(true); + expect(comparison.nodes[0].changed).toBe(true); + }); + + it('puts the changed nodes first', () => { + const comparison = compareExecutions( + execution('run-1', [ + { id: 's1', name: 'Alpha', outputs: ['out'] }, + { id: 's2', name: 'Beta', outputs: ['out'] } + ], { result: { 's1:out': 'same', 's2:out': 'left' } }), + execution('run-2', [ + { id: 's1', name: 'Alpha', outputs: ['out'] }, + { id: 's2', name: 'Beta', outputs: ['out'] } + ], { result: { 's1:out': 'same', 's2:out': 'right' } }) + ); + + expect(comparison.nodes.map((node) => node.title)).toEqual(['Beta', 'Alpha']); + }); + + it('ignores a port that is empty on both sides rather than listing it as equal', () => { + const comparison = compareExecutions( + execution('run-1', [{ id: 's1', outputs: ['unused'] }]), + execution('run-2', [{ id: 's1', outputs: ['unused'] }]) + ); + + expect(comparison.nodes[0].values).toEqual([]); + }); +}); diff --git a/src/app/shared/execution-compare/execution-compare.ts b/src/app/shared/execution-compare/execution-compare.ts new file mode 100644 index 0000000..ec931d4 --- /dev/null +++ b/src/app/shared/execution-compare/execution-compare.ts @@ -0,0 +1,175 @@ +import { TaskExecution, TaskExecutionOutcome, TaskExecutionStep } from '@models/task-execution'; +import { + getExecutionInputValues, + resolveStepOutputs, + stepTitle, + stringifyOutputValue +} from '@shared/task-execution-viewer/execution-viewer.utils'; + +/** Whether a value is the same on both sides, or present on only one of them. */ +export type ComparisonState = 'equal' | 'changed' | 'only-left' | 'only-right'; + +export type ComparedValue = { + /** `inputs` or `outputs` plus the port name, e.g. `outputs.response`. */ + key: string; + kind: 'input' | 'output'; + name: string; + left: string | null; + right: string | null; + state: ComparisonState; +}; + +export type ComparedNode = { + stepId: string; + title: string; + leftStatus: string | null; + rightStatus: string | null; + statusChanged: boolean; + values: ComparedValue[]; + /** True when anything about this node differs, including its status. */ + changed: boolean; + state: ComparisonState; +}; + +export type ComparedOutcome = { + code: string; + label: string | null; + left: string | null; + right: string | null; + state: ComparisonState; +}; + +export type ExecutionComparison = { + nodes: ComparedNode[]; + outcomes: ComparedOutcome[]; + changedNodeCount: number; + /** + * The two runs do not share a node in common. Almost always the flow was edited between them, + * in which case a per-node comparison is meaningless rather than merely empty. + */ + disjoint: boolean; +}; + +function normalize(value: unknown): string | null { + if (value === null || value === undefined) return null; + const text = stringifyOutputValue(value); + return text.trim().length ? text : null; +} + +function stateOf(left: string | null, right: string | null): ComparisonState { + if (left !== null && right === null) return 'only-left'; + if (left === null && right !== null) return 'only-right'; + return left === right ? 'equal' : 'changed'; +} + +function valuesOf(step: TaskExecutionStep | undefined, execution: TaskExecution | undefined) { + if (!step || !execution) return { inputs: {}, outputs: {} }; + return { + inputs: getExecutionInputValues(step, execution.context.inputs ?? {}), + outputs: resolveStepOutputs(step, execution) + }; +} + +function compareOutcomes(left: TaskExecution, right: TaskExecution): ComparedOutcome[] { + const byCode = (execution: TaskExecution) => new Map( + (execution.context.outcomes ?? []).map((outcome) => [outcome.code, outcome]) + ); + const leftOutcomes = byCode(left); + const rightOutcomes = byCode(right); + + return [...new Set([...leftOutcomes.keys(), ...rightOutcomes.keys()])].sort().map((code) => { + const leftValue = normalize(leftOutcomes.get(code)?.payload); + const rightValue = normalize(rightOutcomes.get(code)?.payload); + // An outcome reached by only one run is a difference in itself, even with no payload. + const reachedOnLeft = leftOutcomes.has(code); + const reachedOnRight = rightOutcomes.has(code); + const state: ComparisonState = reachedOnLeft && !reachedOnRight + ? 'only-left' + : !reachedOnLeft && reachedOnRight + ? 'only-right' + : stateOf(leftValue, rightValue); + return { + code, + label: leftOutcomes.get(code)?.label ?? rightOutcomes.get(code)?.label ?? null, + left: leftValue, + right: rightValue, + state + }; + }); +} + +/** + * Joins two runs of the same flow node by node. + * + * The join key is the step id, which is stable across runs of one flow. It is deliberately not + * done by node name: names are not unique and can be edited, and a rename would silently report + * every node as replaced. + * + * This is not the bias comparison. That one scopes itself to the nodes a probe was activated on + * and refuses a run that is not an experiment, so for two ordinary runs it compares nothing. + */ +export function compareExecutions(left: TaskExecution, right: TaskExecution): ExecutionComparison { + const leftSteps = left.context.steps ?? {}; + const rightSteps = right.context.steps ?? {}; + const stepIds = [...new Set([...Object.keys(leftSteps), ...Object.keys(rightSteps)])]; + + const nodes = stepIds.map((stepId) => { + const leftStep = leftSteps[stepId]; + const rightStep = rightSteps[stepId]; + const leftValues = valuesOf(leftStep, left); + const rightValues = valuesOf(rightStep, right); + + const values: ComparedValue[] = []; + for (const kind of ['input', 'output'] as const) { + const leftSide = kind === 'input' ? leftValues.inputs : leftValues.outputs; + const rightSide = kind === 'input' ? rightValues.inputs : rightValues.outputs; + const names = [...new Set([...Object.keys(leftSide), ...Object.keys(rightSide)])].sort(); + for (const name of names) { + const leftValue = normalize(leftSide[name]); + const rightValue = normalize(rightSide[name]); + if (leftValue === null && rightValue === null) continue; + values.push({ + key: `${kind}s.${name}`, + kind, + name, + left: leftValue, + right: rightValue, + state: stateOf(leftValue, rightValue) + }); + } + } + + const leftStatus = leftStep ? String(leftStep.status ?? '') : null; + const rightStatus = rightStep ? String(rightStep.status ?? '') : null; + const statusChanged = !!leftStep && !!rightStep && leftStatus !== rightStatus; + const state: ComparisonState = leftStep && !rightStep + ? 'only-left' + : !leftStep && rightStep + ? 'only-right' + : values.some((value) => value.state !== 'equal') || statusChanged + ? 'changed' + : 'equal'; + + return { + stepId, + title: stepTitle(leftStep ?? rightStep), + leftStatus, + rightStatus, + statusChanged, + values, + changed: state !== 'equal', + state + }; + }); + + // Changed nodes first: on a long flow the differences are what the view exists to show. + nodes.sort((a, b) => Number(b.changed) - Number(a.changed) || a.title.localeCompare(b.title)); + + const sharedStepIds = stepIds.filter((stepId) => leftSteps[stepId] && rightSteps[stepId]); + return { + nodes, + outcomes: compareOutcomes(left, right), + changedNodeCount: nodes.filter((node) => node.changed).length, + disjoint: stepIds.length > 0 && sharedStepIds.length === 0 + }; +} diff --git a/src/app/shared/task-execution-viewer/execution-viewer.utils.ts b/src/app/shared/task-execution-viewer/execution-viewer.utils.ts index fff5562..19998d8 100644 --- a/src/app/shared/task-execution-viewer/execution-viewer.utils.ts +++ b/src/app/shared/task-execution-viewer/execution-viewer.utils.ts @@ -331,6 +331,41 @@ export function getExecutionOutputValues( return result; } +/** + * Everything a step produced, including the outputs that were wired onward. + * + * `context.result` holds only the *unconnected* outputs - it is the flow's result, not a log of + * every node - so on a flow whose nodes all feed one another it is nearly empty, and on one that + * ends in an End node it is empty outright. The value of a connected output is still observable: + * it is what arrived at the input on the other end of the wire. The backend reconstructs node + * outputs the same way when it builds a bias report. + */ +export function resolveStepOutputs( + step: TaskExecutionStep, + execution: TaskExecution +): Record { + const outputs = getExecutionOutputValues(step, execution.context.result ?? {}); + const contextInputs = execution.context.inputs ?? {}; + const steps = execution.context.steps ?? {}; + + for (const connection of execution.stepConnections ?? []) { + if (connection.sourceId !== step.id) continue; + if (Object.prototype.hasOwnProperty.call(outputs, connection.sourceName)) continue; + + const key = `${connection.targetId}:${connection.targetName}`; + if (Object.prototype.hasOwnProperty.call(contextInputs, key)) { + outputs[connection.sourceName] = contextInputs[key]; + continue; + } + const targetInput = (steps[connection.targetId]?.inputs ?? []) + .find((candidate) => candidate.descriptor?.name === connection.targetName); + if (targetInput && targetInput.value != null) { + outputs[connection.sourceName] = targetInput.value; + } + } + return outputs; +} + export function getConnectedInputs(step: TaskExecutionStep): string[] { return (step.inputs ?? []) .filter((input) => input.registered)