Join two runs of a flow node by node

The first half of comparing two executions: a pure function, so the semantics
are testable without a TestBed - the same shape as flow-grouping and
planInputSaves.

It is not the bias comparison. That one scopes itself to the nodes a probe was
activated on and refuses a run that is not an experiment, so for two ordinary
runs its node set is empty and it compares nothing. Generalising it would mean
rewriting its scoping, not relaxing a condition.

The join key is the step id, stable across runs of one flow. Not the node name:
names are not unique and can be edited, and a rename would report every node as
replaced. Two runs sharing no step at all are reported as disjoint - almost
always the flow was edited between them, which makes the join meaningless rather
than merely empty.

resolveStepOutputs is the part that makes this work on a real flow.
context.result holds only the *unconnected* outputs, so on a flow whose nodes
feed one another it is nearly empty, and on one ending in an End node it is
empty outright - the run this was built for has result {} and one outcome. The
value of a connected output is still observable as what arrived at the input on
the other end of the wire, which is how the backend reconstructs node outputs
for a bias report. Outcomes are compared too, for the same reason.

516 frontend tests green.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Lucio Lelii 2026-09-03 22:12:11 +02:00
parent b2d744fe42
commit 381e3d3817
3 changed files with 404 additions and 0 deletions

View File

@ -0,0 +1,194 @@
import { TaskExecution } from '@models/task-execution';
import { compareExecutions } from './execution-compare';
type StepSpec = {
id: string;
name?: string;
status?: string;
inputs?: Record<string, unknown>;
outputs?: string[];
};
function execution(id: string, steps: StepSpec[], overrides: Partial<TaskExecution['context']> = {}): TaskExecution {
return {
id,
name: id,
creationTime: 1,
context: {
inputs: {},
result: {},
errors: {},
warnings: {},
waitingSteps: [],
status: 'SUCCESS',
steps: Object.fromEntries(steps.map((step) => [step.id, {
id: step.id,
status: step.status ?? 'COMPLETED',
simulated: false,
node: {
id: step.id,
name: step.name ?? step.id,
nodeFamily: 'block' as const,
typeName: 'LLMBlock',
inputs: Object.keys(step.inputs ?? {}).map((name) => ({ name, type: 'TEXT' })),
outputs: (step.outputs ?? []).map((name) => ({ name, type: 'TEXT' })),
specificConfiguration: {}
},
inputs: Object.entries(step.inputs ?? {}).map(([name, value]) => ({
descriptor: { name, type: 'TEXT' }, value, set: true, registered: false
})),
outputs: (step.outputs ?? []).map((name) => ({ descriptor: { name, type: 'TEXT' }, connected: false }))
}])),
...overrides
}
} as unknown as TaskExecution;
}
describe('compareExecutions', () => {
it('marks a node whose output differs, and leaves an identical one alone', () => {
const left = execution('run-1', [{ id: 's1', name: 'Evaluate', outputs: ['response'] }],
{ result: { 's1:response': 'Score 7/10' } });
const right = execution('run-2', [{ id: 's1', name: 'Evaluate', outputs: ['response'] }],
{ result: { 's1:response': 'Score 5/10' } });
const { nodes, changedNodeCount } = compareExecutions(left, right);
expect(changedNodeCount).toBe(1);
const value = nodes[0].values.find((one) => one.key === 'outputs.response');
expect(value).toMatchObject({ left: 'Score 7/10', right: 'Score 5/10', state: 'changed' });
});
it('reports an unchanged node as equal', () => {
const step = { id: 's1', outputs: ['response'] };
const result = { 's1:response': 'same' };
const comparison = compareExecutions(
execution('run-1', [step], { result }),
execution('run-2', [step], { result })
);
expect(comparison.changedNodeCount).toBe(0);
expect(comparison.nodes[0].state).toBe('equal');
});
it('reads a connected output from the input it fed, since context.result never holds it', () => {
// This is the case that matters on a real flow: every wired output is absent from
// context.result, so comparing that map alone would report almost nothing.
const wiring = {
stepConnections: [
{ id: 'c1', sourceId: 's1', sourceName: 'response', targetId: 's2', targetName: 'summary' }
]
};
const left = {
...execution('run-1', [
{ id: 's1', name: 'Interview', outputs: ['response'] },
{ id: 's2', name: 'Evaluate', inputs: { summary: 'interview went well' } }
], { inputs: { 's2:summary': 'interview went well' } }),
...wiring
};
const right = {
...execution('run-2', [
{ id: 's1', name: 'Interview', outputs: ['response'] },
{ id: 's2', name: 'Evaluate', inputs: { summary: 'interview went badly' } }
], { inputs: { 's2:summary': 'interview went badly' } }),
...wiring
};
const { nodes } = compareExecutions(left as TaskExecution, right as TaskExecution);
const interview = nodes.find((node) => node.title === 'Interview')!;
expect(interview.values.find((one) => one.key === 'outputs.response')).toMatchObject({
left: 'interview went well',
right: 'interview went badly',
state: 'changed'
});
});
it('compares the outcomes, which is where the answer lives when the flow ends on an End node', () => {
// The real run this was built for has result {} and a single outcome: omitting outcomes would
// make the comparison useless in exactly the case it was asked for.
const outcome = (payload: string) => ({
outcomes: [{ stepId: 's9', code: 'MULTI_CV_RANKING_ACCEPTED', label: 'Final ranking', payload, timestamp: 1 }]
});
const comparison = compareExecutions(
execution('run-1', [{ id: 's1' }], outcome('1. Alice — 7/10')),
execution('run-2', [{ id: 's1' }], outcome('1. Bob — 6/10'))
);
expect(comparison.outcomes).toHaveLength(1);
expect(comparison.outcomes[0]).toMatchObject({
code: 'MULTI_CV_RANKING_ACCEPTED',
left: '1. Alice — 7/10',
right: '1. Bob — 6/10',
state: 'changed'
});
});
it('flags an outcome only one run reached, even with no payload to compare', () => {
const comparison = compareExecutions(
execution('run-1', [{ id: 's1' }], {
outcomes: [{ stepId: 's9', code: 'ACCEPTED', label: 'Accepted', payload: null, timestamp: 1 }]
}),
execution('run-2', [{ id: 's1' }], {
outcomes: [{ stepId: 's9', code: 'REVIEW_REQUESTED', label: 'Review', payload: null, timestamp: 1 }]
})
);
expect(comparison.outcomes.map((one) => [one.code, one.state]))
.toEqual([['ACCEPTED', 'only-left'], ['REVIEW_REQUESTED', 'only-right']]);
});
it('marks a node that only one run has', () => {
const comparison = compareExecutions(
execution('run-1', [{ id: 's1', name: 'Shared' }, { id: 's2', name: 'Dropped' }]),
execution('run-2', [{ id: 's1', name: 'Shared' }])
);
expect(comparison.nodes.find((node) => node.title === 'Dropped')?.state).toBe('only-left');
expect(comparison.disjoint).toBe(false);
});
it('reports two runs with no node in common as disjoint', () => {
// Almost always the flow was edited between the runs, which makes a per-node join meaningless
// rather than merely empty - the view has to say so instead of showing everything as replaced.
const comparison = compareExecutions(
execution('run-1', [{ id: 'old-1' }]),
execution('run-2', [{ id: 'new-1' }])
);
expect(comparison.disjoint).toBe(true);
});
it('treats a differing step status as a change even when the values match', () => {
const comparison = compareExecutions(
execution('run-1', [{ id: 's1', status: 'COMPLETED' }]),
execution('run-2', [{ id: 's1', status: 'SKIPPED' }])
);
expect(comparison.nodes[0].statusChanged).toBe(true);
expect(comparison.nodes[0].changed).toBe(true);
});
it('puts the changed nodes first', () => {
const comparison = compareExecutions(
execution('run-1', [
{ id: 's1', name: 'Alpha', outputs: ['out'] },
{ id: 's2', name: 'Beta', outputs: ['out'] }
], { result: { 's1:out': 'same', 's2:out': 'left' } }),
execution('run-2', [
{ id: 's1', name: 'Alpha', outputs: ['out'] },
{ id: 's2', name: 'Beta', outputs: ['out'] }
], { result: { 's1:out': 'same', 's2:out': 'right' } })
);
expect(comparison.nodes.map((node) => node.title)).toEqual(['Beta', 'Alpha']);
});
it('ignores a port that is empty on both sides rather than listing it as equal', () => {
const comparison = compareExecutions(
execution('run-1', [{ id: 's1', outputs: ['unused'] }]),
execution('run-2', [{ id: 's1', outputs: ['unused'] }])
);
expect(comparison.nodes[0].values).toEqual([]);
});
});

View File

@ -0,0 +1,175 @@
import { TaskExecution, TaskExecutionOutcome, TaskExecutionStep } from '@models/task-execution';
import {
getExecutionInputValues,
resolveStepOutputs,
stepTitle,
stringifyOutputValue
} from '@shared/task-execution-viewer/execution-viewer.utils';
/** Whether a value is the same on both sides, or present on only one of them. */
export type ComparisonState = 'equal' | 'changed' | 'only-left' | 'only-right';
export type ComparedValue = {
/** `inputs` or `outputs` plus the port name, e.g. `outputs.response`. */
key: string;
kind: 'input' | 'output';
name: string;
left: string | null;
right: string | null;
state: ComparisonState;
};
export type ComparedNode = {
stepId: string;
title: string;
leftStatus: string | null;
rightStatus: string | null;
statusChanged: boolean;
values: ComparedValue[];
/** True when anything about this node differs, including its status. */
changed: boolean;
state: ComparisonState;
};
export type ComparedOutcome = {
code: string;
label: string | null;
left: string | null;
right: string | null;
state: ComparisonState;
};
export type ExecutionComparison = {
nodes: ComparedNode[];
outcomes: ComparedOutcome[];
changedNodeCount: number;
/**
* The two runs do not share a node in common. Almost always the flow was edited between them,
* in which case a per-node comparison is meaningless rather than merely empty.
*/
disjoint: boolean;
};
function normalize(value: unknown): string | null {
if (value === null || value === undefined) return null;
const text = stringifyOutputValue(value);
return text.trim().length ? text : null;
}
function stateOf(left: string | null, right: string | null): ComparisonState {
if (left !== null && right === null) return 'only-left';
if (left === null && right !== null) return 'only-right';
return left === right ? 'equal' : 'changed';
}
function valuesOf(step: TaskExecutionStep | undefined, execution: TaskExecution | undefined) {
if (!step || !execution) return { inputs: {}, outputs: {} };
return {
inputs: getExecutionInputValues(step, execution.context.inputs ?? {}),
outputs: resolveStepOutputs(step, execution)
};
}
function compareOutcomes(left: TaskExecution, right: TaskExecution): ComparedOutcome[] {
const byCode = (execution: TaskExecution) => new Map<string, TaskExecutionOutcome>(
(execution.context.outcomes ?? []).map((outcome) => [outcome.code, outcome])
);
const leftOutcomes = byCode(left);
const rightOutcomes = byCode(right);
return [...new Set([...leftOutcomes.keys(), ...rightOutcomes.keys()])].sort().map((code) => {
const leftValue = normalize(leftOutcomes.get(code)?.payload);
const rightValue = normalize(rightOutcomes.get(code)?.payload);
// An outcome reached by only one run is a difference in itself, even with no payload.
const reachedOnLeft = leftOutcomes.has(code);
const reachedOnRight = rightOutcomes.has(code);
const state: ComparisonState = reachedOnLeft && !reachedOnRight
? 'only-left'
: !reachedOnLeft && reachedOnRight
? 'only-right'
: stateOf(leftValue, rightValue);
return {
code,
label: leftOutcomes.get(code)?.label ?? rightOutcomes.get(code)?.label ?? null,
left: leftValue,
right: rightValue,
state
};
});
}
/**
* Joins two runs of the same flow node by node.
*
* The join key is the step id, which is stable across runs of one flow. It is deliberately not
* done by node name: names are not unique and can be edited, and a rename would silently report
* every node as replaced.
*
* This is not the bias comparison. That one scopes itself to the nodes a probe was activated on
* and refuses a run that is not an experiment, so for two ordinary runs it compares nothing.
*/
export function compareExecutions(left: TaskExecution, right: TaskExecution): ExecutionComparison {
const leftSteps = left.context.steps ?? {};
const rightSteps = right.context.steps ?? {};
const stepIds = [...new Set([...Object.keys(leftSteps), ...Object.keys(rightSteps)])];
const nodes = stepIds.map<ComparedNode>((stepId) => {
const leftStep = leftSteps[stepId];
const rightStep = rightSteps[stepId];
const leftValues = valuesOf(leftStep, left);
const rightValues = valuesOf(rightStep, right);
const values: ComparedValue[] = [];
for (const kind of ['input', 'output'] as const) {
const leftSide = kind === 'input' ? leftValues.inputs : leftValues.outputs;
const rightSide = kind === 'input' ? rightValues.inputs : rightValues.outputs;
const names = [...new Set([...Object.keys(leftSide), ...Object.keys(rightSide)])].sort();
for (const name of names) {
const leftValue = normalize(leftSide[name]);
const rightValue = normalize(rightSide[name]);
if (leftValue === null && rightValue === null) continue;
values.push({
key: `${kind}s.${name}`,
kind,
name,
left: leftValue,
right: rightValue,
state: stateOf(leftValue, rightValue)
});
}
}
const leftStatus = leftStep ? String(leftStep.status ?? '') : null;
const rightStatus = rightStep ? String(rightStep.status ?? '') : null;
const statusChanged = !!leftStep && !!rightStep && leftStatus !== rightStatus;
const state: ComparisonState = leftStep && !rightStep
? 'only-left'
: !leftStep && rightStep
? 'only-right'
: values.some((value) => value.state !== 'equal') || statusChanged
? 'changed'
: 'equal';
return {
stepId,
title: stepTitle(leftStep ?? rightStep),
leftStatus,
rightStatus,
statusChanged,
values,
changed: state !== 'equal',
state
};
});
// Changed nodes first: on a long flow the differences are what the view exists to show.
nodes.sort((a, b) => Number(b.changed) - Number(a.changed) || a.title.localeCompare(b.title));
const sharedStepIds = stepIds.filter((stepId) => leftSteps[stepId] && rightSteps[stepId]);
return {
nodes,
outcomes: compareOutcomes(left, right),
changedNodeCount: nodes.filter((node) => node.changed).length,
disjoint: stepIds.length > 0 && sharedStepIds.length === 0
};
}

View File

@ -331,6 +331,41 @@ export function getExecutionOutputValues(
return result;
}
/**
* Everything a step produced, including the outputs that were wired onward.
*
* `context.result` holds only the *unconnected* outputs - it is the flow's result, not a log of
* every node - so on a flow whose nodes all feed one another it is nearly empty, and on one that
* ends in an End node it is empty outright. The value of a connected output is still observable:
* it is what arrived at the input on the other end of the wire. The backend reconstructs node
* outputs the same way when it builds a bias report.
*/
export function resolveStepOutputs(
step: TaskExecutionStep,
execution: TaskExecution
): Record<string, unknown> {
const outputs = getExecutionOutputValues(step, execution.context.result ?? {});
const contextInputs = execution.context.inputs ?? {};
const steps = execution.context.steps ?? {};
for (const connection of execution.stepConnections ?? []) {
if (connection.sourceId !== step.id) continue;
if (Object.prototype.hasOwnProperty.call(outputs, connection.sourceName)) continue;
const key = `${connection.targetId}:${connection.targetName}`;
if (Object.prototype.hasOwnProperty.call(contextInputs, key)) {
outputs[connection.sourceName] = contextInputs[key];
continue;
}
const targetInput = (steps[connection.targetId]?.inputs ?? [])
.find((candidate) => candidate.descriptor?.name === connection.targetName);
if (targetInput && targetInput.value != null) {
outputs[connection.sourceName] = targetInput.value;
}
}
return outputs;
}
export function getConnectedInputs(step: TaskExecutionStep): string[] {
return (step.inputs ?? [])
.filter((input) => input.registered)