Say which subject a bias intervention moved, and let an LLM assess it
The report now reads the per-subject sections the service produces: a row per iterated subject with what its labelled numbers did (Score 7 -> 4), the two texts behind it a click away, and the changed ones listed first. The band at the top leads with what the run actually did - whether the final decision changed, how many subjects moved, the largest delta - because the counts of changed nodes that used to open the report were the least actionable thing in it. A container's iterations are listed with the inner node that changed, which is what a per-subject iterator run needs and the accumulated list could never show. Reports produced before any of this exists still render, from their raw outputs. "Evaluate impact with LLM" sits next to the report it is about, in all three places one is mounted, and opens the provider and model picker the interaction simulator uses - extracted so both call the same dialog rather than two of their own, with temperature 0 offered by default because a verdict that reads differently every time it is asked for is worse than none. The assessment runs as a job, polled like the isolated experiment, and is stored on the report, so reopening it later shows the same verdicts and the model that produced them. It is labelled an assessment throughout, and a pair the model could not answer for is marked without hiding that pair's own figures. A rerun of a simulated run now opens the Simulate dialog on the simulator it inherited, with the inherited sampling out where it can be seen - a seed carried over is the reason the two runs are comparable, and behind a closed section nobody would find it. Before it is started, a run says which simulator the run it repeats used; afterwards, both the bias report and the run-to-run comparison say so when the two sides were not answered the same way, since that difference is not the intervention's doing and nothing said it. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
d11e2f5a9a
commit
f5386cdb7e
|
|
@ -53,6 +53,7 @@ describe('bias impact models', () => {
|
|||
};
|
||||
const job: BiasImpactJob = {
|
||||
id: 'job-1',
|
||||
kind: 'ISOLATED_STEP',
|
||||
status: 'COMPLETED',
|
||||
executionId: 'baseline-1',
|
||||
stepId: 'node-1',
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
import { BiasActivationMode } from './flow';
|
||||
import { BiasActivationMode, LLMDescriptor } from './flow';
|
||||
|
||||
export type BiasInterventionDirection = 'BIAS' | 'MITIGATION' | 'BOTH';
|
||||
|
||||
|
|
@ -111,10 +111,15 @@ export function activeAnnotationIdsFor(
|
|||
|
||||
export type BiasImpactJobStatus = 'QUEUED' | 'RUNNING' | 'COMPLETED' | 'FAILED';
|
||||
|
||||
/** Both kinds of queued bias work: re-running a step, and assessing a comparison with an LLM. */
|
||||
export type BiasImpactJobKind = 'ISOLATED_STEP' | 'REPORT_JUDGE';
|
||||
|
||||
export type BiasImpactJob = {
|
||||
id: string;
|
||||
kind: BiasImpactJobKind;
|
||||
status: BiasImpactJobStatus;
|
||||
executionId: string;
|
||||
/** Empty for a judge job: it is about a whole comparison, not one step. */
|
||||
stepId: string;
|
||||
createdAt: string;
|
||||
startedAt: string | null;
|
||||
|
|
@ -128,12 +133,107 @@ export type BiasImpactJob = {
|
|||
|
||||
export type BiasImpactReportKind = 'ISOLATED_STEP' | 'FULL_FLOW';
|
||||
|
||||
/** A number both runs labelled the same way, and how far the intervention moved it. */
|
||||
export type BiasNumericDelta = {
|
||||
label: string;
|
||||
baseline: number;
|
||||
biased: number;
|
||||
delta: number;
|
||||
};
|
||||
|
||||
export type BiasJudgeImpactLevel = 'NONE' | 'COSMETIC' | 'SUBSTANTIVE' | 'DECISIVE';
|
||||
|
||||
export type BiasJudgeAttribution = 'INJECTION' | 'NON_DETERMINISM' | 'UNCLEAR';
|
||||
|
||||
/**
|
||||
* What a judge model made of one compared pair. An assessment, not a measurement: `error` is set
|
||||
* when the model answered with something unusable, and the pair's own figures stand either way.
|
||||
*/
|
||||
export type BiasJudgeVerdict = {
|
||||
impact: BiasJudgeImpactLevel | null;
|
||||
attribution: BiasJudgeAttribution | null;
|
||||
confidence: number | null;
|
||||
changedAspects: string[];
|
||||
rationale: string | null;
|
||||
error: string | null;
|
||||
};
|
||||
|
||||
export type BiasJudgeSummary = {
|
||||
judge: LLMDescriptor;
|
||||
judgedAt: string;
|
||||
impact: BiasJudgeImpactLevel | null;
|
||||
attribution: BiasJudgeAttribution | null;
|
||||
narrative: string | null;
|
||||
judgedPairs: number;
|
||||
skippedPairs: number;
|
||||
errors: string[];
|
||||
};
|
||||
|
||||
/** One element of a compared list value, which for an iterator container is one subject. */
|
||||
export type BiasItemImpact = {
|
||||
index: number;
|
||||
changed: boolean;
|
||||
textDifference: number;
|
||||
numericDeltas: BiasNumericDelta[];
|
||||
baselineText: string | null;
|
||||
biasedText: string | null;
|
||||
judgeVerdict: BiasJudgeVerdict | null;
|
||||
};
|
||||
|
||||
/** One output field of one node, compared between the two runs. */
|
||||
export type BiasValueImpact = {
|
||||
nodeId: string;
|
||||
nodeName: string;
|
||||
field: string;
|
||||
changed: boolean;
|
||||
textDifference: number;
|
||||
itemsCompared: number;
|
||||
itemsChanged: number;
|
||||
meanItemTextDifference: number;
|
||||
maximumItemTextDifference: number;
|
||||
numericDeltas: BiasNumericDelta[];
|
||||
items: BiasItemImpact[];
|
||||
baselineText: string | null;
|
||||
biasedText: string | null;
|
||||
judgeVerdict: BiasJudgeVerdict | null;
|
||||
};
|
||||
|
||||
/** One iteration of a container step, from the pair of child executions that ran it. */
|
||||
export type BiasIterationImpact = {
|
||||
containerNodeId: string;
|
||||
containerNodeName: string;
|
||||
index: number;
|
||||
baselineExecutionId: string | null;
|
||||
biasedExecutionId: string | null;
|
||||
baselineStatus: string;
|
||||
biasedStatus: string;
|
||||
changed: boolean;
|
||||
values: BiasValueImpact[];
|
||||
};
|
||||
|
||||
/**
|
||||
* How the interactive steps of the two compared runs were answered.
|
||||
*
|
||||
* Absent when neither run was simulated, and on reports produced before it was recorded.
|
||||
*/
|
||||
export type BiasSimulationContext = {
|
||||
baselineSimulated: boolean;
|
||||
baselineSimulator: LLMDescriptor | null;
|
||||
biasedSimulated: boolean;
|
||||
biasedSimulator: LLMDescriptor | null;
|
||||
/** False when only one side was simulated, or the two simulators are not the same one. */
|
||||
comparable: boolean;
|
||||
};
|
||||
|
||||
export type BiasImmediateImpact = {
|
||||
outputChanged: boolean;
|
||||
maximumTextDifference: number;
|
||||
changeRate: number;
|
||||
baselineOutput: unknown;
|
||||
biasedOutputs: unknown[];
|
||||
/** Absent on reports produced before the field-by-field comparison existed. */
|
||||
values?: BiasValueImpact[];
|
||||
iterations?: BiasIterationImpact[];
|
||||
};
|
||||
|
||||
export type BiasDownstreamImpactEntry = {
|
||||
|
|
@ -144,6 +244,8 @@ export type BiasDownstreamImpactEntry = {
|
|||
changed: boolean;
|
||||
baselineOutputs: unknown;
|
||||
biasedOutputs: unknown;
|
||||
values?: BiasValueImpact[];
|
||||
iterations?: BiasIterationImpact[];
|
||||
};
|
||||
|
||||
export type BiasRoutingChangeEntry = {
|
||||
|
|
@ -152,6 +254,17 @@ export type BiasRoutingChangeEntry = {
|
|||
biasedBranch: string;
|
||||
};
|
||||
|
||||
/**
|
||||
* The outcomes each run ended on, when they differ.
|
||||
*
|
||||
* The service has always sent this; nothing read it, so a comparison whose two runs ended on
|
||||
* different outcomes - the largest thing an intervention can do - showed only as a count.
|
||||
*/
|
||||
export type BiasOutcomeChangeEntry = {
|
||||
baselineOutcomeCodes: string[];
|
||||
biasedOutcomeCodes: string[];
|
||||
};
|
||||
|
||||
export type BiasMockedSideEffectKind =
|
||||
| 'HTTP'
|
||||
| 'MCP_AGENT'
|
||||
|
|
@ -178,10 +291,15 @@ export type BiasImpactReport = {
|
|||
immediateImpact: BiasImmediateImpact;
|
||||
downstreamImpact: BiasDownstreamImpactEntry[];
|
||||
routingChanges: BiasRoutingChangeEntry[];
|
||||
outcomeChanges?: BiasOutcomeChangeEntry[];
|
||||
mockedSideEffects: BiasMockedSideEffect[];
|
||||
summary: string;
|
||||
warnings: string[];
|
||||
interventionDirection?: BiasInterventionDirection;
|
||||
schemaVersion?: number;
|
||||
/** Present only once someone has asked a model to assess the comparison. */
|
||||
judge?: BiasJudgeSummary | null;
|
||||
simulation?: BiasSimulationContext | null;
|
||||
};
|
||||
|
||||
export const BIAS_PROBE_ERROR_CODES = [
|
||||
|
|
|
|||
|
|
@ -47,7 +47,7 @@ describe('bias impact main API flow (annotation -> capability -> isolated experi
|
|||
};
|
||||
|
||||
const queuedJob: BiasImpactJob = {
|
||||
id: 'job-1', status: 'QUEUED', executionId: 'execution-1', stepId: 'step-1',
|
||||
id: 'job-1', kind: 'ISOLATED_STEP', status: 'QUEUED', executionId: 'execution-1', stepId: 'step-1',
|
||||
createdAt: '2026-07-21T10:00:00', startedAt: null, completedAt: null,
|
||||
reportId: null, report: null, errorCode: null, errorMessage: null, terminal: false
|
||||
};
|
||||
|
|
|
|||
|
|
@ -0,0 +1,93 @@
|
|||
import { TestBed } from '@angular/core/testing';
|
||||
import { of } from 'rxjs';
|
||||
import { vi } from 'vitest';
|
||||
import { BiasImpactJob, BiasImpactReport } from '@models/bias-impact';
|
||||
import { NodeSettingsDialogService } from '@services/dialogs/node-settings-dialog';
|
||||
import { FieldRetriever } from '@services/retriever/field-retriever';
|
||||
import { TaskExecutionsService } from '@services/task-executions/task-executions';
|
||||
import { BiasReportJudgeService } from './bias-report-judge';
|
||||
|
||||
describe('BiasReportJudgeService', () => {
|
||||
const report = { id: 'report-1', summary: 'judged' } as BiasImpactReport;
|
||||
|
||||
const job = (status: BiasImpactJob['status'], overrides: Partial<BiasImpactJob> = {}): BiasImpactJob => ({
|
||||
id: 'job-1',
|
||||
kind: 'REPORT_JUDGE',
|
||||
status,
|
||||
executionId: 'execution-1',
|
||||
stepId: '',
|
||||
createdAt: '2026-07-21T10:00:00',
|
||||
startedAt: null,
|
||||
completedAt: null,
|
||||
reportId: 'report-1',
|
||||
report: null,
|
||||
errorCode: null,
|
||||
errorMessage: null,
|
||||
terminal: status === 'COMPLETED' || status === 'FAILED',
|
||||
...overrides
|
||||
});
|
||||
|
||||
let judgeReport: ReturnType<typeof vi.fn>;
|
||||
let open: ReturnType<typeof vi.fn>;
|
||||
let service: BiasReportJudgeService;
|
||||
|
||||
function configure(dialogAnswer: Record<string, string> | null, terminalJob: BiasImpactJob) {
|
||||
judgeReport = vi.fn().mockReturnValue(of(job('QUEUED')));
|
||||
open = vi.fn().mockResolvedValue(dialogAnswer);
|
||||
TestBed.configureTestingModule({
|
||||
providers: [
|
||||
BiasReportJudgeService,
|
||||
{ provide: NodeSettingsDialogService, useValue: { open } },
|
||||
{
|
||||
provide: FieldRetriever,
|
||||
useValue: { retrieveValues: vi.fn().mockReturnValue(of(['InternalOllama'])) }
|
||||
},
|
||||
{
|
||||
provide: TaskExecutionsService,
|
||||
useValue: {
|
||||
judgeBiasImpactReport: judgeReport,
|
||||
pollBiasImpactJob: vi.fn().mockReturnValue(of(job('RUNNING'), terminalJob))
|
||||
}
|
||||
}
|
||||
]
|
||||
});
|
||||
service = TestBed.inject(BiasReportJudgeService);
|
||||
}
|
||||
|
||||
it('sends the picked model and answers with the assessed report', async () => {
|
||||
configure({ provider: 'InternalOllama', model: 'gemma:7b', temperature: '0' },
|
||||
job('COMPLETED', { report }));
|
||||
|
||||
const judged = await service.assess('report-1');
|
||||
|
||||
expect(judged).toBe(report);
|
||||
expect(judgeReport).toHaveBeenCalledWith('report-1', {
|
||||
provider: 'InternalOllama',
|
||||
model: 'gemma:7b',
|
||||
parameters: { temperature: 0 }
|
||||
});
|
||||
});
|
||||
|
||||
it('offers temperature 0 by default, so two readings of one comparison agree', async () => {
|
||||
configure({ provider: 'InternalOllama', model: 'gemma:7b' }, job('COMPLETED', { report }));
|
||||
|
||||
await service.assess('report-1');
|
||||
|
||||
expect(open.mock.calls[0][0].title).toBe('Evaluate impact with LLM');
|
||||
expect(open.mock.calls[0][0].initial.temperature).toBe('0');
|
||||
});
|
||||
|
||||
it('asks for nothing when the model picker is dismissed', async () => {
|
||||
configure(null, job('COMPLETED', { report }));
|
||||
|
||||
expect(await service.assess('report-1')).toBeNull();
|
||||
expect(judgeReport).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('surfaces the failure of a job that did not complete', async () => {
|
||||
configure({ provider: 'InternalOllama', model: 'gemma:7b' },
|
||||
job('FAILED', { errorMessage: 'the judge provider is unreachable' }));
|
||||
|
||||
await expect(service.assess('report-1')).rejects.toThrow('the judge provider is unreachable');
|
||||
});
|
||||
});
|
||||
|
|
@ -0,0 +1,42 @@
|
|||
import { inject, Injectable } from '@angular/core';
|
||||
import { lastValueFrom } from 'rxjs';
|
||||
import { BiasImpactReport } from '@models/bias-impact';
|
||||
import { FieldRetriever } from '@services/retriever/field-retriever';
|
||||
import { NodeSettingsDialogService } from '@services/dialogs/node-settings-dialog';
|
||||
import { TaskExecutionsService } from '@services/task-executions/task-executions';
|
||||
import { openLLMDescriptorSettings } from '@shared/llm-descriptor-settings/llm-descriptor-settings';
|
||||
|
||||
/**
|
||||
* Asks a model to assess a comparison, from wherever that comparison is on screen.
|
||||
*
|
||||
* <p>The report viewer is mounted by three different hosts, and all three offer the same action, so
|
||||
* the picking of the model, the queued job and the waiting live here rather than three times over.
|
||||
*
|
||||
* <p>Temperature starts at 0: a judgement that reads differently every time it is asked for is
|
||||
* worse than none, and the seed field next to it is there for the same reason.
|
||||
*/
|
||||
@Injectable({ providedIn: 'root' })
|
||||
export class BiasReportJudgeService {
|
||||
private readonly executions = inject(TaskExecutionsService);
|
||||
private readonly settingsDialog = inject(NodeSettingsDialogService);
|
||||
private readonly fieldRetriever = inject(FieldRetriever);
|
||||
|
||||
/**
|
||||
* Resolves with the assessed report, or null when the model picker was dismissed or no provider
|
||||
* is published at all. Rejects with the failure of the assessment itself.
|
||||
*/
|
||||
async assess(reportId: string): Promise<BiasImpactReport | null> {
|
||||
const judge = await openLLMDescriptorSettings(this.settingsDialog, this.fieldRetriever, {
|
||||
title: 'Evaluate impact with LLM',
|
||||
defaultParameters: { temperature: 0 }
|
||||
});
|
||||
if (!judge) return null;
|
||||
|
||||
const queued = await lastValueFrom(this.executions.judgeBiasImpactReport(reportId, judge));
|
||||
const finished = await lastValueFrom(this.executions.pollBiasImpactJob(queued.id));
|
||||
if (finished.status !== 'COMPLETED' || !finished.report) {
|
||||
throw new Error(finished.errorMessage || 'The LLM assessment did not complete.');
|
||||
}
|
||||
return finished.report;
|
||||
}
|
||||
}
|
||||
|
|
@ -32,6 +32,11 @@ export abstract class TaskExecutionsCallServiceBase {
|
|||
stepId: string,
|
||||
request: BiasImpactExperimentRequest
|
||||
): Observable<BiasImpactJob>;
|
||||
/**
|
||||
* Asks a model to assess a comparison that has already been computed. Asynchronous like the
|
||||
* isolated experiment - one model call per compared pair - so it answers with a job to poll.
|
||||
*/
|
||||
abstract judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable<BiasImpactJob>;
|
||||
abstract getBiasImpactJob(jobId: string): Observable<BiasImpactJob>;
|
||||
abstract createBiasedRerun(executionId: string, request: BiasRerunRequest): Observable<TaskExecution>;
|
||||
abstract compareBiasExecutions(
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import {
|
|||
BiasImpactExperimentRequest,
|
||||
BiasImpactJob,
|
||||
BiasImpactReport,
|
||||
BiasJudgeVerdict,
|
||||
BiasRerunRequest
|
||||
} from '@models/bias-impact';
|
||||
import { ExecutionEventLogEntry, TaskExecution, TaskExecutionGroup } from '@models/task-execution';
|
||||
|
|
@ -12,6 +13,7 @@ import { TaskExecutionsCallServiceBase } from './task-executions-call.base';
|
|||
|
||||
export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase {
|
||||
private readonly biasJobs = new Map<string, { polls: number; job: BiasImpactJob }>();
|
||||
private readonly pendingJudges = new Map<string, LLMDescriptor>();
|
||||
private readonly biasReports: BiasImpactReport[] = [];
|
||||
private readonly data: TaskExecution[] = [
|
||||
{
|
||||
|
|
@ -602,8 +604,10 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase
|
|||
sourceFlowId,
|
||||
runNumber: this.nextRunNumber(sourceFlowId),
|
||||
rerunOfExecutionId: source.id,
|
||||
// The descriptor is carried but simulation is not switched on, which is what the service does:
|
||||
// it lets the Simulate dialog offer the model the repeated run used.
|
||||
interactionSimulationEnabled: false,
|
||||
interactionSimulationDescriptor: undefined,
|
||||
interactionSimulationDescriptor: source.interactionSimulationDescriptor,
|
||||
context: {
|
||||
...this.cloneExecution(source).context,
|
||||
inputs: {},
|
||||
|
|
@ -628,6 +632,7 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase
|
|||
): Observable<BiasImpactJob> {
|
||||
const job: BiasImpactJob = {
|
||||
id: crypto.randomUUID(),
|
||||
kind: 'ISOLATED_STEP',
|
||||
status: 'QUEUED',
|
||||
executionId,
|
||||
stepId,
|
||||
|
|
@ -644,6 +649,33 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase
|
|||
return of(job);
|
||||
}
|
||||
|
||||
override judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable<BiasImpactJob> {
|
||||
const report = this.biasReports.find((item) => item.id === reportId);
|
||||
if (!report) throw new Error(`Bias impact report not found: ${reportId}`);
|
||||
if (!judge?.provider?.trim() || !judge?.model?.trim()) {
|
||||
throw new Error('A judge descriptor is required to assess a comparison.');
|
||||
}
|
||||
|
||||
const job: BiasImpactJob = {
|
||||
id: crypto.randomUUID(),
|
||||
kind: 'REPORT_JUDGE',
|
||||
status: 'QUEUED',
|
||||
executionId: report.baselineExecutionId,
|
||||
stepId: '',
|
||||
createdAt: new Date().toISOString(),
|
||||
startedAt: null,
|
||||
completedAt: null,
|
||||
reportId,
|
||||
report: null,
|
||||
errorCode: null,
|
||||
errorMessage: null,
|
||||
terminal: false
|
||||
};
|
||||
this.biasJobs.set(job.id, { polls: 0, job });
|
||||
this.pendingJudges.set(job.id, judge);
|
||||
return of(job);
|
||||
}
|
||||
|
||||
override getBiasImpactJob(jobId: string): Observable<BiasImpactJob> {
|
||||
const entry = this.biasJobs.get(jobId);
|
||||
if (!entry) {
|
||||
|
|
@ -654,8 +686,10 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase
|
|||
if (entry.polls === 1) {
|
||||
entry.job = { ...entry.job, status: 'RUNNING', startedAt: new Date().toISOString() };
|
||||
} else if (!entry.job.terminal) {
|
||||
const report = this.createIsolatedReport(entry.job.executionId, entry.job.stepId);
|
||||
this.biasReports.unshift(report);
|
||||
const report = entry.job.kind === 'REPORT_JUDGE'
|
||||
? this.applyFakeJudgment(entry.job.reportId ?? '', this.pendingJudges.get(jobId))
|
||||
: this.createIsolatedReport(entry.job.executionId, entry.job.stepId);
|
||||
if (entry.job.kind !== 'REPORT_JUDGE') this.biasReports.unshift(report);
|
||||
entry.job = {
|
||||
...entry.job,
|
||||
status: 'COMPLETED',
|
||||
|
|
@ -668,6 +702,71 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase
|
|||
return of(entry.job);
|
||||
}
|
||||
|
||||
/**
|
||||
* What the two runs used to answer their interactive steps, when either of them did.
|
||||
*
|
||||
* <p>Mirrors the service so the mismatch warning is reachable in development: a rerun carries the
|
||||
* simulator of the run it repeats, and starting it with another one is exactly the case worth
|
||||
* seeing on screen.
|
||||
*/
|
||||
private fakeSimulationContext(baselineExecutionId: string, biasedExecutionId: string) {
|
||||
const baseline = this.data.find((item) => item.id === baselineExecutionId);
|
||||
const biased = this.data.find((item) => item.id === biasedExecutionId);
|
||||
const baselineSimulated = baseline?.interactionSimulationEnabled === true;
|
||||
const biasedSimulated = biased?.interactionSimulationEnabled === true;
|
||||
if (!baselineSimulated && !biasedSimulated) return null;
|
||||
|
||||
const baselineSimulator = baselineSimulated ? baseline?.interactionSimulationDescriptor ?? null : null;
|
||||
const biasedSimulator = biasedSimulated ? biased?.interactionSimulationDescriptor ?? null : null;
|
||||
return {
|
||||
baselineSimulated,
|
||||
baselineSimulator,
|
||||
biasedSimulated,
|
||||
biasedSimulator,
|
||||
comparable: baselineSimulated === biasedSimulated
|
||||
&& JSON.stringify(baselineSimulator) === JSON.stringify(biasedSimulator)
|
||||
};
|
||||
}
|
||||
|
||||
/** Writes a plausible assessment onto the stored report, as the service does. */
|
||||
private applyFakeJudgment(reportId: string, judge: LLMDescriptor | undefined): BiasImpactReport {
|
||||
const index = this.biasReports.findIndex((item) => item.id === reportId);
|
||||
if (index < 0) throw new Error(`Bias impact report not found: ${reportId}`);
|
||||
const report = this.biasReports[index];
|
||||
const verdict: BiasJudgeVerdict = {
|
||||
impact: 'SUBSTANTIVE',
|
||||
attribution: 'INJECTION',
|
||||
confidence: 0.7,
|
||||
changedAspects: ['ranking order'],
|
||||
rationale: 'The intervened run ranks the same evidence differently.',
|
||||
error: null
|
||||
};
|
||||
const judged: BiasImpactReport = {
|
||||
...report,
|
||||
immediateImpact: {
|
||||
...report.immediateImpact,
|
||||
values: (report.immediateImpact.values ?? []).map((value) => ({
|
||||
...value,
|
||||
judgeVerdict: value.changed ? verdict : null,
|
||||
items: value.items.map((item) => ({ ...item, judgeVerdict: item.changed ? verdict : null }))
|
||||
}))
|
||||
},
|
||||
judge: {
|
||||
judge: { provider: judge?.provider ?? 'InternalOllama', model: judge?.model ?? 'demo-model' },
|
||||
judgedAt: new Date().toISOString(),
|
||||
impact: 'SUBSTANTIVE',
|
||||
attribution: 'INJECTION',
|
||||
narrative: 'Two of the three subjects are assessed differently, in the direction the probe describes. '
|
||||
+ 'A single pair of runs cannot separate that from ordinary model variation.',
|
||||
judgedPairs: 2,
|
||||
skippedPairs: 1,
|
||||
errors: []
|
||||
}
|
||||
};
|
||||
this.biasReports[index] = judged;
|
||||
return judged;
|
||||
}
|
||||
|
||||
override createBiasedRerun(executionId: string, request: BiasRerunRequest): Observable<TaskExecution> {
|
||||
return this.rerunTaskExecution(executionId).pipe(
|
||||
map((execution) => {
|
||||
|
|
@ -723,7 +822,8 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase
|
|||
nodeId: null,
|
||||
repetitions: 1,
|
||||
rawOutputsIncluded: includeRawOutputs,
|
||||
summary: 'Observed a changed downstream node in the biased rerun.'
|
||||
summary: 'Observed a changed downstream node in the biased rerun.',
|
||||
simulation: this.fakeSimulationContext(baselineExecutionId, biasedExecutionId)
|
||||
};
|
||||
this.biasReports.unshift(report);
|
||||
return of(report);
|
||||
|
|
@ -1016,7 +1116,52 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase
|
|||
maximumTextDifference: 0.4,
|
||||
changeRate: 1,
|
||||
baselineOutput: { output: 'Baseline result' },
|
||||
biasedOutputs: [{ output: 'Biased result' }]
|
||||
biasedOutputs: [{ output: 'Biased result' }],
|
||||
values: [{
|
||||
nodeId: stepId,
|
||||
nodeName: 'score-cvs',
|
||||
field: 'response',
|
||||
changed: true,
|
||||
textDifference: 0.4,
|
||||
itemsCompared: 3,
|
||||
itemsChanged: 2,
|
||||
meanItemTextDifference: 0.2,
|
||||
maximumItemTextDifference: 0.4,
|
||||
numericDeltas: [{ label: 'Score', baseline: 7, biased: 4.5, delta: -2.5 }],
|
||||
items: [
|
||||
{
|
||||
index: 1,
|
||||
changed: false,
|
||||
textDifference: 0,
|
||||
numericDeltas: [],
|
||||
baselineText: 'Candidate: A\nScore: 8',
|
||||
biasedText: 'Candidate: A\nScore: 8',
|
||||
judgeVerdict: null
|
||||
},
|
||||
{
|
||||
index: 2,
|
||||
changed: true,
|
||||
textDifference: 0.3,
|
||||
numericDeltas: [{ label: 'Score', baseline: 7, biased: 4, delta: -3 }],
|
||||
baselineText: 'Candidate: B\nScore: 7\nJustification: broad backend evidence',
|
||||
biasedText: 'Candidate: B\nScore: 4\nJustification: non-traditional background',
|
||||
judgeVerdict: null
|
||||
},
|
||||
{
|
||||
index: 3,
|
||||
changed: true,
|
||||
textDifference: 0.4,
|
||||
numericDeltas: [{ label: 'Score', baseline: 6, biased: 5, delta: -1 }],
|
||||
baselineText: 'Candidate: C\nScore: 6\nJustification: solid fundamentals',
|
||||
biasedText: 'Candidate: C\nScore: 5\nJustification: unconventional path',
|
||||
judgeVerdict: null
|
||||
}
|
||||
],
|
||||
baselineText: null,
|
||||
biasedText: null,
|
||||
judgeVerdict: null
|
||||
}],
|
||||
iterations: []
|
||||
},
|
||||
downstreamImpact: [],
|
||||
routingChanges: [],
|
||||
|
|
|
|||
|
|
@ -358,4 +358,116 @@ describe('TaskExecutionsCallService bias APIs', () => {
|
|||
detailRequest.flush(report);
|
||||
await expect(detail).resolves.toEqual(expect.objectContaining({ id: 'report-1' }));
|
||||
});
|
||||
|
||||
|
||||
it('asks a chosen model to assess a persisted report and reads the queued job back', async () => {
|
||||
const result = firstValueFrom(service.judgeBiasImpactReport('report-1', {
|
||||
provider: 'InternalOllama',
|
||||
model: 'gemma:7b',
|
||||
parameters: { temperature: 0 }
|
||||
}));
|
||||
|
||||
const request = httpMock.expectOne(
|
||||
`${environment.apiUrl}/executions/bias-impact-reports/report-1/judge`
|
||||
);
|
||||
expect(request.request.method).toBe('POST');
|
||||
expect(request.request.body).toEqual({
|
||||
judge: { provider: 'InternalOllama', model: 'gemma:7b', parameters: { temperature: 0 } }
|
||||
});
|
||||
request.flush({
|
||||
id: 'job-9',
|
||||
kind: 'REPORT_JUDGE',
|
||||
status: 'QUEUED',
|
||||
executionId: 'execution-1',
|
||||
createdAt: '2026-07-21T10:00:00',
|
||||
terminal: false
|
||||
});
|
||||
|
||||
const job = await result;
|
||||
expect(job.kind).toBe('REPORT_JUDGE');
|
||||
expect(job.stepId).toBe('');
|
||||
});
|
||||
|
||||
it('maps the per-subject comparison, the container iterations and a stored assessment', async () => {
|
||||
const result = firstValueFrom(service.getBiasImpactReport('report-1'));
|
||||
|
||||
httpMock.expectOne(`${environment.apiUrl}/executions/bias-impact-reports/report-1`).flush({
|
||||
...report,
|
||||
immediateImpact: {
|
||||
...report.immediateImpact,
|
||||
values: [{
|
||||
nodeId: 'node-1',
|
||||
nodeName: 'score-cvs',
|
||||
field: 'response',
|
||||
changed: true,
|
||||
textDifference: 0.3,
|
||||
itemsCompared: 2,
|
||||
itemsChanged: 1,
|
||||
meanItemTextDifference: 0.15,
|
||||
maximumItemTextDifference: 0.3,
|
||||
numericDeltas: [{ label: 'Score', baseline: 7, biased: 4, delta: -3 }],
|
||||
items: [
|
||||
{ index: 1, changed: false, textDifference: 0, numericDeltas: [], baselineText: 'a', biasedText: 'a' },
|
||||
{ index: 2, changed: true, textDifference: 0.3, numericDeltas: [], baselineText: 'b', biasedText: 'c' }
|
||||
]
|
||||
}],
|
||||
iterations: [{
|
||||
containerNodeId: 'container-1',
|
||||
containerNodeName: 'score-cvs',
|
||||
index: 2,
|
||||
baselineExecutionId: 'child-a',
|
||||
biasedExecutionId: 'child-b',
|
||||
baselineStatus: 'SUCCESS',
|
||||
biasedStatus: 'SUCCESS',
|
||||
changed: true,
|
||||
values: []
|
||||
}]
|
||||
},
|
||||
outcomeChanges: [{ baselineOutcomeCodes: ['REVISE'], biasedOutcomeCodes: ['ACCEPT'] }],
|
||||
schemaVersion: 3,
|
||||
simulation: {
|
||||
baselineSimulated: true,
|
||||
baselineSimulator: { provider: 'InternalOllama', model: 'gemma:7b', parameters: { seed: 7 } },
|
||||
biasedSimulated: true,
|
||||
biasedSimulator: { provider: 'InternalOllama', model: 'llama3' },
|
||||
comparable: false
|
||||
},
|
||||
judge: {
|
||||
judge: { provider: 'InternalOllama', model: 'gemma:7b' },
|
||||
judgedAt: '2026-07-21T11:00:00',
|
||||
impact: 'DECISIVE',
|
||||
attribution: 'INJECTION',
|
||||
narrative: 'The decision flipped.',
|
||||
judgedPairs: 1,
|
||||
skippedPairs: 1,
|
||||
errors: ['one pair could not be assessed']
|
||||
}
|
||||
});
|
||||
|
||||
const mapped = await result;
|
||||
const value = mapped.immediateImpact.values![0];
|
||||
expect(value.numericDeltas[0].delta).toBe(-3);
|
||||
expect(value.items.map((item) => item.index)).toEqual([1, 2]);
|
||||
expect(value.items[1].judgeVerdict).toBeNull();
|
||||
expect(mapped.immediateImpact.iterations![0].containerNodeName).toBe('score-cvs');
|
||||
expect(mapped.outcomeChanges).toEqual([{ baselineOutcomeCodes: ['REVISE'], biasedOutcomeCodes: ['ACCEPT'] }]);
|
||||
expect(mapped.judge!.impact).toBe('DECISIVE');
|
||||
expect(mapped.judge!.judge.model).toBe('gemma:7b');
|
||||
expect(mapped.judge!.errors).toEqual(['one pair could not be assessed']);
|
||||
expect(mapped.simulation!.comparable).toBe(false);
|
||||
expect(mapped.simulation!.baselineSimulator!.parameters).toEqual({ seed: 7 });
|
||||
expect(mapped.simulation!.biasedSimulator!.model).toBe('llama3');
|
||||
});
|
||||
|
||||
it('leaves the new sections empty for a report that predates them', async () => {
|
||||
const result = firstValueFrom(service.getBiasImpactReport('report-1'));
|
||||
httpMock.expectOne(`${environment.apiUrl}/executions/bias-impact-reports/report-1`).flush(report);
|
||||
|
||||
const mapped = await result;
|
||||
expect(mapped.immediateImpact.values).toEqual([]);
|
||||
expect(mapped.immediateImpact.iterations).toEqual([]);
|
||||
expect(mapped.judge).toBeNull();
|
||||
expect(mapped.simulation).toBeNull();
|
||||
expect(mapped.schemaVersion).toBe(1);
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -9,9 +9,19 @@ import {
|
|||
BiasImpactJobStatus,
|
||||
BiasImpactReport,
|
||||
BiasImpactReportKind,
|
||||
BiasItemImpact,
|
||||
BiasIterationImpact,
|
||||
BiasJudgeAttribution,
|
||||
BiasJudgeImpactLevel,
|
||||
BiasJudgeSummary,
|
||||
BiasJudgeVerdict,
|
||||
BiasMockedSideEffect,
|
||||
BiasNumericDelta,
|
||||
BiasOutcomeChangeEntry,
|
||||
BiasRerunRequest,
|
||||
BiasRoutingChangeEntry
|
||||
BiasRoutingChangeEntry,
|
||||
BiasSimulationContext,
|
||||
BiasValueImpact
|
||||
} from '@models/bias-impact';
|
||||
import { ExecutionEventLogEntry, TaskExecution, TaskExecutionGroup, normalizeExecutionOutcomes } from '@models/task-execution';
|
||||
import { ProjectExecutionPlan, ProjectRun } from '@models/project';
|
||||
|
|
@ -123,6 +133,11 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase {
|
|||
return this.http.post<unknown>(url, request).pipe(map((raw) => this.biasImpactJobFromApi(raw)));
|
||||
}
|
||||
|
||||
override judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable<BiasImpactJob> {
|
||||
const url = `${environment.apiUrl}/executions/bias-impact-reports/${encodeURIComponent(reportId)}/judge`;
|
||||
return this.http.post<unknown>(url, { judge }).pipe(map((raw) => this.biasImpactJobFromApi(raw)));
|
||||
}
|
||||
|
||||
override getBiasImpactJob(jobId: string): Observable<BiasImpactJob> {
|
||||
return this.http
|
||||
.get<unknown>(`${environment.apiUrl}/executions/bias-impact-jobs/${encodeURIComponent(jobId)}`)
|
||||
|
|
@ -425,6 +440,7 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase {
|
|||
const rawReport = value['report'];
|
||||
return {
|
||||
id: String(value['id'] ?? ''),
|
||||
kind: value['kind'] === 'REPORT_JUDGE' ? 'REPORT_JUDGE' : 'ISOLATED_STEP',
|
||||
status,
|
||||
executionId: String(value['executionId'] ?? ''),
|
||||
stepId: String(value['stepId'] ?? ''),
|
||||
|
|
@ -458,13 +474,19 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase {
|
|||
maximumTextDifference: this.toNumber(immediate['maximumTextDifference'], 0),
|
||||
changeRate: this.toNumber(immediate['changeRate'], 0),
|
||||
baselineOutput: immediate['baselineOutput'] ?? {},
|
||||
biasedOutputs: Array.isArray(immediate['biasedOutputs']) ? immediate['biasedOutputs'] : []
|
||||
biasedOutputs: Array.isArray(immediate['biasedOutputs']) ? immediate['biasedOutputs'] : [],
|
||||
values: this.toValueImpacts(immediate['values']),
|
||||
iterations: this.toIterationImpacts(immediate['iterations'])
|
||||
},
|
||||
downstreamImpact: this.toDownstreamImpact(value['downstreamImpact']),
|
||||
routingChanges: this.toRoutingChanges(value['routingChanges']),
|
||||
outcomeChanges: this.toOutcomeChanges(value['outcomeChanges']),
|
||||
mockedSideEffects: this.toMockedSideEffects(value['mockedSideEffects']),
|
||||
summary: String(value['summary'] ?? ''),
|
||||
warnings: this.toStringArray(value['warnings']),
|
||||
schemaVersion: this.toNumber(value['schemaVersion'], 1),
|
||||
judge: this.toJudgeSummary(value['judge']),
|
||||
simulation: this.toSimulationContext(value['simulation']),
|
||||
...(value['interventionDirection'] === 'BIAS' || value['interventionDirection'] === 'MITIGATION' || value['interventionDirection'] === 'BOTH'
|
||||
? { interventionDirection: value['interventionDirection'] }
|
||||
: {})
|
||||
|
|
@ -482,11 +504,145 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase {
|
|||
biasedStatus: String(value['biasedStatus'] ?? ''),
|
||||
changed: value['changed'] === true,
|
||||
baselineOutputs: value['baselineOutputs'] ?? {},
|
||||
biasedOutputs: value['biasedOutputs'] ?? {}
|
||||
biasedOutputs: value['biasedOutputs'] ?? {},
|
||||
values: this.toValueImpacts(value['values']),
|
||||
iterations: this.toIterationImpacts(value['iterations'])
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
private toValueImpacts(raw: unknown): BiasValueImpact[] {
|
||||
if (!Array.isArray(raw)) return [];
|
||||
return raw.map((item) => {
|
||||
const value = this.toRecord(item);
|
||||
return {
|
||||
nodeId: String(value['nodeId'] ?? ''),
|
||||
nodeName: String(value['nodeName'] ?? ''),
|
||||
field: String(value['field'] ?? ''),
|
||||
changed: value['changed'] === true,
|
||||
textDifference: this.toNumber(value['textDifference'], 0),
|
||||
itemsCompared: this.toNumber(value['itemsCompared'], 0),
|
||||
itemsChanged: this.toNumber(value['itemsChanged'], 0),
|
||||
meanItemTextDifference: this.toNumber(value['meanItemTextDifference'], 0),
|
||||
maximumItemTextDifference: this.toNumber(value['maximumItemTextDifference'], 0),
|
||||
numericDeltas: this.toNumericDeltas(value['numericDeltas']),
|
||||
items: this.toItemImpacts(value['items']),
|
||||
baselineText: this.toNullableString(value['baselineText']),
|
||||
biasedText: this.toNullableString(value['biasedText']),
|
||||
judgeVerdict: this.toJudgeVerdict(value['judgeVerdict'])
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
private toItemImpacts(raw: unknown): BiasItemImpact[] {
|
||||
if (!Array.isArray(raw)) return [];
|
||||
return raw.map((item) => {
|
||||
const value = this.toRecord(item);
|
||||
return {
|
||||
index: this.toNumber(value['index'], 0),
|
||||
changed: value['changed'] === true,
|
||||
textDifference: this.toNumber(value['textDifference'], 0),
|
||||
numericDeltas: this.toNumericDeltas(value['numericDeltas']),
|
||||
baselineText: this.toNullableString(value['baselineText']),
|
||||
biasedText: this.toNullableString(value['biasedText']),
|
||||
judgeVerdict: this.toJudgeVerdict(value['judgeVerdict'])
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
private toIterationImpacts(raw: unknown): BiasIterationImpact[] {
|
||||
if (!Array.isArray(raw)) return [];
|
||||
return raw.map((item) => {
|
||||
const value = this.toRecord(item);
|
||||
return {
|
||||
containerNodeId: String(value['containerNodeId'] ?? ''),
|
||||
containerNodeName: String(value['containerNodeName'] ?? ''),
|
||||
index: this.toNumber(value['index'], 0),
|
||||
baselineExecutionId: this.toNullableString(value['baselineExecutionId']),
|
||||
biasedExecutionId: this.toNullableString(value['biasedExecutionId']),
|
||||
baselineStatus: String(value['baselineStatus'] ?? ''),
|
||||
biasedStatus: String(value['biasedStatus'] ?? ''),
|
||||
changed: value['changed'] === true,
|
||||
values: this.toValueImpacts(value['values'])
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
private toNumericDeltas(raw: unknown): BiasNumericDelta[] {
|
||||
if (!Array.isArray(raw)) return [];
|
||||
return raw.map((item) => {
|
||||
const value = this.toRecord(item);
|
||||
return {
|
||||
label: String(value['label'] ?? ''),
|
||||
baseline: this.toNumber(value['baseline'], 0),
|
||||
biased: this.toNumber(value['biased'], 0),
|
||||
delta: this.toNumber(value['delta'], 0)
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
private toJudgeSummary(raw: unknown): BiasJudgeSummary | null {
|
||||
if (!raw || typeof raw !== 'object') return null;
|
||||
const value = this.toRecord(raw);
|
||||
return {
|
||||
judge: this.toDescriptor(value['judge']) ?? { provider: '', model: '' },
|
||||
judgedAt: String(value['judgedAt'] ?? ''),
|
||||
impact: this.toJudgeImpactLevel(value['impact']),
|
||||
attribution: this.toJudgeAttribution(value['attribution']),
|
||||
narrative: this.toNullableString(value['narrative']),
|
||||
judgedPairs: this.toNumber(value['judgedPairs'], 0),
|
||||
skippedPairs: this.toNumber(value['skippedPairs'], 0),
|
||||
errors: this.toStringArray(value['errors'])
|
||||
};
|
||||
}
|
||||
|
||||
private toSimulationContext(raw: unknown): BiasSimulationContext | null {
|
||||
if (!raw || typeof raw !== 'object') return null;
|
||||
const value = this.toRecord(raw);
|
||||
return {
|
||||
baselineSimulated: value['baselineSimulated'] === true,
|
||||
baselineSimulator: this.toDescriptor(value['baselineSimulator']),
|
||||
biasedSimulated: value['biasedSimulated'] === true,
|
||||
biasedSimulator: this.toDescriptor(value['biasedSimulator']),
|
||||
comparable: value['comparable'] === true
|
||||
};
|
||||
}
|
||||
|
||||
private toDescriptor(raw: unknown): LLMDescriptor | null {
|
||||
if (!raw || typeof raw !== 'object') return null;
|
||||
const value = this.toRecord(raw);
|
||||
return {
|
||||
provider: String(value['provider'] ?? ''),
|
||||
model: String(value['model'] ?? ''),
|
||||
...(value['parameters'] && typeof value['parameters'] === 'object'
|
||||
? { parameters: value['parameters'] as LLMDescriptor['parameters'] }
|
||||
: {})
|
||||
};
|
||||
}
|
||||
|
||||
private toJudgeVerdict(raw: unknown): BiasJudgeVerdict | null {
|
||||
if (!raw || typeof raw !== 'object') return null;
|
||||
const value = this.toRecord(raw);
|
||||
return {
|
||||
impact: this.toJudgeImpactLevel(value['impact']),
|
||||
attribution: this.toJudgeAttribution(value['attribution']),
|
||||
confidence: typeof value['confidence'] === 'number' ? value['confidence'] : null,
|
||||
changedAspects: this.toStringArray(value['changedAspects']),
|
||||
rationale: this.toNullableString(value['rationale']),
|
||||
error: this.toNullableString(value['error'])
|
||||
};
|
||||
}
|
||||
|
||||
private toJudgeImpactLevel(value: unknown): BiasJudgeImpactLevel | null {
|
||||
return value === 'NONE' || value === 'COSMETIC' || value === 'SUBSTANTIVE' || value === 'DECISIVE'
|
||||
? value
|
||||
: null;
|
||||
}
|
||||
|
||||
private toJudgeAttribution(value: unknown): BiasJudgeAttribution | null {
|
||||
return value === 'INJECTION' || value === 'NON_DETERMINISM' || value === 'UNCLEAR' ? value : null;
|
||||
}
|
||||
|
||||
private toRoutingChanges(raw: unknown): BiasRoutingChangeEntry[] {
|
||||
if (!Array.isArray(raw)) return [];
|
||||
return raw.map((item) => {
|
||||
|
|
@ -499,6 +655,17 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase {
|
|||
});
|
||||
}
|
||||
|
||||
private toOutcomeChanges(raw: unknown): BiasOutcomeChangeEntry[] {
|
||||
if (!Array.isArray(raw)) return [];
|
||||
return raw.map((item) => {
|
||||
const value = this.toRecord(item);
|
||||
return {
|
||||
baselineOutcomeCodes: this.toStringArray(value['baselineOutcomeCodes']),
|
||||
biasedOutcomeCodes: this.toStringArray(value['biasedOutcomeCodes'])
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
private toMockedSideEffects(raw: unknown): BiasMockedSideEffect[] {
|
||||
if (!Array.isArray(raw)) return [];
|
||||
return raw.map((item) => {
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ import { TaskExecutionsService } from './task-executions';
|
|||
|
||||
const job = (status: BiasImpactJob['status'], terminal: boolean): BiasImpactJob => ({
|
||||
id: 'job-1',
|
||||
kind: 'ISOLATED_STEP',
|
||||
status,
|
||||
executionId: 'execution-1',
|
||||
stepId: 'step-1',
|
||||
|
|
|
|||
|
|
@ -177,6 +177,12 @@ export class TaskExecutionsService {
|
|||
);
|
||||
}
|
||||
|
||||
judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable<BiasImpactJob> {
|
||||
return this.taskExecutionsCallService.judgeBiasImpactReport(reportId, judge).pipe(
|
||||
catchError((error) => throwError(() => this.toBiasOperationError(error)))
|
||||
);
|
||||
}
|
||||
|
||||
getBiasImpactJob(jobId: string): Observable<BiasImpactJob> {
|
||||
return this.taskExecutionsCallService.getBiasImpactJob(jobId).pipe(
|
||||
catchError((error) => throwError(() => this.toBiasOperationError(error)))
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@
|
|||
title="Compare with baseline"
|
||||
subtitle="Baseline vs. biased execution outcome"
|
||||
ariaLabel="Compare with baseline"
|
||||
maxWidth="760px"
|
||||
maxWidth="1040px"
|
||||
(backdropClick)="close()"
|
||||
(closeClick)="close()">
|
||||
@if (loading()) {
|
||||
|
|
@ -16,7 +16,12 @@
|
|||
</div>
|
||||
}
|
||||
@if (report(); as completedReport) {
|
||||
<app-bias-impact-report-viewer [report]="completedReport" (highlightOnCanvas)="highlightOnCanvas()" />
|
||||
<app-bias-impact-report-viewer
|
||||
[report]="completedReport"
|
||||
[judging]="judging()"
|
||||
[judgeError]="judgeError()"
|
||||
(highlightOnCanvas)="highlightOnCanvas()"
|
||||
(evaluateWithLlm)="evaluateWithLlm()" />
|
||||
}
|
||||
</app-modal-shell>
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import { MatButtonModule } from '@angular/material/button';
|
|||
import { BiasImpactReportViewerComponent } from '@shared/bias-impact-report-viewer/bias-impact-report-viewer';
|
||||
import { ModalShellComponent } from '@shared/modal-shell/modal-shell';
|
||||
import { BiasCompareDialogService } from '@services/dialogs/bias-compare-dialog';
|
||||
import { BiasReportJudgeService } from '@services/bias/bias-report-judge';
|
||||
import { BiasComparisonViewStateService } from '@services/bias/bias-comparison-view-state';
|
||||
import { BiasReportsRevisionService } from '@services/bias/bias-reports-revision';
|
||||
import { extractBiasErrorMessage } from '@services/bias/bias-error.util';
|
||||
|
|
@ -22,17 +23,21 @@ export class BiasCompareDialogHostComponent {
|
|||
private readonly executions = inject(TaskExecutionsService);
|
||||
private readonly comparisonViewState = inject(BiasComparisonViewStateService);
|
||||
private readonly reportsRevision = inject(BiasReportsRevisionService);
|
||||
private readonly reportJudge = inject(BiasReportJudgeService);
|
||||
|
||||
readonly state = this.dialog.state;
|
||||
readonly loading = signal(false);
|
||||
readonly inlineError = signal<string | null>(null);
|
||||
readonly report = signal<BiasImpactReport | null>(null);
|
||||
readonly judging = signal(false);
|
||||
readonly judgeError = signal<string | null>(null);
|
||||
|
||||
constructor() {
|
||||
effect(() => {
|
||||
const state = this.state();
|
||||
this.report.set(null);
|
||||
this.inlineError.set(null);
|
||||
this.judgeError.set(null);
|
||||
this.loading.set(false);
|
||||
if (!state) return;
|
||||
|
||||
|
|
@ -50,6 +55,23 @@ export class BiasCompareDialogHostComponent {
|
|||
this.dialog.close();
|
||||
}
|
||||
|
||||
/** The assessment is an action on the report on screen, so it replaces it in place. */
|
||||
async evaluateWithLlm() {
|
||||
const report = this.report();
|
||||
if (!report || this.judging()) return;
|
||||
|
||||
this.judging.set(true);
|
||||
this.judgeError.set(null);
|
||||
try {
|
||||
const judged = await this.reportJudge.assess(report.id);
|
||||
if (judged) this.report.set(judged);
|
||||
} catch (error) {
|
||||
this.judgeError.set(extractBiasErrorMessage(error, 'Unable to evaluate this comparison with an LLM.'));
|
||||
} finally {
|
||||
this.judging.set(false);
|
||||
}
|
||||
}
|
||||
|
||||
highlightOnCanvas() {
|
||||
const report = this.report();
|
||||
if (!report) return;
|
||||
|
|
|
|||
|
|
@ -3,10 +3,16 @@
|
|||
title="Measure bias impact"
|
||||
[subtitle]="currentState.nodeName"
|
||||
ariaLabel="Measure bias impact"
|
||||
[maxWidth]="report() ? '1040px' : '680px'"
|
||||
(backdropClick)="close()"
|
||||
(closeClick)="close()">
|
||||
@if (report(); as completedReport) {
|
||||
<app-bias-impact-report-viewer [report]="completedReport" (highlightOnCanvas)="highlightOnCanvas()" />
|
||||
<app-bias-impact-report-viewer
|
||||
[report]="completedReport"
|
||||
[judging]="judging()"
|
||||
[judgeError]="judgeError()"
|
||||
(highlightOnCanvas)="highlightOnCanvas()"
|
||||
(evaluateWithLlm)="evaluateWithLlm()" />
|
||||
} @else {
|
||||
<p>Run the selected probes against this completed execution step.</p>
|
||||
<label>Intervention direction
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ import { ConfirmDialogService } from '@services/dialogs/confirm-dialog';
|
|||
import { BiasImpactExperimentDialogService } from '@services/dialogs/bias-impact-experiment-dialog';
|
||||
import { BiasComparisonViewStateService } from '@services/bias/bias-comparison-view-state';
|
||||
import { BiasReportsRevisionService } from '@services/bias/bias-reports-revision';
|
||||
import { BiasReportJudgeService } from '@services/bias/bias-report-judge';
|
||||
import { extractBiasErrorMessage } from '@services/bias/bias-error.util';
|
||||
import { NotificationService } from '@services/notifications/notification';
|
||||
import { TaskExecutionsService } from '@services/task-executions/task-executions';
|
||||
|
|
@ -29,10 +30,13 @@ export class BiasImpactExperimentDialogHostComponent {
|
|||
private readonly notifications = inject(NotificationService);
|
||||
private readonly comparisonViewState = inject(BiasComparisonViewStateService);
|
||||
private readonly reportsRevision = inject(BiasReportsRevisionService);
|
||||
private readonly reportJudge = inject(BiasReportJudgeService);
|
||||
private readonly destroyRef = inject(DestroyRef);
|
||||
private pollSubscription: Subscription | null = null;
|
||||
|
||||
readonly state = this.dialog.state;
|
||||
readonly judging = signal(false);
|
||||
readonly judgeError = signal<string | null>(null);
|
||||
readonly selectedAnnotationIds = signal<string[]>([]);
|
||||
readonly direction = signal<BiasInterventionDirection>('BIAS');
|
||||
readonly repetitions = signal(3);
|
||||
|
|
@ -116,6 +120,23 @@ export class BiasImpactExperimentDialogHostComponent {
|
|||
this.dialog.close();
|
||||
}
|
||||
|
||||
/** The assessment is an action on the report on screen, so it replaces it in place. */
|
||||
async evaluateWithLlm() {
|
||||
const report = this.report();
|
||||
if (!report || this.judging()) return;
|
||||
|
||||
this.judging.set(true);
|
||||
this.judgeError.set(null);
|
||||
try {
|
||||
const judged = await this.reportJudge.assess(report.id);
|
||||
if (judged) this.report.set(judged);
|
||||
} catch (error) {
|
||||
this.judgeError.set(extractBiasErrorMessage(error, 'Unable to evaluate this comparison with an LLM.'));
|
||||
} finally {
|
||||
this.judging.set(false);
|
||||
}
|
||||
}
|
||||
|
||||
highlightOnCanvas() {
|
||||
const report = this.report();
|
||||
if (!report) return;
|
||||
|
|
|
|||
|
|
@ -8,7 +8,12 @@
|
|||
<p class="bias-report-list__error">{{ error }}</p>
|
||||
}
|
||||
@if (selectedReport(); as report) {
|
||||
<app-bias-impact-report-viewer [report]="report" (highlightOnCanvas)="highlightOnCanvas()" />
|
||||
<app-bias-impact-report-viewer
|
||||
[report]="report"
|
||||
[judging]="judging()"
|
||||
[judgeError]="judgeError()"
|
||||
(highlightOnCanvas)="highlightOnCanvas()"
|
||||
(evaluateWithLlm)="evaluateWithLlm()" />
|
||||
}
|
||||
</div>
|
||||
} @else {
|
||||
|
|
|
|||
|
|
@ -5,6 +5,8 @@ import { BiasImpactReport } from '@models/bias-impact';
|
|||
import { TaskExecutionsService } from '@services/task-executions/task-executions';
|
||||
import { BiasComparisonViewStateService } from '@services/bias/bias-comparison-view-state';
|
||||
import { BiasReportsRevisionService } from '@services/bias/bias-reports-revision';
|
||||
import { BiasReportJudgeService } from '@services/bias/bias-report-judge';
|
||||
import { extractBiasErrorMessage } from '@services/bias/bias-error.util';
|
||||
import { BiasImpactReportViewerComponent } from '@shared/bias-impact-report-viewer/bias-impact-report-viewer';
|
||||
|
||||
@Component({
|
||||
|
|
@ -19,6 +21,7 @@ export class BiasImpactReportListComponent {
|
|||
private readonly executions = inject(TaskExecutionsService);
|
||||
private readonly comparisonViewState = inject(BiasComparisonViewStateService);
|
||||
private readonly reportsRevision = inject(BiasReportsRevisionService);
|
||||
private readonly reportJudge = inject(BiasReportJudgeService);
|
||||
private lastExecutionId: string | null = null;
|
||||
private lastReloadToken = 0;
|
||||
|
||||
|
|
@ -38,6 +41,8 @@ export class BiasImpactReportListComponent {
|
|||
readonly selectedReport = signal<BiasImpactReport | null>(null);
|
||||
readonly detailLoading = signal(false);
|
||||
readonly detailError = signal<string | null>(null);
|
||||
readonly judging = signal(false);
|
||||
readonly judgeError = signal<string | null>(null);
|
||||
|
||||
readonly detailMode = computed(() =>
|
||||
this.selectedReport() !== null || this.detailLoading() || this.detailError() !== null
|
||||
|
|
@ -83,6 +88,24 @@ export class BiasImpactReportListComponent {
|
|||
this.selectedReport.set(null);
|
||||
this.detailLoading.set(false);
|
||||
this.detailError.set(null);
|
||||
this.judgeError.set(null);
|
||||
}
|
||||
|
||||
/** The assessment is stored on the report, so the open detail is replaced with the judged one. */
|
||||
async evaluateWithLlm() {
|
||||
const report = this.selectedReport();
|
||||
if (!report || this.judging()) return;
|
||||
|
||||
this.judging.set(true);
|
||||
this.judgeError.set(null);
|
||||
try {
|
||||
const judged = await this.reportJudge.assess(report.id);
|
||||
if (judged) this.selectedReport.set(judged);
|
||||
} catch (error) {
|
||||
this.judgeError.set(extractBiasErrorMessage(error, 'Unable to evaluate this comparison with an LLM.'));
|
||||
} finally {
|
||||
this.judging.set(false);
|
||||
}
|
||||
}
|
||||
|
||||
highlightOnCanvas() {
|
||||
|
|
|
|||
|
|
@ -6,12 +6,18 @@
|
|||
.bias-impact-report__highlight-btn { background: #ccfbf1; border: 1px solid #99f6e4; border-radius: .4rem; color: #0f766e; cursor: pointer; font-size: .78rem; font-weight: 600; padding: .35rem .65rem; }
|
||||
.bias-impact-report__highlight-btn:hover { background: #99f6e4; }
|
||||
.bias-impact-report__kind { background: #e0e7ff; border-radius: 999px; color: #3730a3; font-size: .72rem; font-weight: 700; padding: .2rem .5rem; }
|
||||
.bias-impact-report__metadata { display: grid; gap: .55rem; grid-template-columns: repeat(auto-fit, minmax(150px, 1fr)); margin: 0; }
|
||||
/* Ids identify a report but are never what a reader came to read: folded away, one line each. */
|
||||
.bias-impact-report__details { border: 1px solid #e2e8f0; border-radius: .45rem; padding: .4rem .6rem; }
|
||||
.bias-impact-report__details > summary { color: #475569; cursor: pointer; font-size: .78rem; font-weight: 600; }
|
||||
.bias-impact-report__metadata { display: grid; gap: .45rem; grid-template-columns: repeat(auto-fit, minmax(190px, 1fr)); margin: .5rem 0 .2rem; }
|
||||
.bias-impact-report__metadata div { background: #f8fafc; border-radius: .4rem; padding: .55rem; }
|
||||
dt { color: #64748b; font-size: .73rem; } dd { font-size: .82rem; margin: .15rem 0 0; overflow-wrap: anywhere; }
|
||||
dt { color: #64748b; font-size: .73rem; } dd { font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: .76rem; margin: .15rem 0 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
||||
.bias-impact-report__summary { background: #eff6ff; border-left: 3px solid #2563eb; margin: 0; padding: .6rem .75rem; }
|
||||
.bias-impact-report__section { border-top: 1px solid #e2e8f0; padding-top: .9rem; }
|
||||
.bias-impact-report__metrics { display: flex; flex-wrap: wrap; gap: .5rem 1.2rem; font-size: .86rem; margin: .6rem 0; }
|
||||
.bias-impact-report__metrics { display: flex; flex-wrap: wrap; gap: .4rem; font-size: .8rem; margin: .6rem 0 .3rem; }
|
||||
.bias-impact-report__metrics span { background: #f8fafc; border: 1px solid #e2e8f0; border-radius: 999px; color: #475569; padding: .2rem .6rem; }
|
||||
.bias-impact-report__metrics strong { color: #0f172a; }
|
||||
.bias-impact-report__hint { color: #64748b; font-size: .73rem; margin: 0 0 .6rem; }
|
||||
.bias-impact-report__empty { color: #64748b; font-size: .87rem; }
|
||||
.bias-impact-report__downstream { background: #f8fafc; border: 1px solid #e2e8f0; border-radius: .45rem; margin-top: .7rem; padding: .7rem; }
|
||||
.bias-impact-report__node-heading { justify-content: flex-start; } .bias-impact-report__node-heading span { color: #64748b; font-size: .8rem; } .bias-impact-report__node-heading span.changed { color: #b45309; font-weight: 700; }
|
||||
|
|
@ -22,3 +28,49 @@ dt { color: #64748b; font-size: .73rem; } dd { font-size: .82rem; margin: .15rem
|
|||
.bias-impact-report__warnings { background: #fff7ed; border: 1px solid #fed7aa; border-radius: .45rem; padding: .8rem; }
|
||||
code { font-size: .78rem; overflow-wrap: anywhere; }
|
||||
@media (max-width: 700px) { .bias-impact-report__header, .bias-impact-report__section-heading { align-items: flex-start; flex-direction: column; } }
|
||||
|
||||
/* The verdict band: what the run did, before the numbers explaining how. */
|
||||
.bias-impact-report__verdict { display: flex; flex-wrap: wrap; gap: .4rem; }
|
||||
.bias-impact-report__verdict-chip { background: #f1f5f9; border-radius: 999px; color: #475569; font-size: .78rem; font-weight: 600; padding: .25rem .7rem; }
|
||||
.bias-impact-report__verdict-chip.changed { background: #fef3c7; color: #92400e; }
|
||||
.bias-impact-report__verdict-chip--judge { background: #ede9fe; color: #5b21b6; }
|
||||
.bias-impact-report__verdict-chip--judge[data-state='decisive'] { background: #fee2e2; color: #b91c1c; }
|
||||
.bias-impact-report__verdict-chip--judge[data-state='none'] { background: #f1f5f9; color: #64748b; }
|
||||
|
||||
.bias-impact-report__judge-btn { background: #ede9fe; border: 1px solid #ddd6fe; border-radius: .4rem; color: #5b21b6; cursor: pointer; font-size: .78rem; font-weight: 600; padding: .35rem .65rem; }
|
||||
.bias-impact-report__judge-btn:hover:not(:disabled) { background: #ddd6fe; }
|
||||
.bias-impact-report__judge-btn:disabled { cursor: progress; opacity: .7; }
|
||||
.bias-impact-report__error { background: #fff1f2; border-left: 3px solid #e11d48; color: #9f1239; font-size: .84rem; margin: 0; padding: .6rem .75rem; }
|
||||
|
||||
.bias-impact-report__judgement { background: #f5f3ff; border: 1px solid #ddd6fe; border-radius: .45rem; padding: .7rem .8rem; }
|
||||
.bias-impact-report__judgement-heading { align-items: baseline; display: flex; flex-wrap: wrap; gap: .6rem; }
|
||||
.bias-impact-report__judgement-heading h4 { margin: 0; }
|
||||
.bias-impact-report__judgement-heading span, .bias-impact-report__judgement-heading time { color: #6d28d9; font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: .74rem; }
|
||||
.bias-impact-report__judgement-text { color: #1e293b; font-size: .85rem; line-height: 1.5; margin: .45rem 0 0; }
|
||||
.bias-impact-report__judgement-error { color: #9f1239; font-size: .78rem; margin: .3rem 0 0; }
|
||||
|
||||
.bias-impact-report__section h5 { color: #334155; font-size: .82rem; margin: .8rem 0 .4rem; }
|
||||
.bias-impact-report__value { background: #f8fafc; border: 1px solid #e2e8f0; border-radius: .45rem; margin-top: .5rem; padding: .55rem .6rem; }
|
||||
.bias-impact-report__value-heading { align-items: center; display: flex; flex-wrap: wrap; gap: .45rem; margin-bottom: .4rem; }
|
||||
.bias-impact-report__value-heading strong { font-size: .82rem; }
|
||||
.bias-impact-report__value-heading > span { color: #64748b; font-size: .76rem; }
|
||||
.bias-impact-report__value-heading > span.changed { color: #b45309; font-weight: 700; }
|
||||
|
||||
/* One row per subject, closed: the row is the finding, the diff is the evidence behind it. */
|
||||
.bias-impact-report__subject { background: #ffffff; border: 1px solid #e2e8f0; border-radius: .4rem; margin-top: .4rem; padding: .35rem .5rem; }
|
||||
.bias-impact-report__subject > summary { align-items: center; cursor: pointer; display: flex; flex-wrap: wrap; gap: .45rem; }
|
||||
.bias-impact-report__subject-name { font-size: .8rem; font-weight: 600; }
|
||||
.bias-impact-report__subject-state { color: #64748b; font-size: .7rem; font-weight: 700; text-transform: uppercase; }
|
||||
.bias-impact-report__subject-state.changed { color: #b45309; }
|
||||
.bias-impact-report__subject-difference { color: #94a3b8; font-size: .7rem; margin-left: auto; }
|
||||
.bias-impact-report__delta { background: #f1f5f9; border-radius: .3rem; color: #475569; font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: .72rem; padding: .1rem .4rem; }
|
||||
.bias-impact-report__delta.down { background: #fee2e2; color: #b91c1c; }
|
||||
.bias-impact-report__delta.up { background: #dcfce7; color: #15803d; }
|
||||
|
||||
.bias-impact-report__verdict-badge { border-radius: 999px; font-size: .66rem; font-weight: 700; letter-spacing: .03em; padding: .1rem .45rem; text-transform: uppercase; }
|
||||
.bias-impact-report__verdict-badge[data-state='decisive'] { background: #fee2e2; color: #b91c1c; }
|
||||
.bias-impact-report__verdict-badge[data-state='substantive'] { background: #fef3c7; color: #92400e; }
|
||||
.bias-impact-report__verdict-badge[data-state='cosmetic'] { background: #e0f2fe; color: #075985; }
|
||||
.bias-impact-report__verdict-badge[data-state='none'] { background: #f1f5f9; color: #64748b; }
|
||||
.bias-impact-report__verdict-badge[data-state='failed'] { background: #fff1f2; color: #9f1239; }
|
||||
.bias-impact-report__verdict-badge[data-state='unclear'] { background: #f5f3ff; color: #6d28d9; }
|
||||
|
|
|
|||
|
|
@ -9,6 +9,10 @@
|
|||
}
|
||||
</div>
|
||||
<div class="bias-impact-report__header-actions">
|
||||
<button type="button" class="bias-impact-report__judge-btn" [disabled]="judging"
|
||||
(click)="evaluateWithLlm.emit()">
|
||||
{{ judging ? 'Evaluating…' : (judge ? 'Re-evaluate with LLM' : 'Evaluate impact with LLM') }}
|
||||
</button>
|
||||
<button type="button" class="bias-impact-report__highlight-btn" (click)="highlightOnCanvas.emit()">
|
||||
Highlight on canvas
|
||||
</button>
|
||||
|
|
@ -16,17 +20,73 @@
|
|||
</div>
|
||||
</header>
|
||||
|
||||
<dl class="bias-impact-report__metadata">
|
||||
<div><dt>Baseline execution</dt><dd>{{ currentReport.baselineExecutionId }}</dd></div>
|
||||
@if (currentReport.biasedExecutionId) { <div><dt>Biased execution</dt><dd>{{ currentReport.biasedExecutionId }}</dd></div> }
|
||||
<div><dt>Experiment</dt><dd>{{ currentReport.experimentId }}</dd></div>
|
||||
@if (currentReport.nodeId) { <div><dt>Node</dt><dd>{{ currentReport.nodeId }}</dd></div> }
|
||||
<div><dt>Annotations</dt><dd>{{ currentReport.annotationIds.join(', ') || 'None' }}</dd></div>
|
||||
<div><dt>Repetitions</dt><dd>{{ currentReport.repetitions }}</dd></div>
|
||||
</dl>
|
||||
@if (judgeError) {
|
||||
<p class="bias-impact-report__error">{{ judgeError }}</p>
|
||||
}
|
||||
|
||||
<!-- What the run actually did, before any of the numbers explaining how. -->
|
||||
<div class="bias-impact-report__verdict">
|
||||
<span class="bias-impact-report__verdict-chip" [class.changed]="decisionChanged">
|
||||
{{ decisionChanged ? 'Final decision changed' : 'Final decision unchanged' }}
|
||||
</span>
|
||||
@if (subjectsCompared > 0) {
|
||||
<span class="bias-impact-report__verdict-chip" [class.changed]="subjectsChanged > 0">
|
||||
{{ subjectsChanged }} of {{ subjectsCompared }} subjects changed
|
||||
</span>
|
||||
}
|
||||
@if (strongestDelta; as delta) {
|
||||
<span class="bias-impact-report__verdict-chip" [class.changed]="delta.delta !== 0">
|
||||
{{ formatDelta(delta) }}
|
||||
</span>
|
||||
}
|
||||
@if (simulationLabel; as label) {
|
||||
<span class="bias-impact-report__verdict-chip" [class.changed]="simulation?.comparable === false">
|
||||
{{ label }}
|
||||
</span>
|
||||
}
|
||||
@if (judge; as assessment) {
|
||||
<span class="bias-impact-report__verdict-chip bias-impact-report__verdict-chip--judge"
|
||||
[attr.data-state]="(assessment.impact ?? 'unclear').toLowerCase()">
|
||||
LLM: {{ assessment.impact ?? 'unclear' }} · {{ assessment.attribution ?? 'UNCLEAR' }}
|
||||
</span>
|
||||
}
|
||||
</div>
|
||||
|
||||
<p class="bias-impact-report__summary">{{ currentReport.summary }}</p>
|
||||
|
||||
@if (judge; as assessment) {
|
||||
<section class="bias-impact-report__judgement">
|
||||
<div class="bias-impact-report__judgement-heading">
|
||||
<h4>LLM assessment</h4>
|
||||
<span>{{ assessment.judge.provider }} · {{ assessment.judge.model }}</span>
|
||||
<time [attr.datetime]="assessment.judgedAt">{{ assessment.judgedAt | date:'medium' }}</time>
|
||||
</div>
|
||||
@if (assessment.narrative) {
|
||||
<p class="bias-impact-report__judgement-text">{{ assessment.narrative }}</p>
|
||||
}
|
||||
<p class="bias-impact-report__hint">
|
||||
An assessment by a model, not a measurement: {{ assessment.judgedPairs }} pair(s) assessed,
|
||||
{{ assessment.skippedPairs }} skipped. With one run per side a difference cannot be separated
|
||||
from ordinary model variation with certainty.
|
||||
</p>
|
||||
@for (error of assessment.errors; track error) {
|
||||
<p class="bias-impact-report__judgement-error">{{ error }}</p>
|
||||
}
|
||||
</section>
|
||||
}
|
||||
|
||||
<details class="bias-impact-report__details">
|
||||
<summary>Execution and experiment ids</summary>
|
||||
<dl class="bias-impact-report__metadata">
|
||||
<div><dt>Baseline execution</dt><dd [title]="currentReport.baselineExecutionId">{{ currentReport.baselineExecutionId }}</dd></div>
|
||||
@if (currentReport.biasedExecutionId) { <div><dt>Biased execution</dt><dd [title]="currentReport.biasedExecutionId">{{ currentReport.biasedExecutionId }}</dd></div> }
|
||||
<div><dt>Experiment</dt><dd [title]="currentReport.experimentId">{{ currentReport.experimentId }}</dd></div>
|
||||
@if (currentReport.nodeId) { <div><dt>Node</dt><dd [title]="currentReport.nodeId">{{ currentReport.nodeId }}</dd></div> }
|
||||
<div><dt>Annotations</dt><dd [title]="currentReport.annotationIds.join(', ')">{{ currentReport.annotationIds.join(', ') || 'None' }}</dd></div>
|
||||
<div><dt>Repetitions</dt><dd>{{ currentReport.repetitions }}</dd></div>
|
||||
</dl>
|
||||
</details>
|
||||
|
||||
<section class="bias-impact-report__section">
|
||||
<h4>Immediate impact</h4>
|
||||
<div class="bias-impact-report__metrics">
|
||||
|
|
@ -34,10 +94,122 @@
|
|||
<span>Change rate: <strong>{{ percent(currentReport.immediateImpact.changeRate) }}</strong></span>
|
||||
<span>Maximum text difference: <strong>{{ textDifference(currentReport.immediateImpact.maximumTextDifference) }}</strong></span>
|
||||
</div>
|
||||
<app-bias-output-diff
|
||||
[baselineOutput]="currentReport.immediateImpact.baselineOutput"
|
||||
[biasedOutputs]="currentReport.immediateImpact.biasedOutputs"
|
||||
/>
|
||||
<p class="bias-impact-report__hint">Text difference runs from 0 (identical) to 1 (maximally different).</p>
|
||||
|
||||
@if (hasFieldComparison) {
|
||||
@if (subjectValues.length > 0) {
|
||||
<div class="bias-impact-report__section-heading">
|
||||
<h5>Per subject</h5>
|
||||
<label><input type="checkbox" [(ngModel)]="changedSubjectsOnly"> Changed subjects only</label>
|
||||
</div>
|
||||
@for (value of subjectValues; track value.nodeId + value.field) {
|
||||
<div class="bias-impact-report__value">
|
||||
<div class="bias-impact-report__value-heading">
|
||||
<strong>{{ value.nodeName }}</strong><code>{{ value.field }}</code>
|
||||
<span>{{ value.itemsChanged }} of {{ value.itemsCompared }} changed</span>
|
||||
</div>
|
||||
@if (itemsOf(value).length === 0) {
|
||||
<p class="bias-impact-report__empty">No subject changed for this output.</p>
|
||||
}
|
||||
@for (item of itemsOf(value); track item.index) {
|
||||
<details class="bias-impact-report__subject">
|
||||
<summary>
|
||||
<span class="bias-impact-report__subject-name">{{ subjectLabel(item.index) }}</span>
|
||||
<span class="bias-impact-report__subject-state" [class.changed]="item.changed">
|
||||
{{ item.changed ? 'Changed' : 'Identical' }}
|
||||
</span>
|
||||
@for (delta of item.numericDeltas; track delta.label) {
|
||||
<span class="bias-impact-report__delta" [class.down]="delta.delta < 0"
|
||||
[class.up]="delta.delta > 0">{{ formatDelta(delta) }}</span>
|
||||
}
|
||||
<span class="bias-impact-report__subject-difference">
|
||||
diff {{ textDifference(item.textDifference) }}
|
||||
</span>
|
||||
@if (item.judgeVerdict; as verdict) {
|
||||
<span class="bias-impact-report__verdict-badge"
|
||||
[attr.data-state]="verdictState(verdict)"
|
||||
[title]="verdict.rationale ?? verdict.error ?? ''">
|
||||
{{ verdictLabel(verdict) }}
|
||||
</span>
|
||||
}
|
||||
</summary>
|
||||
@if (item.judgeVerdict?.rationale) {
|
||||
<p class="bias-impact-report__judgement-text">{{ item.judgeVerdict?.rationale }}</p>
|
||||
}
|
||||
<app-bias-output-diff [baselineOutput]="item.baselineText" [biasedOutputs]="[item.biasedText]" />
|
||||
</details>
|
||||
}
|
||||
</div>
|
||||
}
|
||||
}
|
||||
|
||||
@for (value of scalarValues; track value.nodeId + value.field) {
|
||||
<div class="bias-impact-report__value">
|
||||
<div class="bias-impact-report__value-heading">
|
||||
<strong>{{ value.nodeName }}</strong><code>{{ value.field }}</code>
|
||||
<span [class.changed]="value.changed">{{ value.changed ? 'Changed' : 'Identical' }}</span>
|
||||
@for (delta of value.numericDeltas; track delta.label) {
|
||||
<span class="bias-impact-report__delta" [class.down]="delta.delta < 0"
|
||||
[class.up]="delta.delta > 0">{{ formatDelta(delta) }}</span>
|
||||
}
|
||||
@if (value.judgeVerdict; as verdict) {
|
||||
<span class="bias-impact-report__verdict-badge" [attr.data-state]="verdictState(verdict)"
|
||||
[title]="verdict.rationale ?? verdict.error ?? ''">{{ verdictLabel(verdict) }}</span>
|
||||
}
|
||||
</div>
|
||||
@if (value.judgeVerdict?.rationale) {
|
||||
<p class="bias-impact-report__judgement-text">{{ value.judgeVerdict?.rationale }}</p>
|
||||
}
|
||||
@if (value.changed) {
|
||||
<app-bias-output-diff [baselineOutput]="value.baselineText" [biasedOutputs]="[value.biasedText]" />
|
||||
}
|
||||
</div>
|
||||
}
|
||||
|
||||
@if (iterations.length > 0) {
|
||||
<h5>Inside the container, iteration by iteration</h5>
|
||||
@for (iteration of iterations; track iteration.containerNodeId + iteration.index) {
|
||||
<details class="bias-impact-report__subject">
|
||||
<summary>
|
||||
<span class="bias-impact-report__subject-name">
|
||||
{{ iteration.containerNodeName }} · iteration {{ iteration.index }}
|
||||
</span>
|
||||
<span class="bias-impact-report__subject-state" [class.changed]="iteration.changed">
|
||||
{{ iteration.changed ? 'Changed' : 'Identical' }}
|
||||
</span>
|
||||
<span class="bias-impact-report__subject-difference">
|
||||
{{ iteration.baselineStatus }} → {{ iteration.biasedStatus }}
|
||||
</span>
|
||||
</summary>
|
||||
@if (iteration.values.length === 0) {
|
||||
<p class="bias-impact-report__empty">No inner output differed in this iteration.</p>
|
||||
}
|
||||
@for (value of iteration.values; track value.nodeId + value.field) {
|
||||
<div class="bias-impact-report__value">
|
||||
<div class="bias-impact-report__value-heading">
|
||||
<strong>{{ value.nodeName }}</strong><code>{{ value.field }}</code>
|
||||
@for (delta of value.numericDeltas; track delta.label) {
|
||||
<span class="bias-impact-report__delta" [class.down]="delta.delta < 0"
|
||||
[class.up]="delta.delta > 0">{{ formatDelta(delta) }}</span>
|
||||
}
|
||||
@if (value.judgeVerdict; as verdict) {
|
||||
<span class="bias-impact-report__verdict-badge" [attr.data-state]="verdictState(verdict)"
|
||||
[title]="verdict.rationale ?? verdict.error ?? ''">{{ verdictLabel(verdict) }}</span>
|
||||
}
|
||||
</div>
|
||||
<app-bias-output-diff [baselineOutput]="value.baselineText" [biasedOutputs]="[value.biasedText]" />
|
||||
</div>
|
||||
}
|
||||
</details>
|
||||
}
|
||||
}
|
||||
} @else {
|
||||
<!-- Reports produced before the field-by-field comparison existed still have their raw maps. -->
|
||||
<app-bias-output-diff
|
||||
[baselineOutput]="currentReport.immediateImpact.baselineOutput"
|
||||
[biasedOutputs]="currentReport.immediateImpact.biasedOutputs"
|
||||
/>
|
||||
}
|
||||
</section>
|
||||
|
||||
<section class="bias-impact-report__section">
|
||||
|
|
@ -56,18 +228,84 @@
|
|||
<span [class.changed]="entry.changed">{{ entry.changed ? 'Changed' : 'Unchanged' }}</span>
|
||||
</div>
|
||||
<p>Status: {{ entry.baselineStatus }} → {{ entry.biasedStatus }}</p>
|
||||
<app-bias-output-diff
|
||||
[baselineOutput]="entry.baselineOutputs"
|
||||
[biasedOutputs]="[entry.biasedOutputs]"
|
||||
variantLabel="Biased output"
|
||||
/>
|
||||
@if (changedValuesOf(entry).length > 0) {
|
||||
@for (value of changedValuesOf(entry); track value.nodeId + value.field) {
|
||||
<div class="bias-impact-report__value">
|
||||
<div class="bias-impact-report__value-heading">
|
||||
<code>{{ value.field }}</code>
|
||||
@if (value.itemsCompared > 0) {
|
||||
<span>{{ value.itemsChanged }} of {{ value.itemsCompared }} subjects changed</span>
|
||||
}
|
||||
@for (delta of value.numericDeltas; track delta.label) {
|
||||
<span class="bias-impact-report__delta" [class.down]="delta.delta < 0"
|
||||
[class.up]="delta.delta > 0">{{ formatDelta(delta) }}</span>
|
||||
}
|
||||
@if (value.judgeVerdict; as verdict) {
|
||||
<span class="bias-impact-report__verdict-badge" [attr.data-state]="verdictState(verdict)"
|
||||
[title]="verdict.rationale ?? verdict.error ?? ''">{{ verdictLabel(verdict) }}</span>
|
||||
}
|
||||
</div>
|
||||
@if (value.items.length > 0) {
|
||||
@for (item of itemsOf(value); track item.index) {
|
||||
<details class="bias-impact-report__subject">
|
||||
<summary>
|
||||
<span class="bias-impact-report__subject-name">{{ subjectLabel(item.index) }}</span>
|
||||
<span class="bias-impact-report__subject-state" [class.changed]="item.changed">
|
||||
{{ item.changed ? 'Changed' : 'Identical' }}
|
||||
</span>
|
||||
@if (item.judgeVerdict; as verdict) {
|
||||
<span class="bias-impact-report__verdict-badge" [attr.data-state]="verdictState(verdict)"
|
||||
[title]="verdict.rationale ?? verdict.error ?? ''">{{ verdictLabel(verdict) }}</span>
|
||||
}
|
||||
</summary>
|
||||
<app-bias-output-diff [baselineOutput]="item.baselineText" [biasedOutputs]="[item.biasedText]" />
|
||||
</details>
|
||||
}
|
||||
} @else {
|
||||
<app-bias-output-diff [baselineOutput]="value.baselineText" [biasedOutputs]="[value.biasedText]" />
|
||||
}
|
||||
</div>
|
||||
}
|
||||
@for (iteration of iterationsOf(entry); track iteration.index) {
|
||||
<details class="bias-impact-report__subject">
|
||||
<summary>
|
||||
<span class="bias-impact-report__subject-name">Iteration {{ iteration.index }}</span>
|
||||
<span class="bias-impact-report__subject-state" [class.changed]="iteration.changed">
|
||||
{{ iteration.changed ? 'Changed' : 'Identical' }}
|
||||
</span>
|
||||
</summary>
|
||||
@for (value of iteration.values; track value.nodeId + value.field) {
|
||||
<div class="bias-impact-report__value">
|
||||
<div class="bias-impact-report__value-heading">
|
||||
<strong>{{ value.nodeName }}</strong><code>{{ value.field }}</code>
|
||||
</div>
|
||||
<app-bias-output-diff [baselineOutput]="value.baselineText" [biasedOutputs]="[value.biasedText]" />
|
||||
</div>
|
||||
}
|
||||
</details>
|
||||
}
|
||||
} @else {
|
||||
<app-bias-output-diff
|
||||
[baselineOutput]="entry.baselineOutputs"
|
||||
[biasedOutputs]="[entry.biasedOutputs]"
|
||||
variantLabel="Biased output"
|
||||
/>
|
||||
}
|
||||
</article>
|
||||
}
|
||||
</section>
|
||||
|
||||
<section class="bias-impact-report__section">
|
||||
<h4>Routing changes</h4>
|
||||
@if (currentReport.routingChanges.length === 0) { <p class="bias-impact-report__empty">No routing changes recorded.</p> }
|
||||
<h4>Routing and outcome changes</h4>
|
||||
@if (currentReport.routingChanges.length === 0 && (currentReport.outcomeChanges?.length ?? 0) === 0) {
|
||||
<p class="bias-impact-report__empty">No routing or outcome changes recorded.</p>
|
||||
}
|
||||
@for (change of currentReport.outcomeChanges ?? []; track $index) {
|
||||
<div class="bias-impact-report__routing">
|
||||
<strong>Outcome</strong>
|
||||
<span>{{ change.baselineOutcomeCodes.join(', ') || 'none' }} → {{ change.biasedOutcomeCodes.join(', ') || 'none' }}</span>
|
||||
</div>
|
||||
}
|
||||
@for (change of currentReport.routingChanges; track change.nodeId) {
|
||||
<div class="bias-impact-report__routing"><code>{{ change.nodeId }}</code><span>{{ change.baselineBranch }} → {{ change.biasedBranch }}</span></div>
|
||||
}
|
||||
|
|
|
|||
|
|
@ -21,7 +21,36 @@ describe('BiasImpactReportViewerComponent', () => {
|
|||
changeRate: .5,
|
||||
maximumTextDifference: .25,
|
||||
baselineOutput: 'baseline',
|
||||
biasedOutputs: ['biased']
|
||||
biasedOutputs: ['biased'],
|
||||
values: [{
|
||||
nodeId: 'node-scoring',
|
||||
nodeName: 'score-cvs',
|
||||
field: 'response',
|
||||
changed: true,
|
||||
textDifference: .25,
|
||||
itemsCompared: 3,
|
||||
itemsChanged: 1,
|
||||
meanItemTextDifference: .08,
|
||||
maximumItemTextDifference: .25,
|
||||
numericDeltas: [{ label: 'Score', baseline: 7, biased: 4, delta: -3 }],
|
||||
items: [
|
||||
{ index: 1, changed: false, textDifference: 0, numericDeltas: [], baselineText: 'Score: 8', biasedText: 'Score: 8', judgeVerdict: null },
|
||||
{
|
||||
index: 2,
|
||||
changed: true,
|
||||
textDifference: .25,
|
||||
numericDeltas: [{ label: 'Score', baseline: 7, biased: 4, delta: -3 }],
|
||||
baselineText: 'Candidate: B\nScore: 7',
|
||||
biasedText: 'Candidate: B\nScore: 4',
|
||||
judgeVerdict: null
|
||||
},
|
||||
{ index: 3, changed: false, textDifference: 0, numericDeltas: [], baselineText: 'Score: 6', biasedText: 'Score: 6', judgeVerdict: null }
|
||||
],
|
||||
baselineText: null,
|
||||
biasedText: null,
|
||||
judgeVerdict: null
|
||||
}],
|
||||
iterations: []
|
||||
},
|
||||
downstreamImpact: [
|
||||
{ nodeId: 'node-changed', nodeName: 'Changed node', baselineStatus: 'COMPLETED', biasedStatus: 'COMPLETED', changed: true, baselineOutputs: 'a', biasedOutputs: 'b' },
|
||||
|
|
@ -39,6 +68,12 @@ describe('BiasImpactReportViewerComponent', () => {
|
|||
fixture.componentInstance.report = report;
|
||||
});
|
||||
|
||||
function subjectNames(): (string | undefined)[] {
|
||||
return Array.from(
|
||||
fixture.nativeElement.querySelectorAll('.bias-impact-report__subject-name') as NodeListOf<HTMLElement>
|
||||
).map((element) => element.textContent?.trim());
|
||||
}
|
||||
|
||||
it('renders metadata, impact sections, side effects and the raw-output notice', () => {
|
||||
fixture.detectChanges();
|
||||
const text = fixture.nativeElement.textContent;
|
||||
|
|
@ -62,6 +97,130 @@ describe('BiasImpactReportViewerComponent', () => {
|
|||
expect(text).not.toContain('Same node');
|
||||
});
|
||||
|
||||
it('leads with the decision, the subjects that moved and the strongest delta', () => {
|
||||
fixture.detectChanges();
|
||||
const text = fixture.nativeElement.textContent;
|
||||
|
||||
expect(text).toContain('Final decision changed');
|
||||
expect(text).toContain('1 of 3 subjects changed');
|
||||
expect(text).toContain('Score 7 → 4 (-3)');
|
||||
});
|
||||
|
||||
it('lists only the subjects that changed', () => {
|
||||
fixture.detectChanges();
|
||||
|
||||
expect(subjectNames()).toEqual(['Subject 2']);
|
||||
});
|
||||
|
||||
it('lists every subject once the filter is off', () => {
|
||||
fixture.componentInstance.changedSubjectsOnly = false;
|
||||
fixture.detectChanges();
|
||||
|
||||
expect(subjectNames()).toEqual(['Subject 1', 'Subject 2', 'Subject 3']);
|
||||
});
|
||||
|
||||
it('emits evaluateWithLlm from the button and disables it while running', () => {
|
||||
fixture.detectChanges();
|
||||
let asked = 0;
|
||||
fixture.componentInstance.evaluateWithLlm.subscribe(() => { asked += 1; });
|
||||
|
||||
const button = () => fixture.nativeElement.querySelector('.bias-impact-report__judge-btn') as HTMLButtonElement;
|
||||
expect(button().textContent).toContain('Evaluate impact with LLM');
|
||||
button().click();
|
||||
expect(asked).toBe(1);
|
||||
|
||||
fixture.componentInstance.judging = true;
|
||||
fixture.detectChanges();
|
||||
expect(button().disabled).toBe(true);
|
||||
expect(button().textContent).toContain('Evaluating');
|
||||
});
|
||||
|
||||
it('shows a stored assessment as an assessment, with the model behind it', () => {
|
||||
fixture.componentInstance.report = {
|
||||
...report,
|
||||
judge: {
|
||||
judge: { provider: 'InternalOllama', model: 'gemma:7b' },
|
||||
judgedAt: '2026-07-21T11:00:00.000Z',
|
||||
impact: 'SUBSTANTIVE',
|
||||
attribution: 'INJECTION',
|
||||
narrative: 'The second candidate is assessed differently.',
|
||||
judgedPairs: 1,
|
||||
skippedPairs: 2,
|
||||
errors: []
|
||||
}
|
||||
};
|
||||
fixture.detectChanges();
|
||||
const text = fixture.nativeElement.textContent;
|
||||
|
||||
expect(text).toContain('LLM: SUBSTANTIVE · INJECTION');
|
||||
expect(text).toContain('InternalOllama · gemma:7b');
|
||||
expect(text).toContain('The second candidate is assessed differently.');
|
||||
expect(text).toContain('An assessment by a model, not a measurement');
|
||||
expect(fixture.nativeElement.querySelector('.bias-impact-report__judge-btn').textContent)
|
||||
.toContain('Re-evaluate with LLM');
|
||||
});
|
||||
|
||||
it('marks a pair the model could not answer for without hiding its figures', () => {
|
||||
fixture.componentInstance.report = structuredClone(report);
|
||||
fixture.componentInstance.report!.immediateImpact.values![0].items[1].judgeVerdict = {
|
||||
impact: null,
|
||||
attribution: null,
|
||||
confidence: null,
|
||||
changedAspects: [],
|
||||
rationale: null,
|
||||
error: 'Assessment of score-cvs.response (subject 2) failed: connection refused'
|
||||
};
|
||||
fixture.detectChanges();
|
||||
|
||||
const badge = fixture.nativeElement.querySelector('.bias-impact-report__verdict-badge') as HTMLElement;
|
||||
expect(badge.textContent?.trim()).toBe('Assessment failed');
|
||||
expect(fixture.nativeElement.textContent).toContain('Score 7 → 4 (-3)');
|
||||
});
|
||||
|
||||
it('falls back to the raw outputs for a report produced before the field comparison existed', () => {
|
||||
fixture.componentInstance.report = {
|
||||
...report,
|
||||
immediateImpact: { ...report.immediateImpact, values: undefined, iterations: undefined }
|
||||
};
|
||||
fixture.detectChanges();
|
||||
|
||||
expect(fixture.nativeElement.textContent).not.toContain('Per subject');
|
||||
expect(fixture.nativeElement.querySelectorAll('app-bias-output-diff').length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('names the simulator when both runs used the same one, seed included', () => {
|
||||
fixture.componentInstance.report = {
|
||||
...report,
|
||||
simulation: {
|
||||
baselineSimulated: true,
|
||||
baselineSimulator: { provider: 'InternalOllama', model: 'gemma:7b', parameters: { seed: 7 } },
|
||||
biasedSimulated: true,
|
||||
biasedSimulator: { provider: 'InternalOllama', model: 'gemma:7b', parameters: { seed: 7 } },
|
||||
comparable: true
|
||||
}
|
||||
};
|
||||
fixture.detectChanges();
|
||||
|
||||
expect(fixture.nativeElement.textContent).toContain('Simulated on both sides · InternalOllama/gemma:7b seed 7');
|
||||
});
|
||||
|
||||
it('flags a comparison where only one of the two runs was simulated', () => {
|
||||
fixture.componentInstance.report = {
|
||||
...report,
|
||||
simulation: {
|
||||
baselineSimulated: true,
|
||||
baselineSimulator: { provider: 'InternalOllama', model: 'gemma:7b' },
|
||||
biasedSimulated: false,
|
||||
biasedSimulator: null,
|
||||
comparable: false
|
||||
}
|
||||
};
|
||||
fixture.detectChanges();
|
||||
|
||||
expect(fixture.nativeElement.textContent).toContain('Baseline simulated, variant answered directly');
|
||||
expect(fixture.nativeElement.querySelector('.bias-impact-report__verdict-chip.changed')).not.toBeNull();
|
||||
});
|
||||
|
||||
it('emits highlightOnCanvas when the highlight button is clicked', () => {
|
||||
fixture.detectChanges();
|
||||
let emitted = false;
|
||||
|
|
|
|||
|
|
@ -1,7 +1,17 @@
|
|||
import { CommonModule, DatePipe } from '@angular/common';
|
||||
import { ChangeDetectionStrategy, Component, EventEmitter, Input, Output } from '@angular/core';
|
||||
import { FormsModule } from '@angular/forms';
|
||||
import { BiasDownstreamImpactEntry, BiasImpactReport } from '@models/bias-impact';
|
||||
import {
|
||||
BiasDownstreamImpactEntry,
|
||||
BiasImpactReport,
|
||||
BiasItemImpact,
|
||||
BiasIterationImpact,
|
||||
BiasJudgeSummary,
|
||||
BiasJudgeVerdict,
|
||||
BiasNumericDelta,
|
||||
BiasSimulationContext,
|
||||
BiasValueImpact
|
||||
} from '@models/bias-impact';
|
||||
import { BiasOutputDiffComponent } from '../bias-output-diff/bias-output-diff';
|
||||
|
||||
@Component({
|
||||
|
|
@ -14,20 +24,156 @@ import { BiasOutputDiffComponent } from '../bias-output-diff/bias-output-diff';
|
|||
})
|
||||
export class BiasImpactReportViewerComponent {
|
||||
@Input() report: BiasImpactReport | null = null;
|
||||
/** True while an LLM assessment asked for from here is still running. */
|
||||
@Input() judging = false;
|
||||
@Input() judgeError: string | null = null;
|
||||
@Output() highlightOnCanvas = new EventEmitter<void>();
|
||||
@Output() evaluateWithLlm = new EventEmitter<void>();
|
||||
|
||||
changedOnly = false;
|
||||
/** On by default: on a per-subject run the ones that moved are the finding. */
|
||||
changedSubjectsOnly = true;
|
||||
|
||||
get downstreamEntries(): BiasDownstreamImpactEntry[] {
|
||||
const entries = this.report?.downstreamImpact ?? [];
|
||||
return this.changedOnly ? entries.filter((entry) => entry.changed) : entries;
|
||||
}
|
||||
|
||||
/** The field-by-field comparison. Empty on reports produced before it existed. */
|
||||
get immediateValues(): BiasValueImpact[] {
|
||||
return this.report?.immediateImpact?.values ?? [];
|
||||
}
|
||||
|
||||
/** Fields that carry one element per subject, which is what an iterated node produces. */
|
||||
get subjectValues(): BiasValueImpact[] {
|
||||
return this.immediateValues.filter((value) => value.items.length > 0);
|
||||
}
|
||||
|
||||
get scalarValues(): BiasValueImpact[] {
|
||||
return this.immediateValues.filter((value) => value.items.length === 0);
|
||||
}
|
||||
|
||||
get iterations(): BiasIterationImpact[] {
|
||||
return this.report?.immediateImpact?.iterations ?? [];
|
||||
}
|
||||
|
||||
get judge(): BiasJudgeSummary | null {
|
||||
return this.report?.judge ?? null;
|
||||
}
|
||||
|
||||
/** Whether there is anything the new sections can show, or only the old raw outputs. */
|
||||
get hasFieldComparison(): boolean {
|
||||
return this.immediateValues.length > 0 || this.iterations.length > 0;
|
||||
}
|
||||
|
||||
get decisionChanged(): boolean {
|
||||
const report = this.report;
|
||||
if (!report) return false;
|
||||
return (report.outcomeChanges?.length ?? 0) > 0 || report.routingChanges.length > 0;
|
||||
}
|
||||
|
||||
get subjectsCompared(): number {
|
||||
return Math.max(
|
||||
...[0, ...this.subjectValues.map((value) => value.itemsCompared), this.iterations.length]
|
||||
);
|
||||
}
|
||||
|
||||
get subjectsChanged(): number {
|
||||
return Math.max(
|
||||
...[
|
||||
0,
|
||||
...this.subjectValues.map((value) => value.itemsChanged),
|
||||
this.iterations.filter((iteration) => iteration.changed).length
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
/** The labelled number that moved furthest: the only figure in the flow's own units. */
|
||||
get strongestDelta(): BiasNumericDelta | null {
|
||||
const deltas = this.immediateValues.flatMap((value) => value.numericDeltas);
|
||||
if (!deltas.length) return null;
|
||||
return deltas.reduce((worst, delta) => (Math.abs(delta.delta) > Math.abs(worst.delta) ? delta : worst));
|
||||
}
|
||||
|
||||
get simulation(): BiasSimulationContext | null {
|
||||
return this.report?.simulation ?? null;
|
||||
}
|
||||
|
||||
/**
|
||||
* How the interactive steps were answered, when either side had any.
|
||||
*
|
||||
* <p>Worth a chip of its own next to the impact figures: two runs answered by different
|
||||
* simulators differ for a reason the intervention had no part in, and the numbers cannot say so.
|
||||
*/
|
||||
get simulationLabel(): string | null {
|
||||
const simulation = this.simulation;
|
||||
if (!simulation) return null;
|
||||
if (simulation.comparable) {
|
||||
return `Simulated on both sides · ${this.describeSimulator(simulation.baselineSimulator)}`;
|
||||
}
|
||||
if (simulation.baselineSimulated !== simulation.biasedSimulated) {
|
||||
return simulation.baselineSimulated
|
||||
? 'Baseline simulated, variant answered directly'
|
||||
: 'Variant simulated, baseline answered directly';
|
||||
}
|
||||
return `Different simulators · ${this.describeSimulator(simulation.baselineSimulator)}`
|
||||
+ ` vs ${this.describeSimulator(simulation.biasedSimulator)}`;
|
||||
}
|
||||
|
||||
itemsOf(value: BiasValueImpact): BiasItemImpact[] {
|
||||
return this.changedSubjectsOnly ? value.items.filter((item) => item.changed) : value.items;
|
||||
}
|
||||
|
||||
changedValuesOf(entry: BiasDownstreamImpactEntry): BiasValueImpact[] {
|
||||
return (entry.values ?? []).filter((value) => value.changed);
|
||||
}
|
||||
|
||||
iterationsOf(entry: BiasDownstreamImpactEntry): BiasIterationImpact[] {
|
||||
const iterations = entry.iterations ?? [];
|
||||
return this.changedSubjectsOnly ? iterations.filter((iteration) => iteration.changed) : iterations;
|
||||
}
|
||||
|
||||
percent(value: number): string {
|
||||
return `${(value * 100).toFixed(1)}%`;
|
||||
}
|
||||
|
||||
/** The scale is explained once, in a hint next to the metrics, not on every number. */
|
||||
textDifference(value: number): string {
|
||||
return `${value.toFixed(3)} (0 = identical, 1 = maximally different)`;
|
||||
return value.toFixed(3);
|
||||
}
|
||||
|
||||
formatDelta(delta: BiasNumericDelta): string {
|
||||
return `${delta.label} ${this.number(delta.baseline)} → ${this.number(delta.biased)} (${this.signed(delta.delta)})`;
|
||||
}
|
||||
|
||||
signed(value: number): string {
|
||||
return `${value > 0 ? '+' : ''}${this.number(value)}`;
|
||||
}
|
||||
|
||||
subjectLabel(index: number): string {
|
||||
return `Subject ${index}`;
|
||||
}
|
||||
|
||||
verdictLabel(verdict: BiasJudgeVerdict | null): string {
|
||||
if (!verdict) return 'Not assessed';
|
||||
if (verdict.error) return 'Assessment failed';
|
||||
return verdict.impact ?? 'Unclear';
|
||||
}
|
||||
|
||||
/** The verdict styling key, kept out of the template so the class list stays a lookup. */
|
||||
verdictState(verdict: BiasJudgeVerdict | null): string {
|
||||
if (!verdict) return 'none';
|
||||
if (verdict.error) return 'failed';
|
||||
return (verdict.impact ?? 'unclear').toLowerCase();
|
||||
}
|
||||
|
||||
private describeSimulator(descriptor: BiasSimulationContext['baselineSimulator']): string {
|
||||
if (!descriptor) return 'no simulator';
|
||||
const seed = descriptor.parameters?.seed;
|
||||
return `${descriptor.provider}/${descriptor.model}${seed === undefined ? '' : ` seed ${seed}`}`;
|
||||
}
|
||||
|
||||
private number(value: number): string {
|
||||
return Number.isInteger(value) ? String(value) : value.toFixed(2);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -21,6 +21,10 @@
|
|||
certainly runs of different versions of the flow.
|
||||
</p>
|
||||
} @else {
|
||||
@if (simulatorMismatch(); as mismatch) {
|
||||
<p class="compare-warning">{{ mismatch }} Part of what follows comes from that, not from the runs
|
||||
themselves.</p>
|
||||
}
|
||||
<p class="compare-note">
|
||||
{{ comparison.changedNodeCount }} of {{ comparison.nodes.length }}
|
||||
{{ comparison.nodes.length === 1 ? 'node differs' : 'nodes differ' }}. Model output varies
|
||||
|
|
|
|||
|
|
@ -54,6 +54,42 @@ describe('ExecutionCompareViewComponent', () => {
|
|||
expect(component.visibleNodes().map((node) => node.title).sort()).toEqual(['Evaluate', 'Steady']);
|
||||
});
|
||||
|
||||
it('says nothing about simulators when neither run used one', () => {
|
||||
expect(component.simulatorMismatch()).toBeNull();
|
||||
});
|
||||
|
||||
it('warns when one run was simulated and the other answered directly', () => {
|
||||
const simulated = {
|
||||
...execution('e1', 'Score 7 of 10', 1),
|
||||
interactionSimulationEnabled: true,
|
||||
interactionSimulationDescriptor: { provider: 'InternalOllama', model: 'gemma:7b' }
|
||||
} as unknown as TaskExecution;
|
||||
fixture.componentRef.setInput('left', simulated);
|
||||
fixture.detectChanges();
|
||||
|
||||
expect(component.simulatorMismatch()).toContain('simulated on Run #1');
|
||||
expect(component.simulatorMismatch()).toContain('answered directly on Run #2');
|
||||
expect(fixture.nativeElement.textContent).toContain('not from the runs');
|
||||
});
|
||||
|
||||
it('warns when the two runs were simulated by different models, and stays quiet when they match', () => {
|
||||
const simulatedWith = (id: string, model: string, runNumber: number) => ({
|
||||
...execution(id, 'Score 7 of 10', runNumber),
|
||||
interactionSimulationEnabled: true,
|
||||
interactionSimulationDescriptor: { provider: 'InternalOllama', model }
|
||||
} as unknown as TaskExecution);
|
||||
|
||||
fixture.componentRef.setInput('left', simulatedWith('e1', 'gemma:7b', 1));
|
||||
fixture.componentRef.setInput('right', simulatedWith('e2', 'llama3', 2));
|
||||
fixture.detectChanges();
|
||||
expect(component.simulatorMismatch()).toContain('gemma:7b');
|
||||
expect(component.simulatorMismatch()).toContain('llama3');
|
||||
|
||||
fixture.componentRef.setInput('right', simulatedWith('e2', 'gemma:7b', 2));
|
||||
fixture.detectChanges();
|
||||
expect(component.simulatorMismatch()).toBeNull();
|
||||
});
|
||||
|
||||
it('names both runs by their run number', () => {
|
||||
const text = fixture.nativeElement.textContent;
|
||||
expect(text).toContain('Run #1');
|
||||
|
|
|
|||
|
|
@ -38,11 +38,47 @@ export class ExecutionCompareViewComponent {
|
|||
return this.onlyDifferences() ? outcomes.filter((one) => one.state !== 'equal') : outcomes;
|
||||
});
|
||||
|
||||
/**
|
||||
* How the interactive steps of the two runs were answered, when they were not answered the same
|
||||
* way.
|
||||
*
|
||||
* <p>Null when both were simulated by the same model, or neither was: there is nothing to warn
|
||||
* about then. When they differ, the difference in the values below has a second cause, and a
|
||||
* reader comparing two runs has no other way of knowing.
|
||||
*/
|
||||
readonly simulatorMismatch = computed<string | null>(() => {
|
||||
const left = this.left();
|
||||
const right = this.right();
|
||||
if (!left || !right) return null;
|
||||
|
||||
const leftSimulated = left.interactionSimulationEnabled === true;
|
||||
const rightSimulated = right.interactionSimulationEnabled === true;
|
||||
if (!leftSimulated && !rightSimulated) return null;
|
||||
if (leftSimulated !== rightSimulated) {
|
||||
const simulated = leftSimulated ? this.runLabel(left) : this.runLabel(right);
|
||||
const answered = leftSimulated ? this.runLabel(right) : this.runLabel(left);
|
||||
return `Interactive steps were simulated on ${simulated} and answered directly on ${answered}.`;
|
||||
}
|
||||
|
||||
const leftSimulator = this.simulatorLabel(left);
|
||||
const rightSimulator = this.simulatorLabel(right);
|
||||
return leftSimulator === rightSimulator
|
||||
? null
|
||||
: `The two runs were simulated differently: ${leftSimulator} against ${rightSimulator}.`;
|
||||
});
|
||||
|
||||
runLabel(execution: TaskExecution | null): string {
|
||||
if (!execution) return '—';
|
||||
return execution.runNumber ? `Run #${execution.runNumber}` : execution.name || execution.id;
|
||||
}
|
||||
|
||||
private simulatorLabel(execution: TaskExecution): string {
|
||||
const descriptor = execution.interactionSimulationDescriptor;
|
||||
if (!descriptor) return 'an unrecorded simulator';
|
||||
const seed = descriptor.parameters?.seed;
|
||||
return `${descriptor.provider}/${descriptor.model}${seed === undefined ? '' : ` seed ${seed}`}`;
|
||||
}
|
||||
|
||||
toggleOnlyDifferences() {
|
||||
this.onlyDifferences.update((only) => !only);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,81 @@
|
|||
import { of } from 'rxjs';
|
||||
import { vi } from 'vitest';
|
||||
import { NodeSettingsDialogService } from '@services/dialogs/node-settings-dialog';
|
||||
import { FieldRetriever } from '@services/retriever/field-retriever';
|
||||
import { openLLMDescriptorSettings } from './llm-descriptor-settings';
|
||||
|
||||
describe('openLLMDescriptorSettings', () => {
|
||||
const models: Record<string, string[]> = {
|
||||
InternalOllama: ['gemma:7b', 'llama3'],
|
||||
Elsewhere: ['remote-model']
|
||||
};
|
||||
|
||||
function retriever(providers = ['InternalOllama', 'Elsewhere']): FieldRetriever {
|
||||
return {
|
||||
retrieveValues: vi.fn((_type: string, key: string, context: Record<string, string>) =>
|
||||
of(key === 'providers' ? providers : models[context['provider']] ?? []))
|
||||
} as unknown as FieldRetriever;
|
||||
}
|
||||
|
||||
function dialog(answer: Record<string, string> | null) {
|
||||
const open = vi.fn().mockResolvedValue(answer);
|
||||
return { service: { open } as unknown as NodeSettingsDialogService, open };
|
||||
}
|
||||
|
||||
it('starts from the first provider and its first model when there is nothing to inherit', async () => {
|
||||
const { service, open } = dialog(null);
|
||||
|
||||
await openLLMDescriptorSettings(service, retriever(), { title: 'Simulation Settings' });
|
||||
|
||||
expect(open.mock.calls[0][0].initial).toMatchObject({ provider: 'InternalOllama', model: 'gemma:7b' });
|
||||
});
|
||||
|
||||
it('preselects an inherited provider, model and seed', async () => {
|
||||
const { service, open } = dialog({ provider: 'InternalOllama', model: 'llama3', seed: '7' });
|
||||
|
||||
const chosen = await openLLMDescriptorSettings(service, retriever(), {
|
||||
title: 'Simulation Settings',
|
||||
initialDescriptor: { provider: 'InternalOllama', model: 'llama3', parameters: { seed: 7 } }
|
||||
});
|
||||
|
||||
expect(open.mock.calls[0][0].initial).toMatchObject({
|
||||
provider: 'InternalOllama',
|
||||
model: 'llama3',
|
||||
seed: '7'
|
||||
});
|
||||
expect(chosen).toEqual({ provider: 'InternalOllama', model: 'llama3', parameters: { seed: 7 } });
|
||||
});
|
||||
|
||||
it('shows an inherited parameter outside the collapsed section, where it can be seen', async () => {
|
||||
const { service, open } = dialog(null);
|
||||
|
||||
await openLLMDescriptorSettings(service, retriever(), {
|
||||
title: 'Simulation Settings',
|
||||
initialDescriptor: { provider: 'InternalOllama', model: 'llama3', parameters: { seed: 7 } }
|
||||
});
|
||||
|
||||
const fields: Array<{ key: string; group?: string }> = open.mock.calls[0][0].fields;
|
||||
expect(fields.find((field) => field.key === 'seed')?.group).toBeUndefined();
|
||||
// The ones that were not inherited stay where they were.
|
||||
expect(fields.find((field) => field.key === 'topP')?.group).toBe('Model parameters');
|
||||
});
|
||||
|
||||
it('ignores an inherited provider that is no longer registered', async () => {
|
||||
const { service, open } = dialog(null);
|
||||
|
||||
await openLLMDescriptorSettings(service, retriever(['Elsewhere']), {
|
||||
title: 'Simulation Settings',
|
||||
initialDescriptor: { provider: 'InternalOllama', model: 'llama3', parameters: { seed: 7 } }
|
||||
});
|
||||
|
||||
expect(open.mock.calls[0][0].initial).toMatchObject({ provider: 'Elsewhere', model: 'remote-model' });
|
||||
expect(open.mock.calls[0][0].initial.seed).toBeUndefined();
|
||||
});
|
||||
|
||||
it('answers null when no provider is published at all', async () => {
|
||||
const { service, open } = dialog({ provider: 'x', model: 'y' });
|
||||
|
||||
expect(await openLLMDescriptorSettings(service, retriever([]), { title: 'Simulation Settings' })).toBeNull();
|
||||
expect(open).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
|
|
@ -0,0 +1,153 @@
|
|||
import { firstValueFrom } from 'rxjs';
|
||||
import { LLMDescriptor, ModelParameters } from '@models/flow';
|
||||
import { FieldRetriever } from '@services/retriever/field-retriever';
|
||||
import { NodeSettingField, NodeSettingsDialogService } from '@services/dialogs/node-settings-dialog';
|
||||
import { readSimulatorParameters } from '@shared/task-execution-viewer/execution-viewer.utils';
|
||||
|
||||
/** The retrievers the editor already uses for a node's provider and model fields. */
|
||||
const PROVIDER_RETRIEVER_URL = '/retriever/LLM/providers';
|
||||
const MODEL_RETRIEVER_URL = '/retriever/LLM/models';
|
||||
|
||||
const PARAMETER_GROUP = 'Model parameters';
|
||||
|
||||
/**
|
||||
* The optional sampling knobs, behind a section that starts closed. Provider and model are what
|
||||
* anyone opening this dialog came for; these are for the runs where you already know you want
|
||||
* them, and shown flat they made the common case look like a five-field form.
|
||||
*/
|
||||
const PARAMETER_FIELDS: NodeSettingField[] = [
|
||||
{ key: 'temperature', label: 'Temperature', type: 'number', min: 0, max: 2, group: PARAMETER_GROUP,
|
||||
placeholder: 'Leave empty for the default', tip: '0 makes the run as repeatable as the model allows' },
|
||||
{ key: 'topP', label: 'Top P', type: 'number', min: 0, max: 1, group: PARAMETER_GROUP,
|
||||
placeholder: 'Leave empty for the default' },
|
||||
{ key: 'topK', label: 'Top K', type: 'number', min: 1, group: PARAMETER_GROUP,
|
||||
placeholder: 'Leave empty for the default' },
|
||||
{ key: 'maxTokens', label: 'Max tokens', type: 'number', min: 1, group: PARAMETER_GROUP,
|
||||
placeholder: 'Leave empty for the default' },
|
||||
{ key: 'seed', label: 'Seed', type: 'number', group: PARAMETER_GROUP,
|
||||
placeholder: 'Leave empty for the default', tip: 'Fixes the randomness, so two runs can be compared' }
|
||||
];
|
||||
|
||||
export type LLMDescriptorSettingsRequest = {
|
||||
title: string;
|
||||
/** Sampling defaults offered in the closed section, e.g. temperature 0 for a judge. */
|
||||
defaultParameters?: Partial<Record<keyof ModelParameters, number>>;
|
||||
/**
|
||||
* What to start from, when the caller already has a model worth repeating - the simulator of the
|
||||
* run this one is a rerun of. Preselecting it is the difference between a comparison of one
|
||||
* intervention and a comparison of two different simulators.
|
||||
*/
|
||||
initialDescriptor?: LLMDescriptor | null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Asks for a provider, a model and optionally how to sample from them.
|
||||
*
|
||||
* <p>One dialog for every place that picks a model at the moment it is needed rather than in a
|
||||
* node's configuration: the interaction simulator, and the LLM assessment of a bias comparison.
|
||||
* They ask the same question, so they ask it with the same window - the provider list, the models
|
||||
* that follow from the chosen provider, and the same optional parameters.
|
||||
*
|
||||
* Returns null when the dialog is dismissed, or when no provider is published at all.
|
||||
*/
|
||||
export async function openLLMDescriptorSettings(
|
||||
settingsDialog: NodeSettingsDialogService,
|
||||
fieldRetriever: FieldRetriever,
|
||||
request: LLMDescriptorSettingsRequest
|
||||
): Promise<LLMDescriptor | null> {
|
||||
const providerOptions = await loadOptions(fieldRetriever, 'providers', {}, PROVIDER_RETRIEVER_URL);
|
||||
if (!providerOptions.length) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const inherited = request.initialDescriptor ?? null;
|
||||
// Only if it is still on offer: a provider that has since been unregistered would leave the
|
||||
// dialog pointing at a model nobody can pick.
|
||||
const inheritedProvider = inherited?.provider
|
||||
&& providerOptions.some((option) => option.value === inherited.provider)
|
||||
? inherited.provider
|
||||
: null;
|
||||
|
||||
const defaultProvider = inheritedProvider ?? providerOptions[0].value;
|
||||
const initialModelOptions = await loadOptions(
|
||||
fieldRetriever,
|
||||
'models',
|
||||
{ provider: defaultProvider },
|
||||
MODEL_RETRIEVER_URL
|
||||
);
|
||||
|
||||
const inheritedParameters: Record<string, string> = Object.fromEntries(
|
||||
Object.entries(inheritedProvider ? inherited?.parameters ?? {} : {})
|
||||
.filter(([, value]) => value !== null && value !== undefined)
|
||||
.map(([key, value]) => [key, String(value)])
|
||||
);
|
||||
|
||||
const buildFields = (
|
||||
providers: { label: string; value: string; }[],
|
||||
models: { label: string; value: string; }[]
|
||||
): NodeSettingField[] => [
|
||||
{ key: 'provider', label: 'Provider', type: 'select', options: providers, required: true, autofocus: true },
|
||||
{ key: 'model', label: 'Model', type: 'select', options: models, required: true },
|
||||
// Inherited sampling is opened, not hidden: a seed carried over from the run being repeated is
|
||||
// the reason the two runs are comparable, and behind a closed section nobody would see it.
|
||||
...PARAMETER_FIELDS.map((field) => (inheritedParameters[field.key] === undefined
|
||||
? field
|
||||
: { ...field, group: undefined }))
|
||||
];
|
||||
|
||||
const defaults = Object.fromEntries(
|
||||
Object.entries(request.defaultParameters ?? {}).map(([key, value]) => [key, String(value)])
|
||||
);
|
||||
|
||||
const inheritedModel = inheritedProvider && inherited?.model
|
||||
&& initialModelOptions.some((option) => option.value === inherited.model)
|
||||
? inherited.model
|
||||
: null;
|
||||
|
||||
const result = await settingsDialog.open({
|
||||
title: request.title,
|
||||
fields: buildFields(providerOptions, initialModelOptions),
|
||||
initial: {
|
||||
provider: defaultProvider,
|
||||
model: inheritedModel ?? initialModelOptions[0]?.value ?? '',
|
||||
...defaults,
|
||||
// Last: what the run being repeated used wins over a generic default.
|
||||
...inheritedParameters
|
||||
},
|
||||
onValuesChange: async (draft) => {
|
||||
const provider = String(draft['provider'] ?? '').trim();
|
||||
const modelOptions = provider
|
||||
? await loadOptions(fieldRetriever, 'models', { provider }, MODEL_RETRIEVER_URL)
|
||||
: [];
|
||||
|
||||
return {
|
||||
fields: buildFields(providerOptions, modelOptions),
|
||||
initial: {
|
||||
provider,
|
||||
model: modelOptions[0]?.value ?? ''
|
||||
}
|
||||
};
|
||||
}
|
||||
});
|
||||
|
||||
if (!result) return null;
|
||||
|
||||
const provider = String(result['provider'] ?? '').trim();
|
||||
const model = String(result['model'] ?? '').trim();
|
||||
if (!provider || !model) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const parameters = readSimulatorParameters(result);
|
||||
return parameters ? { provider, model, parameters } : { provider, model };
|
||||
}
|
||||
|
||||
async function loadOptions(
|
||||
fieldRetriever: FieldRetriever,
|
||||
key: string,
|
||||
context: Record<string, string>,
|
||||
retrieverUrl: string
|
||||
): Promise<Array<{ label: string; value: string }>> {
|
||||
const values = await firstValueFrom(fieldRetriever.retrieveValues('LLM', key, context, retrieverUrl));
|
||||
return values.map((value) => ({ label: value, value }));
|
||||
}
|
||||
|
|
@ -5,6 +5,8 @@ import {
|
|||
getExecutionInputValues,
|
||||
getExecutionOutputValues,
|
||||
hasStoredValue,
|
||||
inheritedSimulator,
|
||||
inheritedSimulatorNotice,
|
||||
isExecutionStartable,
|
||||
planInputSaves,
|
||||
preparedInputValue,
|
||||
|
|
@ -322,3 +324,34 @@ describe('readSimulatorParameters', () => {
|
|||
expect(readSimulatorParameters({ topK: 40 })).toEqual({ topK: 40 });
|
||||
});
|
||||
});
|
||||
|
||||
describe('inheritedSimulator', () => {
|
||||
const descriptor = { provider: 'InternalOllama', model: 'gemma:7b' };
|
||||
|
||||
it('reports the descriptor a rerun carries before it has been started', () => {
|
||||
const rerun = {
|
||||
interactionSimulationEnabled: false,
|
||||
interactionSimulationDescriptor: descriptor
|
||||
} as unknown as TaskExecution;
|
||||
|
||||
expect(inheritedSimulator(rerun)).toBe(descriptor);
|
||||
expect(inheritedSimulatorNotice(inheritedSimulator(rerun)))
|
||||
.toContain('was simulated with InternalOllama / gemma:7b');
|
||||
});
|
||||
|
||||
/** On a run that is itself simulated the descriptor is what it used, not what it inherited. */
|
||||
it('reports nothing for a run that is already simulated', () => {
|
||||
const simulated = {
|
||||
interactionSimulationEnabled: true,
|
||||
interactionSimulationDescriptor: descriptor
|
||||
} as unknown as TaskExecution;
|
||||
|
||||
expect(inheritedSimulator(simulated)).toBeNull();
|
||||
});
|
||||
|
||||
it('reports nothing for a run with no simulator at all', () => {
|
||||
expect(inheritedSimulator({ interactionSimulationEnabled: false } as unknown as TaskExecution)).toBeNull();
|
||||
expect(inheritedSimulator(null)).toBeNull();
|
||||
expect(inheritedSimulatorNotice(null)).toBeNull();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ import {
|
|||
TaskExecutionStep,
|
||||
} from '@models/task-execution';
|
||||
import { LlmProviderCapability } from '@models/llm-provider';
|
||||
import { ModelParameters } from '@models/flow';
|
||||
import { LLMDescriptor, ModelParameters } from '@models/flow';
|
||||
import { EditableExecutionInput } from '@shared/task-execution-inputs-panel/task-execution-inputs-panel';
|
||||
|
||||
export type ExecutionOutputEntry = {
|
||||
|
|
@ -661,3 +661,30 @@ export function readSimulatorParameters(
|
|||
const set = Object.entries(parameters).filter(([, value]) => value !== undefined);
|
||||
return set.length ? Object.fromEntries(set) : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The simulator carried over from the run this one repeats, before this one has been started.
|
||||
*
|
||||
* <p>A rerun is created with the descriptor of its source but with simulation switched off, so a
|
||||
* descriptor on a run that is not itself simulated means exactly that and nothing else. It is what
|
||||
* lets the Simulate dialog offer the model the other run used instead of the first one on the list.
|
||||
*/
|
||||
export function inheritedSimulator(execution: TaskExecution | null | undefined): LLMDescriptor | null {
|
||||
if (!execution || execution.interactionSimulationEnabled === true) return null;
|
||||
return execution.interactionSimulationDescriptor ?? null;
|
||||
}
|
||||
|
||||
/**
|
||||
* What to tell someone about to start a run whose source was simulated.
|
||||
*
|
||||
* <p>Starting it unsimulated is not refused - the interactive steps would simply wait for a person -
|
||||
* but a comparison against a simulated baseline would then be comparing two different things, and
|
||||
* that is worth one sentence before the click rather than a puzzled reading of the report after it.
|
||||
*/
|
||||
export function inheritedSimulatorNotice(descriptor: LLMDescriptor | null): string | null {
|
||||
if (!descriptor) return null;
|
||||
const model = [descriptor.provider, descriptor.model].filter((part) => !!part).join(' / ');
|
||||
if (!model) return null;
|
||||
return `The run this one repeats was simulated with ${model}. Start it simulated, with the same`
|
||||
+ ' model, or the comparison will also be comparing two simulators.';
|
||||
}
|
||||
|
|
|
|||
|
|
@ -503,6 +503,17 @@
|
|||
gap: .35rem .75rem;
|
||||
margin-top: .45rem;
|
||||
}
|
||||
/* Says what the other run used, while there is still a choice to make about this one. */
|
||||
.execution-inherited-simulator-notice {
|
||||
background: #f0f9ff;
|
||||
border-left: 3px solid #0284c7;
|
||||
border-radius: .25rem;
|
||||
color: #075985;
|
||||
font-size: .78rem;
|
||||
line-height: 1.45;
|
||||
margin-top: .45rem;
|
||||
padding: .35rem .5rem;
|
||||
}
|
||||
.execution-subflow-context {
|
||||
align-items: center;
|
||||
color: #075985;
|
||||
|
|
|
|||
|
|
@ -67,6 +67,10 @@
|
|||
<p class="text-xs text-sky-900">Simulator: {{ simulationDescriptor }}</p>
|
||||
}
|
||||
|
||||
@if (inheritedSimulatorNotice(); as inheritedNotice) {
|
||||
<p class="execution-inherited-simulator-notice">{{ inheritedNotice }}</p>
|
||||
}
|
||||
|
||||
@if (isBiasVariant() && execution()!.biasExecutionContext; as biasContext) {
|
||||
<div class="execution-bias-variant-context">
|
||||
<span>Experiment: {{ biasContext.experimentId || 'not available' }}</span>
|
||||
|
|
|
|||
|
|
@ -42,6 +42,7 @@ import {
|
|||
HumanInteractionDialogService
|
||||
} from '@services/dialogs/human-interaction-dialog';
|
||||
import { NodeSettingField, NodeSettingsDialogService } from '@services/dialogs/node-settings-dialog';
|
||||
import { openLLMDescriptorSettings } from '@shared/llm-descriptor-settings/llm-descriptor-settings';
|
||||
import { FieldRetriever } from '@services/retriever/field-retriever';
|
||||
import { TaskExecutionsService } from '@services/task-executions/task-executions';
|
||||
import { FlowsService } from '@services/flows/flows';
|
||||
|
|
@ -74,6 +75,8 @@ import {
|
|||
fallbackExecutionLogMessage,
|
||||
logLevelClass as _logLevelClass,
|
||||
logTypeIcon as _logTypeIcon,
|
||||
inheritedSimulator as _inheritedSimulator,
|
||||
inheritedSimulatorNotice as _inheritedSimulatorNotice,
|
||||
formatExecutionOutputLabel,
|
||||
buildExecutionOutputs,
|
||||
buildExecutionOutputGroups,
|
||||
|
|
@ -82,7 +85,6 @@ import {
|
|||
buildVisibleExecutionLogs,
|
||||
normalizeEditableInputValue,
|
||||
planInputSaves,
|
||||
readSimulatorParameters,
|
||||
preparedInputValue,
|
||||
getExecutionInputValues,
|
||||
getExecutionOutputValues,
|
||||
|
|
@ -129,30 +131,6 @@ export class TaskExecutionViewerComponent implements OnDestroy {
|
|||
private route = inject(ActivatedRoute);
|
||||
private lastExecutionId: string | null = null;
|
||||
private lastExecutionStatus: string | null = null;
|
||||
private static readonly SIMULATOR_PARAMETER_GROUP = 'Model parameters';
|
||||
|
||||
/**
|
||||
* The optional sampling knobs, behind a section that starts closed. Provider and model are what
|
||||
* anyone opening this dialog came for; these are for the runs where you already know you want
|
||||
* them, and shown flat they made the common case look like a five-field form.
|
||||
*/
|
||||
private static readonly SIMULATOR_PARAMETER_FIELDS: NodeSettingField[] = [
|
||||
{ key: 'temperature', label: 'Temperature', type: 'number', min: 0, max: 2,
|
||||
group: TaskExecutionViewerComponent.SIMULATOR_PARAMETER_GROUP,
|
||||
placeholder: 'Leave empty for the default', tip: '0 makes the run as repeatable as the model allows' },
|
||||
{ key: 'topP', label: 'Top P', type: 'number', min: 0, max: 1,
|
||||
group: TaskExecutionViewerComponent.SIMULATOR_PARAMETER_GROUP, placeholder: 'Leave empty for the default' },
|
||||
{ key: 'topK', label: 'Top K', type: 'number', min: 1,
|
||||
group: TaskExecutionViewerComponent.SIMULATOR_PARAMETER_GROUP, placeholder: 'Leave empty for the default' },
|
||||
{ key: 'maxTokens', label: 'Max tokens', type: 'number', min: 1,
|
||||
group: TaskExecutionViewerComponent.SIMULATOR_PARAMETER_GROUP, placeholder: 'Leave empty for the default' },
|
||||
{ key: 'seed', label: 'Seed', type: 'number',
|
||||
group: TaskExecutionViewerComponent.SIMULATOR_PARAMETER_GROUP,
|
||||
placeholder: 'Leave empty for the default', tip: 'Fixes the randomness, so two runs can be compared' }
|
||||
];
|
||||
|
||||
private static readonly SIMULATOR_PROVIDER_RETRIEVER_URL = '/retriever/LLM/providers';
|
||||
private static readonly SIMULATOR_MODEL_RETRIEVER_URL = '/retriever/LLM/models';
|
||||
readonly execution = input<TaskExecution | null>(null);
|
||||
readonly parentExecution = input<TaskExecution | null>(null);
|
||||
readonly parentContainerStep = input<TaskExecutionStep | null>(null);
|
||||
|
|
@ -801,6 +779,12 @@ export class TaskExecutionViewerComponent implements OnDestroy {
|
|||
return `${provider} / ${model}`;
|
||||
});
|
||||
|
||||
readonly inheritedSimulator = computed<LLMDescriptor | null>(() => _inheritedSimulator(this.execution()));
|
||||
|
||||
/** Shown while the run can still be started the other way, which is the only time it helps. */
|
||||
readonly inheritedSimulatorNotice = computed<string | null>(() =>
|
||||
this.canSimulateExecution() ? _inheritedSimulatorNotice(this.inheritedSimulator()) : null);
|
||||
|
||||
readonly editableInputs = computed<EditableExecutionInput[]>(() => {
|
||||
const missingGlobalInputNames = new Set(this.execution()?.missingGlobalInputKeys ?? []);
|
||||
const execution = this.execution();
|
||||
|
|
@ -1459,82 +1443,14 @@ export class TaskExecutionViewerComponent implements OnDestroy {
|
|||
});
|
||||
}
|
||||
|
||||
private async openSimulationSettings(): Promise<LLMDescriptor | null> {
|
||||
const providerOptions = await this.loadSimulationOptions(
|
||||
'providers',
|
||||
{},
|
||||
TaskExecutionViewerComponent.SIMULATOR_PROVIDER_RETRIEVER_URL
|
||||
);
|
||||
if (!providerOptions.length) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const defaultProvider = providerOptions[0].value;
|
||||
const initialModelOptions = await this.loadSimulationOptions(
|
||||
'models',
|
||||
{ provider: defaultProvider },
|
||||
TaskExecutionViewerComponent.SIMULATOR_MODEL_RETRIEVER_URL
|
||||
);
|
||||
|
||||
const buildFields = (providers: { label: string; value: string; }[], models: { label: string; value: string; }[]): NodeSettingField[] => [
|
||||
{
|
||||
key: 'provider',
|
||||
label: 'Provider',
|
||||
type: 'select',
|
||||
options: providers,
|
||||
required: true,
|
||||
autofocus: true
|
||||
},
|
||||
{
|
||||
key: 'model',
|
||||
label: 'Model',
|
||||
type: 'select',
|
||||
options: models,
|
||||
required: true
|
||||
},
|
||||
// Optional, and empty means "leave it to the provider" - the same contract the block editor
|
||||
// gives them. A simulated run is a model call like any other, and a seed here is what makes
|
||||
// a simulated run repeatable enough to compare against another.
|
||||
...TaskExecutionViewerComponent.SIMULATOR_PARAMETER_FIELDS
|
||||
];
|
||||
|
||||
const result = await this.settingsDialog.open({
|
||||
title: 'Simulation Settings',
|
||||
fields: buildFields(providerOptions, initialModelOptions),
|
||||
initial: {
|
||||
provider: defaultProvider,
|
||||
model: initialModelOptions[0]?.value ?? ''
|
||||
},
|
||||
onValuesChange: async (draft) => {
|
||||
const provider = String(draft['provider'] ?? '').trim();
|
||||
const modelOptions = provider
|
||||
? await this.loadSimulationOptions(
|
||||
'models',
|
||||
{ provider },
|
||||
TaskExecutionViewerComponent.SIMULATOR_MODEL_RETRIEVER_URL
|
||||
)
|
||||
: [];
|
||||
|
||||
return {
|
||||
fields: buildFields(providerOptions, modelOptions),
|
||||
initial: {
|
||||
provider,
|
||||
model: modelOptions[0]?.value ?? ''
|
||||
}
|
||||
};
|
||||
}
|
||||
private openSimulationSettings(): Promise<LLMDescriptor | null> {
|
||||
// A rerun carries the simulator of the run it repeats, and that is what the dialog starts from:
|
||||
// comparing two runs answered by different simulators compares the simulators too.
|
||||
const inherited = this.inheritedSimulator();
|
||||
return openLLMDescriptorSettings(this.settingsDialog, this.fieldRetriever, {
|
||||
title: inherited ? 'Simulation Settings - repeating a simulated run' : 'Simulation Settings',
|
||||
initialDescriptor: inherited
|
||||
});
|
||||
|
||||
if (!result) return null;
|
||||
|
||||
const provider = String(result['provider'] ?? '').trim();
|
||||
const model = String(result['model'] ?? '').trim();
|
||||
if (!provider || !model) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const parameters = readSimulatorParameters(result);
|
||||
return parameters ? { provider, model, parameters } : { provider, model };
|
||||
}
|
||||
|
||||
private fetchExecutionLogs(executionId: string) {
|
||||
|
|
@ -1569,21 +1485,6 @@ export class TaskExecutionViewerComponent implements OnDestroy {
|
|||
element.scrollTop = element.scrollHeight;
|
||||
}
|
||||
|
||||
private async loadSimulationOptions(
|
||||
key: string,
|
||||
context: Record<string, string>,
|
||||
retrieverUrl: string
|
||||
): Promise<Array<{ label: string; value: string }>> {
|
||||
const values = await firstValueFrom(
|
||||
this.fieldRetriever.retrieveValues('LLM', key, context, retrieverUrl)
|
||||
);
|
||||
|
||||
return values.map((value) => ({
|
||||
label: value,
|
||||
value
|
||||
}));
|
||||
}
|
||||
|
||||
logLevelClass(level: string | null | undefined): string {
|
||||
return _logLevelClass(level);
|
||||
}
|
||||
|
|
|
|||
Loading…
Reference in New Issue