diff --git a/src/main/java/it/cnr/isti/workflow/manager/controllers/BiasExperimentsController.java b/src/main/java/it/cnr/isti/workflow/manager/controllers/BiasExperimentsController.java index 63e6408..99df3f7 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/controllers/BiasExperimentsController.java +++ b/src/main/java/it/cnr/isti/workflow/manager/controllers/BiasExperimentsController.java @@ -24,6 +24,7 @@ import it.cnr.isti.workflow.manager.executions.bias.BiasImpactReport; import it.cnr.isti.workflow.manager.executions.bias.BiasImpactJob; import it.cnr.isti.workflow.manager.executions.bias.BiasImpactJobService; import it.cnr.isti.workflow.manager.executions.bias.BiasImpactService; +import it.cnr.isti.workflow.manager.executions.bias.BiasJudgeRequest; import it.cnr.isti.workflow.manager.executions.bias.BiasRerunRequest; import jakarta.validation.Valid; @@ -94,6 +95,19 @@ public class BiasExperimentsController { return biasImpactService.getReports(executionId, owner(userDetails)); } + @PostMapping("/bias-impact-reports/{reportId}/judge") + @Operation(summary = "Ask an LLM to assess a persisted bias impact report", + description = "Queues an assessment of the comparison, one call per compared pair, with the provider and " + + "model chosen by the caller the way the interaction simulator's are. Poll the returned job; the " + + "verdicts are stored on the report itself, replacing any previous assessment.") + public ResponseEntity judgeReport( + @PathVariable String reportId, + @RequestBody @Valid BiasJudgeRequest request, + @AuthenticationPrincipal LoginEntity userDetails) { + BiasImpactJob job = biasImpactJobService.createJudgeJob(reportId, request.judge(), owner(userDetails)); + return ResponseEntity.accepted().body(job); + } + @GetMapping("/bias-impact-reports/{reportId}") @Operation(summary = "Get a persisted bias impact report") public BiasImpactReport getReport( diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/ExecutionsService.java b/src/main/java/it/cnr/isti/workflow/manager/executions/ExecutionsService.java index 2f851b2..7d711bc 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/executions/ExecutionsService.java +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/ExecutionsService.java @@ -514,10 +514,31 @@ public class ExecutionsService { ExecutionObject rerun = createExecution(source.getName(), source.getFlow(), owner, runGroupId, source.getSourceFlowId(), source.getId(), nextRunNumber); copyReusableInputs(source, rerun); + inheritSimulationSettings(source, rerun); persist(rerun); return rerun; } + /** + * Carries the simulator of the run being repeated onto the repetition. + * + *

Only the descriptor, never {@code interactionSimulationEnabled}: simulating stays an + * explicit act, and with the flag off the descriptor does nothing. On a rerun that has not been + * started, its presence is what says "the run I repeat was simulated, and with this" - which is + * what lets the client offer the same model instead of the first one on the list. + * + *

It matters most for a bias rerun. Two runs whose interactive steps were answered by + * different simulators - or one by a simulator and the other by a person - differ for a reason + * that has nothing to do with the intervention, and the comparison cannot tell the two apart. + */ + private void inheritSimulationSettings(ExecutionObject source, ExecutionObject rerun) { + LLMDescriptor simulator = source.getInteractionSimulationDescriptor(); + if (simulator == null) { + return; + } + rerun.setInteractionSimulationDescriptor(simulator); + } + @Transactional public ExecutionObject createBiasRerun(String id, String owner, BiasRerunRequest request) { ExecutionObject source = getExecutionByOwner(id, owner); @@ -544,6 +565,7 @@ public class ExecutionsService { ExecutionObject rerun = createExecution(source.getName(), source.getFlow(), owner, runGroupId, source.getSourceFlowId(), source.getId(), nextRunNumber, biasContext); copyReusableInputs(source, rerun); + inheritSimulationSettings(source, rerun); persist(rerun); return rerun; } diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasDownstreamImpact.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasDownstreamImpact.java index 48e8190..5cdaec4 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasDownstreamImpact.java +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasDownstreamImpact.java @@ -1,5 +1,6 @@ package it.cnr.isti.workflow.manager.executions.bias; +import java.util.List; import java.util.Map; public record BiasDownstreamImpact( @@ -9,10 +10,31 @@ public record BiasDownstreamImpact( String biasedStatus, boolean changed, Map baselineOutputs, - Map biasedOutputs) { + Map biasedOutputs, + /** The same field-by-field comparison the immediate impact carries, for this node. */ + List values, + /** + * One entry per iteration when this node is a container, empty otherwise. + * + *

Without it a container reports that its accumulated output list changed and stops + * there, which on a per-subject iteration is precisely the information that matters. + */ + List iterations) { public BiasDownstreamImpact { baselineOutputs = baselineOutputs == null ? Map.of() : Map.copyOf(baselineOutputs); biasedOutputs = biasedOutputs == null ? Map.of() : Map.copyOf(biasedOutputs); + values = values == null ? List.of() : List.copyOf(values); + iterations = iterations == null ? List.of() : List.copyOf(iterations); + } + + public BiasDownstreamImpact withValues(List replacements) { + return new BiasDownstreamImpact(nodeId, nodeName, baselineStatus, biasedStatus, changed, baselineOutputs, + biasedOutputs, replacements, iterations); + } + + public BiasDownstreamImpact withIterations(List replacements) { + return new BiasDownstreamImpact(nodeId, nodeName, baselineStatus, biasedStatus, changed, baselineOutputs, + biasedOutputs, values, replacements); } } diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJob.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJob.java index 370ae5b..29f1f52 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJob.java +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJob.java @@ -4,6 +4,7 @@ import java.time.LocalDateTime; public record BiasImpactJob( String id, + BiasImpactJobKind kind, BiasImpactJobStatus status, String executionId, String stepId, diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJobKind.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJobKind.java new file mode 100644 index 0000000..643fd62 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJobKind.java @@ -0,0 +1,15 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +/** + * What a queued bias job does. + * + *

Both kinds end with a persisted report and are polled the same way, so they share one job + * resource rather than duplicating the queue, the recovery of interrupted work and the polling + * contract the editor already speaks. + */ +public enum BiasImpactJobKind { + /** Re-runs one block with the intervention active, against a captured baseline output. */ + ISOLATED_STEP, + /** Asks a model to assess a comparison that has already been computed. */ + REPORT_JUDGE +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJobService.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJobService.java index 1aabf15..9601abe 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJobService.java +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJobService.java @@ -12,6 +12,7 @@ import org.springframework.stereotype.Service; import it.cnr.isti.workflow.manager.executions.bias.persistence.BiasImpactJobEntity; import it.cnr.isti.workflow.manager.executions.bias.persistence.BiasImpactJobRepository; import it.cnr.isti.workflow.manager.flows.validation.ValidationErrorCode; +import it.cnr.isti.workflow.manager.llms.LLMDescriptor; import jakarta.annotation.PostConstruct; import jakarta.annotation.PreDestroy; @@ -32,6 +33,7 @@ public class BiasImpactJobService { BiasImpactJobEntity entity = repository.save(BiasImpactJobEntity.builder() .id(UUID.randomUUID().toString()) .owner(owner) + .kind(BiasImpactJobKind.ISOLATED_STEP) .executionId(executionId) .stepId(stepId) .status(BiasImpactJobStatus.QUEUED) @@ -42,6 +44,30 @@ public class BiasImpactJobService { return toView(entity); } + /** + * Queues the LLM assessment of an existing report. + * + *

Asynchronous for the same reason the isolated experiment is: one model call per compared + * pair, and a per-subject comparison has as many pairs as there were subjects. The report is + * validated here so that a report that cannot be judged is refused now, rather than by a job + * that fails a minute later. + */ + public BiasImpactJob createJudgeJob(String reportId, LLMDescriptor descriptor, String owner) { + BiasImpactReport report = impactService.getReport(reportId, owner); + BiasImpactJobEntity entity = repository.save(BiasImpactJobEntity.builder() + .id(UUID.randomUUID().toString()) + .owner(owner) + .kind(BiasImpactJobKind.REPORT_JUDGE) + .executionId(report.baselineExecutionId()) + .reportTargetId(reportId) + .judge(descriptor) + .status(BiasImpactJobStatus.QUEUED) + .createdAt(LocalDateTime.now()) + .build()); + submit(entity.getId()); + return toView(entity); + } + public BiasImpactJob get(String jobId, String owner) { return repository.findByIdAndOwner(jobId, owner) .map(this::toView) @@ -80,8 +106,10 @@ public class BiasImpactJobService { entity.setStartedAt(LocalDateTime.now()); repository.save(entity); try { - BiasImpactReport report = impactService.runIsolatedStepExperiment( - entity.getExecutionId(), entity.getStepId(), entity.getRequest(), entity.getOwner()); + BiasImpactReport report = entity.getKind() == BiasImpactJobKind.REPORT_JUDGE + ? impactService.judgeReport(entity.getReportTargetId(), entity.getJudge(), entity.getOwner()) + : impactService.runIsolatedStepExperiment( + entity.getExecutionId(), entity.getStepId(), entity.getRequest(), entity.getOwner()); entity.setReportId(report.id()); entity.setStatus(BiasImpactJobStatus.COMPLETED); } catch (BiasApiException exception) { @@ -89,8 +117,13 @@ public class BiasImpactJobService { entity.setErrorMessage(exception.getReason()); entity.setStatus(BiasImpactJobStatus.FAILED); } catch (Exception exception) { - entity.setErrorCode(ValidationErrorCode.BIAS_EXPERIMENT_FAILED.name()); - entity.setErrorMessage(exception.getMessage() == null ? "Bias impact experiment failed" : exception.getMessage()); + boolean judging = entity.getKind() == BiasImpactJobKind.REPORT_JUDGE; + entity.setErrorCode(judging + ? ValidationErrorCode.BIAS_JUDGE_FAILED.name() + : ValidationErrorCode.BIAS_EXPERIMENT_FAILED.name()); + entity.setErrorMessage(exception.getMessage() == null + ? (judging ? "The LLM assessment failed" : "Bias impact experiment failed") + : exception.getMessage()); entity.setStatus(BiasImpactJobStatus.FAILED); } finally { entity.setCompletedAt(LocalDateTime.now()); @@ -104,6 +137,7 @@ public class BiasImpactJobService { : impactService.getReport(entity.getReportId(), entity.getOwner()); return new BiasImpactJob( entity.getId(), + entity.getKind(), entity.getStatus(), entity.getExecutionId(), entity.getStepId(), diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJudge.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJudge.java new file mode 100644 index 0000000..89925af --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactJudge.java @@ -0,0 +1,492 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.time.LocalDateTime; +import java.util.ArrayList; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Locale; +import java.util.Map; +import java.util.Objects; + +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; +import org.springframework.http.HttpStatus; +import org.springframework.stereotype.Component; + +import it.cnr.isti.workflow.manager.app.JacksonConverterSupport; +import it.cnr.isti.workflow.manager.blocks.Block; +import it.cnr.isti.workflow.manager.containers.Container; +import it.cnr.isti.workflow.manager.containers.configurations.ContainerConfiguration; +import it.cnr.isti.workflow.manager.containers.configurations.LoopContainerConfiguration; +import it.cnr.isti.workflow.manager.executions.ExecutionObject; +import it.cnr.isti.workflow.manager.flows.model.FlowData; +import it.cnr.isti.workflow.manager.flows.model.FlowNode; +import it.cnr.isti.workflow.manager.flows.model.bias.BiasBehavioralProbe; +import it.cnr.isti.workflow.manager.flows.model.bias.BlockBiasAnnotation; +import it.cnr.isti.workflow.manager.flows.validation.ValidationErrorCode; +import it.cnr.isti.workflow.manager.llms.LLMCredentialResolver; +import it.cnr.isti.workflow.manager.llms.LLMDescriptor; +import it.cnr.isti.workflow.manager.llms.providers.LLMProvider; +import tools.jackson.databind.JsonNode; + +/** + * Asks a model what a comparison means, one aligned pair of outputs at a time. + * + *

The deterministic figures say a candidate's score fell by three points and that 0.42 of the + * text changed. They cannot say whether the change is the intervention doing what its probe + * described or the same model answering differently twice, and that distinction is the whole + * question behind running the experiment. This is where it gets asked - and the answer is recorded + * as an assessment, next to the model that gave it, never as a measurement. + * + *

Provider, model and sampling parameters arrive as an {@link LLMDescriptor} and are resolved the + * way the interaction simulator resolves its own: by provider name against the registered providers, + * with the credential taken from the execution being judged. + */ +@Component +public class BiasImpactJudge { + + private static final Logger log = LoggerFactory.getLogger(BiasImpactJudge.class); + + /** + * How many pairs one evaluation will spend model calls on. + * + * A full-flow report on a fifty-subject run has hundreds of comparable pairs, and a local + * model answers in seconds each. The changed pairs come first, so the cap falls on the ones + * that had nothing to say. + */ + public static final int MAX_JUDGED_PAIRS = 20; + + /** Per side of a pair. Two truncated answers still show what moved; two whole ones may not fit. */ + static final int MAX_TEXT_CHARACTERS = 4000; + + private final Map llmProviders; + private final LLMCredentialResolver credentialResolver; + + public BiasImpactJudge(Map llmProviders, LLMCredentialResolver credentialResolver) { + this.llmProviders = llmProviders; + this.credentialResolver = credentialResolver; + } + + /** The verdicts, addressed by target path, plus what to put at the top of the report. */ + public record Judgment(Map verdicts, BiasJudgeSummary summary) { + + public Judgment { + verdicts = verdicts == null ? Map.of() : Map.copyOf(verdicts); + } + } + + /** + * @param comparison the freshly recomputed comparison, which still carries the raw texts a + * persisted report may have been stripped of + */ + public Judgment evaluate(BiasImpactReport comparison, LLMDescriptor descriptor, ExecutionObject baseline, + ExecutionObject biased) { + LLMProvider provider = resolveProvider(descriptor.provider()); + String authorization = resolveAuthorization(provider, baseline); + String interventions = describeInterventions(biased, comparison.annotationIds()); + + List targets = BiasJudgeTarget.collect(comparison); + Map verdicts = new LinkedHashMap<>(); + List errors = new ArrayList<>(); + int judged = 0; + int skipped = 0; + + for (BiasJudgeTarget target : targets) { + if (!target.changed()) { + // Nothing moved: the answer is known, and asking would spend a call to be told so. + verdicts.put(target.path(), BiasJudgeVerdict.identical()); + skipped++; + continue; + } + if (judged >= MAX_JUDGED_PAIRS) { + skipped++; + continue; + } + judged++; + verdicts.put(target.path(), judgeTarget(provider, descriptor, authorization, interventions, target, errors)); + } + if (skipped > 0 && judged >= MAX_JUDGED_PAIRS) { + errors.add("Only the first " + MAX_JUDGED_PAIRS + " changed pairs were assessed."); + } + + BiasJudgeImpactLevel impact = rollUpImpact(verdicts.values(), comparison); + BiasJudgeAttribution attribution = rollUpAttribution(verdicts.values()); + String narrative = judged == 0 + ? "No pair differed between the two runs, so nothing was submitted for assessment." + : narrate(provider, descriptor, authorization, interventions, comparison, verdicts, impact, + attribution, errors); + + return new Judgment(verdicts, new BiasJudgeSummary(descriptor, LocalDateTime.now(), impact, attribution, + narrative, judged, skipped, errors)); + } + + private BiasJudgeVerdict judgeTarget(LLMProvider provider, LLMDescriptor descriptor, String authorization, + String interventions, BiasJudgeTarget target, List errors) { + String prompt = pairPrompt(interventions, target); + try { + String response = provider.generateJson(descriptor.model(), prompt, authorization, descriptor.parameters()); + try { + return parseVerdict(response); + } catch (RuntimeException firstFailure) { + // One repair attempt: a model that wrapped its JSON in prose usually complies when + // told exactly what came back and what was wrong with it. + String repaired = provider.generateJson(descriptor.model(), + repairPrompt(prompt, response, firstFailure.getMessage()), authorization, + descriptor.parameters()); + return parseVerdict(repaired); + } + } catch (RuntimeException exception) { + String message = "Assessment of " + describeTarget(target) + " failed: " + rootMessage(exception); + log.warn("Bias impact judge failed for {}: {}", target.path(), rootMessage(exception)); + errors.add(message); + return BiasJudgeVerdict.failed(message); + } + } + + /** + * The narration of the whole comparison. + * + *

Only the prose is asked for. The level and the attribution above it are rolled up from the + * per-pair verdicts in code, so that re-reading a report cannot show a different headline than + * the pairs it is made of. + */ + private String narrate(LLMProvider provider, LLMDescriptor descriptor, String authorization, String interventions, + BiasImpactReport comparison, Map verdicts, BiasJudgeImpactLevel impact, + BiasJudgeAttribution attribution, List errors) { + String prompt = """ + You are reviewing an experiment on an automated workflow. An intervention was activated on one + run and not on the other, and the two runs have already been compared pair by pair. + + Activated intervention(s): + %s + + Deterministic findings: + %s + + Per-pair assessments already made: + %s + + Rolled-up level: %s. Rolled-up attribution: %s. + + Write two to four sentences for a reviewer: what changed in substance, for whom, and whether it + looks like the intervention's doing. Do not restate the numbers. Do not recommend anything. Say + plainly when a single pair of runs cannot separate the intervention from ordinary model variation. + Answer with prose only, no JSON and no headings. + """.formatted(interventions, describeFindings(comparison), describeVerdicts(verdicts), impact, + attribution); + try { + String response = provider.generate(descriptor.model(), prompt, authorization, descriptor.parameters()); + return response == null || response.isBlank() ? null : response.strip(); + } catch (RuntimeException exception) { + errors.add("The narrative could not be produced: " + rootMessage(exception)); + return null; + } + } + + private String pairPrompt(String interventions, BiasJudgeTarget target) { + return """ + You are assessing one output of an automated workflow, produced twice: once on a baseline run and + once on a run where a bias intervention was deliberately activated. + + Activated intervention(s): + %s + + Output under assessment: %s of node "%s"%s + + Already measured, do not recompute: normalized text difference %.3f%s + + BASELINE OUTPUT: + <<< + %s + >>> + + INTERVENED OUTPUT: + <<< + %s + >>> + + Answer with this JSON object and nothing else: + {"impact":"NONE|COSMETIC|SUBSTANTIVE|DECISIVE", + "attribution":"INJECTION|NON_DETERMINISM|UNCLEAR", + "confidence":0.0, + "changedAspects":["short phrase"], + "rationale":"one or two sentences"} + + impact is about meaning, not wording: NONE nothing of substance, COSMETIC same content phrased + differently, SUBSTANTIVE the assessment or its justification changed, DECISIVE the decision this + output carries would differ. + attribution is INJECTION when the change is what the intervention's instruction asked for, + NON_DETERMINISM when it looks like the same model answering differently, UNCLEAR when one pair of + runs cannot tell them apart. + """.formatted( + interventions, + target.field(), + target.nodeName(), + target.subjectIndex() == null ? "" : " for subject " + target.subjectIndex(), + target.textDifference(), + describeDeltas(target.numericDeltas()), + truncate(target.baselineText()), + truncate(target.biasedText())); + } + + private String repairPrompt(String originalPrompt, String response, String failure) { + return """ + %s + + Your previous answer could not be read as the requested JSON object (%s). It was: + <<< + %s + >>> + + Answer again with the JSON object alone, no prose, no code fences. + """.formatted(originalPrompt, failure == null ? "unknown parsing error" : failure, + truncate(response)); + } + + private BiasJudgeVerdict parseVerdict(String response) { + JsonNode root = JacksonConverterSupport.mapper().readTree(extractJsonObject(response)); + BiasJudgeImpactLevel impact = readEnum(root.path("impact").asString(null), BiasJudgeImpactLevel.class); + BiasJudgeAttribution attribution = readEnum(root.path("attribution").asString(null), + BiasJudgeAttribution.class); + if (impact == null) { + throw new IllegalArgumentException("the impact field was missing or not one of the four levels"); + } + List aspects = new ArrayList<>(); + root.path("changedAspects").forEach(node -> { + String aspect = node.asString(null); + if (aspect != null && !aspect.isBlank()) { + aspects.add(aspect.strip()); + } + }); + String rationale = root.path("rationale").asString(null); + Double confidence = root.path("confidence").isNumber() ? root.path("confidence").asDouble() : null; + return new BiasJudgeVerdict( + impact, + attribution == null ? BiasJudgeAttribution.UNCLEAR : attribution, + confidence == null ? null : Math.max(0.0, Math.min(1.0, confidence)), + aspects, + rationale == null || rationale.isBlank() ? null : rationale.strip(), + null); + } + + private > E readEnum(String value, Class type) { + if (value == null || value.isBlank()) { + return null; + } + try { + return Enum.valueOf(type, value.strip().toUpperCase(Locale.ROOT)); + } catch (IllegalArgumentException exception) { + return null; + } + } + + /** The first JSON object in a response, so a model that wrapped it in prose is still readable. */ + private String extractJsonObject(String response) { + if (response == null || response.isBlank()) { + throw new IllegalArgumentException("the model returned an empty answer"); + } + int start = response.indexOf('{'); + int end = response.lastIndexOf('}'); + if (start < 0 || end <= start) { + throw new IllegalArgumentException("the answer contained no JSON object"); + } + return response.substring(start, end + 1); + } + + /** + * The worst per-pair level, raised to DECISIVE when the run's own outcome or routing changed. + * + *

The escalation is not the model's opinion but a fact of the two runs: an experiment that + * ended on a different outcome was decisive whatever the wording of the outputs suggests. + */ + private BiasJudgeImpactLevel rollUpImpact(java.util.Collection verdicts, + BiasImpactReport comparison) { + if (!comparison.outcomeChanges().isEmpty() || !comparison.routingChanges().isEmpty()) { + return BiasJudgeImpactLevel.DECISIVE; + } + BiasJudgeImpactLevel worst = null; + for (BiasJudgeVerdict verdict : verdicts) { + worst = BiasJudgeImpactLevel.worst(worst, verdict.impact()); + } + return worst == null ? BiasJudgeImpactLevel.NONE : worst; + } + + private BiasJudgeAttribution rollUpAttribution(java.util.Collection verdicts) { + List considered = verdicts.stream() + .filter(verdict -> !verdict.failure()) + .filter(verdict -> verdict.impact() != null && verdict.impact() != BiasJudgeImpactLevel.NONE) + .map(BiasJudgeVerdict::attribution) + .filter(Objects::nonNull) + .toList(); + if (considered.isEmpty()) { + return BiasJudgeAttribution.UNCLEAR; + } + if (considered.contains(BiasJudgeAttribution.INJECTION)) { + return BiasJudgeAttribution.INJECTION; + } + return considered.stream().allMatch(one -> one == BiasJudgeAttribution.NON_DETERMINISM) + ? BiasJudgeAttribution.NON_DETERMINISM + : BiasJudgeAttribution.UNCLEAR; + } + + /** + * The interventions in the words of whoever annotated the flow. + * + *

The probe's instruction is the most useful line in the prompt: it is the only statement of + * what the change is supposed to look like, which is what separates attributing a difference to + * the intervention from noticing that a difference exists. + */ + private String describeInterventions(ExecutionObject biased, List annotationIds) { + List described = new ArrayList<>(); + collectAnnotations(biased.getFlow(), annotationIds, described); + return described.isEmpty() + ? "(the activated annotations are no longer part of the flow definition)" + : String.join(System.lineSeparator(), described); + } + + private void collectAnnotations(FlowData flow, List annotationIds, List described) { + if (flow == null || flow.getNodes() == null) { + return; + } + for (FlowNode node : flow.getNodes()) { + for (BlockBiasAnnotation annotation : node.getBiasAnnotations()) { + if (annotation == null || !annotationIds.contains(annotation.id())) { + continue; + } + described.add(describeAnnotation(node, annotation)); + } + if (node instanceof Container container) { + subFlowsOf(container).forEach(subFlow -> collectAnnotations(subFlow, annotationIds, described)); + } + } + } + + private String describeAnnotation(FlowNode node, BlockBiasAnnotation annotation) { + StringBuilder text = new StringBuilder("- on node \"" + node.getName() + "\" (") + .append(node instanceof Block ? "block" : "container").append("): ") + .append(annotation.category()).append(", severity ").append(annotation.severity()) + .append(". Issue: ").append(annotation.issue()); + BiasBehavioralProbe probe = annotation.biasProbe() == null ? annotation.mitigationProbe() + : annotation.biasProbe(); + if (probe != null) { + text.append(" Activated as ").append(probe.activationMode()) + .append(" with the instruction: \"").append(truncate(probe.instruction())).append('"'); + if (probe.expectedImpact() != null && !probe.expectedImpact().isBlank()) { + text.append(" Expected: ").append(probe.expectedImpact()); + } + } + return text.toString(); + } + + private List subFlowsOf(Container container) { + ContainerConfiguration configuration = container.getSpecificConfiguration(); + List subFlows = new ArrayList<>(); + if (configuration == null) { + return subFlows; + } + if (configuration.getSubFlow() != null) { + subFlows.add(configuration.getSubFlow()); + } + if (configuration instanceof LoopContainerConfiguration loop && loop.getGuardSubFlow() != null) { + subFlows.add(loop.getGuardSubFlow()); + } + return subFlows; + } + + private String describeFindings(BiasImpactReport comparison) { + List lines = new ArrayList<>(); + lines.add("- " + comparison.summary()); + // A model asked to attribute a change must know when something other than the intervention + // also differed between the two runs. + if (comparison.simulation() != null && !comparison.simulation().comparable()) { + lines.add("- the interactive steps were not answered the same way on both runs: " + + comparison.simulation().describe()); + } + comparison.outcomeChanges().forEach(change -> lines.add("- outcomes " + change.baselineOutcomeCodes() + + " became " + change.biasedOutcomeCodes())); + comparison.routingChanges().forEach(change -> lines.add("- node " + change.nodeId() + " routed to " + + change.biasedBranch() + " instead of " + change.baselineBranch())); + comparison.immediateImpact().values().stream() + .filter(BiasValueImpact::changed) + .forEach(value -> lines.add("- " + value.nodeName() + "." + value.field() + ": " + + value.itemsChanged() + " of " + value.itemsCompared() + " subjects changed" + + describeDeltas(value.numericDeltas()))); + return String.join(System.lineSeparator(), lines); + } + + private String describeVerdicts(Map verdicts) { + List lines = new ArrayList<>(); + verdicts.forEach((path, verdict) -> { + if (verdict.failure() || verdict.impact() == null || verdict.impact() == BiasJudgeImpactLevel.NONE) { + return; + } + lines.add("- " + path + ": " + verdict.impact() + ", " + verdict.attribution() + + (verdict.rationale() == null ? "" : " - " + verdict.rationale())); + }); + return lines.isEmpty() ? "(none of substance)" : String.join(System.lineSeparator(), lines); + } + + private String describeDeltas(List deltas) { + if (deltas.isEmpty()) { + return ""; + } + List described = deltas.stream() + .map(delta -> String.format(Locale.ROOT, "%s %.2f to %.2f (%+.2f)", delta.label(), delta.baseline(), + delta.biased(), delta.delta())) + .toList(); + return "; " + String.join(", ", described); + } + + private String describeTarget(BiasJudgeTarget target) { + return target.nodeName() + "." + target.field() + + (target.subjectIndex() == null ? "" : " (subject " + target.subjectIndex() + ")"); + } + + private String truncate(String text) { + if (text == null) { + return ""; + } + return text.length() <= MAX_TEXT_CHARACTERS + ? text + : text.substring(0, MAX_TEXT_CHARACTERS) + System.lineSeparator() + "[truncated]"; + } + + private LLMProvider resolveProvider(String providerName) { + LLMProvider provider = llmProviders.get(providerName); + if (provider != null) { + return provider; + } + return llmProviders.values().stream() + .filter(candidate -> providerName != null && providerName.equalsIgnoreCase(candidate.getName())) + .findFirst() + .orElseThrow(() -> new BiasApiException(HttpStatus.BAD_REQUEST, + ValidationErrorCode.BIAS_JUDGE_PROVIDER_NOT_FOUND, "judge", providerName, + "No LLM provider is registered under the name " + providerName)); + } + + /** + * The credential the judged run itself used. + * + *

Taken from the baseline execution rather than asked for again: the model that assesses a + * run is reached the same way as the models that produced it, and the internal provider - the + * one a local install has - needs no credential at all. + */ + private String resolveAuthorization(LLMProvider provider, ExecutionObject baseline) { + try { + return credentialResolver.resolve(provider, baseline.getProvidedAuthorizations(), + baseline.getContext().getResolvedExecutionVariables()); + } catch (IllegalArgumentException exception) { + throw new BiasApiException(HttpStatus.CONFLICT, ValidationErrorCode.BIAS_JUDGE_UNAVAILABLE, + "judge", provider.getName(), + "The judge provider needs a credential this execution does not carry: " + exception.getMessage()); + } + } + + private String rootMessage(RuntimeException exception) { + Throwable cause = exception; + while (cause.getCause() != null && cause.getMessage() == null) { + cause = cause.getCause(); + } + return cause.getMessage() == null ? cause.getClass().getSimpleName() : cause.getMessage(); + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactReport.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactReport.java index 3d9674c..e9eba4d 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactReport.java +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactReport.java @@ -21,7 +21,22 @@ public record BiasImpactReport( List outcomeChanges, List mockedSideEffects, String summary, - List warnings) { + List warnings, + /** + * Which shape this report was computed in. + * + *

A comparison is persisted once and served from storage afterwards, so a report produced + * before the per-subject figures existed would keep answering with them empty forever. This + * is what lets the service recompute one instead. + */ + int schemaVersion, + /** Present only once someone has asked a model to assess this comparison. */ + BiasJudgeSummary judge, + /** How the interactive steps were answered on each side; null when neither was simulated. */ + BiasSimulationContext simulation) { + + /** Bumped whenever a new computed section would be missing from an already persisted report. */ + public static final int CURRENT_SCHEMA_VERSION = 3; public BiasImpactReport { interventionDirection = java.util.Objects.requireNonNull(interventionDirection, "interventionDirection"); @@ -31,5 +46,25 @@ public record BiasImpactReport( outcomeChanges = outcomeChanges == null ? List.of() : List.copyOf(outcomeChanges); mockedSideEffects = mockedSideEffects == null ? List.of() : List.copyOf(mockedSideEffects); warnings = warnings == null ? List.of() : List.copyOf(warnings); + // Absent in the JSON of every report written before this field existed. + schemaVersion = schemaVersion < 1 ? 1 : schemaVersion; + } + + public boolean outdated() { + return schemaVersion < CURRENT_SCHEMA_VERSION; + } + + public BiasImpactReport withJudge(BiasJudgeSummary judgeSummary) { + return new BiasImpactReport(id, experimentId, kind, interventionDirection, baselineExecutionId, + biasedExecutionId, nodeId, annotationIds, repetitions, createdAt, rawOutputsIncluded, immediateImpact, + downstreamImpact, routingChanges, outcomeChanges, mockedSideEffects, summary, warnings, schemaVersion, + judgeSummary, simulation); + } + + public BiasImpactReport withImpact(BiasOutputImpact immediate, List downstream) { + return new BiasImpactReport(id, experimentId, kind, interventionDirection, baselineExecutionId, + biasedExecutionId, nodeId, annotationIds, repetitions, createdAt, rawOutputsIncluded, immediate, + downstream, routingChanges, outcomeChanges, mockedSideEffects, summary, warnings, schemaVersion, judge, + simulation); } } diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactService.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactService.java index 098bb91..e08504a 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactService.java +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasImpactService.java @@ -19,6 +19,7 @@ import org.springframework.stereotype.Service; import org.springframework.transaction.annotation.Transactional; import it.cnr.isti.workflow.manager.blocks.Block; +import it.cnr.isti.workflow.manager.containers.Container; import it.cnr.isti.workflow.manager.executions.ExecutionObject; import it.cnr.isti.workflow.manager.executions.ExecutionEventType; import it.cnr.isti.workflow.manager.executions.ExecutionsService; @@ -29,6 +30,7 @@ import it.cnr.isti.workflow.manager.executions.executors.NodeExecutors; import it.cnr.isti.workflow.manager.executions.steps.Step; import it.cnr.isti.workflow.manager.flows.model.Connection; import it.cnr.isti.workflow.manager.flows.model.Dependency; +import it.cnr.isti.workflow.manager.llms.LLMDescriptor; import it.cnr.isti.workflow.manager.flows.validation.ValidationErrorCode; @Service @@ -37,12 +39,17 @@ public class BiasImpactService { private final ExecutionsService executionsService; private final BiasImpactReportRepository reportRepository; private final BiasBehaviorAdapterRegistry adapterRegistry; + private final BiasIterationComparator iterationComparator; + private final BiasImpactJudge judge; public BiasImpactService(ExecutionsService executionsService, BiasImpactReportRepository reportRepository, - BiasBehaviorAdapterRegistry adapterRegistry) { + BiasBehaviorAdapterRegistry adapterRegistry, BiasIterationComparator iterationComparator, + BiasImpactJudge judge) { this.executionsService = executionsService; this.reportRepository = reportRepository; this.adapterRegistry = adapterRegistry; + this.iterationComparator = iterationComparator; + this.judge = judge; } @Transactional @@ -77,7 +84,12 @@ public class BiasImpactService { variantContext)); } - BiasOutputImpact impact = outputImpact(baselineOutput, biasedOutputs, request.includeRawOutputs()); + List warnings = new ArrayList<>(List.of( + "The baseline output was captured from the selected completed execution; only experimental variants were repeated.")); + List values = fieldImpacts(block.getId(), block.getName(), baselineOutput, biasedOutputs, + warnings); + BiasOutputImpact impact = outputImpact(baselineOutput, biasedOutputs, values, List.of(), + request.includeRawOutputs()); List routingChanges = routingChanges(block.getId(), baselineOutput, biasedOutputs.isEmpty() ? Map.of() : biasedOutputs.getFirst()); List mockedSideEffects = request.externalSideEffectPolicy() == ExternalSideEffectPolicy.MOCK @@ -101,11 +113,13 @@ public class BiasImpactService { routingChanges, List.of(), mockedSideEffects, - impact.outputChanged() - ? "The activated " + request.direction().name().toLowerCase() - + " intervention changed the observed output of the selected block." - : "No output change was observed for the selected block.", - List.of("The baseline output was captured from the selected completed execution; only experimental variants were repeated.")); + isolatedSummary(request.direction(), impact), + warnings, + BiasImpactReport.CURRENT_SCHEMA_VERSION, + null, + // An isolated experiment re-runs one block with the inputs it had: no interaction to + // simulate, so there is nothing to declare here. + null); return persist(report, owner, request.includeRawOutputs()); } @@ -174,11 +188,40 @@ public class BiasImpactService { "execution", biasedExecutionId, "Executions do not belong to the same history"); } - var existingReport = reportRepository + var existingEntity = reportRepository .findFirstByBaselineExecutionIdAndBiasedExecutionIdAndOwnerAndRawOutputsIncludedOrderByCreatedAtDesc( baselineExecutionId, biasedExecutionId, owner, includeRawOutputs); - if (existingReport.isPresent()) { - return existingReport.get().getReport(); + BiasImpactReport existingReport = existingEntity.map(BiasImpactReportEntity::getReport).orElse(null); + // A report computed before the per-subject sections existed would keep serving them empty: + // recompute it in place instead, keeping the id so links to it stay valid. + if (existingReport != null && !existingReport.outdated()) { + return existingReport; + } + + // Reusing the id of an outdated report makes the save an update, which the unique + // constraint on (baseline, biased, owner, rawOutputsIncluded) requires anyway. + BiasImpactReport report = computeFullFlowReport(baseline, biased, includeRawOutputs, owner, + existingReport == null ? UUID.randomUUID().toString() : existingReport.id()); + return persist(report, owner, includeRawOutputs); + } + + /** + * The comparison itself, without persistence. + * + *

Separate because the LLM evaluation needs the same aligned pairs with their texts, which a + * report persisted with {@code rawOutputsIncluded = false} no longer carries. + */ + BiasImpactReport computeFullFlowReport(ExecutionObject baseline, ExecutionObject biased, + boolean includeRawOutputs, String owner, String reportId) { + BiasExecutionContext context = biased.getBiasExecutionContext(); + List warnings = new ArrayList<>(List.of( + "Observed differences may also include normal non-determinism from external models.")); + + // Whoever answered the interactive steps is part of what produced these outputs, so a + // mismatch there belongs at the top of the report and not in the reader's assumptions. + BiasSimulationContext simulation = BiasSimulationContext.orNull(baseline, biased); + if (simulation != null && simulation.warning() != null) { + warnings.add(simulation.warning()); } Set activatedNodeIds = new LinkedHashSet<>(context.activeBiasAnnotationIdsByNode().keySet()); @@ -187,12 +230,21 @@ public class BiasImpactService { activatedNodeIds.addAll(context.mitigationSubflowActivatedContainerIds()); Set downstreamIds = downstreamNodeIds(baseline, activatedNodeIds); List downstream = downstreamIds.stream() - .map(nodeId -> compareStep(baseline, biased, nodeId, includeRawOutputs)) + .map(nodeId -> compareStep(baseline, biased, nodeId, includeRawOutputs, owner, warnings)) .toList(); Map baselineImmediate = combinedOutputs(baseline, activatedNodeIds); Map biasedImmediate = combinedOutputs(biased, activatedNodeIds); - BiasOutputImpact immediate = outputImpact(baselineImmediate, List.of(biasedImmediate), includeRawOutputs); + List immediateValues = new ArrayList<>(); + List immediateIterations = new ArrayList<>(); + for (String nodeId : activatedNodeIds) { + immediateValues.addAll(fieldImpacts(nodeId, BiasOutputs.nodeName(baseline, nodeId), + BiasOutputs.outputsOf(baseline, nodeId), + List.of(BiasOutputs.outputsOf(biased, nodeId)), warnings)); + immediateIterations.addAll(iterationsOf(baseline, biased, nodeId, owner, warnings)); + } + BiasOutputImpact immediate = outputImpact(baselineImmediate, List.of(biasedImmediate), immediateValues, + immediateIterations, includeRawOutputs); List routing = activatedNodeIds.stream() .flatMap(nodeId -> routingChanges(nodeId, outputsOf(baseline, nodeId), outputsOf(biased, nodeId)).stream()) .toList(); @@ -216,8 +268,8 @@ public class BiasImpactService { List outcomeChanges = Objects.equals(baselineOutcomeCodes, biasedOutcomeCodes) ? List.of() : List.of(new BiasOutcomeChange(baselineOutcomeCodes, biasedOutcomeCodes)); - BiasImpactReport report = new BiasImpactReport( - UUID.randomUUID().toString(), + return new BiasImpactReport( + reportId, context.experimentId(), BiasExperimentKind.FULL_FLOW, context.interventionDirection(), @@ -233,10 +285,11 @@ public class BiasImpactService { routing, outcomeChanges, mockedSideEffects, - "Observed " + changedDownstream + " changed downstream node(s), " + routing.size() - + " routing change(s), and " + outcomeChanges.size() + " outcome change(s).", - List.of("Observed differences may also include normal non-determinism from external models.")); - return persist(report, owner, includeRawOutputs); + fullFlowSummary(immediate, downstream, routing, outcomeChanges, changedDownstream), + warnings, + BiasImpactReport.CURRENT_SCHEMA_VERSION, + null, + simulation); } private List outcomeCodes(ExecutionObject execution) { @@ -306,13 +359,13 @@ public class BiasImpactService { } private BiasDownstreamImpact compareStep(ExecutionObject baseline, ExecutionObject biased, String nodeId, - boolean includeRawOutputs) { + boolean includeRawOutputs, String owner, List warnings) { Step baselineStep = baseline.getContext().getSteps().get(nodeId); Step biasedStep = biased.getContext().getSteps().get(nodeId); Map baselineOutputs = outputsOf(baseline, nodeId); Map biasedOutputs = outputsOf(biased, nodeId); - String baselineStatus = baselineStep == null ? "MISSING" : baselineStep.getStatus().name(); - String biasedStatus = biasedStep == null ? "MISSING" : biasedStep.getStatus().name(); + String baselineStatus = baselineStep == null ? BiasIterationImpact.MISSING_STATUS : baselineStep.getStatus().name(); + String biasedStatus = biasedStep == null ? BiasIterationImpact.MISSING_STATUS : biasedStep.getStatus().name(); boolean changed = !Objects.equals(baselineStatus, biasedStatus) || !Objects.equals(baselineOutputs, biasedOutputs); String nodeName = baselineStep == null ? nodeId : baselineStep.getNode().getName(); return new BiasDownstreamImpact( @@ -322,7 +375,58 @@ public class BiasImpactService { biasedStatus, changed, includeRawOutputs ? baselineOutputs : Map.of(), - includeRawOutputs ? biasedOutputs : Map.of()); + includeRawOutputs ? biasedOutputs : Map.of(), + fieldImpacts(nodeId, nodeName, baselineOutputs, List.of(biasedOutputs), warnings), + iterationsOf(baseline, biased, nodeId, owner, warnings)); + } + + /** + * The per-iteration comparison of a container step, empty for anything else. + * + *

Only asked for when both runs know the node as a container: the comparator would answer + * empty anyway, and this keeps a comparison of two long flows from loading every child run. + */ + private List iterationsOf(ExecutionObject baseline, ExecutionObject biased, String nodeId, + String owner, List warnings) { + Step baselineStep = baseline.getContext().getSteps().get(nodeId); + Step biasedStep = biased.getContext().getSteps().get(nodeId); + if (baselineStep == null || biasedStep == null || !(baselineStep.getNode() instanceof Container container)) { + return List.of(); + } + BiasIterationComparator.Comparison comparison = iterationComparator.compare(baseline.getId(), biased.getId(), + nodeId, container.getName(), owner); + warnings.addAll(comparison.warnings()); + return comparison.iterations(); + } + + /** + * One {@link BiasValueImpact} per output field of a node, against the variant that moved it most. + * + *

The worst variant rather than the first: an isolated experiment repeats the intervention to + * see how consistently it lands, and a field's headline figure that ignored the repetition where + * it landed hardest would understate exactly what the repetitions were for. + */ + private List fieldImpacts(String nodeId, String nodeName, Map baseline, + List> variants, List warnings) { + Set fields = new LinkedHashSet<>(baseline.keySet()); + variants.forEach(variant -> fields.addAll(variant.keySet())); + + List values = new ArrayList<>(); + for (String field : fields) { + BiasValueImpact worst = null; + for (Map variant : variants) { + BiasValueAnalyzer.Analysis analysis = BiasValueAnalyzer.analyze(nodeId, nodeName, field, + baseline.get(field), variant.get(field)); + warnings.addAll(analysis.warnings()); + if (worst == null || analysis.impact().textDifference() > worst.textDifference()) { + worst = analysis.impact(); + } + } + if (worst != null) { + values.add(worst); + } + } + return values; } private Set downstreamNodeIds(ExecutionObject execution, Set sourceIds) { @@ -355,34 +459,23 @@ public class BiasImpactService { } private Map outputsOf(ExecutionObject execution, String nodeId) { - Map outputs = new LinkedHashMap<>(); - execution.getContext().getResult().forEach((key, value) -> { - if (nodeId.equals(key.nodeId())) { - outputs.put(key.fieldId(), value); - } - }); - for (Connection connection : execution.getStepConnections()) { - if (!nodeId.equals(connection.getSourceId())) { - continue; - } - Step targetStep = execution.getContext().getSteps().get(connection.getTargetId()); - if (targetStep == null) { - continue; - } - targetStep.getInputs().stream() - .filter(input -> connection.getTargetName().equals(input.getDescriptor().getName())) - .filter(input -> input.getValue() != null) - .findFirst() - .ifPresent(input -> outputs.put(connection.getSourceName(), input.getValue())); - } - return Collections.unmodifiableMap(outputs); + return BiasOutputs.outputsOf(execution, nodeId); } + /** + * The headline figures, computed from the field-level comparison rather than from the outputs + * printed as one string. + * + *

{@code maximumTextDifference} used to be an edit distance between two stringified maps. On + * a flow whose activated node emits several fields - or a list of them, one per subject - that + * number was mostly a measure of the map's punctuation, and it could not be traced back to + * anything a reader could act on. + */ private BiasOutputImpact outputImpact(Map baseline, List> variants, - boolean includeRawOutputs) { + List values, List iterations, boolean includeRawOutputs) { long changed = variants.stream().filter(variant -> !Objects.equals(baseline, variant)).count(); - double maximumDifference = variants.stream() - .mapToDouble(variant -> textDifference(baseline.toString(), variant.toString())) + double maximumDifference = values.stream() + .mapToDouble(BiasValueImpact::textDifference) .max() .orElse(0.0); return new BiasOutputImpact( @@ -390,7 +483,15 @@ public class BiasImpactService { maximumDifference, variants.isEmpty() ? 0.0 : (double) changed / variants.size(), includeRawOutputs ? baseline : Map.of(), - includeRawOutputs ? variants : List.of()); + includeRawOutputs ? variants : List.of(), + includeRawOutputs ? values : values.stream().map(BiasValueImpact::withoutTexts).toList(), + includeRawOutputs ? iterations + : iterations.stream().map(BiasImpactService::withoutTexts).toList()); + } + + /** The same comparison with the raw texts dropped, for a report asked not to carry them. */ + private static BiasIterationImpact withoutTexts(BiasIterationImpact iteration) { + return iteration.withValues(iteration.values().stream().map(BiasValueImpact::withoutTexts).toList()); } private List routingChanges(String nodeId, Map baseline, @@ -403,28 +504,134 @@ public class BiasImpactService { return List.of(new BiasRoutingChange(nodeId, baselineBranch, biasedBranch)); } - private double textDifference(String left, String right) { - if (Objects.equals(left, right)) { - return 0.0; + /** + * The one-line reading of a comparison, in the order a reviewer needs it. + * + *

It leads with the decision - a changed outcome or a changed branch - because that is the + * only part of a run that leaves the system. Counting changed nodes first, as this used to, put + * the least actionable number in the most prominent place. + */ + private String fullFlowSummary(BiasOutputImpact immediate, List downstream, + List routing, List outcomeChanges, long changedDownstream) { + List sentences = new ArrayList<>(); + if (!outcomeChanges.isEmpty()) { + BiasOutcomeChange change = outcomeChanges.getFirst(); + sentences.add("The final outcome changed from " + describeCodes(change.baselineOutcomeCodes()) + + " to " + describeCodes(change.biasedOutcomeCodes()) + "."); + } else if (!routing.isEmpty()) { + BiasRoutingChange change = routing.getFirst(); + sentences.add("The flow took a different branch at " + change.nodeId() + ": " + + change.baselineBranch() + " became " + change.biasedBranch() + "."); + } else { + sentences.add("The final outcome did not change."); } - int maximumLength = Math.max(left.length(), right.length()); - if (maximumLength == 0) { - return 0.0; + + int subjects = subjectsCompared(immediate); + int changedSubjects = subjectsChanged(immediate); + if (subjects > 0) { + sentences.add(changedSubjects + " of " + subjects + + (subjects == 1 ? " evaluated subject changed." : " evaluated subjects changed.")); } - int[] previous = new int[right.length() + 1]; - for (int column = 0; column <= right.length(); column++) { - previous[column] = column; + + strongestDelta(immediate).ifPresent(delta -> sentences.add(delta.label() + " moved by " + + String.format(java.util.Locale.ROOT, "%+.2f", delta.delta()) + " on average.")); + + sentences.add("Observed " + changedDownstream + " changed downstream node(s) out of " + downstream.size() + + ", and " + routing.size() + " routing change(s)."); + return String.join(" ", sentences); + } + + private String isolatedSummary(BiasInterventionDirection direction, BiasOutputImpact impact) { + if (!impact.outputChanged()) { + return "No output change was observed for the selected block."; } - for (int row = 1; row <= left.length(); row++) { - int[] current = new int[right.length() + 1]; - current[0] = row; - for (int column = 1; column <= right.length(); column++) { - int substitution = previous[column - 1] - + (left.charAt(row - 1) == right.charAt(column - 1) ? 0 : 1); - current[column] = Math.min(Math.min(current[column - 1] + 1, previous[column] + 1), substitution); - } - previous = current; + StringBuilder summary = new StringBuilder("The activated " + direction.name().toLowerCase() + + " intervention changed the observed output of the selected block."); + strongestDelta(impact).ifPresent(delta -> summary.append(" ").append(delta.label()).append(" moved by ") + .append(String.format(java.util.Locale.ROOT, "%+.2f", delta.delta())).append(" on average.")); + return summary.toString(); + } + + private String describeCodes(List codes) { + return codes.isEmpty() ? "none" : String.join(", ", codes); + } + + /** Every list element of an activated node, plus every iteration of an activated container. */ + private int subjectsCompared(BiasOutputImpact immediate) { + int fromValues = immediate.values().stream().mapToInt(BiasValueImpact::itemsCompared).max().orElse(0); + return Math.max(fromValues, immediate.iterations().size()); + } + + private int subjectsChanged(BiasOutputImpact immediate) { + int fromValues = immediate.values().stream().mapToInt(BiasValueImpact::itemsChanged).max().orElse(0); + return Math.max(fromValues, (int) immediate.iterations().stream().filter(BiasIterationImpact::changed).count()); + } + + /** The labelled number that moved furthest, which is the only figure in the flow's own units. */ + private java.util.Optional strongestDelta(BiasOutputImpact immediate) { + return immediate.values().stream() + .flatMap(value -> value.numericDeltas().stream()) + .max(java.util.Comparator.comparingDouble(delta -> Math.abs(delta.delta()))); + } + + /** + * Asks a model to assess an already persisted comparison, and stores what it said. + * + *

The comparison is recomputed rather than read back: a report saved with + * {@code rawOutputsIncluded = false} no longer carries the two texts, and judging without them + * would be judging the figures rather than the outputs. Both executions are final, so the + * recomputation is the same comparison, and only the verdicts are written onto the stored report. + * + *

The evaluation replaces any previous one. The descriptor and timestamp on it say which model + * produced the verdicts now on display, which is what makes a second opinion worth asking for. + */ + @Transactional + public BiasImpactReport judgeReport(String reportId, LLMDescriptor descriptor, String owner) { + BiasImpactReportEntity entity = reportRepository.findByIdAndOwner(reportId, owner) + .orElseThrow(() -> new BiasApiException(HttpStatus.NOT_FOUND, ValidationErrorCode.BIAS_REPORT_NOT_FOUND, + "biasImpactReport", reportId, "Bias impact report not found: " + reportId)); + BiasImpactReport stored = entity.getReport(); + + BiasImpactReport comparison = comparisonForJudging(stored, owner); + ExecutionObject baseline = executionsService.getExecutionByOwner(stored.baselineExecutionId(), owner); + ExecutionObject biased = stored.biasedExecutionId() == null + ? baseline + : executionsService.getExecutionByOwner(stored.biasedExecutionId(), owner); + + BiasImpactJudge.Judgment judgment = judge.evaluate(comparison, descriptor, baseline, biased); + BiasImpactReport judged = BiasJudgeTarget.apply(stored, judgment.verdicts()).withJudge(judgment.summary()); + entity.setReport(judged); + reportRepository.save(entity); + return judged; + } + + /** + * The comparison to judge, with its texts. + * + *

A full-flow report is recomputed from the two executions. An isolated experiment cannot be: + * its variants were produced by running the block again and were never persisted anywhere but in + * the report itself, so there the stored report is the only account of what happened - which is + * also why judging one is refused when it was saved without its raw outputs. + */ + private BiasImpactReport comparisonForJudging(BiasImpactReport stored, String owner) { + if (stored.kind() == BiasExperimentKind.FULL_FLOW && stored.biasedExecutionId() != null) { + ExecutionObject baseline = executionsService.getExecutionByOwner(stored.baselineExecutionId(), owner); + ExecutionObject biased = executionsService.getExecutionByOwner(stored.biasedExecutionId(), owner); + return computeFullFlowReport(baseline, biased, true, owner, stored.id()); } - return (double) previous[right.length()] / maximumLength; + if (!stored.rawOutputsIncluded()) { + throw new BiasApiException(HttpStatus.CONFLICT, ValidationErrorCode.BIAS_JUDGE_UNAVAILABLE, + "biasImpactReport", stored.id(), + "This experiment was saved without its raw outputs, so there is nothing left to assess. " + + "Run it again with raw outputs included."); + } + // An isolated report from before the field-by-field comparison existed has no pairs to hand + // the judge, but it does still carry the outputs they are made of. + ExecutionObject baseline = executionsService.getExecutionByOwner(stored.baselineExecutionId(), owner); + String nodeId = stored.nodeId() == null ? "" : stored.nodeId(); + List values = fieldImpacts(nodeId, BiasOutputs.nodeName(baseline, nodeId), + stored.immediateImpact().baselineOutput(), stored.immediateImpact().biasedOutputs(), + new ArrayList<>()); + return stored.withImpact(stored.immediateImpact().withValues(values), stored.downstreamImpact()); } } diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasItemImpact.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasItemImpact.java new file mode 100644 index 0000000..3855d13 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasItemImpact.java @@ -0,0 +1,34 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.util.List; + +/** + * One element of a compared list value, which for an iterator container is one participant. + * + *

Elements are paired by position: the iterated input list is the same on both sides of a bias + * rerun, so index {@code i} is the same subject in both runs. No attempt is made to recognise the + * subject inside the text - a name match would silently repair nothing and invent pairings when it + * failed. + */ +public record BiasItemImpact( + int index, + boolean changed, + double textDifference, + List numericDeltas, + String baselineText, + String biasedText, + BiasJudgeVerdict judgeVerdict) { + + public BiasItemImpact { + numericDeltas = numericDeltas == null ? List.of() : List.copyOf(numericDeltas); + } + + public BiasItemImpact withVerdict(BiasJudgeVerdict verdict) { + return new BiasItemImpact(index, changed, textDifference, numericDeltas, baselineText, biasedText, verdict); + } + + /** The same item without its raw texts, for a report that was asked not to carry them. */ + public BiasItemImpact withoutTexts() { + return new BiasItemImpact(index, changed, textDifference, numericDeltas, null, null, judgeVerdict); + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasIterationComparator.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasIterationComparator.java new file mode 100644 index 0000000..c5a8f96 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasIterationComparator.java @@ -0,0 +1,149 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.util.ArrayList; +import java.util.LinkedHashMap; +import java.util.LinkedHashSet; +import java.util.List; +import java.util.Map; +import java.util.Set; + +import org.springframework.stereotype.Component; +import org.springframework.web.server.ResponseStatusException; + +import it.cnr.isti.workflow.manager.executions.ContainerSubflowRole; +import it.cnr.isti.workflow.manager.executions.ExecutionObject; +import it.cnr.isti.workflow.manager.executions.ExecutionsService; + +/** + * Lines up the two runs of a container step iteration by iteration. + * + *

An iterator container evaluates one subject per iteration, and its own output is the + * accumulated list. Comparing only that says a list changed; comparing the child executions says + * which iteration changed and which node inside the subflow changed it - which is the difference + * between a report that flags a run and one that points at a candidate. + * + *

Iterations are joined on {@code parentIterationIndex}, which each child already records. Guard + * subflows are left out: a loop creates one alongside every main iteration, and pairing a guard + * with a main run reported every loop as rewritten. + */ +@Component +public class BiasIterationComparator { + + /** + * Iterations analyzed per container step. A hundred-subject run would otherwise turn one + * comparison into a hundred nested ones, and the first fifty already answer the question. + */ + public static final int MAX_ITERATIONS = 50; + + private final ExecutionsService executionsService; + + public BiasIterationComparator(ExecutionsService executionsService) { + this.executionsService = executionsService; + } + + /** The per-iteration comparison of one container step, plus what the caller must be told. */ + public record Comparison(List iterations, List warnings) { + + public Comparison { + iterations = iterations == null ? List.of() : List.copyOf(iterations); + warnings = warnings == null ? List.of() : List.copyOf(warnings); + } + + static Comparison empty() { + return new Comparison(List.of(), List.of()); + } + } + + public Comparison compare(String baselineExecutionId, String biasedExecutionId, String stepId, String stepName, + String owner) { + Map baselineIterations = mainIterationsOf(baselineExecutionId, stepId, owner); + Map biasedIterations = mainIterationsOf(biasedExecutionId, stepId, owner); + if (baselineIterations.isEmpty() && biasedIterations.isEmpty()) { + return Comparison.empty(); + } + + Set indexes = new LinkedHashSet<>(baselineIterations.keySet()); + indexes.addAll(biasedIterations.keySet()); + List ordered = indexes.stream().sorted().toList(); + + List warnings = new ArrayList<>(); + if (ordered.size() > MAX_ITERATIONS) { + warnings.add("Only the first " + MAX_ITERATIONS + " of " + ordered.size() + + " iterations of the container step were compared."); + } + if (!baselineIterations.keySet().equals(biasedIterations.keySet())) { + warnings.add("The two runs did not perform the same iterations of the container step: " + + baselineIterations.size() + " against " + biasedIterations.size() + "."); + } + + List impacts = new ArrayList<>(); + for (Integer index : ordered.subList(0, Math.min(ordered.size(), MAX_ITERATIONS))) { + impacts.add(compareIteration(stepId, stepName, index, baselineIterations.get(index), + biasedIterations.get(index), warnings)); + } + return new Comparison(impacts, warnings); + } + + private BiasIterationImpact compareIteration(String stepId, String stepName, int index, ExecutionObject baseline, + ExecutionObject biased, List warnings) { + if (baseline == null || biased == null) { + return new BiasIterationImpact( + stepId, + stepName, + index, + baseline == null ? null : baseline.getId(), + biased == null ? null : biased.getId(), + baseline == null ? BiasIterationImpact.MISSING_STATUS : baseline.getContext().getStatus().name(), + biased == null ? BiasIterationImpact.MISSING_STATUS : biased.getContext().getStatus().name(), + true, + List.of()); + } + + Set nodeIds = new LinkedHashSet<>(BiasOutputs.nodeIdsWithResults(baseline)); + nodeIds.addAll(BiasOutputs.nodeIdsWithResults(biased)); + + List values = new ArrayList<>(); + for (String nodeId : nodeIds) { + Map baselineOutputs = BiasOutputs.outputsOf(baseline, nodeId); + Map biasedOutputs = BiasOutputs.outputsOf(biased, nodeId); + Set fields = new LinkedHashSet<>(baselineOutputs.keySet()); + fields.addAll(biasedOutputs.keySet()); + for (String field : fields) { + BiasValueAnalyzer.Analysis analysis = BiasValueAnalyzer.analyze(nodeId, + BiasOutputs.nodeName(baseline, nodeId), field, + baselineOutputs.get(field), biasedOutputs.get(field)); + warnings.addAll(analysis.warnings()); + // Only what moved: an iteration listing every unchanged inner output is the wall of + // text this view exists to replace. + if (analysis.impact().changed()) { + values.add(analysis.impact()); + } + } + } + + String baselineStatus = baseline.getContext().getStatus().name(); + String biasedStatus = biased.getContext().getStatus().name(); + return new BiasIterationImpact(stepId, stepName, index, baseline.getId(), biased.getId(), baselineStatus, + biasedStatus, !values.isEmpty() || !baselineStatus.equals(biasedStatus), values); + } + + private Map mainIterationsOf(String executionId, String stepId, String owner) { + Map byIndex = new LinkedHashMap<>(); + List children; + try { + children = executionsService.getContainerIterationsByOwner(executionId, stepId, owner); + } catch (ResponseStatusException exception) { + // How the service says the step is not a container, or is not in this run at all. The + // rest of the comparison is still worth having; anything else is a fault, not an answer. + return byIndex; + } + for (ExecutionObject child : children) { + if (child.getSubflowRole() != null && child.getSubflowRole() != ContainerSubflowRole.MAIN) { + continue; + } + Integer index = child.getParentIterationIndex(); + byIndex.putIfAbsent(index == null ? byIndex.size() + 1 : index, child); + } + return byIndex; + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasIterationImpact.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasIterationImpact.java new file mode 100644 index 0000000..0720e86 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasIterationImpact.java @@ -0,0 +1,40 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.util.List; + +/** + * One iteration of a container step, from the pair of child executions that ran it. + * + *

A container's own output is the accumulated list, which shows that something moved but not + * where inside the subflow it moved. The child executions carry that, and they are already + * addressable: each records the {@code parentIterationIndex} it ran for. + */ +public record BiasIterationImpact( + /** The container step these iterations belong to: a flow can activate more than one. */ + String containerNodeId, + String containerNodeName, + int index, + String baselineExecutionId, + String biasedExecutionId, + String baselineStatus, + String biasedStatus, + boolean changed, + List values) { + + /** The status recorded for a side that has no child execution for this index at all. */ + public static final String MISSING_STATUS = "MISSING"; + + public BiasIterationImpact { + values = values == null ? List.of() : List.copyOf(values); + } + + public BiasIterationImpact withValues(List replacements) { + return new BiasIterationImpact(containerNodeId, containerNodeName, index, baselineExecutionId, + biasedExecutionId, baselineStatus, biasedStatus, changed, replacements); + } + + /** The path this iteration is addressed by when verdicts are attached back to a report. */ + public String key() { + return containerNodeId + "#" + index; + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeAttribution.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeAttribution.java new file mode 100644 index 0000000..4f68503 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeAttribution.java @@ -0,0 +1,14 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +/** + * Whether the change carries the intervention's fingerprint, or looks like ordinary model variation. + * + *

Kept separate from {@link BiasJudgeImpactLevel} on purpose: a large change that the model would + * have made anyway and a small change that does exactly what the probe asked for are different + * findings, and one axis cannot say both. + */ +public enum BiasJudgeAttribution { + INJECTION, + NON_DETERMINISM, + UNCLEAR +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeImpactLevel.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeImpactLevel.java new file mode 100644 index 0000000..4429c88 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeImpactLevel.java @@ -0,0 +1,20 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +/** How far the meaning of an output moved, as assessed by the judge model. */ +public enum BiasJudgeImpactLevel { + NONE, + COSMETIC, + SUBSTANTIVE, + DECISIVE; + + /** Ordering is the declaration order, so the worst level of a set is its maximum. */ + public static BiasJudgeImpactLevel worst(BiasJudgeImpactLevel left, BiasJudgeImpactLevel right) { + if (left == null) { + return right; + } + if (right == null) { + return left; + } + return left.compareTo(right) >= 0 ? left : right; + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeRequest.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeRequest.java new file mode 100644 index 0000000..c609ee6 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeRequest.java @@ -0,0 +1,15 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import it.cnr.isti.workflow.manager.llms.LLMDescriptor; +import jakarta.validation.Valid; +import jakarta.validation.constraints.NotNull; + +/** + * Which model to ask for an assessment of an already computed comparison. + * + *

Deliberately shaped like {@code ExecutionSimulationRequest}: the model that judges a run is + * picked when the judging is asked for, the same way the model that stands in for a person is picked + * when a simulated run is launched, and the editor reuses the same provider and model pickers. + */ +public record BiasJudgeRequest(@Valid @NotNull LLMDescriptor judge) { +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeSummary.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeSummary.java new file mode 100644 index 0000000..31eff01 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeSummary.java @@ -0,0 +1,31 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.time.LocalDateTime; +import java.util.List; + +import it.cnr.isti.workflow.manager.llms.LLMDescriptor; + +/** + * The report-level result of an LLM evaluation, and the model that produced it. + * + *

The descriptor is stored, not just used: a verdict without the model and sampling parameters + * behind it cannot be checked or reproduced, which would make it advice rather than evidence. + * + *

{@code impact} and {@code attribution} are rolled up from the per-pair verdicts in code, not + * asked of the model a second time - only {@code narrative} comes from it, so the headline of a + * report cannot drift between two readings of the same comparison. + */ +public record BiasJudgeSummary( + LLMDescriptor judge, + LocalDateTime judgedAt, + BiasJudgeImpactLevel impact, + BiasJudgeAttribution attribution, + String narrative, + int judgedPairs, + int skippedPairs, + List errors) { + + public BiasJudgeSummary { + errors = errors == null ? List.of() : List.copyOf(errors); + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeTarget.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeTarget.java new file mode 100644 index 0000000..5dc575e --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeTarget.java @@ -0,0 +1,132 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.util.ArrayList; +import java.util.List; +import java.util.Map; + +/** + * One aligned pair of outputs to be assessed, and where it sits in the report. + * + *

The paths are what let a verdict be written back onto a persisted report: the comparison is + * recomputed with its raw texts to build the prompts, and only the verdicts travel back, addressed + * by path. + */ +public record BiasJudgeTarget( + String path, + String nodeName, + String field, + /** Which subject this pair belongs to: the list element or the iteration index. */ + Integer subjectIndex, + boolean changed, + double textDifference, + List numericDeltas, + String baselineText, + String biasedText) { + + /** + * Every pair worth an opinion, in the order they should be spent on. + * + *

Subjects first: on a per-subject run they are the finding, and the aggregated field that + * follows from them is only worth judging on its own when there are no subjects to judge. + */ + public static List collect(BiasImpactReport report) { + List targets = new ArrayList<>(); + for (BiasValueImpact value : report.immediateImpact().values()) { + targets.addAll(fromValue("immediate", value)); + } + for (BiasIterationImpact iteration : report.immediateImpact().iterations()) { + for (BiasValueImpact value : iteration.values()) { + targets.addAll(fromIterationValue("immediate", iteration, value)); + } + } + for (BiasDownstreamImpact downstream : report.downstreamImpact()) { + for (BiasValueImpact value : downstream.values()) { + targets.addAll(fromValue("downstream", value)); + } + for (BiasIterationImpact iteration : downstream.iterations()) { + for (BiasValueImpact value : iteration.values()) { + targets.addAll(fromIterationValue("downstream", iteration, value)); + } + } + } + // A changed pair is the point; the unchanged ones are kept last so a cap spends the calls + // where something actually moved. + List ordered = new ArrayList<>(targets.stream().filter(BiasJudgeTarget::changed).toList()); + ordered.addAll(targets.stream().filter(target -> !target.changed()).toList()); + return ordered; + } + + private static List fromValue(String scope, BiasValueImpact value) { + String base = scope + ":" + value.key(); + if (value.items().isEmpty()) { + return List.of(new BiasJudgeTarget(base, value.nodeName(), value.field(), null, value.changed(), + value.textDifference(), value.numericDeltas(), value.baselineText(), value.biasedText())); + } + return value.items().stream() + .map(item -> new BiasJudgeTarget(base + "#" + item.index(), value.nodeName(), value.field(), + item.index(), item.changed(), item.textDifference(), item.numericDeltas(), + item.baselineText(), item.biasedText())) + .toList(); + } + + private static List fromIterationValue(String scope, BiasIterationImpact iteration, + BiasValueImpact value) { + String base = scope + "-iteration:" + iteration.key() + ":" + value.key(); + if (value.items().isEmpty()) { + return List.of(new BiasJudgeTarget(base, value.nodeName(), value.field(), iteration.index(), + value.changed(), value.textDifference(), value.numericDeltas(), value.baselineText(), + value.biasedText())); + } + return value.items().stream() + .map(item -> new BiasJudgeTarget(base + "#" + item.index(), value.nodeName(), value.field(), + iteration.index(), item.changed(), item.textDifference(), item.numericDeltas(), + item.baselineText(), item.biasedText())) + .toList(); + } + + /** Writes verdicts back onto a report, matching the paths {@link #collect} handed out. */ + public static BiasImpactReport apply(BiasImpactReport report, Map verdicts) { + if (verdicts.isEmpty()) { + return report; + } + BiasOutputImpact immediate = report.immediateImpact() + .withValues(report.immediateImpact().values().stream() + .map(value -> applyToValue("immediate", value, verdicts)) + .toList()) + .withIterations(report.immediateImpact().iterations().stream() + .map(iteration -> applyToIteration("immediate", iteration, verdicts)) + .toList()); + List downstream = report.downstreamImpact().stream() + .map(entry -> entry + .withValues(entry.values().stream() + .map(value -> applyToValue("downstream", value, verdicts)) + .toList()) + .withIterations(entry.iterations().stream() + .map(iteration -> applyToIteration("downstream", iteration, verdicts)) + .toList())) + .toList(); + return report.withImpact(immediate, downstream); + } + + private static BiasIterationImpact applyToIteration(String scope, BiasIterationImpact iteration, + Map verdicts) { + return iteration.withValues(iteration.values().stream() + .map(value -> applyToValue(scope + "-iteration:" + iteration.key(), value, verdicts)) + .toList()); + } + + private static BiasValueImpact applyToValue(String scope, BiasValueImpact value, + Map verdicts) { + String base = scope + ":" + value.key(); + BiasValueImpact judged = verdicts.containsKey(base) ? value.withVerdict(verdicts.get(base)) : value; + if (judged.items().isEmpty()) { + return judged; + } + return judged.withItems(judged.items().stream() + .map(item -> { + BiasJudgeVerdict verdict = verdicts.get(base + "#" + item.index()); + return verdict == null ? item : item.withVerdict(verdict); + }) + .toList()); + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeVerdict.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeVerdict.java new file mode 100644 index 0000000..940be0f --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeVerdict.java @@ -0,0 +1,37 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.util.List; + +/** + * What a judge model made of one aligned pair of outputs. + * + *

Every field is nullable and {@code error} exists because a model can always answer with + * something unusable: a comparison whose numbers are sound must not be thrown away because the + * narration of it failed, so a failed verdict is recorded as a failure next to the pair it belongs + * to rather than propagated. + */ +public record BiasJudgeVerdict( + BiasJudgeImpactLevel impact, + BiasJudgeAttribution attribution, + Double confidence, + List changedAspects, + String rationale, + String error) { + + public BiasJudgeVerdict { + changedAspects = changedAspects == null ? List.of() : List.copyOf(changedAspects); + } + + public static BiasJudgeVerdict identical() { + return new BiasJudgeVerdict(BiasJudgeImpactLevel.NONE, BiasJudgeAttribution.UNCLEAR, 1.0, + List.of(), "The two sides are identical, so no model was asked.", null); + } + + public static BiasJudgeVerdict failed(String error) { + return new BiasJudgeVerdict(null, null, null, List.of(), null, error); + } + + public boolean failure() { + return error != null && !error.isBlank(); + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasNumericDelta.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasNumericDelta.java new file mode 100644 index 0000000..da44912 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasNumericDelta.java @@ -0,0 +1,16 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +/** + * One labelled number that both sides of a comparison produced, and how far it moved. + * + *

This is the only impact figure stated in the units the flow itself uses: a scoring step that + * answers {@code Score: 8} against {@code Score: 5} moved by three points, which is a sentence a + * reviewer can act on. A normalized text distance over the same two answers is 0.04 and says + * nothing. + */ +public record BiasNumericDelta(String label, double baseline, double biased, double delta) { + + public static BiasNumericDelta of(String label, double baseline, double biased) { + return new BiasNumericDelta(label, baseline, biased, biased - baseline); + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasOutputImpact.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasOutputImpact.java index 96e6cc2..2eebf60 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasOutputImpact.java +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasOutputImpact.java @@ -8,10 +8,35 @@ public record BiasOutputImpact( double maximumTextDifference, double changeRate, Map baselineOutput, - List> biasedOutputs) { + List> biasedOutputs, + /** + * The comparison field by field, and element by element inside a list. + * + *

{@code maximumTextDifference} is the worst of these, not a distance over every output + * stringified into one map: that number moved with the shape of the map and could not tell a + * rewritten answer from a reordered one. + */ + List values, + /** + * One entry per iteration when the intervention was activated on a container, empty + * otherwise. This is where a per-subject iterator run says which subject moved. + */ + List iterations) { public BiasOutputImpact { baselineOutput = baselineOutput == null ? Map.of() : Map.copyOf(baselineOutput); biasedOutputs = biasedOutputs == null ? List.of() : List.copyOf(biasedOutputs); + values = values == null ? List.of() : List.copyOf(values); + iterations = iterations == null ? List.of() : List.copyOf(iterations); + } + + public BiasOutputImpact withValues(List replacements) { + return new BiasOutputImpact(outputChanged, maximumTextDifference, changeRate, baselineOutput, biasedOutputs, + replacements, iterations); + } + + public BiasOutputImpact withIterations(List replacements) { + return new BiasOutputImpact(outputChanged, maximumTextDifference, changeRate, baselineOutput, biasedOutputs, + values, replacements); } } diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasOutputs.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasOutputs.java new file mode 100644 index 0000000..dd26e64 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasOutputs.java @@ -0,0 +1,67 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.util.Collections; +import java.util.LinkedHashMap; +import java.util.LinkedHashSet; +import java.util.Map; +import java.util.Set; + +import it.cnr.isti.workflow.manager.executions.ExecutionObject; +import it.cnr.isti.workflow.manager.executions.steps.Step; +import it.cnr.isti.workflow.manager.flows.model.Connection; + +/** Reading a node's outputs out of a finished execution, shared by everything that compares two. */ +final class BiasOutputs { + + private BiasOutputs() { + } + + /** + * What one node produced: its own recorded results, plus the values that reached the inputs it + * is wired to. + * + *

The second half is not redundant. A node whose result was consumed and not persisted under + * its own key is still observable on the other end of its connection, and dropping that made + * whole branches of a comparison look empty. + */ + static Map outputsOf(ExecutionObject execution, String nodeId) { + Map outputs = new LinkedHashMap<>(); + execution.getContext().getResult().forEach((key, value) -> { + if (nodeId.equals(key.nodeId())) { + outputs.put(key.fieldId(), value); + } + }); + for (Connection connection : execution.getStepConnections()) { + if (!nodeId.equals(connection.getSourceId())) { + continue; + } + Step targetStep = execution.getContext().getSteps().get(connection.getTargetId()); + if (targetStep == null) { + continue; + } + targetStep.getInputs().stream() + .filter(input -> connection.getTargetName().equals(input.getDescriptor().getName())) + .filter(input -> input.getValue() != null) + .findFirst() + .ifPresent(input -> outputs.put(connection.getSourceName(), input.getValue())); + } + return Collections.unmodifiableMap(outputs); + } + + /** Every node of an execution that recorded a result, in step order. */ + static Set nodeIdsWithResults(ExecutionObject execution) { + Set nodeIds = new LinkedHashSet<>(); + execution.getContext().getResult().keySet().forEach(key -> nodeIds.add(key.nodeId())); + return nodeIds; + } + + static String nodeName(ExecutionObject execution, String nodeId) { + Step step = execution.getContext().getSteps().get(nodeId); + return step == null || step.getNode() == null ? nodeId : step.getNode().getName(); + } + + static String statusOf(ExecutionObject execution, String nodeId) { + Step step = execution.getContext().getSteps().get(nodeId); + return step == null || step.getStatus() == null ? BiasIterationImpact.MISSING_STATUS : step.getStatus().name(); + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasSimulationContext.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasSimulationContext.java new file mode 100644 index 0000000..8b4cfea --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasSimulationContext.java @@ -0,0 +1,77 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.util.Objects; + +import it.cnr.isti.workflow.manager.executions.ExecutionObject; +import it.cnr.isti.workflow.manager.llms.LLMDescriptor; + +/** + * How the interactive steps of the two compared runs were answered. + * + *

A run whose human decisions came from a simulator and one whose came from a different simulator + * differ for a reason the intervention had no part in, and the figures elsewhere in the report cannot + * separate the two causes. Recorded here so the report can say so, months later, when the executions + * behind it may no longer exist. + */ +public record BiasSimulationContext( + boolean baselineSimulated, + LLMDescriptor baselineSimulator, + boolean biasedSimulated, + LLMDescriptor biasedSimulator, + /** False when only one side was simulated, or when the two simulators are not the same one. */ + boolean comparable) { + + public static BiasSimulationContext of(ExecutionObject baseline, ExecutionObject biased) { + boolean baselineSimulated = baseline.isInteractionSimulationEnabled(); + boolean biasedSimulated = biased.isInteractionSimulationEnabled(); + LLMDescriptor baselineSimulator = baselineSimulated ? baseline.getInteractionSimulationDescriptor() : null; + LLMDescriptor biasedSimulator = biasedSimulated ? biased.getInteractionSimulationDescriptor() : null; + // Same provider, same model, same sampling: a different seed is a different simulator, which + // is exactly the case worth flagging rather than glossing over. + boolean comparable = baselineSimulated == biasedSimulated + && Objects.equals(baselineSimulator, biasedSimulator); + return new BiasSimulationContext(baselineSimulated, baselineSimulator, biasedSimulated, biasedSimulator, + comparable); + } + + /** Null when neither run simulated anything: there is nothing to declare. */ + public static BiasSimulationContext orNull(ExecutionObject baseline, ExecutionObject biased) { + BiasSimulationContext context = of(baseline, biased); + return context.baselineSimulated() || context.biasedSimulated() ? context : null; + } + + /** The caveat to put among the report's warnings, or null when the two sides match. */ + public String warning() { + if (comparable) { + return null; + } + if (baselineSimulated != biasedSimulated) { + String simulated = baselineSimulated ? "baseline" : "variant"; + String answered = baselineSimulated ? "variant" : "baseline"; + return "Interactive steps were simulated on the " + simulated + " run and answered directly on the " + + answered + " run, so part of the difference does not come from the intervention."; + } + return "The two runs were simulated with different models or settings (" + describe(baselineSimulator) + + " against " + describe(biasedSimulator) + + "), so part of the difference may come from the simulator rather than the intervention."; + } + + public String describe() { + if (!baselineSimulated && !biasedSimulated) { + return "not simulated"; + } + return comparable + ? "simulated on both sides with " + describe(baselineSimulator) + : describe(baselineSimulator) + " against " + describe(biasedSimulator); + } + + private static String describe(LLMDescriptor descriptor) { + if (descriptor == null) { + return "no simulator"; + } + String seed = descriptor.parameters() == null || descriptor.parameters().seed() == null + ? "" + : " seed " + descriptor.parameters().seed(); + return descriptor.provider() + "/" + descriptor.model() + seed; + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasValueAnalyzer.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasValueAnalyzer.java new file mode 100644 index 0000000..a781533 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasValueAnalyzer.java @@ -0,0 +1,267 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.util.ArrayList; +import java.util.LinkedHashMap; +import java.util.LinkedHashSet; +import java.util.List; +import java.util.Map; +import java.util.Objects; +import java.util.Set; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +import it.cnr.isti.workflow.manager.app.JacksonConverterSupport; + +/** + * Compares two values that the same output port produced in two runs. + * + *

The one place where a difference is quantified. It exists because the comparison used to + * stringify every activated node's outputs into one map and measure that: the resulting number was + * dominated by the map's own punctuation, and on a flow that evaluates one subject per iteration it + * could not say which subject had moved, by how much, or whether anything of substance had changed + * at all. + * + *

Three figures, in increasing order of usefulness: a normalized edit distance on the text, the + * share of list elements that changed, and the movement of numbers that both sides label the same + * way. The last one is the only one in the flow's own units, and it is also the only one that can be + * absent - no extraction is guessed at. + */ +public final class BiasValueAnalyzer { + + /** + * Beyond this many elements the per-element detail costs more than it informs, so only the + * aggregates are kept. Reported as a warning rather than silently truncated. + */ + public static final int MAX_ITEMS = 200; + + /** + * The edit distance is quadratic in the two lengths. Two long model answers are exactly the case + * this runs on, so beyond this many characters per side it measures a prefix and says so. + */ + static final int MAX_DISTANCE_CHARACTERS = 20_000; + + /** + * A label and the number it carries: {@code Score: 7}, {@code confidence = 0.8}. + * + *

Deliberately narrow. It matches a short label immediately before a separator and a number, + * which is the shape a prompt asking for {@code Score: <1-10>} produces; anything looser starts + * pairing years, list indices and ids across two texts and reporting their difference as impact. + */ + private static final Pattern LABELLED_NUMBER = + Pattern.compile("(?m)(?:^|[\\n;,])\\s*([A-Za-z][A-Za-z ._-]{0,30}?)\\s*[:=]\\s*(-?\\d+(?:[.,]\\d+)?)"); + + private BiasValueAnalyzer() { + } + + /** What the caller must be told about a value it asked to analyze, beyond the numbers. */ + public record Analysis(BiasValueImpact impact, List warnings) { + + public Analysis { + warnings = warnings == null ? List.of() : List.copyOf(warnings); + } + } + + public static Analysis analyze(String nodeId, String nodeName, String field, Object baseline, Object biased) { + List warnings = new ArrayList<>(); + + if (baseline instanceof List baselineItems && biased instanceof List biasedItems) { + return new Analysis(analyzeLists(nodeId, nodeName, field, baselineItems, biasedItems, warnings), warnings); + } + + String baselineText = asText(baseline); + String biasedText = asText(biased); + double difference = textDifference(baselineText, biasedText, warnings, field); + return new Analysis(new BiasValueImpact( + nodeId, + nodeName, + field, + !Objects.equals(baselineText, biasedText), + difference, + 0, + 0, + 0.0, + 0.0, + numericDeltas(baselineText, biasedText), + List.of(), + baselineText, + biasedText, + null), warnings); + } + + private static BiasValueImpact analyzeLists(String nodeId, String nodeName, String field, + List baselineItems, List biasedItems, List warnings) { + int compared = Math.max(baselineItems.size(), biasedItems.size()); + boolean capped = compared > MAX_ITEMS; + if (capped) { + warnings.add("Only the first " + MAX_ITEMS + " of " + compared + " elements of " + field + + " were compared element by element."); + } + int analyzed = Math.min(compared, MAX_ITEMS); + + List items = new ArrayList<>(analyzed); + int changedItems = 0; + double totalDifference = 0.0; + double maximumDifference = 0.0; + for (int index = 0; index < analyzed; index++) { + String baselineText = index < baselineItems.size() ? asText(baselineItems.get(index)) : ""; + String biasedText = index < biasedItems.size() ? asText(biasedItems.get(index)) : ""; + boolean changed = !Objects.equals(baselineText, biasedText); + double difference = textDifference(baselineText, biasedText, warnings, field + "[" + (index + 1) + "]"); + if (changed) { + changedItems++; + } + totalDifference += difference; + maximumDifference = Math.max(maximumDifference, difference); + items.add(new BiasItemImpact( + index + 1, + changed, + difference, + numericDeltas(baselineText, biasedText), + baselineText, + biasedText, + null)); + } + + // The field-level figure is the worst element, not a distance over the two lists printed as + // one string: one changed candidate out of five is a real finding, and an average would bury + // it under four identical ones. + return new BiasValueImpact( + nodeId, + nodeName, + field, + !Objects.equals(baselineItems, biasedItems), + maximumDifference, + analyzed, + changedItems, + analyzed == 0 ? 0.0 : totalDifference / analyzed, + maximumDifference, + aggregateNumericDeltas(items), + items, + null, + null, + null); + } + + /** + * The deltas of a whole list: one entry per label, averaged over the elements that carry it. + * + *

Kept because "Score moved by -2.3 on average across five candidates" is the sentence the + * headline of a report needs, while the per-element deltas answer "which candidate". + */ + private static List aggregateNumericDeltas(List items) { + Map> byLabel = new LinkedHashMap<>(); + for (BiasItemImpact item : items) { + for (BiasNumericDelta delta : item.numericDeltas()) { + byLabel.computeIfAbsent(delta.label(), ignored -> new ArrayList<>()).add(delta); + } + } + List aggregated = new ArrayList<>(); + byLabel.forEach((label, deltas) -> aggregated.add(new BiasNumericDelta( + label, + deltas.stream().mapToDouble(BiasNumericDelta::baseline).average().orElse(0.0), + deltas.stream().mapToDouble(BiasNumericDelta::biased).average().orElse(0.0), + deltas.stream().mapToDouble(BiasNumericDelta::delta).average().orElse(0.0)))); + return aggregated; + } + + /** + * Numbers labelled the same way on both sides. + * + *

A label that appears more than once on either side is skipped: with two occurrences there + * is no way to know which one to subtract from which, and picking the first would report a + * confident delta between unrelated numbers. + */ + public static List numericDeltas(String baseline, String biased) { + Map baselineNumbers = uniqueLabelledNumbers(baseline); + Map biasedNumbers = uniqueLabelledNumbers(biased); + Set shared = new LinkedHashSet<>(baselineNumbers.keySet()); + shared.retainAll(biasedNumbers.keySet()); + return shared.stream() + .map(label -> BiasNumericDelta.of(label, baselineNumbers.get(label), biasedNumbers.get(label))) + .toList(); + } + + private static Map uniqueLabelledNumbers(String text) { + Map values = new LinkedHashMap<>(); + Set ambiguous = new LinkedHashSet<>(); + if (text == null || text.isBlank()) { + return values; + } + Matcher matcher = LABELLED_NUMBER.matcher(text); + while (matcher.find()) { + String label = matcher.group(1).trim(); + if (label.isEmpty()) { + continue; + } + if (values.containsKey(label)) { + ambiguous.add(label); + continue; + } + try { + values.put(label, Double.parseDouble(matcher.group(2).replace(',', '.'))); + } catch (NumberFormatException ignored) { + // A number the pattern accepted but Java will not parse is not a measurement. + } + } + ambiguous.forEach(values::remove); + return values; + } + + /** Normalized edit distance: 0 identical, 1 maximally different. */ + public static double textDifference(String left, String right) { + return textDifference(left, right, new ArrayList<>(), null); + } + + private static double textDifference(String left, String right, List warnings, String field) { + String leftText = left == null ? "" : left; + String rightText = right == null ? "" : right; + if (leftText.equals(rightText)) { + return 0.0; + } + if (leftText.length() > MAX_DISTANCE_CHARACTERS || rightText.length() > MAX_DISTANCE_CHARACTERS) { + if (field != null) { + warnings.add("The text difference for " + field + " was measured on the first " + + MAX_DISTANCE_CHARACTERS + " characters of each side."); + } + leftText = leftText.substring(0, Math.min(leftText.length(), MAX_DISTANCE_CHARACTERS)); + rightText = rightText.substring(0, Math.min(rightText.length(), MAX_DISTANCE_CHARACTERS)); + if (leftText.equals(rightText)) { + return 0.0; + } + } + int maximumLength = Math.max(leftText.length(), rightText.length()); + if (maximumLength == 0) { + return 0.0; + } + int[] previous = new int[rightText.length() + 1]; + for (int column = 0; column <= rightText.length(); column++) { + previous[column] = column; + } + for (int row = 1; row <= leftText.length(); row++) { + int[] current = new int[rightText.length() + 1]; + current[0] = row; + for (int column = 1; column <= rightText.length(); column++) { + int substitution = previous[column - 1] + + (leftText.charAt(row - 1) == rightText.charAt(column - 1) ? 0 : 1); + current[column] = Math.min(Math.min(current[column - 1] + 1, previous[column] + 1), substitution); + } + previous = current; + } + return (double) previous[rightText.length()] / maximumLength; + } + + /** The text a value is compared as. Structure is pretty-printed so it diffs line by line. */ + public static String asText(Object value) { + if (value == null) { + return ""; + } + if (value instanceof String text) { + return text; + } + try { + return JacksonConverterSupport.mapper().writerWithDefaultPrettyPrinter().writeValueAsString(value); + } catch (RuntimeException exception) { + return String.valueOf(value); + } + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasValueImpact.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasValueImpact.java new file mode 100644 index 0000000..4533ce5 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/BiasValueImpact.java @@ -0,0 +1,61 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import java.util.List; + +/** + * One output field of one node, compared between the baseline and the variant. + * + *

This is what replaced a single distance computed over every activated node's outputs + * stringified into one map: that number was dominated by the shape of the map, and on a flow that + * scores five candidates it could not say which of them moved. + */ +public record BiasValueImpact( + String nodeId, + String nodeName, + String field, + boolean changed, + double textDifference, + int itemsCompared, + int itemsChanged, + double meanItemTextDifference, + double maximumItemTextDifference, + List numericDeltas, + List items, + /** + * The two texts compared, for a field that is not a list; a list keeps its texts on its + * items instead. Dropped, like the items' own, when the report was asked not to carry raw + * outputs - the figures above stay either way. + */ + String baselineText, + String biasedText, + BiasJudgeVerdict judgeVerdict) { + + public BiasValueImpact { + numericDeltas = numericDeltas == null ? List.of() : List.copyOf(numericDeltas); + items = items == null ? List.of() : List.copyOf(items); + } + + /** The path this value is addressed by when verdicts are attached back to a report. */ + public String key() { + return nodeId + "." + field; + } + + public BiasValueImpact withVerdict(BiasJudgeVerdict verdict) { + return new BiasValueImpact(nodeId, nodeName, field, changed, textDifference, itemsCompared, itemsChanged, + meanItemTextDifference, maximumItemTextDifference, numericDeltas, items, baselineText, biasedText, + verdict); + } + + public BiasValueImpact withItems(List replacements) { + return new BiasValueImpact(nodeId, nodeName, field, changed, textDifference, itemsCompared, itemsChanged, + meanItemTextDifference, maximumItemTextDifference, numericDeltas, replacements, baselineText, + biasedText, judgeVerdict); + } + + /** The same field without its raw texts, for a report that was asked not to carry them. */ + public BiasValueImpact withoutTexts() { + return new BiasValueImpact(nodeId, nodeName, field, changed, textDifference, itemsCompared, itemsChanged, + meanItemTextDifference, maximumItemTextDifference, numericDeltas, + items.stream().map(BiasItemImpact::withoutTexts).toList(), null, null, judgeVerdict); + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/persistence/BiasImpactJobEntity.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/persistence/BiasImpactJobEntity.java index deec63f..feffa55 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/persistence/BiasImpactJobEntity.java +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/persistence/BiasImpactJobEntity.java @@ -3,7 +3,9 @@ package it.cnr.isti.workflow.manager.executions.bias.persistence; import java.time.LocalDateTime; import it.cnr.isti.workflow.manager.executions.bias.BiasImpactExperimentRequest; +import it.cnr.isti.workflow.manager.executions.bias.BiasImpactJobKind; import it.cnr.isti.workflow.manager.executions.bias.BiasImpactJobStatus; +import it.cnr.isti.workflow.manager.llms.LLMDescriptor; import jakarta.persistence.Column; import jakarta.persistence.Convert; import jakarta.persistence.Entity; @@ -37,9 +39,14 @@ public class BiasImpactJobEntity { @Column(name = "execution_id", nullable = false) private String executionId; - @Column(name = "step_id", nullable = false) + /** Null for a judge job: it is about a whole comparison, not one step. */ + @Column(name = "step_id") private String stepId; + @Enumerated(EnumType.STRING) + @Column(nullable = false) + private BiasImpactJobKind kind; + @Enumerated(EnumType.STRING) @Column(nullable = false) private BiasImpactJobStatus status; @@ -62,7 +69,15 @@ public class BiasImpactJobEntity { @Column(name = "error_message", columnDefinition = "TEXT") private String errorMessage; - @Column(name = "request_data", nullable = false, columnDefinition = "TEXT") + @Column(name = "request_data", columnDefinition = "TEXT") @Convert(converter = BiasImpactExperimentRequestConverter.class) private BiasImpactExperimentRequest request; + + /** The report a judge job assesses. It is also the report the job completes with. */ + @Column(name = "report_target_id") + private String reportTargetId; + + @Column(name = "judge_data", columnDefinition = "TEXT") + @Convert(converter = LLMDescriptorConverter.class) + private LLMDescriptor judge; } diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/bias/persistence/LLMDescriptorConverter.java b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/persistence/LLMDescriptorConverter.java new file mode 100644 index 0000000..c3bbea4 --- /dev/null +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/bias/persistence/LLMDescriptorConverter.java @@ -0,0 +1,35 @@ +package it.cnr.isti.workflow.manager.executions.bias.persistence; + +import it.cnr.isti.workflow.manager.app.JacksonConverterSupport; +import it.cnr.isti.workflow.manager.llms.LLMDescriptor; +import jakarta.persistence.AttributeConverter; +import jakarta.persistence.Converter; +import tools.jackson.core.JacksonException; + +@Converter(autoApply = false) +public class LLMDescriptorConverter implements AttributeConverter { + + @Override + public String convertToDatabaseColumn(LLMDescriptor descriptor) { + if (descriptor == null) { + return null; + } + try { + return JacksonConverterSupport.mapper().writeValueAsString(descriptor); + } catch (JacksonException exception) { + throw new IllegalArgumentException("Unable to serialize the judge descriptor", exception); + } + } + + @Override + public LLMDescriptor convertToEntityAttribute(String value) { + if (value == null || value.isBlank()) { + return null; + } + try { + return JacksonConverterSupport.mapper().readValue(value, LLMDescriptor.class); + } catch (Exception exception) { + throw new IllegalArgumentException("Unable to deserialize the judge descriptor", exception); + } + } +} diff --git a/src/main/java/it/cnr/isti/workflow/manager/flows/validation/ValidationErrorCode.java b/src/main/java/it/cnr/isti/workflow/manager/flows/validation/ValidationErrorCode.java index f7f3876..26b8d8f 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/flows/validation/ValidationErrorCode.java +++ b/src/main/java/it/cnr/isti/workflow/manager/flows/validation/ValidationErrorCode.java @@ -38,6 +38,9 @@ public enum ValidationErrorCode { BIAS_SIDE_EFFECT_BLOCKED, BIAS_SIDE_EFFECT_CONFIRMATION_REQUIRED, BIAS_JOB_NOT_FOUND, + BIAS_JUDGE_PROVIDER_NOT_FOUND, + BIAS_JUDGE_UNAVAILABLE, + BIAS_JUDGE_FAILED, BIAS_REPORT_NOT_FOUND, BIAS_EXPERIMENT_FAILED, CONTAINER_CONFIGURATION_MISSING, diff --git a/src/main/resources/db/migration/V9__add_bias_judge_job.sql b/src/main/resources/db/migration/V9__add_bias_judge_job.sql new file mode 100644 index 0000000..4280748 --- /dev/null +++ b/src/main/resources/db/migration/V9__add_bias_judge_job.sql @@ -0,0 +1,18 @@ +-- A bias job used to be one thing: re-run a step with the intervention active. Asking a model to +-- assess an existing comparison is the second, and it reuses the same queue, recovery and polling +-- rather than growing a parallel one - so the columns that only the first kind fills stop being +-- mandatory. +ALTER TABLE bias_impact_job_entity + ADD COLUMN kind VARCHAR(32) NOT NULL DEFAULT 'ISOLATED_STEP'; + +ALTER TABLE bias_impact_job_entity + ALTER COLUMN step_id DROP NOT NULL; + +ALTER TABLE bias_impact_job_entity + ALTER COLUMN request_data DROP NOT NULL; + +ALTER TABLE bias_impact_job_entity + ADD COLUMN report_target_id VARCHAR(255); + +ALTER TABLE bias_impact_job_entity + ADD COLUMN judge_data TEXT; diff --git a/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasExperimentsIntegrationTest.java b/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasExperimentsIntegrationTest.java index 7c3835b..8887662 100644 --- a/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasExperimentsIntegrationTest.java +++ b/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasExperimentsIntegrationTest.java @@ -3,6 +3,7 @@ package it.cnr.isti.workflow.manager.executions.bias; import static org.junit.jupiter.api.Assertions.assertEquals; import static org.junit.jupiter.api.Assertions.assertFalse; import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertNull; import static org.junit.jupiter.api.Assertions.assertTrue; import static org.junit.jupiter.api.Assertions.assertThrows; @@ -24,6 +25,10 @@ import it.cnr.isti.workflow.manager.blocks.configurations.ConditionalBlockConfig import it.cnr.isti.workflow.manager.blocks.factories.HTTPServerCallBlockFactory; import it.cnr.isti.workflow.manager.blocks.factories.LLMBlockFactory; import it.cnr.isti.workflow.manager.blocks.factories.ConditionalBlockFactory; +import it.cnr.isti.workflow.manager.blocks.configurations.HumanDecisionBlockConfiguration; +import it.cnr.isti.workflow.manager.blocks.configurations.HumanDecisionOption; +import it.cnr.isti.workflow.manager.blocks.factories.HumanDecisionBlockFactory; +import it.cnr.isti.workflow.manager.blocks.types.HumanDecisionBlockType; import it.cnr.isti.workflow.manager.blocks.types.HTTPServerCallBlockType; import it.cnr.isti.workflow.manager.blocks.types.LLMBlockType; import it.cnr.isti.workflow.manager.blocks.types.ConditionalBlockType; @@ -56,6 +61,16 @@ class BiasExperimentsIntegrationTest { private static final String OWNER = "bias-experiment-user"; + private static final LLMDescriptor SIMULATOR = LLMDescriptor.builder() + .provider("biasExperimentProvider") + .model("simulate-model") + .build(); + + private static final LLMDescriptor OTHER_SIMULATOR = LLMDescriptor.builder() + .provider("biasExperimentProvider") + .model("other-simulator-model") + .build(); + @TestConfiguration static class TestConfig { @Bean @@ -73,6 +88,10 @@ class BiasExperimentsIntegrationTest { @Override public String generate(String model, String prompt) { + // The one prompt shape only a simulated HumanDecision produces. + if (prompt.contains("Choose exactly one of the following options")) { + return "CHOICE: accept" + System.lineSeparator() + "RATIONALE: the evidence holds."; + } if (prompt.contains("BIAS_DIRECTIVE") && prompt.contains("MITIGATION_DIRECTIVE") && prompt.indexOf("BIAS_DIRECTIVE") > prompt.indexOf("MITIGATION_DIRECTIVE")) { return "WRONG_ORDER"; @@ -84,6 +103,56 @@ class BiasExperimentsIntegrationTest { } }; } + + @Bean + LLMProvider biasJudgeProvider() { + return new LLMProvider() { + @Override + public String getName() { + return "biasJudgeProvider"; + } + + @Override + public List getRegisteredModels() { + return List.of("judge-test-model"); + } + + @Override + public String generate(String model, String prompt) { + return "The intervened run reached a different conclusion for the subject at hand."; + } + + /** Wrapped in prose on purpose: that is what a small local model actually returns. */ + @Override + public String generateJson(String model, String prompt) { + return """ + Sure, here is my assessment: + {"impact":"SUBSTANTIVE","attribution":"INJECTION","confidence":0.7, + "changedAspects":["conclusion"],"rationale":"The intervened output states the opposite."} + """; + } + }; + } + + @Bean + LLMProvider brokenJudgeProvider() { + return new LLMProvider() { + @Override + public String getName() { + return "brokenJudgeProvider"; + } + + @Override + public List getRegisteredModels() { + return List.of("broken-judge-model"); + } + + @Override + public String generate(String model, String prompt) { + return "no json here either"; + } + }; + } } @Autowired @@ -122,6 +191,9 @@ class BiasExperimentsIntegrationTest { @Autowired GenericContainerFactory genericContainerFactory; + @Autowired + HumanDecisionBlockFactory humanDecisionBlockFactory; + @Test void discoveryExposesBehavioralProbeAndPerTypeCapabilities() { var descriptor = annotationsController.getDescriptor(); @@ -217,6 +289,16 @@ class BiasExperimentsIntegrationTest { BiasImpactReport report = biasImpactService.compareFullFlow(baseline.getId(), variant.getId(), true, OWNER); assertTrue(report.immediateImpact().outputChanged()); assertEquals(container.getId(), report.nodeId()); + + // The container's own output only says a value changed. The child runs say which node inside + // the subflow changed it, which is the whole point of comparing them iteration by iteration. + BiasIterationImpact iteration = report.immediateImpact().iterations().getFirst(); + assertEquals(container.getId(), iteration.containerNodeId()); + assertTrue(iteration.changed()); + BiasValueImpact innerValue = iteration.values().getFirst(); + assertEquals(innerBlock.getId(), innerValue.nodeId()); + assertEquals("BASELINE", innerValue.baselineText()); + assertEquals("BIASED", innerValue.biasedText()); } @Test @@ -478,6 +560,213 @@ class BiasExperimentsIntegrationTest { assertEquals(ValidationErrorCode.BIAS_SIDE_EFFECT_BLOCKED.name(), blocked.getErrorCode()); } + + @Test + void llmAssessmentIsStoredOnTheReportNextToTheModelThatProducedIt() { + Block block = annotatedLlmBlock(); + ExecutionObject baseline = completedExecution(block); + String annotationId = block.getBiasAnnotations().getFirst().id(); + ExecutionObject variant = createAndRunVariant(baseline, block, annotationId, BiasInterventionDirection.BIAS); + BiasImpactReport report = biasImpactService.compareFullFlow(baseline.getId(), variant.getId(), true, OWNER); + + // The comparison stands on its own before anyone asks a model about it. + assertFalse(report.immediateImpact().values().isEmpty()); + assertNull(report.judge()); + + LLMDescriptor judge = LLMDescriptor.builder() + .provider("biasJudgeProvider") + .model("judge-test-model") + .build(); + BiasImpactReport judged = biasImpactService.judgeReport(report.id(), judge, OWNER); + + assertNotNull(judged.judge()); + assertEquals("biasJudgeProvider", judged.judge().judge().provider()); + assertEquals("judge-test-model", judged.judge().judge().model()); + assertEquals(BiasJudgeAttribution.INJECTION, judged.judge().attribution()); + assertTrue(judged.judge().judgedPairs() >= 1); + assertNotNull(judged.judge().narrative()); + assertTrue(judged.judge().errors().isEmpty()); + + BiasJudgeVerdict verdict = judged.immediateImpact().values().stream() + .filter(value -> value.judgeVerdict() != null) + .findFirst() + .orElseThrow() + .judgeVerdict(); + assertEquals(BiasJudgeImpactLevel.SUBSTANTIVE, verdict.impact()); + assertEquals("The intervened output states the opposite.", verdict.rationale()); + + // Stored on the report itself, so reopening it shows the same assessment. + BiasImpactReport reloaded = biasImpactService.getReport(report.id(), OWNER); + assertEquals(BiasJudgeImpactLevel.SUBSTANTIVE, reloaded.judge().impact()); + assertEquals(judged.judge().narrative(), reloaded.judge().narrative()); + } + + @Test + void anUnreadableAssessmentIsRecordedAsAFailureAndLeavesTheComparisonIntact() { + Block block = annotatedLlmBlock(); + ExecutionObject baseline = completedExecution(block); + String annotationId = block.getBiasAnnotations().getFirst().id(); + ExecutionObject variant = createAndRunVariant(baseline, block, annotationId, BiasInterventionDirection.BIAS); + BiasImpactReport report = biasImpactService.compareFullFlow(baseline.getId(), variant.getId(), true, OWNER); + + BiasImpactReport judged = biasImpactService.judgeReport(report.id(), + LLMDescriptor.builder().provider("brokenJudgeProvider").model("broken-judge-model").build(), OWNER); + + assertFalse(judged.judge().errors().isEmpty()); + assertTrue(judged.immediateImpact().outputChanged()); + assertEquals(report.immediateImpact().maximumTextDifference(), + judged.immediateImpact().maximumTextDifference()); + assertTrue(judged.immediateImpact().values().stream() + .map(BiasValueImpact::judgeVerdict) + .filter(java.util.Objects::nonNull) + .anyMatch(BiasJudgeVerdict::failure)); + } + + @Test + void anIsolatedExperimentIsAssessedFromTheOutputsItsOwnReportCarries() { + Block block = annotatedLlmBlock(); + ExecutionObject baseline = completedExecution(block); + BiasImpactReport report = biasImpactService.runIsolatedStepExperiment( + baseline.getId(), + block.getId(), + new BiasImpactExperimentRequest( + List.of(block.getBiasAnnotations().getFirst().id()), + 2, + true, + ExternalSideEffectPolicy.BLOCK, + false, + BiasInterventionDirection.BIAS), + OWNER); + + BiasImpactReport judged = biasImpactService.judgeReport(report.id(), + LLMDescriptor.builder().provider("biasJudgeProvider").model("judge-test-model").build(), OWNER); + + assertEquals(BiasJudgeImpactLevel.SUBSTANTIVE, judged.judge().impact()); + assertTrue(judged.judge().judgedPairs() >= 1); + } + + @Test + void judgingIsRefusedForAProviderThatIsNotRegistered() { + Block block = annotatedLlmBlock(); + ExecutionObject baseline = completedExecution(block); + String annotationId = block.getBiasAnnotations().getFirst().id(); + ExecutionObject variant = createAndRunVariant(baseline, block, annotationId, BiasInterventionDirection.BIAS); + BiasImpactReport report = biasImpactService.compareFullFlow(baseline.getId(), variant.getId(), true, OWNER); + + BiasApiException exception = assertThrows(BiasApiException.class, () -> biasImpactService.judgeReport( + report.id(), LLMDescriptor.builder().provider("nowhere").model("nothing").build(), OWNER)); + assertEquals(ValidationErrorCode.BIAS_JUDGE_PROVIDER_NOT_FOUND.name(), exception.getErrorCode()); + } + + @Test + void anAssessmentAskedForThroughAJobEndsUpOnTheReportItPolled() { + Block block = annotatedLlmBlock(); + ExecutionObject baseline = completedExecution(block); + String annotationId = block.getBiasAnnotations().getFirst().id(); + ExecutionObject variant = createAndRunVariant(baseline, block, annotationId, BiasInterventionDirection.BIAS); + BiasImpactReport report = biasImpactService.compareFullFlow(baseline.getId(), variant.getId(), true, OWNER); + + BiasImpactJob queued = biasImpactJobService.createJudgeJob(report.id(), + LLMDescriptor.builder().provider("biasJudgeProvider").model("judge-test-model").build(), OWNER); + assertEquals(BiasImpactJobKind.REPORT_JUDGE, queued.kind()); + + BiasImpactJob completed = waitUntilJobFinal(queued.id()); + assertEquals(BiasImpactJobStatus.COMPLETED, completed.status()); + assertEquals(report.id(), completed.reportId()); + assertNotNull(completed.report().judge()); + } + + @Test + void aBiasRerunInheritsTheSimulatorOfTheRunItRepeatsWithoutBeingSimulatedItself() { + SimulatedBaseline baseline = simulatedBaseline(SIMULATOR); + + ExecutionObject variant = executionsService.createBiasRerun( + baseline.execution().getId(), + OWNER, + new BiasRerunRequest( + List.of(new BiasActivation(baseline.annotatedBlock().getId(), + List.of(baseline.annotationId()), false, BiasInterventionDirection.BIAS)), + ExternalSideEffectPolicy.BLOCK, + false)); + + assertEquals(SIMULATOR, variant.getInteractionSimulationDescriptor()); + // Carried, not switched on: simulating stays something the caller asks for. + assertFalse(variant.isInteractionSimulationEnabled()); + } + + @Test + void aComparisonSaysSoWhenTheTwoRunsWereSimulatedByDifferentModels() { + SimulatedBaseline baseline = simulatedBaseline(SIMULATOR); + ExecutionObject variant = simulatedVariant(baseline, OTHER_SIMULATOR); + + BiasImpactReport report = biasImpactService.compareFullFlow( + baseline.execution().getId(), variant.getId(), true, OWNER); + + assertFalse(report.simulation().comparable()); + assertTrue(report.warnings().stream() + .anyMatch(warning -> warning.contains("different models") && warning.contains("other-simulator-model"))); + } + + @Test + void aComparisonOfTwoRunsSimulatedTheSameWayHasNothingToWarnAbout() { + SimulatedBaseline baseline = simulatedBaseline(SIMULATOR); + ExecutionObject variant = simulatedVariant(baseline, SIMULATOR); + + BiasImpactReport report = biasImpactService.compareFullFlow( + baseline.execution().getId(), variant.getId(), true, OWNER); + + assertTrue(report.simulation().comparable()); + assertEquals("simulate-model", report.simulation().baselineSimulator().model()); + assertTrue(report.warnings().stream().noneMatch(warning -> warning.contains("simulator"))); + } + + private record SimulatedBaseline(ExecutionObject execution, Block annotatedBlock, + String annotationId) { + } + + /** A run of a flow that has something to simulate, answered by the given simulator. */ + private SimulatedBaseline simulatedBaseline(LLMDescriptor simulator) { + Block annotated = annotatedLlmBlock(); + Block decision = humanDecisionBlockFactory.create( + HumanDecisionBlockConfiguration.builder() + .name("Review the answer") + .question("Does the answer hold up?") + .options(List.of( + new HumanDecisionOption("accept", "Accept"), + new HumanDecisionOption("revise", "Request revision"))) + .rationaleRequired(true) + .build()); + + ExecutionObject execution = executionsService.createExecutionForFlow( + "simulated-bias-flow-" + annotated.getId(), + "Simulated bias flow", + FlowData.builder().block(annotated).block(decision).build(), + OWNER); + executionsService.prepareInput(execution.getId(), decision.getId(), + HumanDecisionBlockFactory.INPUT_NAME, "the answer under review"); + executionsService.startSimulationExecution(execution.getId(), simulator); + waitUntilFinal(execution); + assertEquals(ExecutionStatus.SUCCESS, execution.getContext().getStatus()); + assertTrue(execution.isInteractionSimulationEnabled()); + + return new SimulatedBaseline(execution, annotated, annotated.getBiasAnnotations().getFirst().id()); + } + + private ExecutionObject simulatedVariant(SimulatedBaseline baseline, LLMDescriptor simulator) { + ExecutionObject variant = executionsService.createBiasRerun( + baseline.execution().getId(), + OWNER, + new BiasRerunRequest( + List.of(new BiasActivation(baseline.annotatedBlock().getId(), + List.of(baseline.annotationId()), false, BiasInterventionDirection.BIAS)), + ExternalSideEffectPolicy.BLOCK, + false)); + executionsService.startSimulationExecution(variant.getId(), simulator); + waitUntilFinal(variant); + assertEquals(ExecutionStatus.SUCCESS, variant.getContext().getStatus()); + return variant; + } + private ExecutionObject completedExecution(Block block) { ExecutionObject execution = executionsService.createExecutionForFlow( "bias-flow-" + block.getId(), diff --git a/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeTargetTest.java b/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeTargetTest.java new file mode 100644 index 0000000..e173db8 --- /dev/null +++ b/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasJudgeTargetTest.java @@ -0,0 +1,86 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertNull; +import static org.junit.jupiter.api.Assertions.assertTrue; + +import java.time.LocalDateTime; +import java.util.List; +import java.util.Map; + +import org.junit.jupiter.api.Test; + +class BiasJudgeTargetTest { + + @Test + void collectsOnePairPerSubjectAndPutsTheChangedOnesFirst() { + BiasImpactReport report = reportWith(listValue()); + + List targets = BiasJudgeTarget.collect(report); + + assertEquals(3, targets.size()); + assertTrue(targets.getFirst().changed()); + assertEquals(2, targets.getFirst().subjectIndex()); + assertTrue(targets.stream().skip(1).noneMatch(BiasJudgeTarget::changed)); + } + + @Test + void writesVerdictsBackOntoTheSubjectsTheyWereAskedAbout() { + BiasImpactReport report = reportWith(listValue()); + Map verdicts = BiasJudgeTarget.collect(report).stream() + .collect(java.util.stream.Collectors.toMap(BiasJudgeTarget::path, + target -> target.changed() + ? new BiasJudgeVerdict(BiasJudgeImpactLevel.SUBSTANTIVE, + BiasJudgeAttribution.INJECTION, 0.8, List.of("score"), "moved", null) + : BiasJudgeVerdict.identical())); + + BiasImpactReport judged = BiasJudgeTarget.apply(report, verdicts); + + List items = judged.immediateImpact().values().getFirst().items(); + assertEquals(BiasJudgeImpactLevel.NONE, items.get(0).judgeVerdict().impact()); + assertEquals(BiasJudgeImpactLevel.SUBSTANTIVE, items.get(1).judgeVerdict().impact()); + assertEquals(BiasJudgeAttribution.INJECTION, items.get(1).judgeVerdict().attribution()); + } + + @Test + void addressesAnIterationOfAContainerSeparatelyFromTheContainersOwnOutput() { + BiasValueImpact inner = new BiasValueImpact("inner", "score-cv", "response", true, 0.4, 0, 0, 0.0, 0.0, + List.of(), List.of(), "8", "3", null); + BiasIterationImpact iteration = new BiasIterationImpact("container", "score-cvs", 2, "child-a", "child-b", + "SUCCESS", "SUCCESS", true, List.of(inner)); + BiasImpactReport report = reportWith(listValue()).withImpact( + reportWith(listValue()).immediateImpact().withIterations(List.of(iteration)), List.of()); + + List targets = BiasJudgeTarget.collect(report); + BiasJudgeTarget iterationTarget = targets.stream() + .filter(target -> target.path().contains("-iteration:")) + .findFirst() + .orElseThrow(); + assertEquals("immediate-iteration:container#2:inner.response", iterationTarget.path()); + assertEquals(2, iterationTarget.subjectIndex()); + + BiasImpactReport judged = BiasJudgeTarget.apply(report, + Map.of(iterationTarget.path(), BiasJudgeVerdict.identical())); + assertNotNull(judged.immediateImpact().iterations().getFirst().values().getFirst().judgeVerdict()); + assertNull(judged.immediateImpact().values().getFirst().judgeVerdict()); + } + + private BiasValueImpact listValue() { + return new BiasValueImpact("node", "score-cvs", "response", true, 0.5, 3, 1, 0.16, 0.5, List.of(), + List.of( + new BiasItemImpact(1, false, 0.0, List.of(), "a", "a", null), + new BiasItemImpact(2, true, 0.5, List.of(), "b", "changed", null), + new BiasItemImpact(3, false, 0.0, List.of(), "c", "c", null)), + null, null, null); + } + + private BiasImpactReport reportWith(BiasValueImpact value) { + BiasOutputImpact immediate = new BiasOutputImpact(true, 0.5, 1.0, Map.of(), List.of(), List.of(value), + List.of()); + return new BiasImpactReport("report", "experiment", BiasExperimentKind.FULL_FLOW, + BiasInterventionDirection.BIAS, "baseline", "biased", "node", List.of("annotation"), 1, + LocalDateTime.now(), true, immediate, List.of(), List.of(), List.of(), List.of(), "summary", + List.of(), BiasImpactReport.CURRENT_SCHEMA_VERSION, null, null); + } +} diff --git a/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasSimulationContextTest.java b/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasSimulationContextTest.java new file mode 100644 index 0000000..150a7e4 --- /dev/null +++ b/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasSimulationContextTest.java @@ -0,0 +1,61 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNull; +import static org.junit.jupiter.api.Assertions.assertTrue; + +import org.junit.jupiter.api.Test; + +import it.cnr.isti.workflow.manager.llms.LLMDescriptor; +import it.cnr.isti.workflow.manager.llms.ModelParameters; + +class BiasSimulationContextTest { + + private static final LLMDescriptor GEMMA = LLMDescriptor.builder() + .provider("InternalOllama") + .model("gemma:7b") + .build(); + + @Test + void twoRunsSimulatedTheSameWayAreComparableAndSayNothingMore() { + BiasSimulationContext context = new BiasSimulationContext(true, GEMMA, true, GEMMA, true); + + assertTrue(context.comparable()); + assertNull(context.warning()); + assertTrue(context.describe().contains("both sides")); + } + + @Test + void twoDifferentSimulatorsAreNotComparableAndBothAreNamed() { + LLMDescriptor other = LLMDescriptor.builder().provider("InternalOllama").model("llama3").build(); + BiasSimulationContext context = new BiasSimulationContext(true, GEMMA, true, other, false); + + assertFalse(context.comparable()); + assertTrue(context.warning().contains("gemma:7b")); + assertTrue(context.warning().contains("llama3")); + assertTrue(context.warning().contains("rather than the intervention")); + } + + @Test + void aSimulatedRunAgainstOneAnsweredDirectlySaysWhichWasWhich() { + BiasSimulationContext context = new BiasSimulationContext(true, GEMMA, false, null, false); + + assertFalse(context.comparable()); + assertTrue(context.warning().contains("simulated on the baseline run")); + assertTrue(context.warning().contains("answered directly on the variant run")); + } + + /** The seed is the whole reason the parameters travel with the descriptor. */ + @Test + void theSameModelWithADifferentSeedIsADifferentSimulator() { + LLMDescriptor seeded = LLMDescriptor.builder() + .provider("InternalOllama") + .model("gemma:7b") + .parameters(ModelParameters.builder().seed(7L).build()) + .build(); + + assertFalse(seeded.equals(GEMMA)); + BiasSimulationContext context = new BiasSimulationContext(true, seeded, true, GEMMA, false); + assertTrue(context.warning().contains("seed 7")); + } +} diff --git a/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasValueAnalyzerTest.java b/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasValueAnalyzerTest.java new file mode 100644 index 0000000..5b8d177 --- /dev/null +++ b/src/test/java/it/cnr/isti/workflow/manager/executions/bias/BiasValueAnalyzerTest.java @@ -0,0 +1,130 @@ +package it.cnr.isti.workflow.manager.executions.bias; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertTrue; + +import java.util.List; +import java.util.Map; + +import org.junit.jupiter.api.Test; + +class BiasValueAnalyzerTest { + + @Test + void pairsListElementsByPositionAndCountsOnlyTheOnesThatMoved() { + BiasValueImpact impact = BiasValueAnalyzer.analyze("node", "score-cvs", "response", + List.of("Candidate: A\nScore: 8", "Candidate: B\nScore: 7", "Candidate: C\nScore: 6"), + List.of("Candidate: A\nScore: 8", "Candidate: B\nScore: 3", "Candidate: C\nScore: 6")).impact(); + + assertEquals(3, impact.itemsCompared()); + assertEquals(1, impact.itemsChanged()); + assertTrue(impact.changed()); + assertFalse(impact.items().get(0).changed()); + assertTrue(impact.items().get(1).changed()); + assertEquals(2, impact.items().get(1).index()); + } + + @Test + void reportsTheMovementOfNumbersBothSidesLabelTheSameWay() { + BiasValueImpact impact = BiasValueAnalyzer.analyze("node", "score-cv", "response", + "Candidate: B\nScore: 7\nJustification: broad backend evidence", + "Candidate: B\nScore: 3\nJustification: non-traditional background").impact(); + + BiasNumericDelta score = impact.numericDeltas().stream() + .filter(delta -> delta.label().equals("Score")) + .findFirst() + .orElseThrow(); + assertEquals(7.0, score.baseline()); + assertEquals(3.0, score.biased()); + assertEquals(-4.0, score.delta()); + } + + @Test + void aggregatesElementDeltasIntoOneFigurePerLabel() { + BiasValueImpact impact = BiasValueAnalyzer.analyze("node", "score-cvs", "response", + List.of("Score: 8", "Score: 6"), + List.of("Score: 6", "Score: 2")).impact(); + + BiasNumericDelta score = impact.numericDeltas().getFirst(); + assertEquals("Score", score.label()); + assertEquals(-3.0, score.delta()); + } + + @Test + void skipsALabelThatAppearsTwiceRatherThanSubtractingUnrelatedNumbers() { + BiasValueImpact impact = BiasValueAnalyzer.analyze("node", "aggregate", "response", + "Score: 8\nScore: 6", + "Score: 2\nScore: 1").impact(); + + assertTrue(impact.numericDeltas().isEmpty()); + } + + @Test + void extractsNothingFromProseThatCarriesNoLabelledNumber() { + BiasValueImpact impact = BiasValueAnalyzer.analyze("node", "review", "response", + "The candidate looks strong on backend work.", + "The candidate looks weak on backend work.").impact(); + + assertTrue(impact.numericDeltas().isEmpty()); + assertTrue(impact.textDifference() > 0.0); + assertTrue(impact.textDifference() < 1.0); + } + + @Test + void treatsAnElementOnlyOneSideProducedAsChanged() { + BiasValueImpact impact = BiasValueAnalyzer.analyze("node", "score-cvs", "response", + List.of("a", "b"), + List.of("a")).impact(); + + assertEquals(2, impact.itemsCompared()); + assertEquals(1, impact.itemsChanged()); + assertEquals("", impact.items().get(1).biasedText()); + } + + @Test + void keepsTheAggregatesAndWarnsWhenThereAreMoreElementsThanItCompares() { + List baseline = java.util.stream.IntStream.range(0, BiasValueAnalyzer.MAX_ITEMS + 5) + .mapToObj(index -> "value " + index) + .toList(); + List biased = baseline.stream().map(value -> value + " changed").toList(); + + BiasValueAnalyzer.Analysis analysis = BiasValueAnalyzer.analyze("node", "many", "response", baseline, biased); + + assertEquals(BiasValueAnalyzer.MAX_ITEMS, analysis.impact().items().size()); + assertEquals(BiasValueAnalyzer.MAX_ITEMS, analysis.impact().itemsChanged()); + assertTrue(analysis.warnings().stream().anyMatch(warning -> warning.contains("element by element"))); + } + + @Test + void measuresAPrefixWhenTheTextsAreTooLongToCompareWholeAndSaysSo() { + String baseline = "a".repeat(BiasValueAnalyzer.MAX_DISTANCE_CHARACTERS + 100); + String biased = "b".repeat(BiasValueAnalyzer.MAX_DISTANCE_CHARACTERS + 100); + + BiasValueAnalyzer.Analysis analysis = BiasValueAnalyzer.analyze("node", "long", "response", baseline, biased); + + assertEquals(1.0, analysis.impact().textDifference()); + assertTrue(analysis.warnings().stream().anyMatch(warning -> warning.contains("first 20000 characters"))); + } + + @Test + void comparesStructureAsPrettyPrintedTextSoItDiffsLineByLine() { + BiasValueImpact impact = BiasValueAnalyzer.analyze("node", "structured", "response", + Map.of("score", 8), + Map.of("score", 3)).impact(); + + assertTrue(impact.changed()); + assertNotNull(impact.baselineText()); + assertTrue(impact.baselineText().contains("score")); + } + + @Test + void callsTwoIdenticalValuesUnchangedWithNoDistanceAtAll() { + BiasValueImpact impact = BiasValueAnalyzer.analyze("node", "same", "response", "identical", "identical") + .impact(); + + assertFalse(impact.changed()); + assertEquals(0.0, impact.textDifference()); + } +}