Add a node for a person to evaluate software against named criteria

The human nodes could ask a question or collect a paragraph, so a flow
that wanted a judgement about a running piece of software got back prose:
not comparable between two runs, and nothing a later node could branch on.

HumanEvaluationBlock asks instead for a verdict per named criterion, with
an optional script so two testers exercise the same thing, and evidence
files. It points at the target rather than hosting it - provisioning
environments or driving browsers belongs outside the engine, and the flow
already knows the URL.

blindUntilSubmitted keeps a reference verdict, usually an agent's, out of
sight until the person commits to their own, and the execution records
which came first. That turns the node from a place to rubber-stamp an
agent into something that measures the automation bias the catalogue
already names.

The engine needed only a way to submit several fields as one act, which
Step.interact already accepted; and the interaction contract now comes
from the block type, so the editor and the assistant stop keeping their
own table of block names.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Lucio Lelii 2026-09-20 23:24:33 +02:00
parent 52b02ce349
commit 6032622c67
24 changed files with 1269 additions and 56 deletions

View File

@ -255,33 +255,16 @@ public class BlockCatalogService {
}
private AssistantInteractionContractDescriptor resolveInteractionContract(BlockType blockType) {
if ("ChatInteraction".equals(blockType.getName()) || "MCPAgentChat".equals(blockType.getName())) {
return new AssistantInteractionContractDescriptor(
"chat-session",
"message",
"response",
"history",
"response",
true);
it.cnr.isti.workflow.manager.blocks.types.InteractionContract contract = blockType.getInteractionContract();
if (contract == null) {
return null;
}
if ("HumanInteractionBlock".equals(blockType.getName())) {
return new AssistantInteractionContractDescriptor(
"single-response",
null,
"output",
null,
"output",
false);
}
if ("HumanDecisionBlock".equals(blockType.getName())) {
return new AssistantInteractionContractDescriptor(
"human-decision",
"rationale",
"choice",
null,
"choice",
true);
}
return null;
return new AssistantInteractionContractDescriptor(
contract.kind(),
contract.messageField(),
contract.completionField(),
contract.historyField(),
contract.responseField(),
contract.supportsPartialResult());
}
}

View File

@ -0,0 +1,59 @@
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
// SPDX-License-Identifier: AGPL-3.0-or-later
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
package it.cnr.isti.workflow.manager.blocks.configurations;
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
import com.fasterxml.jackson.annotation.JsonProperty;
import it.cnr.isti.workflow.manager.configurations.annotations.SchemaAllowedValues;
import it.cnr.isti.workflow.manager.configurations.annotations.UiDescription;
import it.cnr.isti.workflow.manager.configurations.annotations.UiLabel;
import it.cnr.isti.workflow.manager.configurations.annotations.UiOrder;
import jakarta.validation.constraints.NotBlank;
import jakarta.validation.constraints.Size;
/**
* One thing the tester is asked to judge, and how.
*
* <p>Criteria are what make two evaluations of the same software comparable: a free-text "it works"
* cannot be compared, a verdict per named criterion can. {@code routing} decides whether the
* criterion also becomes an output port - twelve criteria would otherwise leave the node bristling
* with ports nobody wires.
*/
@JsonIgnoreProperties(ignoreUnknown = true)
public record EvaluationCriterion(
@NotBlank
@Size(max = 64)
@UiLabel("Criterion")
@UiDescription("Short name, also the output port name when this criterion routes.")
@UiOrder(10)
@JsonProperty(required = true)
String name,
@Size(max = 500)
@UiLabel("What to judge")
@UiDescription("What the tester should look at to answer. Vague wording here is the usual reason two testers disagree.")
@UiOrder(20)
String description,
@UiLabel("Scale")
@UiDescription("pass/fail, or a 1-5 score.")
@UiOrder(30)
@SchemaAllowedValues(value = { "PASS_FAIL", "SCORE_1_5" }, defaultValue = "PASS_FAIL")
EvaluationScale scale,
@UiLabel("Routes the flow")
@UiDescription("Gives this criterion its own output port, so a later node can branch on it.")
@UiOrder(40)
boolean routing) {
public EvaluationCriterion {
scale = scale == null ? EvaluationScale.PASS_FAIL : scale;
}
public EvaluationCriterion(String name) {
this(name, null, EvaluationScale.PASS_FAIL, false);
}
}

View File

@ -0,0 +1,54 @@
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
// SPDX-License-Identifier: AGPL-3.0-or-later
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
package it.cnr.isti.workflow.manager.blocks.configurations;
import com.fasterxml.jackson.annotation.JsonCreator;
import com.fasterxml.jackson.annotation.JsonValue;
/**
* How one evaluation criterion is scored.
*
* <p>The choice is per criterion rather than per node because the two scales answer different
* questions: PASS_FAIL is easier for two people to agree on, SCORE_1_5 carries more information at
* the cost of that agreement. A node usually wants both - "does it work at all" next to "how good
* is it".
*/
public enum EvaluationScale {
PASS_FAIL,
SCORE_1_5;
@JsonCreator
public static EvaluationScale fromString(String key) {
return EvaluationScale.valueOf(key.toUpperCase());
}
@JsonValue
public String toValue() {
return name();
}
/** Whether {@code value} is one of the verdicts this scale accepts, case-insensitively. */
public boolean accepts(String value) {
if (value == null || value.isBlank()) {
return false;
}
String normalized = value.strip();
return switch (this) {
case PASS_FAIL -> "pass".equalsIgnoreCase(normalized) || "fail".equalsIgnoreCase(normalized);
case SCORE_1_5 -> switch (normalized) {
case "1", "2", "3", "4", "5" -> true;
default -> false;
};
};
}
/** What the tester may answer, spelled out for the person and for the simulator prompt. */
public String allowedValuesDescription() {
return switch (this) {
case PASS_FAIL -> "pass or fail";
case SCORE_1_5 -> "a whole number from 1 to 5";
};
}
}

View File

@ -0,0 +1,192 @@
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
// SPDX-License-Identifier: AGPL-3.0-or-later
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
package it.cnr.isti.workflow.manager.blocks.configurations;
import java.util.List;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.fasterxml.jackson.annotation.JsonProperty;
import it.cnr.isti.workflow.manager.blocks.factories.TemplateInputs;
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
import it.cnr.isti.workflow.manager.configurations.annotations.LongText;
import it.cnr.isti.workflow.manager.configurations.annotations.Structural;
import it.cnr.isti.workflow.manager.configurations.annotations.UiDescription;
import it.cnr.isti.workflow.manager.configurations.annotations.UiLabel;
import it.cnr.isti.workflow.manager.configurations.annotations.UiOrder;
import it.cnr.isti.workflow.manager.configurations.annotations.UiUniqueItemsBy;
import jakarta.validation.Valid;
import jakarta.validation.constraints.AssertTrue;
import jakarta.validation.constraints.NotBlank;
import jakarta.validation.constraints.Size;
import lombok.Builder;
import lombok.EqualsAndHashCode;
import lombok.Getter;
import lombok.NoArgsConstructor;
import lombok.NonNull;
/**
* A person exercises something - typically a running piece of software - and scores it against
* named criteria.
*
* <p>It differs from {@link HumanInteractiveBlockConfiguration}, which returns one blob of text, in
* that the answer is structured: a verdict per criterion, so two evaluations of the same target can
* be compared and a later node can branch on one of them without parsing prose.
*
* <p>The node deliberately does not host the thing under test. {@code target} points at it - a URL
* a previous node produced, or an artefact in a workspace - and the tester opens it however they
* normally would. Hosting browsers or provisioning environments belongs outside the workflow engine.
*/
@Getter
@EqualsAndHashCode(callSuper = true)
@NoArgsConstructor(access = lombok.AccessLevel.PROTECTED)
public class HumanEvaluationBlockConfiguration extends BlockConfiguration<HumanEvaluationBlockType> {
public static final int MAX_CRITERIA = 12;
public static final int MAX_TASK_STEPS = 20;
@UiOrder(30)
@NotBlank
@Size(max = 2000)
@LongText(
placeholder = "What the tester should open, and where",
tip = "Use ${{}} to pull in something an earlier block produced, for example: Open ${{deployUrl}} and sign in as the demo user.",
acceptVariableAsPlaceholder = true)
@UiLabel("What to evaluate")
@UiDescription("Where the thing under test lives and how to reach it. The node points at it, it does not host it.")
@JsonProperty(required = true)
private String target;
@UiOrder(40)
@Size(max = MAX_TASK_STEPS)
@UiLabel("Steps to perform")
@UiDescription("The same script for every tester. Without it two people try different things and their verdicts are not comparable.")
private List<@NotBlank @Size(max = 500) String> taskScript = List.of();
@UiOrder(50)
@Structural
@Valid
@Size(min = 1, max = MAX_CRITERIA)
@UiUniqueItemsBy("name")
@UiLabel("Criteria")
@JsonProperty(required = true)
private List<EvaluationCriterion> criteria = List.of();
@UiOrder(60)
@UiLabel("Require evidence")
@UiDescription("Refuses a verdict with no attachment. Worth it when someone will have to review the judgement later.")
private boolean evidenceRequired;
@UiOrder(70)
@UiLabel("Hide the reference verdict until submitted")
@UiDescription("Keeps an AI's own verdict out of sight until the person has committed to theirs, so the judgement is not anchored to it. The execution records which came first.")
private boolean blindUntilSubmitted;
@UiOrder(80)
@Size(max = 64)
@UiLabel("Reference input")
@UiDescription("Name of the input holding the verdict to compare against, usually an agent's. Must be one of the ${{}} names used above.")
private String referenceInput;
@Builder
public HumanEvaluationBlockConfiguration(@NonNull String name, String target, List<String> taskScript,
List<EvaluationCriterion> criteria, boolean evidenceRequired, boolean blindUntilSubmitted,
String referenceInput) {
super(name);
this.target = target;
this.taskScript = taskScript == null ? List.of() : List.copyOf(taskScript);
this.criteria = criteria == null ? List.of() : List.copyOf(criteria);
this.evidenceRequired = evidenceRequired;
this.blindUntilSubmitted = blindUntilSubmitted;
this.referenceInput = referenceInput;
}
@Override
public Class<HumanEvaluationBlockType> getBlockType() {
return HumanEvaluationBlockType.class;
}
public static HumanEvaluationBlockConfiguration empty() {
return HumanEvaluationBlockConfiguration.builder()
.name(HumanEvaluationBlockType.TYPE)
.target("Open ${{targetUrl}} and try the main flow")
.taskScript(List.of())
.criteria(List.of(new EvaluationCriterion("works", "Does the main flow complete without errors?",
EvaluationScale.PASS_FAIL, true)))
.build();
}
/** The criteria that also become output ports, in declaration order. */
@JsonIgnore
public List<EvaluationCriterion> routingCriteria() {
return criteria.stream().filter(EvaluationCriterion::routing).toList();
}
@AssertTrue(message = "target placeholder names must start with a letter and contain only letters, digits, '-', '_' or '.'")
@JsonIgnore
boolean isTargetPlaceholderNamesValid() {
return TemplateInputs.hasValidPlaceholderNames(target);
}
@AssertTrue(message = "target placeholder '[]' (array) marker must be used consistently for every occurrence of the same name")
@JsonIgnore
boolean isTargetPlaceholderMultiplicityConsistent() {
return TemplateInputs.hasConsistentPlaceholderMultiplicity(target);
}
@AssertTrue(message = "criteria must have unique non-blank names")
@JsonIgnore
boolean areCriteriaNamesUnique() {
if (criteria == null || criteria.isEmpty()) {
return true;
}
long named = criteria.stream()
.filter(criterion -> criterion != null && criterion.name() != null && !criterion.name().isBlank())
.count();
long distinct = criteria.stream()
.filter(criterion -> criterion != null && criterion.name() != null && !criterion.name().isBlank())
.map(criterion -> criterion.name().strip())
.distinct()
.count();
return named == criteria.size() && distinct == named;
}
/**
* A criterion named like one of the node's own outputs would produce two ports with one name,
* and the second would silently win.
*/
@AssertTrue(message = "criteria cannot be named verdict, notes, evidence or blind: those are the node's own outputs")
@JsonIgnore
boolean areCriteriaNamesFree() {
if (criteria == null) {
return true;
}
return criteria.stream()
.filter(criterion -> criterion != null && criterion.name() != null)
.noneMatch(criterion -> RESERVED_OUTPUT_NAMES.contains(criterion.name().strip().toLowerCase()));
}
/**
* Blind evaluation only means something when there is a reference to hide; without one the flag
* would be a switch that does nothing, which is worse than a refused save.
*/
@AssertTrue(message = "referenceInput is required when the reference verdict is hidden until submitted")
@JsonIgnore
boolean isReferenceInputPresentWhenBlind() {
return !blindUntilSubmitted || (referenceInput != null && !referenceInput.isBlank());
}
/** The reference has to be one of the inputs the node actually has, or nothing is ever hidden. */
@AssertTrue(message = "referenceInput must be one of the ${{}} placeholder names used in the target")
@JsonIgnore
boolean isReferenceInputAmongPlaceholders() {
if (referenceInput == null || referenceInput.isBlank()) {
return true;
}
return TemplateInputs.declaresPlaceholder(target, referenceInput);
}
private static final List<String> RESERVED_OUTPUT_NAMES = List.of("verdict", "notes", "evidence", "blind");
}

View File

@ -0,0 +1,88 @@
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
// SPDX-License-Identifier: AGPL-3.0-or-later
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
package it.cnr.isti.workflow.manager.blocks.factories;
import java.util.ArrayList;
import java.util.List;
import java.util.Set;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Component;
import it.cnr.isti.workflow.manager.blocks.Block;
import it.cnr.isti.workflow.manager.blocks.IOCapability;
import it.cnr.isti.workflow.manager.blocks.IOCapabilityType;
import it.cnr.isti.workflow.manager.blocks.configurations.HumanEvaluationBlockConfiguration;
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
import it.cnr.isti.workflow.manager.ios.IODescriptor;
import it.cnr.isti.workflow.manager.ios.IOType;
@Component
public class HumanEvaluationBlockFactory
implements BlockFactory<HumanEvaluationBlockType, HumanEvaluationBlockConfiguration> {
/** The overall judgement, and the fields the interaction fills besides the criteria. */
public static final String VERDICT_FIELD = "verdict";
public static final String NOTES_FIELD = "notes";
public static final String EVIDENCE_FIELD = "evidence";
public static final String BLIND_FIELD = "blind";
private static final List<IOCapability> INPUT_CAPABILITIES = List.of(
new IOCapability(IOCapabilityType.ANY, false),
new IOCapability(IOCapabilityType.ANY, true));
private static final List<IOCapability> ANY_CAPABILITY = List.of(new IOCapability(IOCapabilityType.ANY, false));
private static final List<IOCapability> FILE_CAPABILITY = List.of(
new IOCapability(IOCapabilityType.FILE, true));
@Autowired
HumanEvaluationBlockType blockType;
@Override
public Block<HumanEvaluationBlockType> create(HumanEvaluationBlockConfiguration configuration) {
List<IODescriptor> inputs = new ArrayList<>();
Set<TemplateInputs.Placeholder> placeholders = TemplateInputs.extractNames(configuration.getTarget());
placeholders.forEach(placeholder -> inputs.add(
IODescriptor.input(placeholder.name(), IOType.ANY, placeholder.multiple(), INPUT_CAPABILITIES)));
Block.BlockBuilder<HumanEvaluationBlockType> builder = Block.<HumanEvaluationBlockType>builder()
.inputs(inputs)
.specificConfiguration(configuration)
.type(blockType);
// Only the criteria marked as routing get a port of their own: a dozen criteria would
// otherwise leave the node bristling with ports nobody wires. Every criterion is still in
// the verdict payload.
configuration.routingCriteria().forEach(criterion -> builder.output(
IODescriptor.output(criterion.name(), IOType.TEXT, false, ANY_CAPABILITY)));
builder.output(IODescriptor.output(VERDICT_FIELD, IOType.JSON, false, ANY_CAPABILITY));
builder.output(IODescriptor.output(NOTES_FIELD, IOType.TEXT, false, ANY_CAPABILITY));
builder.output(IODescriptor.output(EVIDENCE_FIELD, IOType.FILE, true, FILE_CAPABILITY));
// Whether the reference verdict was still hidden when the person committed to theirs. An
// output rather than only an event, so a flow can keep the anchored and unanchored runs apart.
builder.output(IODescriptor.output(BLIND_FIELD, IOType.BOOLEAN, false, ANY_CAPABILITY));
return builder.build();
}
@Override
public Block<HumanEvaluationBlockType> createEmpty() {
return create(HumanEvaluationBlockConfiguration.empty());
}
@Override
public Class<HumanEvaluationBlockType> getBlockType() {
return HumanEvaluationBlockType.class;
}
@Override
public List<IOCapability> supportedInputCapabilities() {
return INPUT_CAPABILITIES;
}
@Override
public List<IOCapability> supportedOutputCapabilities() {
return ANY_CAPABILITY;
}
}

View File

@ -105,4 +105,19 @@ public final class TemplateInputs {
}
return multiplicitiesByName.values().stream().allMatch(multiplicities -> multiplicities.size() == 1);
}
/**
* Whether the text declares a placeholder with exactly this name, so a configuration can check
* that a field naming one of its own inputs names one that will actually exist.
*
* <p>Public where {@link #extractNames(String)} is not, because a configuration asking "is this
* an input of mine?" needs the answer, not the {@link Placeholder} type behind it.
*/
public static boolean declaresPlaceholder(String text, String name) {
if (name == null || name.isBlank()) {
return false;
}
String wanted = name.strip();
return extractNames(text).stream().anyMatch(placeholder -> placeholder.name().equals(wanted));
}
}

View File

@ -23,6 +23,16 @@ public interface BlockType {
return NodeTypeCapabilities.activity();
}
/**
* How a person answers this block, or null when it takes no interaction. An interactive block
* that returns null renders with no form at all, so this belongs beside
* {@link #isUserInteractive()} rather than in a table kept elsewhere.
*/
@JsonIgnore
default InteractionContract getInteractionContract() {
return null;
}
@JsonIgnore
Class<? extends BlockConfiguration<?>> getBlockConfigurationClass();

View File

@ -34,6 +34,11 @@ public class ChatInteractionBlockType implements BlockType {
return true;
}
@Override
public InteractionContract getInteractionContract() {
return InteractionContract.chatSession();
}
@Override
public Class<? extends BlockConfiguration<?>> getBlockConfigurationClass() {
return ChatInteractionBlockConfiguration.class;

View File

@ -40,6 +40,11 @@ public class HumanDecisionBlockType implements BlockType {
return NodeTypeCapabilities.decision();
}
@Override
public InteractionContract getInteractionContract() {
return InteractionContract.humanDecision();
}
@Override
public Class<? extends BlockConfiguration<?>> getBlockConfigurationClass() {
return HumanDecisionBlockConfiguration.class;

View File

@ -0,0 +1,58 @@
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
// SPDX-License-Identifier: AGPL-3.0-or-later
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
package it.cnr.isti.workflow.manager.blocks.types;
import org.springframework.stereotype.Component;
import it.cnr.isti.workflow.manager.blocks.configurations.BlockConfiguration;
import it.cnr.isti.workflow.manager.blocks.configurations.HumanEvaluationBlockConfiguration;
import it.cnr.isti.workflow.manager.flows.model.capabilities.NodeTypeCapabilities;
@Component(HumanEvaluationBlockType.TYPE)
public class HumanEvaluationBlockType implements BlockType {
public static final String TYPE = "HumanEvaluationBlock";
@Override
public String getName() {
return TYPE;
}
@Override
public String getDescription() {
return "A person exercises something - typically running software - and scores it against named criteria. "
+ "Use it when the flow needs a judgement it can act on: the answer is a verdict per criterion plus "
+ "optional evidence, not free text. The node points at what to evaluate, it does not host it.";
}
@Override
public boolean validate() {
return true;
}
@Override
public boolean isUserInteractive() {
return true;
}
/**
* An activity, not a decision: a criterion may route the flow, but the node's purpose is to
* record a judgement, and it stays useful with no routing criterion at all.
*/
@Override
public NodeTypeCapabilities getCapabilities() {
return NodeTypeCapabilities.activity();
}
@Override
public InteractionContract getInteractionContract() {
return InteractionContract.evaluationForm();
}
@Override
public Class<? extends BlockConfiguration<?>> getBlockConfigurationClass() {
return HumanEvaluationBlockConfiguration.class;
}
}

View File

@ -30,6 +30,11 @@ public class HumanInteractionBlockType implements BlockType {
return true;
}
@Override
public InteractionContract getInteractionContract() {
return InteractionContract.singleResponse();
}
@Override
public Class<? extends BlockConfiguration<?>> getBlockConfigurationClass() {
return HumanInteractiveBlockConfiguration.class;

View File

@ -0,0 +1,51 @@
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
// SPDX-License-Identifier: AGPL-3.0-or-later
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
package it.cnr.isti.workflow.manager.blocks.types;
/**
* How a person answers this block: which shape the editor should render, and the interaction fields
* behind it.
*
* <p>Declared by the block type rather than looked up by name at the edges. The same table used to
* be written out twice - once for the editor, once for the assistant catalogue - and a new
* interactive block only worked once someone remembered to extend both.
*
* @param kind the shape to render: {@code chat-session}, {@code single-response},
* {@code human-decision}, {@code evaluation-form}
* @param messageField field carrying what the person writes along the way, when there is one
* @param completionField field whose arrival completes the step
* @param historyField field holding the exchange so far, for the shapes that have one
* @param responseField field the answer is read back from
* @param supportsPartialResult whether an answer may arrive in more than one piece
*/
public record InteractionContract(
String kind,
String messageField,
String completionField,
String historyField,
String responseField,
boolean supportsPartialResult) {
public static InteractionContract chatSession() {
return new InteractionContract("chat-session", "message", "response", "history", "response", true);
}
public static InteractionContract singleResponse() {
return new InteractionContract("single-response", null, "output", null, "output", false);
}
public static InteractionContract humanDecision() {
return new InteractionContract("human-decision", "rationale", "choice", null, "choice", true);
}
/**
* A form of named criteria. It completes on {@code verdict} because that is the field the whole
* judgement is submitted under, and it accepts partial results: revealing the reference verdict
* arrives before the judgement does.
*/
public static InteractionContract evaluationForm() {
return new InteractionContract("evaluation-form", "notes", "verdict", null, "verdict", true);
}
}

View File

@ -34,6 +34,11 @@ public class MCPAgentChatBlockType implements BlockType {
return true;
}
@Override
public InteractionContract getInteractionContract() {
return InteractionContract.chatSession();
}
@Override
public Class<? extends BlockConfiguration<?>> getBlockConfigurationClass() {
return MCPAgentChatBlockConfiguration.class;

View File

@ -186,34 +186,17 @@ public class BlocksController {
}
private InteractionContractDescriptor resolveInteractionContract(BlockType blockType) {
if ("ChatInteraction".equals(blockType.getName()) || "MCPAgentChat".equals(blockType.getName())) {
return new InteractionContractDescriptor(
"chat-session",
"message",
"response",
"history",
"response",
true);
it.cnr.isti.workflow.manager.blocks.types.InteractionContract contract = blockType.getInteractionContract();
if (contract == null) {
return null;
}
if ("HumanInteractionBlock".equals(blockType.getName())) {
return new InteractionContractDescriptor(
"single-response",
null,
"output",
null,
"output",
false);
}
if ("HumanDecisionBlock".equals(blockType.getName())) {
return new InteractionContractDescriptor(
"human-decision",
"rationale",
"choice",
null,
"choice",
true);
}
return null;
return new InteractionContractDescriptor(
contract.kind(),
contract.messageField(),
contract.completionField(),
contract.historyField(),
contract.responseField(),
contract.supportsPartialResult());
}
private BlockType resolveBlockType(String type) {

View File

@ -9,6 +9,7 @@ import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.stream.Collectors;
@ -39,6 +40,9 @@ import it.cnr.isti.workflow.manager.executions.ExecutionEvent;
import it.cnr.isti.workflow.manager.executions.ExecutionKind;
import it.cnr.isti.workflow.manager.executions.ExecutionObject;
import it.cnr.isti.workflow.manager.executions.ExecutionAuthorizationValueRequest;
import it.cnr.isti.workflow.manager.executions.HumanEvaluationRequest;
import it.cnr.isti.workflow.manager.blocks.factories.HumanEvaluationBlockFactory;
import it.cnr.isti.workflow.manager.executions.executors.blocks.HumanEvaluationExecutor;
import it.cnr.isti.workflow.manager.executions.ExecutionSimulationRequest;
import it.cnr.isti.workflow.manager.executions.ExecutionsService;
import it.cnr.isti.workflow.manager.executions.api.ExecutionContextView;
@ -410,6 +414,56 @@ public class ExecutionsController {
executionService.setInteractionValue(visibleExecution(executionId, userDetails).getId(), nodeId, fieldName, value));
}
@PutMapping(path = "{executionId}/node/{nodeId}/evaluation", consumes = "application/json")
@Operation(summary = "Submits a human evaluation",
description = "Records one person's judgement of a HumanEvaluation node, whole: every criterion and the "
+ "notes in one act. Refused if a criterion is unknown to the node or its verdict is outside the "
+ "scale the criterion declares, in which case nothing is recorded.")
public ExecutionView submitEvaluation(@PathVariable String executionId, @PathVariable String nodeId,
@RequestBody @jakarta.validation.Valid HumanEvaluationRequest request,
@AuthenticationPrincipal LoginEntity userDetails) {
Map<String, Object> interaction = new LinkedHashMap<>(request.verdict());
if (request.notes() != null) {
interaction.put(HumanEvaluationBlockFactory.NOTES_FIELD, request.notes());
}
return ExecutionView.fromExecution(executionService.setInteractionValues(
visibleExecution(executionId, userDetails).getId(), nodeId, interaction));
}
@PutMapping(path = "{executionId}/node/{nodeId}/evaluation/evidence", consumes = "multipart/form-data")
@Operation(summary = "Attaches evidence to a human evaluation",
description = "Adds files - screenshots, recordings, exports - to a waiting HumanEvaluation step. They "
+ "accumulate across calls and leave on the node's evidence output once the judgement is submitted.")
public ExecutionView attachEvaluationEvidence(@PathVariable String executionId, @PathVariable String nodeId,
@RequestParam List<MultipartFile> files, @AuthenticationPrincipal LoginEntity userDetails) {
List<File> stored = new ArrayList<>();
try {
for (MultipartFile file : files) {
File target = createUploadTempFile(HumanEvaluationBlockFactory.EVIDENCE_FIELD,
file.getOriginalFilename());
file.transferTo(target);
stored.add(target);
}
} catch (IOException e) {
throw new WebServerException("Error while creating file", e);
}
return ExecutionView.fromExecution(executionService.setInteractionValues(
visibleExecution(executionId, userDetails).getId(), nodeId,
Map.of(HumanEvaluationBlockFactory.EVIDENCE_FIELD, stored)));
}
@PutMapping(path = "{executionId}/node/{nodeId}/evaluation/reveal")
@Operation(summary = "Reveals the reference verdict",
description = "Marks that the person evaluating asked to see the reference verdict before judging, on a "
+ "node configured to keep it hidden. The execution records the reveal, so a judgement made after "
+ "it can be told from one made blind - which is the whole point of hiding it.")
public ExecutionView revealEvaluationReference(@PathVariable String executionId, @PathVariable String nodeId,
@AuthenticationPrincipal LoginEntity userDetails) {
return ExecutionView.fromExecution(executionService.setInteractionValues(
visibleExecution(executionId, userDetails).getId(), nodeId,
Map.of(HumanEvaluationExecutor.REFERENCE_REVEALED_FIELD, Boolean.TRUE)));
}
@PutMapping(path = "{executionId}/authorizations")
@Operation(summary = "Provides execution authorization",
description = "Stores the selected saved credential reference for a provider authorization required by an "

View File

@ -390,13 +390,23 @@ public class ExecutionContext implements ExecutionListener {
}
protected void setInteractionValue(String stepId, String fieldName, Object value) {
setInteractionValues(stepId, Map.of(fieldName, value));
}
/**
* Hands a whole interaction to the step at once.
*
* <p>A judgement made of several fields has to arrive together: submitted one field at a time,
* a failure halfway leaves the step holding half an answer, and no caller can tell which half.
*/
protected void setInteractionValues(String stepId, Map<String, Object> values) {
Step<?> step = this.steps.get(stepId);
if (step == null)
throw new IllegalArgumentException("Step with id " + stepId + " not found");
if (step.getStatus() != StepStatus.WAITING_FOR_INTERACTION)
throw new IllegalStateException("Step with id " + stepId
+ " is not in WAITING_FOR_INTERACTION status (CURRENT STATUS is " + step.getStatus() + ")");
step.interact(Map.of(fieldName, value));
step.interact(values);
}
protected void setContainerContinuation(String stepId, ContainerContinuationSnapshot continuation) {

View File

@ -21,6 +21,12 @@ public enum ExecutionEventType {
STEP_RESUMED,
HUMAN_DECISION_REQUESTED,
HUMAN_DECISION_RECORDED,
HUMAN_EVALUATION_RECORDED,
/**
* The reference verdict was shown to the person evaluating. Logged before the judgement to keep
* the order recoverable: an evaluation made after this event saw the other verdict first.
*/
HUMAN_EVALUATION_REFERENCE_REVEALED,
FLOW_PATH_ENDED,
FLOW_OUTCOME_RECORDED,
CONTAINER_SUBFLOW_STARTED,

View File

@ -188,8 +188,12 @@ public class ExecutionObject {
}
protected void setInteractionValue(String blockId, String fieldName, Object value) {
setInteractionValues(blockId, Map.of(fieldName, value));
}
protected void setInteractionValues(String blockId, Map<String, Object> values) {
if (this.context.getStatus() == ExecutionStatus.WAITING){
this.context.setInteractionValue(blockId, fieldName, value);
this.context.setInteractionValues(blockId, values);
} else
throw new IllegalStateException("Execution with id " + this.getId()
+ " is not in WAITING status (CURRENT STATUS is " + this.getContext().getStatus() + ")");

View File

@ -708,6 +708,13 @@ public class ExecutionsService {
return eo;
}
/** Submits a multi-field interaction as one act, so a step never holds half an answer. */
public ExecutionObject setInteractionValues(String executionId, String blockId, Map<String, Object> values) {
ExecutionObject eo = getExecution(executionId);
eo.setInteractionValues(blockId, values);
return eo;
}
public ExecutionObject setContainerContinuation(String executionId, String stepId,
ContainerContinuationSnapshot continuation) {
ExecutionObject execution = getExecution(executionId);

View File

@ -0,0 +1,30 @@
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
// SPDX-License-Identifier: AGPL-3.0-or-later
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
package it.cnr.isti.workflow.manager.executions;
import java.util.Map;
import jakarta.validation.constraints.NotEmpty;
import jakarta.validation.constraints.Size;
/**
* One person's judgement, submitted whole.
*
* <p>Whole, because the fields are one answer: sent a criterion at a time, a failure partway
* through leaves the step holding a judgement nobody made, with no way to tell which half is
* missing. The criterion names and their allowed values are the node's own, and the executor
* refuses anything else.
*
* @param verdict criterion name to the verdict given for it
* @param notes what the tester wants to say beyond the scores, optional
*/
public record HumanEvaluationRequest(
@NotEmpty Map<String, @Size(max = 64) String> verdict,
@Size(max = 4000) String notes) {
public HumanEvaluationRequest {
verdict = verdict == null ? Map.of() : Map.copyOf(verdict);
}
}

View File

@ -0,0 +1,311 @@
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
// SPDX-License-Identifier: AGPL-3.0-or-later
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
package it.cnr.isti.workflow.manager.executions.executors.blocks;
import java.util.ArrayList;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.stream.Collectors;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Component;
import org.springframework.util.StringUtils;
import it.cnr.isti.workflow.manager.blocks.Block;
import it.cnr.isti.workflow.manager.blocks.configurations.EvaluationCriterion;
import it.cnr.isti.workflow.manager.blocks.configurations.HumanEvaluationBlockConfiguration;
import it.cnr.isti.workflow.manager.blocks.factories.HumanEvaluationBlockFactory;
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
import it.cnr.isti.workflow.manager.executions.ExecutionEventLogger;
import it.cnr.isti.workflow.manager.executions.ExecutionEventType;
import it.cnr.isti.workflow.manager.executions.ExecutionVariableDescriptor;
import it.cnr.isti.workflow.manager.executions.InteractionResult;
import it.cnr.isti.workflow.manager.executions.NodeExecutionException;
import it.cnr.isti.workflow.manager.executions.bias.runtime.BiasRuntimeSupport;
import it.cnr.isti.workflow.manager.executions.steps.Input;
import it.cnr.isti.workflow.manager.llms.LLMCredentialResolver;
import it.cnr.isti.workflow.manager.llms.LLMDescriptor;
import it.cnr.isti.workflow.manager.llms.ProviderCredential;
import it.cnr.isti.workflow.manager.llms.providers.LLMProvider;
/**
* Turns one person's judgement of a piece of software into outputs the flow can act on.
*
* <p>Like every human block, {@link #execute} refuses: the engine never runs this node, a person
* does, through {@link #interact}. {@link #simulate} exists so bias experiments, which run with no
* humans available, still have something to run.
*/
@Component
public class HumanEvaluationExecutor implements BlockExecutor<HumanEvaluationBlockType> {
/** Marks, in the step's partial results, that the reference verdict was already shown. */
public static final String REFERENCE_REVEALED_FIELD = "__referenceRevealed";
@Autowired
private Map<String, LLMProvider> llmProviders;
@Autowired
private LLMCredentialResolver credentialResolver;
@Override
public Map<String, Object> execute(Block<HumanEvaluationBlockType> block, List<Input> inputs,
Map<String, Object> authorizations, Map<String, Object> executionVariables,
Map<String, ExecutionVariableDescriptor> executionVariableDescriptors, ExecutionEventLogger eventLogger) {
throw new UnsupportedOperationException("HumanEvaluation blocks require a person to evaluate the target");
}
@Override
public InteractionResult interact(Block<HumanEvaluationBlockType> block, List<Input> inputs,
Map<String, Object> interaction, Map<String, Object> partialResults, Map<String, Object> authorizations,
Map<String, Object> executionVariables,
Map<String, ExecutionVariableDescriptor> executionVariableDescriptors, ExecutionEventLogger eventLogger) {
HumanEvaluationBlockConfiguration configuration =
(HumanEvaluationBlockConfiguration) block.getSpecificConfiguration();
Map<String, Object> collected = new LinkedHashMap<>(partialResults == null ? Map.of() : partialResults);
if (interaction != null) {
interaction.forEach((field, value) -> {
if (!isKnownField(configuration, field)) {
throw new IllegalArgumentException("Unsupported HumanEvaluation interaction field: " + field);
}
if (HumanEvaluationBlockFactory.EVIDENCE_FIELD.equals(field)) {
// Evidence arrives upload by upload, so a second screenshot must not replace the first.
List<Object> merged = new ArrayList<>(asList(collected.get(field)));
merged.addAll(asList(value));
collected.put(field, merged);
} else {
collected.put(field, value);
}
});
boolean revealedNow = Boolean.TRUE.equals(interaction.get(REFERENCE_REVEALED_FIELD));
if (revealedNow && eventLogger != null) {
// Logged as it happens rather than with the verdict: the order of the two events is
// what says whether the judgement that follows was anchored to the reference.
eventLogger.info(ExecutionEventType.HUMAN_EVALUATION_REFERENCE_REVEALED,
"Reference verdict revealed before the evaluation was submitted",
Map.of("referenceInput", configuration.getReferenceInput() == null
? "" : configuration.getReferenceInput()));
}
}
// The judgement is one answer: until every criterion has a verdict, the step keeps waiting.
// A reveal or an evidence upload arrives on its own and lands here.
if (!missingCriteria(configuration, collected).isEmpty()) {
return InteractionResult.partial(collected);
}
for (EvaluationCriterion criterion : configuration.getCriteria()) {
String value = asText(collected.get(criterion.name()));
if (!criterion.scale().accepts(value)) {
throw new HumanEvaluationException("HUMAN_EVALUATION_INVALID_VERDICT",
"Criterion '" + criterion.name() + "' expects " + criterion.scale().allowedValuesDescription()
+ ", got: " + value);
}
}
List<Object> evidence = asList(collected.get(HumanEvaluationBlockFactory.EVIDENCE_FIELD));
if (configuration.isEvidenceRequired() && evidence.isEmpty()) {
throw new HumanEvaluationException("HUMAN_EVALUATION_EVIDENCE_REQUIRED",
"This evaluation requires at least one piece of evidence");
}
boolean blind = configuration.isBlindUntilSubmitted()
&& !Boolean.TRUE.equals(collected.get(REFERENCE_REVEALED_FIELD));
Map<String, Object> verdict = new LinkedHashMap<>();
Map<String, Object> outputs = new LinkedHashMap<>();
for (EvaluationCriterion criterion : configuration.getCriteria()) {
String value = asText(collected.get(criterion.name())).strip();
verdict.put(criterion.name(), value);
if (criterion.routing()) {
outputs.put(criterion.name(), value);
}
}
String notes = asText(collected.get(HumanEvaluationBlockFactory.NOTES_FIELD));
outputs.put(HumanEvaluationBlockFactory.VERDICT_FIELD, verdict);
outputs.put(HumanEvaluationBlockFactory.NOTES_FIELD, notes);
outputs.put(HumanEvaluationBlockFactory.EVIDENCE_FIELD, evidence);
outputs.put(HumanEvaluationBlockFactory.BLIND_FIELD, blind);
if (eventLogger != null) {
// The verdicts themselves are the point of the record; the notes are not, and may quote
// whatever the tester saw on screen.
eventLogger.info(ExecutionEventType.HUMAN_EVALUATION_RECORDED,
"Recorded human evaluation",
Map.of("verdict", verdict,
"blind", blind,
"evidenceCount", evidence.size(),
"notesPresent", StringUtils.hasText(notes)));
}
return InteractionResult.completed(outputs);
}
@Override
public Map<String, Object> simulate(Block<HumanEvaluationBlockType> block, List<Input> inputs,
Map<String, Object> authorizations, Map<String, Object> executionVariables,
Map<String, ExecutionVariableDescriptor> executionVariableDescriptors,
LLMDescriptor simulatorDescriptor, ExecutionEventLogger eventLogger) {
HumanEvaluationBlockConfiguration configuration =
(HumanEvaluationBlockConfiguration) block.getSpecificConfiguration();
if (simulatorDescriptor == null) {
throw new IllegalArgumentException("Missing simulation descriptor for HumanEvaluation execution");
}
LLMProvider llmProvider = resolveProvider(simulatorDescriptor);
ProviderCredential credential = credentialResolver.resolve(llmProvider, authorizations, executionVariables);
String prompt = BiasRuntimeSupport.decoratePrompt(simulationPrompt(configuration, inputs), executionVariables);
ModelParameterReporting.reportUnsupported(llmProvider, simulatorDescriptor.parameters(),
simulatorDescriptor.model(), eventLogger);
String response = llmProvider.generate(simulatorDescriptor.model(), prompt, credential,
simulatorDescriptor.parameters());
if (!StringUtils.hasText(response)) {
throw new IllegalArgumentException("Simulated HumanEvaluation produced an empty response");
}
Map<String, String> answered = parseSimulatedVerdict(response);
Map<String, Object> verdict = new LinkedHashMap<>();
Map<String, Object> outputs = new LinkedHashMap<>();
for (EvaluationCriterion criterion : configuration.getCriteria()) {
String value = answered.get(criterion.name().toLowerCase());
if (!criterion.scale().accepts(value)) {
throw new HumanEvaluationException("HUMAN_EVALUATION_SIMULATION_FAILED",
"Simulator did not return a usable verdict for criterion '" + criterion.name() + "'");
}
String normalized = value.strip();
verdict.put(criterion.name(), normalized);
if (criterion.routing()) {
outputs.put(criterion.name(), normalized);
}
}
outputs.put(HumanEvaluationBlockFactory.VERDICT_FIELD, verdict);
outputs.put(HumanEvaluationBlockFactory.NOTES_FIELD, answered.getOrDefault("notes", ""));
outputs.put(HumanEvaluationBlockFactory.EVIDENCE_FIELD, List.of());
// A simulator is never anchored to a reference it was not shown, and this node's whole
// reason for recording the flag is to tell anchored judgements from unanchored ones.
outputs.put(HumanEvaluationBlockFactory.BLIND_FIELD, configuration.isBlindUntilSubmitted());
return outputs;
}
@Override
public Class<HumanEvaluationBlockType> getBlockType() {
return HumanEvaluationBlockType.class;
}
@Override
public boolean isInteractive() {
return true;
}
private LLMProvider resolveProvider(LLMDescriptor descriptor) {
LLMProvider provider = llmProviders.get(descriptor.provider());
if (provider == null) {
provider = llmProviders.values().stream()
.filter(candidate -> candidate.getName().equals(descriptor.provider()))
.findFirst()
.orElse(null);
}
if (provider == null) {
throw new IllegalArgumentException("Provider not found: " + descriptor.provider());
}
return provider;
}
private String simulationPrompt(HumanEvaluationBlockConfiguration configuration, List<Input> inputs) {
String context = inputs.stream()
.map(input -> "%s= %s".formatted(input.getDescriptor().getName(), input.getValue()))
.collect(Collectors.joining(", "));
String steps = configuration.getTaskScript().isEmpty()
? "(no script given: judge on the description alone)"
: configuration.getTaskScript().stream()
.map(step -> "- " + step)
.collect(Collectors.joining("\n"));
String criteria = configuration.getCriteria().stream()
.map(criterion -> "- %s (%s): %s".formatted(criterion.name(),
criterion.scale().allowedValuesDescription(),
StringUtils.hasText(criterion.description()) ? criterion.description() : criterion.name()))
.collect(Collectors.joining("\n"));
String answerLines = configuration.getCriteria().stream()
.map(criterion -> "%s: <%s>".formatted(criterion.name(), criterion.scale().allowedValuesDescription()))
.collect(Collectors.joining("\n"));
return """
You are standing in for a human tester. Given the context { %s }, evaluate this target:
"%s"
Steps performed:
%s
Score every criterion:
%s
Answer in exactly this format, one line per criterion, nothing else before them:
%s
NOTES: <one short line, or empty>
""".formatted(context, configuration.getTarget(), steps, criteria, answerLines);
}
/** Reads back the {@code name: value} lines the prompt asks for, case-insensitively. */
private Map<String, String> parseSimulatedVerdict(String response) {
Map<String, String> answered = new LinkedHashMap<>();
for (String line : response.lines().toList()) {
String trimmed = line.strip();
int separator = trimmed.indexOf(':');
if (separator <= 0) {
continue;
}
String key = trimmed.substring(0, separator).strip().toLowerCase();
String value = trimmed.substring(separator + 1).strip();
if (!key.isEmpty() && !answered.containsKey(key)) {
answered.put(key, value);
}
}
return answered;
}
private boolean isKnownField(HumanEvaluationBlockConfiguration configuration, String field) {
if (REFERENCE_REVEALED_FIELD.equals(field)
|| HumanEvaluationBlockFactory.NOTES_FIELD.equals(field)
|| HumanEvaluationBlockFactory.EVIDENCE_FIELD.equals(field)) {
return true;
}
return configuration.getCriteria().stream()
.anyMatch(criterion -> criterion.name().equals(field));
}
private List<String> missingCriteria(HumanEvaluationBlockConfiguration configuration,
Map<String, Object> collected) {
List<String> missing = new ArrayList<>();
for (EvaluationCriterion criterion : configuration.getCriteria()) {
if (!StringUtils.hasText(asText(collected.get(criterion.name())))) {
missing.add(criterion.name());
}
}
return missing;
}
private static String asText(Object value) {
return value == null ? "" : value.toString();
}
private static List<Object> asList(Object value) {
if (value == null) {
return List.of();
}
if (value instanceof List<?> list) {
return List.copyOf(list);
}
return List.of(value);
}
private static final class HumanEvaluationException extends NodeExecutionException {
private HumanEvaluationException(String errorCode, String message) {
super(errorCode, message);
}
}
}

View File

@ -0,0 +1,88 @@
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
// SPDX-License-Identifier: AGPL-3.0-or-later
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
package it.cnr.isti.workflow.manager.blocks.factories;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.util.List;
import org.junit.jupiter.api.BeforeEach;
import org.junit.jupiter.api.Test;
import it.cnr.isti.workflow.manager.blocks.Block;
import it.cnr.isti.workflow.manager.blocks.configurations.EvaluationCriterion;
import it.cnr.isti.workflow.manager.blocks.configurations.EvaluationScale;
import it.cnr.isti.workflow.manager.blocks.configurations.HumanEvaluationBlockConfiguration;
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
import it.cnr.isti.workflow.manager.ios.IODescriptor;
import it.cnr.isti.workflow.manager.ios.IOType;
class HumanEvaluationBlockFactoryTest {
private HumanEvaluationBlockFactory factory;
@BeforeEach
void setUp() {
factory = new HumanEvaluationBlockFactory();
factory.blockType = new HumanEvaluationBlockType();
}
private List<String> outputNames(Block<HumanEvaluationBlockType> block) {
return block.getOutputs().stream().map(IODescriptor::getName).toList();
}
@Test
void givesAPortOnlyToTheCriteriaThatRoute() {
// Twelve criteria would otherwise leave the node bristling with ports nobody wires; the
// ones that do not route are still in the verdict payload.
Block<HumanEvaluationBlockType> block = factory.create(HumanEvaluationBlockConfiguration.builder()
.name("evaluation")
.target("Open ${{deployUrl}}")
.criteria(List.of(
new EvaluationCriterion("works", null, EvaluationScale.PASS_FAIL, true),
new EvaluationCriterion("clarity", null, EvaluationScale.SCORE_1_5, false)))
.build());
assertTrue(outputNames(block).contains("works"));
assertFalse(outputNames(block).contains("clarity"));
}
@Test
void alwaysCarriesTheVerdictNotesEvidenceAndBlindOutputs() {
Block<HumanEvaluationBlockType> block = factory.createEmpty();
assertTrue(outputNames(block).containsAll(List.of(
HumanEvaluationBlockFactory.VERDICT_FIELD,
HumanEvaluationBlockFactory.NOTES_FIELD,
HumanEvaluationBlockFactory.EVIDENCE_FIELD,
HumanEvaluationBlockFactory.BLIND_FIELD)));
}
@Test
void evidenceLeavesAsSeveralFiles() {
// One screenshot is the common case, but a tester who took four must not lose three.
IODescriptor evidence = factory.createEmpty().getOutputs().stream()
.filter(output -> HumanEvaluationBlockFactory.EVIDENCE_FIELD.equals(output.getName()))
.findFirst()
.orElseThrow();
assertEquals(IOType.FILE, evidence.getType());
assertTrue(evidence.isMultiple());
}
@Test
void turnsEveryTargetPlaceholderIntoAnInput() {
Block<HumanEvaluationBlockType> block = factory.create(HumanEvaluationBlockConfiguration.builder()
.name("evaluation")
.target("Open ${{deployUrl}} and compare with ${{agentReport}}")
.criteria(List.of(new EvaluationCriterion("works")))
.build());
List<String> inputs = block.getInputs().stream().map(IODescriptor::getName).toList();
assertTrue(inputs.containsAll(List.of("deployUrl", "agentReport")));
}
}

View File

@ -28,6 +28,7 @@ import it.cnr.isti.workflow.manager.blocks.IOCapabilityType;
import it.cnr.isti.workflow.manager.blocks.configurations.ChatInteractionBlockConfiguration;
import it.cnr.isti.workflow.manager.blocks.configurations.ChatInteractionInput;
import it.cnr.isti.workflow.manager.blocks.types.HumanInteractionBlockType;
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
import it.cnr.isti.workflow.manager.blocks.types.ChatInteractionBlockType;
import it.cnr.isti.workflow.manager.blocks.types.ConditionalBlockType;
import it.cnr.isti.workflow.manager.blocks.configurations.HTTPServerCallBlockConfiguration;
@ -89,6 +90,38 @@ public class BlocksControllerTest {
System.out.println("Type: " + type.type());
}
@Test
public void theEvaluationNodeReachesTheEditorWithTheFormItNeeds() {
// What the editor renders comes entirely from this descriptor: without the contract it would
// show an interactive node with no way to answer it.
BlockConfigurationDescriptor evaluation = blocksController.getTypes().stream()
.filter(type -> HumanEvaluationBlockType.TYPE.equals(type.type()))
.findFirst()
.orElseThrow(() -> new AssertionError("HumanEvaluationBlock is not in the catalog"));
assertTrue(evaluation.userInteractive());
assertNotNull(evaluation.interactionContract());
assertEquals("evaluation-form", evaluation.interactionContract().kind());
assertEquals("verdict", evaluation.interactionContract().completionField());
// The judgement arrives after a reveal or an evidence upload, which are interactions of
// their own: a contract that refused partial results would drop them.
assertTrue(evaluation.interactionContract().supportsPartialResult());
}
@Test
public void everyInteractiveBlockSaysHowItIsAnswered() {
// The two readers of this used to keep their own table of block names, so a new interactive
// block rendered as unsupported until someone remembered to extend both.
List<BlockConfigurationDescriptor> unanswerable = blocksController.getTypes().stream()
.filter(BlockConfigurationDescriptor::userInteractive)
.filter(type -> type.interactionContract() == null)
.toList();
assertTrue(unanswerable.isEmpty(),
() -> "Interactive blocks with no interaction contract: "
+ unanswerable.stream().map(BlockConfigurationDescriptor::type).toList());
}
@Test
public void getTypeCatalogExtractsSharedBlockDefinitions() {
BlocksController.BlockConfigurationCatalog catalog = blocksController.getConfigurationCatalog();

View File

@ -0,0 +1,157 @@
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
// SPDX-License-Identifier: AGPL-3.0-or-later
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
package it.cnr.isti.workflow.manager.executions.executors.blocks;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertThrows;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.File;
import java.util.List;
import java.util.Map;
import org.junit.jupiter.api.Test;
import it.cnr.isti.workflow.manager.blocks.Block;
import it.cnr.isti.workflow.manager.blocks.configurations.EvaluationCriterion;
import it.cnr.isti.workflow.manager.blocks.configurations.EvaluationScale;
import it.cnr.isti.workflow.manager.blocks.configurations.HumanEvaluationBlockConfiguration;
import it.cnr.isti.workflow.manager.blocks.factories.HumanEvaluationBlockFactory;
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
import it.cnr.isti.workflow.manager.executions.InteractionResult;
import it.cnr.isti.workflow.manager.executions.NodeExecutionException;
class HumanEvaluationExecutorTest {
private final HumanEvaluationExecutor executor = new HumanEvaluationExecutor();
private HumanEvaluationBlockConfiguration config(boolean evidenceRequired, boolean blind) {
return HumanEvaluationBlockConfiguration.builder()
.name("evaluation")
.target("Open ${{deployUrl}} and compare with ${{agentReport}}")
.taskScript(List.of("Sign in", "Create an order"))
.criteria(List.of(
new EvaluationCriterion("works", "Does the order go through?", EvaluationScale.PASS_FAIL, true),
new EvaluationCriterion("clarity", "How clear are the errors?", EvaluationScale.SCORE_1_5,
false)))
.evidenceRequired(evidenceRequired)
.blindUntilSubmitted(blind)
.referenceInput(blind ? "agentReport" : null)
.build();
}
private Block<HumanEvaluationBlockType> block(HumanEvaluationBlockConfiguration configuration) {
return Block.<HumanEvaluationBlockType>builder()
.type(new HumanEvaluationBlockType())
.specificConfiguration(configuration)
.build();
}
private InteractionResult interact(HumanEvaluationBlockConfiguration configuration,
Map<String, Object> interaction, Map<String, Object> partialResults) {
return executor.interact(block(configuration), List.of(), interaction, partialResults, Map.of(), Map.of(),
Map.of(), null);
}
@Test
void keepsWaitingUntilEveryCriterionHasAVerdict() {
// A judgement is one answer: half of it is not a smaller judgement, it is no judgement.
InteractionResult result = interact(config(false, false), Map.of("works", "pass"), Map.of());
assertFalse(result.completed());
assertEquals("pass", result.partialResults().get("works"));
}
@Test
void completesWithAVerdictPerCriterionAndRoutesOnlyWhereAsked() {
InteractionResult result = interact(config(false, false),
Map.of("works", "pass", "clarity", "4", HumanEvaluationBlockFactory.NOTES_FIELD, "Slow but fine"),
Map.of());
assertTrue(result.completed());
assertEquals(Map.of("works", "pass", "clarity", "4"),
result.outputs().get(HumanEvaluationBlockFactory.VERDICT_FIELD));
// works routes, clarity does not: it is in the verdict but has no port of its own.
assertEquals("pass", result.outputs().get("works"));
assertFalse(result.outputs().containsKey("clarity"));
assertEquals("Slow but fine", result.outputs().get(HumanEvaluationBlockFactory.NOTES_FIELD));
}
@Test
void refusesAVerdictOutsideTheCriterionScale() {
// The scale is the whole reason two evaluations can be compared; "mostly" is not on it.
NodeExecutionException failure = assertThrows(NodeExecutionException.class,
() -> interact(config(false, false), Map.of("works", "mostly", "clarity", "4"), Map.of()));
assertTrue(failure.getMessage().contains("works"));
}
@Test
void refusesAScoreOutsideTheOneToFiveRange() {
assertThrows(NodeExecutionException.class,
() -> interact(config(false, false), Map.of("works", "pass", "clarity", "9"), Map.of()));
}
@Test
void refusesAFieldTheNodeDoesNotHave() {
// Silently dropping it would record a judgement the person did not give.
assertThrows(IllegalArgumentException.class,
() -> interact(config(false, false), Map.of("speed", "pass"), Map.of()));
}
@Test
void evidenceFromSeveralUploadsAccumulates() {
Map<String, Object> afterFirst = interact(config(false, false),
Map.of(HumanEvaluationBlockFactory.EVIDENCE_FIELD, List.of(new File("one.png"))), Map.of())
.partialResults();
InteractionResult result = interact(config(false, false),
Map.of(HumanEvaluationBlockFactory.EVIDENCE_FIELD, List.of(new File("two.png")),
"works", "pass", "clarity", "3"),
afterFirst);
assertTrue(result.completed());
assertEquals(2, ((List<?>) result.outputs().get(HumanEvaluationBlockFactory.EVIDENCE_FIELD)).size());
}
@Test
void refusesAVerdictWithNoEvidenceWhenEvidenceIsRequired() {
assertThrows(NodeExecutionException.class,
() -> interact(config(true, false), Map.of("works", "pass", "clarity", "3"), Map.of()));
}
@Test
void recordsAJudgementAsBlindWhenTheReferenceWasNeverRevealed() {
InteractionResult result = interact(config(false, true), Map.of("works", "fail", "clarity", "2"), Map.of());
assertEquals(true, result.outputs().get(HumanEvaluationBlockFactory.BLIND_FIELD));
}
@Test
void recordsAJudgementAsAnchoredOnceTheReferenceHasBeenRevealed() {
// This flag is the measurement: without it, an anchored verdict and a blind one are the
// same row in the results.
Map<String, Object> afterReveal = interact(config(false, true),
Map.of(HumanEvaluationExecutor.REFERENCE_REVEALED_FIELD, Boolean.TRUE), Map.of()).partialResults();
InteractionResult result = interact(config(false, true), Map.of("works", "fail", "clarity", "2"), afterReveal);
assertEquals(false, result.outputs().get(HumanEvaluationBlockFactory.BLIND_FIELD));
}
@Test
void aNodeThatHidesNothingIsNeverReportedAsBlind() {
InteractionResult result = interact(config(false, false), Map.of("works", "pass", "clarity", "5"), Map.of());
assertEquals(false, result.outputs().get(HumanEvaluationBlockFactory.BLIND_FIELD));
}
@Test
void theEngineNeverRunsThisNodeOnItsOwn() {
assertThrows(UnsupportedOperationException.class,
() -> executor.execute(block(config(false, false)), List.of(), Map.of(), Map.of(), Map.of(), null));
}
}