Add a node for a person to evaluate software against named criteria
The human nodes could ask a question or collect a paragraph, so a flow that wanted a judgement about a running piece of software got back prose: not comparable between two runs, and nothing a later node could branch on. HumanEvaluationBlock asks instead for a verdict per named criterion, with an optional script so two testers exercise the same thing, and evidence files. It points at the target rather than hosting it - provisioning environments or driving browsers belongs outside the engine, and the flow already knows the URL. blindUntilSubmitted keeps a reference verdict, usually an agent's, out of sight until the person commits to their own, and the execution records which came first. That turns the node from a place to rubber-stamp an agent into something that measures the automation bias the catalogue already names. The engine needed only a way to submit several fields as one act, which Step.interact already accepted; and the interaction contract now comes from the block type, so the editor and the assistant stop keeping their own table of block names. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
52b02ce349
commit
6032622c67
|
|
@ -255,33 +255,16 @@ public class BlockCatalogService {
|
|||
}
|
||||
|
||||
private AssistantInteractionContractDescriptor resolveInteractionContract(BlockType blockType) {
|
||||
if ("ChatInteraction".equals(blockType.getName()) || "MCPAgentChat".equals(blockType.getName())) {
|
||||
return new AssistantInteractionContractDescriptor(
|
||||
"chat-session",
|
||||
"message",
|
||||
"response",
|
||||
"history",
|
||||
"response",
|
||||
true);
|
||||
it.cnr.isti.workflow.manager.blocks.types.InteractionContract contract = blockType.getInteractionContract();
|
||||
if (contract == null) {
|
||||
return null;
|
||||
}
|
||||
if ("HumanInteractionBlock".equals(blockType.getName())) {
|
||||
return new AssistantInteractionContractDescriptor(
|
||||
"single-response",
|
||||
null,
|
||||
"output",
|
||||
null,
|
||||
"output",
|
||||
false);
|
||||
}
|
||||
if ("HumanDecisionBlock".equals(blockType.getName())) {
|
||||
return new AssistantInteractionContractDescriptor(
|
||||
"human-decision",
|
||||
"rationale",
|
||||
"choice",
|
||||
null,
|
||||
"choice",
|
||||
true);
|
||||
}
|
||||
return null;
|
||||
return new AssistantInteractionContractDescriptor(
|
||||
contract.kind(),
|
||||
contract.messageField(),
|
||||
contract.completionField(),
|
||||
contract.historyField(),
|
||||
contract.responseField(),
|
||||
contract.supportsPartialResult());
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,59 @@
|
|||
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
|
||||
|
||||
package it.cnr.isti.workflow.manager.blocks.configurations;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
|
||||
import com.fasterxml.jackson.annotation.JsonProperty;
|
||||
|
||||
import it.cnr.isti.workflow.manager.configurations.annotations.SchemaAllowedValues;
|
||||
import it.cnr.isti.workflow.manager.configurations.annotations.UiDescription;
|
||||
import it.cnr.isti.workflow.manager.configurations.annotations.UiLabel;
|
||||
import it.cnr.isti.workflow.manager.configurations.annotations.UiOrder;
|
||||
import jakarta.validation.constraints.NotBlank;
|
||||
import jakarta.validation.constraints.Size;
|
||||
|
||||
/**
|
||||
* One thing the tester is asked to judge, and how.
|
||||
*
|
||||
* <p>Criteria are what make two evaluations of the same software comparable: a free-text "it works"
|
||||
* cannot be compared, a verdict per named criterion can. {@code routing} decides whether the
|
||||
* criterion also becomes an output port - twelve criteria would otherwise leave the node bristling
|
||||
* with ports nobody wires.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record EvaluationCriterion(
|
||||
@NotBlank
|
||||
@Size(max = 64)
|
||||
@UiLabel("Criterion")
|
||||
@UiDescription("Short name, also the output port name when this criterion routes.")
|
||||
@UiOrder(10)
|
||||
@JsonProperty(required = true)
|
||||
String name,
|
||||
|
||||
@Size(max = 500)
|
||||
@UiLabel("What to judge")
|
||||
@UiDescription("What the tester should look at to answer. Vague wording here is the usual reason two testers disagree.")
|
||||
@UiOrder(20)
|
||||
String description,
|
||||
|
||||
@UiLabel("Scale")
|
||||
@UiDescription("pass/fail, or a 1-5 score.")
|
||||
@UiOrder(30)
|
||||
@SchemaAllowedValues(value = { "PASS_FAIL", "SCORE_1_5" }, defaultValue = "PASS_FAIL")
|
||||
EvaluationScale scale,
|
||||
|
||||
@UiLabel("Routes the flow")
|
||||
@UiDescription("Gives this criterion its own output port, so a later node can branch on it.")
|
||||
@UiOrder(40)
|
||||
boolean routing) {
|
||||
|
||||
public EvaluationCriterion {
|
||||
scale = scale == null ? EvaluationScale.PASS_FAIL : scale;
|
||||
}
|
||||
|
||||
public EvaluationCriterion(String name) {
|
||||
this(name, null, EvaluationScale.PASS_FAIL, false);
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,54 @@
|
|||
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
|
||||
|
||||
package it.cnr.isti.workflow.manager.blocks.configurations;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonCreator;
|
||||
import com.fasterxml.jackson.annotation.JsonValue;
|
||||
|
||||
/**
|
||||
* How one evaluation criterion is scored.
|
||||
*
|
||||
* <p>The choice is per criterion rather than per node because the two scales answer different
|
||||
* questions: PASS_FAIL is easier for two people to agree on, SCORE_1_5 carries more information at
|
||||
* the cost of that agreement. A node usually wants both - "does it work at all" next to "how good
|
||||
* is it".
|
||||
*/
|
||||
public enum EvaluationScale {
|
||||
PASS_FAIL,
|
||||
SCORE_1_5;
|
||||
|
||||
@JsonCreator
|
||||
public static EvaluationScale fromString(String key) {
|
||||
return EvaluationScale.valueOf(key.toUpperCase());
|
||||
}
|
||||
|
||||
@JsonValue
|
||||
public String toValue() {
|
||||
return name();
|
||||
}
|
||||
|
||||
/** Whether {@code value} is one of the verdicts this scale accepts, case-insensitively. */
|
||||
public boolean accepts(String value) {
|
||||
if (value == null || value.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
String normalized = value.strip();
|
||||
return switch (this) {
|
||||
case PASS_FAIL -> "pass".equalsIgnoreCase(normalized) || "fail".equalsIgnoreCase(normalized);
|
||||
case SCORE_1_5 -> switch (normalized) {
|
||||
case "1", "2", "3", "4", "5" -> true;
|
||||
default -> false;
|
||||
};
|
||||
};
|
||||
}
|
||||
|
||||
/** What the tester may answer, spelled out for the person and for the simulator prompt. */
|
||||
public String allowedValuesDescription() {
|
||||
return switch (this) {
|
||||
case PASS_FAIL -> "pass or fail";
|
||||
case SCORE_1_5 -> "a whole number from 1 to 5";
|
||||
};
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,192 @@
|
|||
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
|
||||
|
||||
package it.cnr.isti.workflow.manager.blocks.configurations;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonIgnore;
|
||||
import com.fasterxml.jackson.annotation.JsonProperty;
|
||||
|
||||
import it.cnr.isti.workflow.manager.blocks.factories.TemplateInputs;
|
||||
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
|
||||
import it.cnr.isti.workflow.manager.configurations.annotations.LongText;
|
||||
import it.cnr.isti.workflow.manager.configurations.annotations.Structural;
|
||||
import it.cnr.isti.workflow.manager.configurations.annotations.UiDescription;
|
||||
import it.cnr.isti.workflow.manager.configurations.annotations.UiLabel;
|
||||
import it.cnr.isti.workflow.manager.configurations.annotations.UiOrder;
|
||||
import it.cnr.isti.workflow.manager.configurations.annotations.UiUniqueItemsBy;
|
||||
import jakarta.validation.Valid;
|
||||
import jakarta.validation.constraints.AssertTrue;
|
||||
import jakarta.validation.constraints.NotBlank;
|
||||
import jakarta.validation.constraints.Size;
|
||||
import lombok.Builder;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.Getter;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.NonNull;
|
||||
|
||||
/**
|
||||
* A person exercises something - typically a running piece of software - and scores it against
|
||||
* named criteria.
|
||||
*
|
||||
* <p>It differs from {@link HumanInteractiveBlockConfiguration}, which returns one blob of text, in
|
||||
* that the answer is structured: a verdict per criterion, so two evaluations of the same target can
|
||||
* be compared and a later node can branch on one of them without parsing prose.
|
||||
*
|
||||
* <p>The node deliberately does not host the thing under test. {@code target} points at it - a URL
|
||||
* a previous node produced, or an artefact in a workspace - and the tester opens it however they
|
||||
* normally would. Hosting browsers or provisioning environments belongs outside the workflow engine.
|
||||
*/
|
||||
@Getter
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
@NoArgsConstructor(access = lombok.AccessLevel.PROTECTED)
|
||||
public class HumanEvaluationBlockConfiguration extends BlockConfiguration<HumanEvaluationBlockType> {
|
||||
|
||||
public static final int MAX_CRITERIA = 12;
|
||||
public static final int MAX_TASK_STEPS = 20;
|
||||
|
||||
@UiOrder(30)
|
||||
@NotBlank
|
||||
@Size(max = 2000)
|
||||
@LongText(
|
||||
placeholder = "What the tester should open, and where",
|
||||
tip = "Use ${{}} to pull in something an earlier block produced, for example: Open ${{deployUrl}} and sign in as the demo user.",
|
||||
acceptVariableAsPlaceholder = true)
|
||||
@UiLabel("What to evaluate")
|
||||
@UiDescription("Where the thing under test lives and how to reach it. The node points at it, it does not host it.")
|
||||
@JsonProperty(required = true)
|
||||
private String target;
|
||||
|
||||
@UiOrder(40)
|
||||
@Size(max = MAX_TASK_STEPS)
|
||||
@UiLabel("Steps to perform")
|
||||
@UiDescription("The same script for every tester. Without it two people try different things and their verdicts are not comparable.")
|
||||
private List<@NotBlank @Size(max = 500) String> taskScript = List.of();
|
||||
|
||||
@UiOrder(50)
|
||||
@Structural
|
||||
@Valid
|
||||
@Size(min = 1, max = MAX_CRITERIA)
|
||||
@UiUniqueItemsBy("name")
|
||||
@UiLabel("Criteria")
|
||||
@JsonProperty(required = true)
|
||||
private List<EvaluationCriterion> criteria = List.of();
|
||||
|
||||
@UiOrder(60)
|
||||
@UiLabel("Require evidence")
|
||||
@UiDescription("Refuses a verdict with no attachment. Worth it when someone will have to review the judgement later.")
|
||||
private boolean evidenceRequired;
|
||||
|
||||
@UiOrder(70)
|
||||
@UiLabel("Hide the reference verdict until submitted")
|
||||
@UiDescription("Keeps an AI's own verdict out of sight until the person has committed to theirs, so the judgement is not anchored to it. The execution records which came first.")
|
||||
private boolean blindUntilSubmitted;
|
||||
|
||||
@UiOrder(80)
|
||||
@Size(max = 64)
|
||||
@UiLabel("Reference input")
|
||||
@UiDescription("Name of the input holding the verdict to compare against, usually an agent's. Must be one of the ${{}} names used above.")
|
||||
private String referenceInput;
|
||||
|
||||
@Builder
|
||||
public HumanEvaluationBlockConfiguration(@NonNull String name, String target, List<String> taskScript,
|
||||
List<EvaluationCriterion> criteria, boolean evidenceRequired, boolean blindUntilSubmitted,
|
||||
String referenceInput) {
|
||||
super(name);
|
||||
this.target = target;
|
||||
this.taskScript = taskScript == null ? List.of() : List.copyOf(taskScript);
|
||||
this.criteria = criteria == null ? List.of() : List.copyOf(criteria);
|
||||
this.evidenceRequired = evidenceRequired;
|
||||
this.blindUntilSubmitted = blindUntilSubmitted;
|
||||
this.referenceInput = referenceInput;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Class<HumanEvaluationBlockType> getBlockType() {
|
||||
return HumanEvaluationBlockType.class;
|
||||
}
|
||||
|
||||
public static HumanEvaluationBlockConfiguration empty() {
|
||||
return HumanEvaluationBlockConfiguration.builder()
|
||||
.name(HumanEvaluationBlockType.TYPE)
|
||||
.target("Open ${{targetUrl}} and try the main flow")
|
||||
.taskScript(List.of())
|
||||
.criteria(List.of(new EvaluationCriterion("works", "Does the main flow complete without errors?",
|
||||
EvaluationScale.PASS_FAIL, true)))
|
||||
.build();
|
||||
}
|
||||
|
||||
/** The criteria that also become output ports, in declaration order. */
|
||||
@JsonIgnore
|
||||
public List<EvaluationCriterion> routingCriteria() {
|
||||
return criteria.stream().filter(EvaluationCriterion::routing).toList();
|
||||
}
|
||||
|
||||
@AssertTrue(message = "target placeholder names must start with a letter and contain only letters, digits, '-', '_' or '.'")
|
||||
@JsonIgnore
|
||||
boolean isTargetPlaceholderNamesValid() {
|
||||
return TemplateInputs.hasValidPlaceholderNames(target);
|
||||
}
|
||||
|
||||
@AssertTrue(message = "target placeholder '[]' (array) marker must be used consistently for every occurrence of the same name")
|
||||
@JsonIgnore
|
||||
boolean isTargetPlaceholderMultiplicityConsistent() {
|
||||
return TemplateInputs.hasConsistentPlaceholderMultiplicity(target);
|
||||
}
|
||||
|
||||
@AssertTrue(message = "criteria must have unique non-blank names")
|
||||
@JsonIgnore
|
||||
boolean areCriteriaNamesUnique() {
|
||||
if (criteria == null || criteria.isEmpty()) {
|
||||
return true;
|
||||
}
|
||||
long named = criteria.stream()
|
||||
.filter(criterion -> criterion != null && criterion.name() != null && !criterion.name().isBlank())
|
||||
.count();
|
||||
long distinct = criteria.stream()
|
||||
.filter(criterion -> criterion != null && criterion.name() != null && !criterion.name().isBlank())
|
||||
.map(criterion -> criterion.name().strip())
|
||||
.distinct()
|
||||
.count();
|
||||
return named == criteria.size() && distinct == named;
|
||||
}
|
||||
|
||||
/**
|
||||
* A criterion named like one of the node's own outputs would produce two ports with one name,
|
||||
* and the second would silently win.
|
||||
*/
|
||||
@AssertTrue(message = "criteria cannot be named verdict, notes, evidence or blind: those are the node's own outputs")
|
||||
@JsonIgnore
|
||||
boolean areCriteriaNamesFree() {
|
||||
if (criteria == null) {
|
||||
return true;
|
||||
}
|
||||
return criteria.stream()
|
||||
.filter(criterion -> criterion != null && criterion.name() != null)
|
||||
.noneMatch(criterion -> RESERVED_OUTPUT_NAMES.contains(criterion.name().strip().toLowerCase()));
|
||||
}
|
||||
|
||||
/**
|
||||
* Blind evaluation only means something when there is a reference to hide; without one the flag
|
||||
* would be a switch that does nothing, which is worse than a refused save.
|
||||
*/
|
||||
@AssertTrue(message = "referenceInput is required when the reference verdict is hidden until submitted")
|
||||
@JsonIgnore
|
||||
boolean isReferenceInputPresentWhenBlind() {
|
||||
return !blindUntilSubmitted || (referenceInput != null && !referenceInput.isBlank());
|
||||
}
|
||||
|
||||
/** The reference has to be one of the inputs the node actually has, or nothing is ever hidden. */
|
||||
@AssertTrue(message = "referenceInput must be one of the ${{}} placeholder names used in the target")
|
||||
@JsonIgnore
|
||||
boolean isReferenceInputAmongPlaceholders() {
|
||||
if (referenceInput == null || referenceInput.isBlank()) {
|
||||
return true;
|
||||
}
|
||||
return TemplateInputs.declaresPlaceholder(target, referenceInput);
|
||||
}
|
||||
|
||||
private static final List<String> RESERVED_OUTPUT_NAMES = List.of("verdict", "notes", "evidence", "blind");
|
||||
}
|
||||
|
|
@ -0,0 +1,88 @@
|
|||
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
|
||||
|
||||
package it.cnr.isti.workflow.manager.blocks.factories;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import it.cnr.isti.workflow.manager.blocks.Block;
|
||||
import it.cnr.isti.workflow.manager.blocks.IOCapability;
|
||||
import it.cnr.isti.workflow.manager.blocks.IOCapabilityType;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.HumanEvaluationBlockConfiguration;
|
||||
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
|
||||
import it.cnr.isti.workflow.manager.ios.IODescriptor;
|
||||
import it.cnr.isti.workflow.manager.ios.IOType;
|
||||
|
||||
@Component
|
||||
public class HumanEvaluationBlockFactory
|
||||
implements BlockFactory<HumanEvaluationBlockType, HumanEvaluationBlockConfiguration> {
|
||||
|
||||
/** The overall judgement, and the fields the interaction fills besides the criteria. */
|
||||
public static final String VERDICT_FIELD = "verdict";
|
||||
public static final String NOTES_FIELD = "notes";
|
||||
public static final String EVIDENCE_FIELD = "evidence";
|
||||
public static final String BLIND_FIELD = "blind";
|
||||
|
||||
private static final List<IOCapability> INPUT_CAPABILITIES = List.of(
|
||||
new IOCapability(IOCapabilityType.ANY, false),
|
||||
new IOCapability(IOCapabilityType.ANY, true));
|
||||
private static final List<IOCapability> ANY_CAPABILITY = List.of(new IOCapability(IOCapabilityType.ANY, false));
|
||||
private static final List<IOCapability> FILE_CAPABILITY = List.of(
|
||||
new IOCapability(IOCapabilityType.FILE, true));
|
||||
|
||||
@Autowired
|
||||
HumanEvaluationBlockType blockType;
|
||||
|
||||
@Override
|
||||
public Block<HumanEvaluationBlockType> create(HumanEvaluationBlockConfiguration configuration) {
|
||||
List<IODescriptor> inputs = new ArrayList<>();
|
||||
Set<TemplateInputs.Placeholder> placeholders = TemplateInputs.extractNames(configuration.getTarget());
|
||||
placeholders.forEach(placeholder -> inputs.add(
|
||||
IODescriptor.input(placeholder.name(), IOType.ANY, placeholder.multiple(), INPUT_CAPABILITIES)));
|
||||
|
||||
Block.BlockBuilder<HumanEvaluationBlockType> builder = Block.<HumanEvaluationBlockType>builder()
|
||||
.inputs(inputs)
|
||||
.specificConfiguration(configuration)
|
||||
.type(blockType);
|
||||
|
||||
// Only the criteria marked as routing get a port of their own: a dozen criteria would
|
||||
// otherwise leave the node bristling with ports nobody wires. Every criterion is still in
|
||||
// the verdict payload.
|
||||
configuration.routingCriteria().forEach(criterion -> builder.output(
|
||||
IODescriptor.output(criterion.name(), IOType.TEXT, false, ANY_CAPABILITY)));
|
||||
|
||||
builder.output(IODescriptor.output(VERDICT_FIELD, IOType.JSON, false, ANY_CAPABILITY));
|
||||
builder.output(IODescriptor.output(NOTES_FIELD, IOType.TEXT, false, ANY_CAPABILITY));
|
||||
builder.output(IODescriptor.output(EVIDENCE_FIELD, IOType.FILE, true, FILE_CAPABILITY));
|
||||
// Whether the reference verdict was still hidden when the person committed to theirs. An
|
||||
// output rather than only an event, so a flow can keep the anchored and unanchored runs apart.
|
||||
builder.output(IODescriptor.output(BLIND_FIELD, IOType.BOOLEAN, false, ANY_CAPABILITY));
|
||||
return builder.build();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Block<HumanEvaluationBlockType> createEmpty() {
|
||||
return create(HumanEvaluationBlockConfiguration.empty());
|
||||
}
|
||||
|
||||
@Override
|
||||
public Class<HumanEvaluationBlockType> getBlockType() {
|
||||
return HumanEvaluationBlockType.class;
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<IOCapability> supportedInputCapabilities() {
|
||||
return INPUT_CAPABILITIES;
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<IOCapability> supportedOutputCapabilities() {
|
||||
return ANY_CAPABILITY;
|
||||
}
|
||||
}
|
||||
|
|
@ -105,4 +105,19 @@ public final class TemplateInputs {
|
|||
}
|
||||
return multiplicitiesByName.values().stream().allMatch(multiplicities -> multiplicities.size() == 1);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the text declares a placeholder with exactly this name, so a configuration can check
|
||||
* that a field naming one of its own inputs names one that will actually exist.
|
||||
*
|
||||
* <p>Public where {@link #extractNames(String)} is not, because a configuration asking "is this
|
||||
* an input of mine?" needs the answer, not the {@link Placeholder} type behind it.
|
||||
*/
|
||||
public static boolean declaresPlaceholder(String text, String name) {
|
||||
if (name == null || name.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
String wanted = name.strip();
|
||||
return extractNames(text).stream().anyMatch(placeholder -> placeholder.name().equals(wanted));
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -23,6 +23,16 @@ public interface BlockType {
|
|||
return NodeTypeCapabilities.activity();
|
||||
}
|
||||
|
||||
/**
|
||||
* How a person answers this block, or null when it takes no interaction. An interactive block
|
||||
* that returns null renders with no form at all, so this belongs beside
|
||||
* {@link #isUserInteractive()} rather than in a table kept elsewhere.
|
||||
*/
|
||||
@JsonIgnore
|
||||
default InteractionContract getInteractionContract() {
|
||||
return null;
|
||||
}
|
||||
|
||||
@JsonIgnore
|
||||
Class<? extends BlockConfiguration<?>> getBlockConfigurationClass();
|
||||
|
||||
|
|
|
|||
|
|
@ -34,6 +34,11 @@ public class ChatInteractionBlockType implements BlockType {
|
|||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public InteractionContract getInteractionContract() {
|
||||
return InteractionContract.chatSession();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Class<? extends BlockConfiguration<?>> getBlockConfigurationClass() {
|
||||
return ChatInteractionBlockConfiguration.class;
|
||||
|
|
|
|||
|
|
@ -40,6 +40,11 @@ public class HumanDecisionBlockType implements BlockType {
|
|||
return NodeTypeCapabilities.decision();
|
||||
}
|
||||
|
||||
@Override
|
||||
public InteractionContract getInteractionContract() {
|
||||
return InteractionContract.humanDecision();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Class<? extends BlockConfiguration<?>> getBlockConfigurationClass() {
|
||||
return HumanDecisionBlockConfiguration.class;
|
||||
|
|
|
|||
|
|
@ -0,0 +1,58 @@
|
|||
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
|
||||
|
||||
package it.cnr.isti.workflow.manager.blocks.types;
|
||||
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.BlockConfiguration;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.HumanEvaluationBlockConfiguration;
|
||||
import it.cnr.isti.workflow.manager.flows.model.capabilities.NodeTypeCapabilities;
|
||||
|
||||
@Component(HumanEvaluationBlockType.TYPE)
|
||||
public class HumanEvaluationBlockType implements BlockType {
|
||||
|
||||
public static final String TYPE = "HumanEvaluationBlock";
|
||||
|
||||
@Override
|
||||
public String getName() {
|
||||
return TYPE;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String getDescription() {
|
||||
return "A person exercises something - typically running software - and scores it against named criteria. "
|
||||
+ "Use it when the flow needs a judgement it can act on: the answer is a verdict per criterion plus "
|
||||
+ "optional evidence, not free text. The node points at what to evaluate, it does not host it.";
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean validate() {
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isUserInteractive() {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* An activity, not a decision: a criterion may route the flow, but the node's purpose is to
|
||||
* record a judgement, and it stays useful with no routing criterion at all.
|
||||
*/
|
||||
@Override
|
||||
public NodeTypeCapabilities getCapabilities() {
|
||||
return NodeTypeCapabilities.activity();
|
||||
}
|
||||
|
||||
@Override
|
||||
public InteractionContract getInteractionContract() {
|
||||
return InteractionContract.evaluationForm();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Class<? extends BlockConfiguration<?>> getBlockConfigurationClass() {
|
||||
return HumanEvaluationBlockConfiguration.class;
|
||||
}
|
||||
}
|
||||
|
|
@ -30,6 +30,11 @@ public class HumanInteractionBlockType implements BlockType {
|
|||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public InteractionContract getInteractionContract() {
|
||||
return InteractionContract.singleResponse();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Class<? extends BlockConfiguration<?>> getBlockConfigurationClass() {
|
||||
return HumanInteractiveBlockConfiguration.class;
|
||||
|
|
|
|||
|
|
@ -0,0 +1,51 @@
|
|||
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
|
||||
|
||||
package it.cnr.isti.workflow.manager.blocks.types;
|
||||
|
||||
/**
|
||||
* How a person answers this block: which shape the editor should render, and the interaction fields
|
||||
* behind it.
|
||||
*
|
||||
* <p>Declared by the block type rather than looked up by name at the edges. The same table used to
|
||||
* be written out twice - once for the editor, once for the assistant catalogue - and a new
|
||||
* interactive block only worked once someone remembered to extend both.
|
||||
*
|
||||
* @param kind the shape to render: {@code chat-session}, {@code single-response},
|
||||
* {@code human-decision}, {@code evaluation-form}
|
||||
* @param messageField field carrying what the person writes along the way, when there is one
|
||||
* @param completionField field whose arrival completes the step
|
||||
* @param historyField field holding the exchange so far, for the shapes that have one
|
||||
* @param responseField field the answer is read back from
|
||||
* @param supportsPartialResult whether an answer may arrive in more than one piece
|
||||
*/
|
||||
public record InteractionContract(
|
||||
String kind,
|
||||
String messageField,
|
||||
String completionField,
|
||||
String historyField,
|
||||
String responseField,
|
||||
boolean supportsPartialResult) {
|
||||
|
||||
public static InteractionContract chatSession() {
|
||||
return new InteractionContract("chat-session", "message", "response", "history", "response", true);
|
||||
}
|
||||
|
||||
public static InteractionContract singleResponse() {
|
||||
return new InteractionContract("single-response", null, "output", null, "output", false);
|
||||
}
|
||||
|
||||
public static InteractionContract humanDecision() {
|
||||
return new InteractionContract("human-decision", "rationale", "choice", null, "choice", true);
|
||||
}
|
||||
|
||||
/**
|
||||
* A form of named criteria. It completes on {@code verdict} because that is the field the whole
|
||||
* judgement is submitted under, and it accepts partial results: revealing the reference verdict
|
||||
* arrives before the judgement does.
|
||||
*/
|
||||
public static InteractionContract evaluationForm() {
|
||||
return new InteractionContract("evaluation-form", "notes", "verdict", null, "verdict", true);
|
||||
}
|
||||
}
|
||||
|
|
@ -34,6 +34,11 @@ public class MCPAgentChatBlockType implements BlockType {
|
|||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public InteractionContract getInteractionContract() {
|
||||
return InteractionContract.chatSession();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Class<? extends BlockConfiguration<?>> getBlockConfigurationClass() {
|
||||
return MCPAgentChatBlockConfiguration.class;
|
||||
|
|
|
|||
|
|
@ -186,34 +186,17 @@ public class BlocksController {
|
|||
}
|
||||
|
||||
private InteractionContractDescriptor resolveInteractionContract(BlockType blockType) {
|
||||
if ("ChatInteraction".equals(blockType.getName()) || "MCPAgentChat".equals(blockType.getName())) {
|
||||
return new InteractionContractDescriptor(
|
||||
"chat-session",
|
||||
"message",
|
||||
"response",
|
||||
"history",
|
||||
"response",
|
||||
true);
|
||||
it.cnr.isti.workflow.manager.blocks.types.InteractionContract contract = blockType.getInteractionContract();
|
||||
if (contract == null) {
|
||||
return null;
|
||||
}
|
||||
if ("HumanInteractionBlock".equals(blockType.getName())) {
|
||||
return new InteractionContractDescriptor(
|
||||
"single-response",
|
||||
null,
|
||||
"output",
|
||||
null,
|
||||
"output",
|
||||
false);
|
||||
}
|
||||
if ("HumanDecisionBlock".equals(blockType.getName())) {
|
||||
return new InteractionContractDescriptor(
|
||||
"human-decision",
|
||||
"rationale",
|
||||
"choice",
|
||||
null,
|
||||
"choice",
|
||||
true);
|
||||
}
|
||||
return null;
|
||||
return new InteractionContractDescriptor(
|
||||
contract.kind(),
|
||||
contract.messageField(),
|
||||
contract.completionField(),
|
||||
contract.historyField(),
|
||||
contract.responseField(),
|
||||
contract.supportsPartialResult());
|
||||
}
|
||||
|
||||
private BlockType resolveBlockType(String type) {
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ import java.io.IOException;
|
|||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.stream.Collectors;
|
||||
|
|
@ -39,6 +40,9 @@ import it.cnr.isti.workflow.manager.executions.ExecutionEvent;
|
|||
import it.cnr.isti.workflow.manager.executions.ExecutionKind;
|
||||
import it.cnr.isti.workflow.manager.executions.ExecutionObject;
|
||||
import it.cnr.isti.workflow.manager.executions.ExecutionAuthorizationValueRequest;
|
||||
import it.cnr.isti.workflow.manager.executions.HumanEvaluationRequest;
|
||||
import it.cnr.isti.workflow.manager.blocks.factories.HumanEvaluationBlockFactory;
|
||||
import it.cnr.isti.workflow.manager.executions.executors.blocks.HumanEvaluationExecutor;
|
||||
import it.cnr.isti.workflow.manager.executions.ExecutionSimulationRequest;
|
||||
import it.cnr.isti.workflow.manager.executions.ExecutionsService;
|
||||
import it.cnr.isti.workflow.manager.executions.api.ExecutionContextView;
|
||||
|
|
@ -410,6 +414,56 @@ public class ExecutionsController {
|
|||
executionService.setInteractionValue(visibleExecution(executionId, userDetails).getId(), nodeId, fieldName, value));
|
||||
}
|
||||
|
||||
@PutMapping(path = "{executionId}/node/{nodeId}/evaluation", consumes = "application/json")
|
||||
@Operation(summary = "Submits a human evaluation",
|
||||
description = "Records one person's judgement of a HumanEvaluation node, whole: every criterion and the "
|
||||
+ "notes in one act. Refused if a criterion is unknown to the node or its verdict is outside the "
|
||||
+ "scale the criterion declares, in which case nothing is recorded.")
|
||||
public ExecutionView submitEvaluation(@PathVariable String executionId, @PathVariable String nodeId,
|
||||
@RequestBody @jakarta.validation.Valid HumanEvaluationRequest request,
|
||||
@AuthenticationPrincipal LoginEntity userDetails) {
|
||||
Map<String, Object> interaction = new LinkedHashMap<>(request.verdict());
|
||||
if (request.notes() != null) {
|
||||
interaction.put(HumanEvaluationBlockFactory.NOTES_FIELD, request.notes());
|
||||
}
|
||||
return ExecutionView.fromExecution(executionService.setInteractionValues(
|
||||
visibleExecution(executionId, userDetails).getId(), nodeId, interaction));
|
||||
}
|
||||
|
||||
@PutMapping(path = "{executionId}/node/{nodeId}/evaluation/evidence", consumes = "multipart/form-data")
|
||||
@Operation(summary = "Attaches evidence to a human evaluation",
|
||||
description = "Adds files - screenshots, recordings, exports - to a waiting HumanEvaluation step. They "
|
||||
+ "accumulate across calls and leave on the node's evidence output once the judgement is submitted.")
|
||||
public ExecutionView attachEvaluationEvidence(@PathVariable String executionId, @PathVariable String nodeId,
|
||||
@RequestParam List<MultipartFile> files, @AuthenticationPrincipal LoginEntity userDetails) {
|
||||
List<File> stored = new ArrayList<>();
|
||||
try {
|
||||
for (MultipartFile file : files) {
|
||||
File target = createUploadTempFile(HumanEvaluationBlockFactory.EVIDENCE_FIELD,
|
||||
file.getOriginalFilename());
|
||||
file.transferTo(target);
|
||||
stored.add(target);
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new WebServerException("Error while creating file", e);
|
||||
}
|
||||
return ExecutionView.fromExecution(executionService.setInteractionValues(
|
||||
visibleExecution(executionId, userDetails).getId(), nodeId,
|
||||
Map.of(HumanEvaluationBlockFactory.EVIDENCE_FIELD, stored)));
|
||||
}
|
||||
|
||||
@PutMapping(path = "{executionId}/node/{nodeId}/evaluation/reveal")
|
||||
@Operation(summary = "Reveals the reference verdict",
|
||||
description = "Marks that the person evaluating asked to see the reference verdict before judging, on a "
|
||||
+ "node configured to keep it hidden. The execution records the reveal, so a judgement made after "
|
||||
+ "it can be told from one made blind - which is the whole point of hiding it.")
|
||||
public ExecutionView revealEvaluationReference(@PathVariable String executionId, @PathVariable String nodeId,
|
||||
@AuthenticationPrincipal LoginEntity userDetails) {
|
||||
return ExecutionView.fromExecution(executionService.setInteractionValues(
|
||||
visibleExecution(executionId, userDetails).getId(), nodeId,
|
||||
Map.of(HumanEvaluationExecutor.REFERENCE_REVEALED_FIELD, Boolean.TRUE)));
|
||||
}
|
||||
|
||||
@PutMapping(path = "{executionId}/authorizations")
|
||||
@Operation(summary = "Provides execution authorization",
|
||||
description = "Stores the selected saved credential reference for a provider authorization required by an "
|
||||
|
|
|
|||
|
|
@ -390,13 +390,23 @@ public class ExecutionContext implements ExecutionListener {
|
|||
}
|
||||
|
||||
protected void setInteractionValue(String stepId, String fieldName, Object value) {
|
||||
setInteractionValues(stepId, Map.of(fieldName, value));
|
||||
}
|
||||
|
||||
/**
|
||||
* Hands a whole interaction to the step at once.
|
||||
*
|
||||
* <p>A judgement made of several fields has to arrive together: submitted one field at a time,
|
||||
* a failure halfway leaves the step holding half an answer, and no caller can tell which half.
|
||||
*/
|
||||
protected void setInteractionValues(String stepId, Map<String, Object> values) {
|
||||
Step<?> step = this.steps.get(stepId);
|
||||
if (step == null)
|
||||
throw new IllegalArgumentException("Step with id " + stepId + " not found");
|
||||
if (step.getStatus() != StepStatus.WAITING_FOR_INTERACTION)
|
||||
throw new IllegalStateException("Step with id " + stepId
|
||||
+ " is not in WAITING_FOR_INTERACTION status (CURRENT STATUS is " + step.getStatus() + ")");
|
||||
step.interact(Map.of(fieldName, value));
|
||||
step.interact(values);
|
||||
}
|
||||
|
||||
protected void setContainerContinuation(String stepId, ContainerContinuationSnapshot continuation) {
|
||||
|
|
|
|||
|
|
@ -21,6 +21,12 @@ public enum ExecutionEventType {
|
|||
STEP_RESUMED,
|
||||
HUMAN_DECISION_REQUESTED,
|
||||
HUMAN_DECISION_RECORDED,
|
||||
HUMAN_EVALUATION_RECORDED,
|
||||
/**
|
||||
* The reference verdict was shown to the person evaluating. Logged before the judgement to keep
|
||||
* the order recoverable: an evaluation made after this event saw the other verdict first.
|
||||
*/
|
||||
HUMAN_EVALUATION_REFERENCE_REVEALED,
|
||||
FLOW_PATH_ENDED,
|
||||
FLOW_OUTCOME_RECORDED,
|
||||
CONTAINER_SUBFLOW_STARTED,
|
||||
|
|
|
|||
|
|
@ -188,8 +188,12 @@ public class ExecutionObject {
|
|||
}
|
||||
|
||||
protected void setInteractionValue(String blockId, String fieldName, Object value) {
|
||||
setInteractionValues(blockId, Map.of(fieldName, value));
|
||||
}
|
||||
|
||||
protected void setInteractionValues(String blockId, Map<String, Object> values) {
|
||||
if (this.context.getStatus() == ExecutionStatus.WAITING){
|
||||
this.context.setInteractionValue(blockId, fieldName, value);
|
||||
this.context.setInteractionValues(blockId, values);
|
||||
} else
|
||||
throw new IllegalStateException("Execution with id " + this.getId()
|
||||
+ " is not in WAITING status (CURRENT STATUS is " + this.getContext().getStatus() + ")");
|
||||
|
|
|
|||
|
|
@ -708,6 +708,13 @@ public class ExecutionsService {
|
|||
return eo;
|
||||
}
|
||||
|
||||
/** Submits a multi-field interaction as one act, so a step never holds half an answer. */
|
||||
public ExecutionObject setInteractionValues(String executionId, String blockId, Map<String, Object> values) {
|
||||
ExecutionObject eo = getExecution(executionId);
|
||||
eo.setInteractionValues(blockId, values);
|
||||
return eo;
|
||||
}
|
||||
|
||||
public ExecutionObject setContainerContinuation(String executionId, String stepId,
|
||||
ContainerContinuationSnapshot continuation) {
|
||||
ExecutionObject execution = getExecution(executionId);
|
||||
|
|
|
|||
|
|
@ -0,0 +1,30 @@
|
|||
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
|
||||
|
||||
package it.cnr.isti.workflow.manager.executions;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
import jakarta.validation.constraints.NotEmpty;
|
||||
import jakarta.validation.constraints.Size;
|
||||
|
||||
/**
|
||||
* One person's judgement, submitted whole.
|
||||
*
|
||||
* <p>Whole, because the fields are one answer: sent a criterion at a time, a failure partway
|
||||
* through leaves the step holding a judgement nobody made, with no way to tell which half is
|
||||
* missing. The criterion names and their allowed values are the node's own, and the executor
|
||||
* refuses anything else.
|
||||
*
|
||||
* @param verdict criterion name to the verdict given for it
|
||||
* @param notes what the tester wants to say beyond the scores, optional
|
||||
*/
|
||||
public record HumanEvaluationRequest(
|
||||
@NotEmpty Map<String, @Size(max = 64) String> verdict,
|
||||
@Size(max = 4000) String notes) {
|
||||
|
||||
public HumanEvaluationRequest {
|
||||
verdict = verdict == null ? Map.of() : Map.copyOf(verdict);
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,311 @@
|
|||
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
|
||||
|
||||
package it.cnr.isti.workflow.manager.executions.executors.blocks;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.stereotype.Component;
|
||||
import org.springframework.util.StringUtils;
|
||||
|
||||
import it.cnr.isti.workflow.manager.blocks.Block;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.EvaluationCriterion;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.HumanEvaluationBlockConfiguration;
|
||||
import it.cnr.isti.workflow.manager.blocks.factories.HumanEvaluationBlockFactory;
|
||||
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
|
||||
import it.cnr.isti.workflow.manager.executions.ExecutionEventLogger;
|
||||
import it.cnr.isti.workflow.manager.executions.ExecutionEventType;
|
||||
import it.cnr.isti.workflow.manager.executions.ExecutionVariableDescriptor;
|
||||
import it.cnr.isti.workflow.manager.executions.InteractionResult;
|
||||
import it.cnr.isti.workflow.manager.executions.NodeExecutionException;
|
||||
import it.cnr.isti.workflow.manager.executions.bias.runtime.BiasRuntimeSupport;
|
||||
import it.cnr.isti.workflow.manager.executions.steps.Input;
|
||||
import it.cnr.isti.workflow.manager.llms.LLMCredentialResolver;
|
||||
import it.cnr.isti.workflow.manager.llms.LLMDescriptor;
|
||||
import it.cnr.isti.workflow.manager.llms.ProviderCredential;
|
||||
import it.cnr.isti.workflow.manager.llms.providers.LLMProvider;
|
||||
|
||||
/**
|
||||
* Turns one person's judgement of a piece of software into outputs the flow can act on.
|
||||
*
|
||||
* <p>Like every human block, {@link #execute} refuses: the engine never runs this node, a person
|
||||
* does, through {@link #interact}. {@link #simulate} exists so bias experiments, which run with no
|
||||
* humans available, still have something to run.
|
||||
*/
|
||||
@Component
|
||||
public class HumanEvaluationExecutor implements BlockExecutor<HumanEvaluationBlockType> {
|
||||
|
||||
/** Marks, in the step's partial results, that the reference verdict was already shown. */
|
||||
public static final String REFERENCE_REVEALED_FIELD = "__referenceRevealed";
|
||||
|
||||
@Autowired
|
||||
private Map<String, LLMProvider> llmProviders;
|
||||
|
||||
@Autowired
|
||||
private LLMCredentialResolver credentialResolver;
|
||||
|
||||
@Override
|
||||
public Map<String, Object> execute(Block<HumanEvaluationBlockType> block, List<Input> inputs,
|
||||
Map<String, Object> authorizations, Map<String, Object> executionVariables,
|
||||
Map<String, ExecutionVariableDescriptor> executionVariableDescriptors, ExecutionEventLogger eventLogger) {
|
||||
throw new UnsupportedOperationException("HumanEvaluation blocks require a person to evaluate the target");
|
||||
}
|
||||
|
||||
@Override
|
||||
public InteractionResult interact(Block<HumanEvaluationBlockType> block, List<Input> inputs,
|
||||
Map<String, Object> interaction, Map<String, Object> partialResults, Map<String, Object> authorizations,
|
||||
Map<String, Object> executionVariables,
|
||||
Map<String, ExecutionVariableDescriptor> executionVariableDescriptors, ExecutionEventLogger eventLogger) {
|
||||
HumanEvaluationBlockConfiguration configuration =
|
||||
(HumanEvaluationBlockConfiguration) block.getSpecificConfiguration();
|
||||
|
||||
Map<String, Object> collected = new LinkedHashMap<>(partialResults == null ? Map.of() : partialResults);
|
||||
if (interaction != null) {
|
||||
interaction.forEach((field, value) -> {
|
||||
if (!isKnownField(configuration, field)) {
|
||||
throw new IllegalArgumentException("Unsupported HumanEvaluation interaction field: " + field);
|
||||
}
|
||||
if (HumanEvaluationBlockFactory.EVIDENCE_FIELD.equals(field)) {
|
||||
// Evidence arrives upload by upload, so a second screenshot must not replace the first.
|
||||
List<Object> merged = new ArrayList<>(asList(collected.get(field)));
|
||||
merged.addAll(asList(value));
|
||||
collected.put(field, merged);
|
||||
} else {
|
||||
collected.put(field, value);
|
||||
}
|
||||
});
|
||||
|
||||
boolean revealedNow = Boolean.TRUE.equals(interaction.get(REFERENCE_REVEALED_FIELD));
|
||||
if (revealedNow && eventLogger != null) {
|
||||
// Logged as it happens rather than with the verdict: the order of the two events is
|
||||
// what says whether the judgement that follows was anchored to the reference.
|
||||
eventLogger.info(ExecutionEventType.HUMAN_EVALUATION_REFERENCE_REVEALED,
|
||||
"Reference verdict revealed before the evaluation was submitted",
|
||||
Map.of("referenceInput", configuration.getReferenceInput() == null
|
||||
? "" : configuration.getReferenceInput()));
|
||||
}
|
||||
}
|
||||
|
||||
// The judgement is one answer: until every criterion has a verdict, the step keeps waiting.
|
||||
// A reveal or an evidence upload arrives on its own and lands here.
|
||||
if (!missingCriteria(configuration, collected).isEmpty()) {
|
||||
return InteractionResult.partial(collected);
|
||||
}
|
||||
|
||||
for (EvaluationCriterion criterion : configuration.getCriteria()) {
|
||||
String value = asText(collected.get(criterion.name()));
|
||||
if (!criterion.scale().accepts(value)) {
|
||||
throw new HumanEvaluationException("HUMAN_EVALUATION_INVALID_VERDICT",
|
||||
"Criterion '" + criterion.name() + "' expects " + criterion.scale().allowedValuesDescription()
|
||||
+ ", got: " + value);
|
||||
}
|
||||
}
|
||||
|
||||
List<Object> evidence = asList(collected.get(HumanEvaluationBlockFactory.EVIDENCE_FIELD));
|
||||
if (configuration.isEvidenceRequired() && evidence.isEmpty()) {
|
||||
throw new HumanEvaluationException("HUMAN_EVALUATION_EVIDENCE_REQUIRED",
|
||||
"This evaluation requires at least one piece of evidence");
|
||||
}
|
||||
|
||||
boolean blind = configuration.isBlindUntilSubmitted()
|
||||
&& !Boolean.TRUE.equals(collected.get(REFERENCE_REVEALED_FIELD));
|
||||
|
||||
Map<String, Object> verdict = new LinkedHashMap<>();
|
||||
Map<String, Object> outputs = new LinkedHashMap<>();
|
||||
for (EvaluationCriterion criterion : configuration.getCriteria()) {
|
||||
String value = asText(collected.get(criterion.name())).strip();
|
||||
verdict.put(criterion.name(), value);
|
||||
if (criterion.routing()) {
|
||||
outputs.put(criterion.name(), value);
|
||||
}
|
||||
}
|
||||
|
||||
String notes = asText(collected.get(HumanEvaluationBlockFactory.NOTES_FIELD));
|
||||
outputs.put(HumanEvaluationBlockFactory.VERDICT_FIELD, verdict);
|
||||
outputs.put(HumanEvaluationBlockFactory.NOTES_FIELD, notes);
|
||||
outputs.put(HumanEvaluationBlockFactory.EVIDENCE_FIELD, evidence);
|
||||
outputs.put(HumanEvaluationBlockFactory.BLIND_FIELD, blind);
|
||||
|
||||
if (eventLogger != null) {
|
||||
// The verdicts themselves are the point of the record; the notes are not, and may quote
|
||||
// whatever the tester saw on screen.
|
||||
eventLogger.info(ExecutionEventType.HUMAN_EVALUATION_RECORDED,
|
||||
"Recorded human evaluation",
|
||||
Map.of("verdict", verdict,
|
||||
"blind", blind,
|
||||
"evidenceCount", evidence.size(),
|
||||
"notesPresent", StringUtils.hasText(notes)));
|
||||
}
|
||||
return InteractionResult.completed(outputs);
|
||||
}
|
||||
|
||||
@Override
|
||||
public Map<String, Object> simulate(Block<HumanEvaluationBlockType> block, List<Input> inputs,
|
||||
Map<String, Object> authorizations, Map<String, Object> executionVariables,
|
||||
Map<String, ExecutionVariableDescriptor> executionVariableDescriptors,
|
||||
LLMDescriptor simulatorDescriptor, ExecutionEventLogger eventLogger) {
|
||||
HumanEvaluationBlockConfiguration configuration =
|
||||
(HumanEvaluationBlockConfiguration) block.getSpecificConfiguration();
|
||||
if (simulatorDescriptor == null) {
|
||||
throw new IllegalArgumentException("Missing simulation descriptor for HumanEvaluation execution");
|
||||
}
|
||||
LLMProvider llmProvider = resolveProvider(simulatorDescriptor);
|
||||
ProviderCredential credential = credentialResolver.resolve(llmProvider, authorizations, executionVariables);
|
||||
|
||||
String prompt = BiasRuntimeSupport.decoratePrompt(simulationPrompt(configuration, inputs), executionVariables);
|
||||
ModelParameterReporting.reportUnsupported(llmProvider, simulatorDescriptor.parameters(),
|
||||
simulatorDescriptor.model(), eventLogger);
|
||||
String response = llmProvider.generate(simulatorDescriptor.model(), prompt, credential,
|
||||
simulatorDescriptor.parameters());
|
||||
if (!StringUtils.hasText(response)) {
|
||||
throw new IllegalArgumentException("Simulated HumanEvaluation produced an empty response");
|
||||
}
|
||||
|
||||
Map<String, String> answered = parseSimulatedVerdict(response);
|
||||
Map<String, Object> verdict = new LinkedHashMap<>();
|
||||
Map<String, Object> outputs = new LinkedHashMap<>();
|
||||
for (EvaluationCriterion criterion : configuration.getCriteria()) {
|
||||
String value = answered.get(criterion.name().toLowerCase());
|
||||
if (!criterion.scale().accepts(value)) {
|
||||
throw new HumanEvaluationException("HUMAN_EVALUATION_SIMULATION_FAILED",
|
||||
"Simulator did not return a usable verdict for criterion '" + criterion.name() + "'");
|
||||
}
|
||||
String normalized = value.strip();
|
||||
verdict.put(criterion.name(), normalized);
|
||||
if (criterion.routing()) {
|
||||
outputs.put(criterion.name(), normalized);
|
||||
}
|
||||
}
|
||||
|
||||
outputs.put(HumanEvaluationBlockFactory.VERDICT_FIELD, verdict);
|
||||
outputs.put(HumanEvaluationBlockFactory.NOTES_FIELD, answered.getOrDefault("notes", ""));
|
||||
outputs.put(HumanEvaluationBlockFactory.EVIDENCE_FIELD, List.of());
|
||||
// A simulator is never anchored to a reference it was not shown, and this node's whole
|
||||
// reason for recording the flag is to tell anchored judgements from unanchored ones.
|
||||
outputs.put(HumanEvaluationBlockFactory.BLIND_FIELD, configuration.isBlindUntilSubmitted());
|
||||
return outputs;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Class<HumanEvaluationBlockType> getBlockType() {
|
||||
return HumanEvaluationBlockType.class;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isInteractive() {
|
||||
return true;
|
||||
}
|
||||
|
||||
private LLMProvider resolveProvider(LLMDescriptor descriptor) {
|
||||
LLMProvider provider = llmProviders.get(descriptor.provider());
|
||||
if (provider == null) {
|
||||
provider = llmProviders.values().stream()
|
||||
.filter(candidate -> candidate.getName().equals(descriptor.provider()))
|
||||
.findFirst()
|
||||
.orElse(null);
|
||||
}
|
||||
if (provider == null) {
|
||||
throw new IllegalArgumentException("Provider not found: " + descriptor.provider());
|
||||
}
|
||||
return provider;
|
||||
}
|
||||
|
||||
private String simulationPrompt(HumanEvaluationBlockConfiguration configuration, List<Input> inputs) {
|
||||
String context = inputs.stream()
|
||||
.map(input -> "%s= %s".formatted(input.getDescriptor().getName(), input.getValue()))
|
||||
.collect(Collectors.joining(", "));
|
||||
String steps = configuration.getTaskScript().isEmpty()
|
||||
? "(no script given: judge on the description alone)"
|
||||
: configuration.getTaskScript().stream()
|
||||
.map(step -> "- " + step)
|
||||
.collect(Collectors.joining("\n"));
|
||||
String criteria = configuration.getCriteria().stream()
|
||||
.map(criterion -> "- %s (%s): %s".formatted(criterion.name(),
|
||||
criterion.scale().allowedValuesDescription(),
|
||||
StringUtils.hasText(criterion.description()) ? criterion.description() : criterion.name()))
|
||||
.collect(Collectors.joining("\n"));
|
||||
String answerLines = configuration.getCriteria().stream()
|
||||
.map(criterion -> "%s: <%s>".formatted(criterion.name(), criterion.scale().allowedValuesDescription()))
|
||||
.collect(Collectors.joining("\n"));
|
||||
|
||||
return """
|
||||
You are standing in for a human tester. Given the context { %s }, evaluate this target:
|
||||
"%s"
|
||||
|
||||
Steps performed:
|
||||
%s
|
||||
|
||||
Score every criterion:
|
||||
%s
|
||||
|
||||
Answer in exactly this format, one line per criterion, nothing else before them:
|
||||
%s
|
||||
NOTES: <one short line, or empty>
|
||||
""".formatted(context, configuration.getTarget(), steps, criteria, answerLines);
|
||||
}
|
||||
|
||||
/** Reads back the {@code name: value} lines the prompt asks for, case-insensitively. */
|
||||
private Map<String, String> parseSimulatedVerdict(String response) {
|
||||
Map<String, String> answered = new LinkedHashMap<>();
|
||||
for (String line : response.lines().toList()) {
|
||||
String trimmed = line.strip();
|
||||
int separator = trimmed.indexOf(':');
|
||||
if (separator <= 0) {
|
||||
continue;
|
||||
}
|
||||
String key = trimmed.substring(0, separator).strip().toLowerCase();
|
||||
String value = trimmed.substring(separator + 1).strip();
|
||||
if (!key.isEmpty() && !answered.containsKey(key)) {
|
||||
answered.put(key, value);
|
||||
}
|
||||
}
|
||||
return answered;
|
||||
}
|
||||
|
||||
private boolean isKnownField(HumanEvaluationBlockConfiguration configuration, String field) {
|
||||
if (REFERENCE_REVEALED_FIELD.equals(field)
|
||||
|| HumanEvaluationBlockFactory.NOTES_FIELD.equals(field)
|
||||
|| HumanEvaluationBlockFactory.EVIDENCE_FIELD.equals(field)) {
|
||||
return true;
|
||||
}
|
||||
return configuration.getCriteria().stream()
|
||||
.anyMatch(criterion -> criterion.name().equals(field));
|
||||
}
|
||||
|
||||
private List<String> missingCriteria(HumanEvaluationBlockConfiguration configuration,
|
||||
Map<String, Object> collected) {
|
||||
List<String> missing = new ArrayList<>();
|
||||
for (EvaluationCriterion criterion : configuration.getCriteria()) {
|
||||
if (!StringUtils.hasText(asText(collected.get(criterion.name())))) {
|
||||
missing.add(criterion.name());
|
||||
}
|
||||
}
|
||||
return missing;
|
||||
}
|
||||
|
||||
private static String asText(Object value) {
|
||||
return value == null ? "" : value.toString();
|
||||
}
|
||||
|
||||
private static List<Object> asList(Object value) {
|
||||
if (value == null) {
|
||||
return List.of();
|
||||
}
|
||||
if (value instanceof List<?> list) {
|
||||
return List.copyOf(list);
|
||||
}
|
||||
return List.of(value);
|
||||
}
|
||||
|
||||
private static final class HumanEvaluationException extends NodeExecutionException {
|
||||
private HumanEvaluationException(String errorCode, String message) {
|
||||
super(errorCode, message);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,88 @@
|
|||
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
|
||||
|
||||
package it.cnr.isti.workflow.manager.blocks.factories;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import it.cnr.isti.workflow.manager.blocks.Block;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.EvaluationCriterion;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.EvaluationScale;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.HumanEvaluationBlockConfiguration;
|
||||
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
|
||||
import it.cnr.isti.workflow.manager.ios.IODescriptor;
|
||||
import it.cnr.isti.workflow.manager.ios.IOType;
|
||||
|
||||
class HumanEvaluationBlockFactoryTest {
|
||||
|
||||
private HumanEvaluationBlockFactory factory;
|
||||
|
||||
@BeforeEach
|
||||
void setUp() {
|
||||
factory = new HumanEvaluationBlockFactory();
|
||||
factory.blockType = new HumanEvaluationBlockType();
|
||||
}
|
||||
|
||||
private List<String> outputNames(Block<HumanEvaluationBlockType> block) {
|
||||
return block.getOutputs().stream().map(IODescriptor::getName).toList();
|
||||
}
|
||||
|
||||
@Test
|
||||
void givesAPortOnlyToTheCriteriaThatRoute() {
|
||||
// Twelve criteria would otherwise leave the node bristling with ports nobody wires; the
|
||||
// ones that do not route are still in the verdict payload.
|
||||
Block<HumanEvaluationBlockType> block = factory.create(HumanEvaluationBlockConfiguration.builder()
|
||||
.name("evaluation")
|
||||
.target("Open ${{deployUrl}}")
|
||||
.criteria(List.of(
|
||||
new EvaluationCriterion("works", null, EvaluationScale.PASS_FAIL, true),
|
||||
new EvaluationCriterion("clarity", null, EvaluationScale.SCORE_1_5, false)))
|
||||
.build());
|
||||
|
||||
assertTrue(outputNames(block).contains("works"));
|
||||
assertFalse(outputNames(block).contains("clarity"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void alwaysCarriesTheVerdictNotesEvidenceAndBlindOutputs() {
|
||||
Block<HumanEvaluationBlockType> block = factory.createEmpty();
|
||||
|
||||
assertTrue(outputNames(block).containsAll(List.of(
|
||||
HumanEvaluationBlockFactory.VERDICT_FIELD,
|
||||
HumanEvaluationBlockFactory.NOTES_FIELD,
|
||||
HumanEvaluationBlockFactory.EVIDENCE_FIELD,
|
||||
HumanEvaluationBlockFactory.BLIND_FIELD)));
|
||||
}
|
||||
|
||||
@Test
|
||||
void evidenceLeavesAsSeveralFiles() {
|
||||
// One screenshot is the common case, but a tester who took four must not lose three.
|
||||
IODescriptor evidence = factory.createEmpty().getOutputs().stream()
|
||||
.filter(output -> HumanEvaluationBlockFactory.EVIDENCE_FIELD.equals(output.getName()))
|
||||
.findFirst()
|
||||
.orElseThrow();
|
||||
|
||||
assertEquals(IOType.FILE, evidence.getType());
|
||||
assertTrue(evidence.isMultiple());
|
||||
}
|
||||
|
||||
@Test
|
||||
void turnsEveryTargetPlaceholderIntoAnInput() {
|
||||
Block<HumanEvaluationBlockType> block = factory.create(HumanEvaluationBlockConfiguration.builder()
|
||||
.name("evaluation")
|
||||
.target("Open ${{deployUrl}} and compare with ${{agentReport}}")
|
||||
.criteria(List.of(new EvaluationCriterion("works")))
|
||||
.build());
|
||||
|
||||
List<String> inputs = block.getInputs().stream().map(IODescriptor::getName).toList();
|
||||
assertTrue(inputs.containsAll(List.of("deployUrl", "agentReport")));
|
||||
}
|
||||
}
|
||||
|
|
@ -28,6 +28,7 @@ import it.cnr.isti.workflow.manager.blocks.IOCapabilityType;
|
|||
import it.cnr.isti.workflow.manager.blocks.configurations.ChatInteractionBlockConfiguration;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.ChatInteractionInput;
|
||||
import it.cnr.isti.workflow.manager.blocks.types.HumanInteractionBlockType;
|
||||
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
|
||||
import it.cnr.isti.workflow.manager.blocks.types.ChatInteractionBlockType;
|
||||
import it.cnr.isti.workflow.manager.blocks.types.ConditionalBlockType;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.HTTPServerCallBlockConfiguration;
|
||||
|
|
@ -89,6 +90,38 @@ public class BlocksControllerTest {
|
|||
System.out.println("Type: " + type.type());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void theEvaluationNodeReachesTheEditorWithTheFormItNeeds() {
|
||||
// What the editor renders comes entirely from this descriptor: without the contract it would
|
||||
// show an interactive node with no way to answer it.
|
||||
BlockConfigurationDescriptor evaluation = blocksController.getTypes().stream()
|
||||
.filter(type -> HumanEvaluationBlockType.TYPE.equals(type.type()))
|
||||
.findFirst()
|
||||
.orElseThrow(() -> new AssertionError("HumanEvaluationBlock is not in the catalog"));
|
||||
|
||||
assertTrue(evaluation.userInteractive());
|
||||
assertNotNull(evaluation.interactionContract());
|
||||
assertEquals("evaluation-form", evaluation.interactionContract().kind());
|
||||
assertEquals("verdict", evaluation.interactionContract().completionField());
|
||||
// The judgement arrives after a reveal or an evidence upload, which are interactions of
|
||||
// their own: a contract that refused partial results would drop them.
|
||||
assertTrue(evaluation.interactionContract().supportsPartialResult());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void everyInteractiveBlockSaysHowItIsAnswered() {
|
||||
// The two readers of this used to keep their own table of block names, so a new interactive
|
||||
// block rendered as unsupported until someone remembered to extend both.
|
||||
List<BlockConfigurationDescriptor> unanswerable = blocksController.getTypes().stream()
|
||||
.filter(BlockConfigurationDescriptor::userInteractive)
|
||||
.filter(type -> type.interactionContract() == null)
|
||||
.toList();
|
||||
|
||||
assertTrue(unanswerable.isEmpty(),
|
||||
() -> "Interactive blocks with no interaction contract: "
|
||||
+ unanswerable.stream().map(BlockConfigurationDescriptor::type).toList());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void getTypeCatalogExtractsSharedBlockDefinitions() {
|
||||
BlocksController.BlockConfigurationCatalog catalog = blocksController.getConfigurationCatalog();
|
||||
|
|
|
|||
|
|
@ -0,0 +1,157 @@
|
|||
// SPDX-FileCopyrightText: 2025-2026 Lucio Lelii <lucio.lelii@isti.cnr.it> - ISTI-CNR
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
// Attribution term under AGPL-3.0 section 7(b): see LICENSE-ADDENDUM.
|
||||
|
||||
package it.cnr.isti.workflow.manager.executions.executors.blocks;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.io.File;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import it.cnr.isti.workflow.manager.blocks.Block;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.EvaluationCriterion;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.EvaluationScale;
|
||||
import it.cnr.isti.workflow.manager.blocks.configurations.HumanEvaluationBlockConfiguration;
|
||||
import it.cnr.isti.workflow.manager.blocks.factories.HumanEvaluationBlockFactory;
|
||||
import it.cnr.isti.workflow.manager.blocks.types.HumanEvaluationBlockType;
|
||||
import it.cnr.isti.workflow.manager.executions.InteractionResult;
|
||||
import it.cnr.isti.workflow.manager.executions.NodeExecutionException;
|
||||
|
||||
class HumanEvaluationExecutorTest {
|
||||
|
||||
private final HumanEvaluationExecutor executor = new HumanEvaluationExecutor();
|
||||
|
||||
private HumanEvaluationBlockConfiguration config(boolean evidenceRequired, boolean blind) {
|
||||
return HumanEvaluationBlockConfiguration.builder()
|
||||
.name("evaluation")
|
||||
.target("Open ${{deployUrl}} and compare with ${{agentReport}}")
|
||||
.taskScript(List.of("Sign in", "Create an order"))
|
||||
.criteria(List.of(
|
||||
new EvaluationCriterion("works", "Does the order go through?", EvaluationScale.PASS_FAIL, true),
|
||||
new EvaluationCriterion("clarity", "How clear are the errors?", EvaluationScale.SCORE_1_5,
|
||||
false)))
|
||||
.evidenceRequired(evidenceRequired)
|
||||
.blindUntilSubmitted(blind)
|
||||
.referenceInput(blind ? "agentReport" : null)
|
||||
.build();
|
||||
}
|
||||
|
||||
private Block<HumanEvaluationBlockType> block(HumanEvaluationBlockConfiguration configuration) {
|
||||
return Block.<HumanEvaluationBlockType>builder()
|
||||
.type(new HumanEvaluationBlockType())
|
||||
.specificConfiguration(configuration)
|
||||
.build();
|
||||
}
|
||||
|
||||
private InteractionResult interact(HumanEvaluationBlockConfiguration configuration,
|
||||
Map<String, Object> interaction, Map<String, Object> partialResults) {
|
||||
return executor.interact(block(configuration), List.of(), interaction, partialResults, Map.of(), Map.of(),
|
||||
Map.of(), null);
|
||||
}
|
||||
|
||||
@Test
|
||||
void keepsWaitingUntilEveryCriterionHasAVerdict() {
|
||||
// A judgement is one answer: half of it is not a smaller judgement, it is no judgement.
|
||||
InteractionResult result = interact(config(false, false), Map.of("works", "pass"), Map.of());
|
||||
|
||||
assertFalse(result.completed());
|
||||
assertEquals("pass", result.partialResults().get("works"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void completesWithAVerdictPerCriterionAndRoutesOnlyWhereAsked() {
|
||||
InteractionResult result = interact(config(false, false),
|
||||
Map.of("works", "pass", "clarity", "4", HumanEvaluationBlockFactory.NOTES_FIELD, "Slow but fine"),
|
||||
Map.of());
|
||||
|
||||
assertTrue(result.completed());
|
||||
assertEquals(Map.of("works", "pass", "clarity", "4"),
|
||||
result.outputs().get(HumanEvaluationBlockFactory.VERDICT_FIELD));
|
||||
// works routes, clarity does not: it is in the verdict but has no port of its own.
|
||||
assertEquals("pass", result.outputs().get("works"));
|
||||
assertFalse(result.outputs().containsKey("clarity"));
|
||||
assertEquals("Slow but fine", result.outputs().get(HumanEvaluationBlockFactory.NOTES_FIELD));
|
||||
}
|
||||
|
||||
@Test
|
||||
void refusesAVerdictOutsideTheCriterionScale() {
|
||||
// The scale is the whole reason two evaluations can be compared; "mostly" is not on it.
|
||||
NodeExecutionException failure = assertThrows(NodeExecutionException.class,
|
||||
() -> interact(config(false, false), Map.of("works", "mostly", "clarity", "4"), Map.of()));
|
||||
|
||||
assertTrue(failure.getMessage().contains("works"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void refusesAScoreOutsideTheOneToFiveRange() {
|
||||
assertThrows(NodeExecutionException.class,
|
||||
() -> interact(config(false, false), Map.of("works", "pass", "clarity", "9"), Map.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
void refusesAFieldTheNodeDoesNotHave() {
|
||||
// Silently dropping it would record a judgement the person did not give.
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> interact(config(false, false), Map.of("speed", "pass"), Map.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
void evidenceFromSeveralUploadsAccumulates() {
|
||||
Map<String, Object> afterFirst = interact(config(false, false),
|
||||
Map.of(HumanEvaluationBlockFactory.EVIDENCE_FIELD, List.of(new File("one.png"))), Map.of())
|
||||
.partialResults();
|
||||
|
||||
InteractionResult result = interact(config(false, false),
|
||||
Map.of(HumanEvaluationBlockFactory.EVIDENCE_FIELD, List.of(new File("two.png")),
|
||||
"works", "pass", "clarity", "3"),
|
||||
afterFirst);
|
||||
|
||||
assertTrue(result.completed());
|
||||
assertEquals(2, ((List<?>) result.outputs().get(HumanEvaluationBlockFactory.EVIDENCE_FIELD)).size());
|
||||
}
|
||||
|
||||
@Test
|
||||
void refusesAVerdictWithNoEvidenceWhenEvidenceIsRequired() {
|
||||
assertThrows(NodeExecutionException.class,
|
||||
() -> interact(config(true, false), Map.of("works", "pass", "clarity", "3"), Map.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
void recordsAJudgementAsBlindWhenTheReferenceWasNeverRevealed() {
|
||||
InteractionResult result = interact(config(false, true), Map.of("works", "fail", "clarity", "2"), Map.of());
|
||||
|
||||
assertEquals(true, result.outputs().get(HumanEvaluationBlockFactory.BLIND_FIELD));
|
||||
}
|
||||
|
||||
@Test
|
||||
void recordsAJudgementAsAnchoredOnceTheReferenceHasBeenRevealed() {
|
||||
// This flag is the measurement: without it, an anchored verdict and a blind one are the
|
||||
// same row in the results.
|
||||
Map<String, Object> afterReveal = interact(config(false, true),
|
||||
Map.of(HumanEvaluationExecutor.REFERENCE_REVEALED_FIELD, Boolean.TRUE), Map.of()).partialResults();
|
||||
|
||||
InteractionResult result = interact(config(false, true), Map.of("works", "fail", "clarity", "2"), afterReveal);
|
||||
|
||||
assertEquals(false, result.outputs().get(HumanEvaluationBlockFactory.BLIND_FIELD));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNodeThatHidesNothingIsNeverReportedAsBlind() {
|
||||
InteractionResult result = interact(config(false, false), Map.of("works", "pass", "clarity", "5"), Map.of());
|
||||
|
||||
assertEquals(false, result.outputs().get(HumanEvaluationBlockFactory.BLIND_FIELD));
|
||||
}
|
||||
|
||||
@Test
|
||||
void theEngineNeverRunsThisNodeOnItsOwn() {
|
||||
assertThrows(UnsupportedOperationException.class,
|
||||
() -> executor.execute(block(config(false, false)), List.of(), Map.of(), Map.of(), Map.of(), null));
|
||||
}
|
||||
}
|
||||
Loading…
Reference in New Issue