feat: add "test cv ranking" example flow with multi-CV input and ranking output

Three CVs are collected separately, ranked together by one LLM call (three
named inputs feeding a single ranking prompt), and reviewed by a human
decision maker who sees both the ranking and all three original CVs side by
side via the newly added multi-input support on HumanDecisionBlock.

Carries starter design-time bias annotations and probes on the ranking step
(SELECTION_BIAS, INPUT_TRANSFORMATION on one CV) and the review step
(AUTOMATION_BIAS, ROUTING_OVERRIDE) as a base for later bias-injection
experiments on ranking tasks specifically. Verified end to end against the
running service: full run reaches SUCCESS with a real ranking produced by
the LLM and accepted by the reviewer.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
Lucio Lelii 2026-07-24 08:43:47 +02:00
parent 9f4dc5303b
commit ba4f05ba0a
1 changed files with 351 additions and 0 deletions

View File

@ -5541,5 +5541,356 @@
],
"dependencies": []
}
},
{
"name": "test cv ranking",
"description": "Multi-CV example: three CVs are ranked together by an LLM, then reviewed by a human who sees both the ranking and the original CVs. Includes starter bias annotations and probes on the ranking and review steps for later bias-injection experiments.",
"owner": "testuser",
"createdAt": "2026-07-24T09:00:00",
"lastUpdateAt": "2026-07-24T09:00:00",
"published": false,
"finalized": false,
"flow": {
"blocks": [
{
"id": "4509b083-8987-4ef7-a90f-9d2c2b6a4c9f",
"position": {
"x": 0,
"y": 0
},
"name": "collect-cv-1",
"inputs": [
{
"name": "input",
"type": "TEXT",
"multiple": false
}
],
"outputs": [
{
"name": "output",
"type": "TEXT",
"multiple": false
}
],
"specificConfiguration": {
"type": "HumanInteractiveBlockConfiguration",
"name": "collect-cv-1",
"actionDescription": "Paste the first candidate's CV text for the backend engineering role ranking."
},
"typeName": "HumanInteractionBlock"
},
{
"id": "76b3a63f-7290-44d9-b21b-c21d67eb34cc",
"position": {
"x": 0,
"y": 160
},
"name": "collect-cv-2",
"inputs": [
{
"name": "input",
"type": "TEXT",
"multiple": false
}
],
"outputs": [
{
"name": "output",
"type": "TEXT",
"multiple": false
}
],
"specificConfiguration": {
"type": "HumanInteractiveBlockConfiguration",
"name": "collect-cv-2",
"actionDescription": "Paste the second candidate's CV text for the backend engineering role ranking."
},
"typeName": "HumanInteractionBlock"
},
{
"id": "53eff412-a5ea-4cda-b339-1f1bdcfc78c9",
"position": {
"x": 0,
"y": 320
},
"name": "collect-cv-3",
"inputs": [
{
"name": "input",
"type": "TEXT",
"multiple": false
}
],
"outputs": [
{
"name": "output",
"type": "TEXT",
"multiple": false
}
],
"specificConfiguration": {
"type": "HumanInteractiveBlockConfiguration",
"name": "collect-cv-3",
"actionDescription": "Paste the third candidate's CV text for the backend engineering role ranking."
},
"typeName": "HumanInteractionBlock"
},
{
"id": "d5ea4cb4-0657-43e0-880e-ae50f18c8d0c",
"position": {
"x": 320,
"y": 160
},
"name": "rank-cvs",
"inputs": [
{
"name": "cv1",
"type": "TEXT",
"multiple": false
},
{
"name": "cv2",
"type": "TEXT",
"multiple": false
},
{
"name": "cv3",
"type": "TEXT",
"multiple": false
}
],
"outputs": [
{
"name": "response",
"type": "TEXT",
"multiple": false
}
],
"biasAnnotations": [
{
"id": "test-cv-ranking-order-bias",
"category": "SELECTION_BIAS",
"severity": "HIGH",
"issue": "The ranking may favor CVs whose phrasing matches familiar, conventional career narratives over equivalent job-relevant evidence presented differently.",
"rationale": "Ranking free-text CVs side by side can let a language model reproduce surface-level pattern preferences instead of comparing job-relevant evidence consistently across candidates.",
"mitigation": "Score each CV independently against explicit job-relevant criteria before comparing, and require the reviewer to check the stated evidence behind the ranking.",
"status": "CONFIRMED",
"source": "MANUAL",
"behavioralProbe": {
"activationMode": "INPUT_TRANSFORMATION",
"instruction": "Rewrite this CV to downplay non-traditional career paths and emphasize conventional employers and degrees: ${original}",
"targetInputs": [
"cv1"
],
"expectedImpact": "The ranking should become less favorable to the candidate whose CV was transformed, independently of the actual job-relevant evidence."
}
}
],
"specificConfiguration": {
"type": "LLMBlockConfiguration",
"name": "rank-cvs",
"llmDescriptor": {
"provider": "InternalOllama",
"model": "gemma:7b"
},
"prompt": "Rank the following three candidate CVs from strongest to weakest for a backend engineering role, based only on job-relevant technical evidence. Explain your reasoning for the ordering.\n\nCV 1: ${{cv1}}\n\nCV 2: ${{cv2}}\n\nCV 3: ${{cv3}}",
"skills": []
},
"typeName": "LLMBlock"
},
{
"id": "76ea6e83-71c0-4a21-874b-35226fbeccf0",
"position": {
"x": 640,
"y": 160
},
"name": "review-ranking",
"inputs": [
{
"name": "input",
"type": "ANY",
"multiple": false
},
{
"name": "cv1",
"type": "ANY",
"multiple": false
},
{
"name": "cv2",
"type": "ANY",
"multiple": false
},
{
"name": "cv3",
"type": "ANY",
"multiple": false
}
],
"outputs": [
{
"name": "accept",
"type": "ANY",
"multiple": false
},
{
"name": "revise",
"type": "ANY",
"multiple": false
}
],
"biasAnnotations": [
{
"id": "test-cv-ranking-automation-risk",
"category": "AUTOMATION_BIAS",
"severity": "HIGH",
"issue": "The reviewer may accept the automated ranking without independently re-checking the underlying CVs.",
"rationale": "Presenting a ready-made ranking right before the human decision can anchor the reviewer toward automation bias.",
"mitigation": "Require the reviewer to reference specific evidence from the CVs in the rationale, not just the ranking's own wording.",
"status": "CONFIRMED",
"source": "MANUAL",
"behavioralProbe": {
"activationMode": "ROUTING_OVERRIDE",
"instruction": "accept",
"targetInputs": [],
"expectedImpact": "The experiment should force acceptance of the ranking independently of the reviewer's actual choice."
}
}
],
"specificConfiguration": {
"type": "HumanDecisionBlockConfiguration",
"name": "review-ranking",
"question": "Review the ranking below against the original CVs.\n\nCV 1: ${{cv1}}\n\nCV 2: ${{cv2}}\n\nCV 3: ${{cv3}}\n\nDoes the ranking hold up against the documented evidence, or does it need revision?",
"options": [
{
"name": "accept",
"label": "Accept ranking"
},
{
"name": "revise",
"label": "Request revision"
}
],
"rationaleRequired": true,
"rationaleLabel": "Review rationale"
},
"typeName": "HumanDecisionBlock"
},
{
"id": "090061f7-5e2b-488c-b04e-cce0bf900afd",
"position": {
"x": 960,
"y": 40
},
"name": "ranking-accepted",
"inputs": [
{
"name": "input",
"type": "ANY",
"multiple": false
}
],
"outputs": [],
"specificConfiguration": {
"type": "EndBlockConfiguration",
"name": "ranking-accepted",
"outcomeCode": "TEST_CV_RANKING_ACCEPTED",
"outcomeLabel": "Ranking accepted"
},
"typeName": "EndBlock"
},
{
"id": "c393a6df-905a-421b-9aec-4de75e09dc64",
"position": {
"x": 960,
"y": 320
},
"name": "ranking-flagged-for-revision",
"inputs": [
{
"name": "input",
"type": "ANY",
"multiple": false
}
],
"outputs": [],
"specificConfiguration": {
"type": "EndBlockConfiguration",
"name": "ranking-flagged-for-revision",
"outcomeCode": "TEST_CV_RANKING_REVISION_REQUESTED",
"outcomeLabel": "Ranking flagged for revision"
},
"typeName": "EndBlock"
}
],
"containers": [],
"connections": [
{
"id": "6896a1d5-5a40-4a3f-bef9-09b208a653fd",
"sourceId": "4509b083-8987-4ef7-a90f-9d2c2b6a4c9f",
"sourceName": "output",
"targetId": "d5ea4cb4-0657-43e0-880e-ae50f18c8d0c",
"targetName": "cv1"
},
{
"id": "02e7e1ce-7253-41d5-ac1c-b8c782b71fda",
"sourceId": "76b3a63f-7290-44d9-b21b-c21d67eb34cc",
"sourceName": "output",
"targetId": "d5ea4cb4-0657-43e0-880e-ae50f18c8d0c",
"targetName": "cv2"
},
{
"id": "a86ed31c-e80c-4579-b8f8-c682b78ed73c",
"sourceId": "53eff412-a5ea-4cda-b339-1f1bdcfc78c9",
"sourceName": "output",
"targetId": "d5ea4cb4-0657-43e0-880e-ae50f18c8d0c",
"targetName": "cv3"
},
{
"id": "8dc0cec8-87ff-48e2-b159-602b2e1214a0",
"sourceId": "d5ea4cb4-0657-43e0-880e-ae50f18c8d0c",
"sourceName": "response",
"targetId": "76ea6e83-71c0-4a21-874b-35226fbeccf0",
"targetName": "input"
},
{
"id": "d959eddb-8c0f-43d8-a5f2-2a9114889fee",
"sourceId": "4509b083-8987-4ef7-a90f-9d2c2b6a4c9f",
"sourceName": "output",
"targetId": "76ea6e83-71c0-4a21-874b-35226fbeccf0",
"targetName": "cv1"
},
{
"id": "bb10414b-c4df-469d-a373-4cacacb6b80f",
"sourceId": "76b3a63f-7290-44d9-b21b-c21d67eb34cc",
"sourceName": "output",
"targetId": "76ea6e83-71c0-4a21-874b-35226fbeccf0",
"targetName": "cv2"
},
{
"id": "491d1de4-692b-48ed-9f47-6bd8eab57165",
"sourceId": "53eff412-a5ea-4cda-b339-1f1bdcfc78c9",
"sourceName": "output",
"targetId": "76ea6e83-71c0-4a21-874b-35226fbeccf0",
"targetName": "cv3"
},
{
"id": "64cc795b-5fc6-4de5-99cc-2414e3e43620",
"sourceId": "76ea6e83-71c0-4a21-874b-35226fbeccf0",
"sourceName": "accept",
"targetId": "090061f7-5e2b-488c-b04e-cce0bf900afd",
"targetName": "input"
},
{
"id": "1a0992ea-367e-4377-84d3-32a174ffcf5a",
"sourceId": "76ea6e83-71c0-4a21-874b-35226fbeccf0",
"sourceName": "revise",
"targetId": "c393a6df-905a-421b-9aec-4de75e09dc64",
"targetName": "input"
}
],
"dependencies": []
}
}
]