Merge branch 'main' into feature/human-evaluation-node

This commit is contained in:
Lucio Lelii 2026-09-21 14:58:20 +02:00
commit 51651a2013
9 changed files with 101 additions and 24 deletions

7
.gitignore vendored
View File

@ -36,3 +36,10 @@ build/
### macOS ###
.DS_Store
### Local environment and tooling ###
# Holds DB and provider credentials: never tracked.
local.env
.claude/
# Working notes, not part of the published source.
docs/

View File

@ -13,7 +13,7 @@ authors:
affiliation: >-
Institute of Information Science and Technologies "A. Faedo" (ISTI-CNR), Pisa, Italy
license: AGPL-3.0-or-later
repository-code: "https://github.com/luciolelii/humainflow-service"
repository-code: "https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service"
keywords:
- workflow
- human-in-the-loop

View File

@ -19,7 +19,7 @@ or unmodified, you must preserve the following attribution:
Based on HumAIn Flow, originally developed by Lucio Lelii
(Institute of Information Science and Technologies "A. Faedo" (ISTI-CNR), Pisa, Italy)
https://github.com/luciolelii/humainflow-service
https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service
The attribution must be preserved in both of the following places:

2
NOTICE
View File

@ -23,4 +23,4 @@ one:
Based on HumAIn Flow, originally developed by Lucio Lelii
(Institute of Information Science and Technologies "A. Faedo" (ISTI-CNR), Pisa, Italy)
https://github.com/luciolelii/humainflow-service
https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service

View File

@ -6,11 +6,6 @@
docker buildx build --platform linux/amd64,linux/arm64 -t luciolelii/humainflow:latest . --push
## HOST
lelii@build-host.internal
## License
Copyright (C) 2025-2026 Lucio Lelii — ISTI-CNR.
@ -28,7 +23,7 @@ In practice this means:
and where the interface shows its legal notices:
> Based on HumAIn Flow, originally developed by Lucio Lelii (ISTI-CNR)
> — https://github.com/luciolelii/humainflow-service
> — https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service
If you use HumAIn Flow in academic work, please cite it — see
[CITATION.cff](CITATION.cff).

View File

@ -15,7 +15,7 @@
<packaging>jar</packaging>
<name>workflow-manager</name>
<description>Workflow server project</description>
<url>https://github.com/luciolelii/humainflow-service</url>
<url>https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service</url>
<licenses>
<license>
<name>GNU Affero General Public License v3.0 or later</name>
@ -34,10 +34,10 @@
</developer>
</developers>
<scm>
<connection>scm:git:https://github.com/luciolelii/humainflow-service.git</connection>
<developerConnection>scm:git:git@github.com:luciolelii/humainflow-service.git</developerConnection>
<connection>scm:git:https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service.git</connection>
<developerConnection>scm:git:git@gitea-s2i2s.isti.cnr.it:lelii/humainflow-service.git</developerConnection>
<tag>HEAD</tag>
<url>https://github.com/luciolelii/humainflow-service</url>
<url>https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service</url>
</scm>
<properties>
<java.version>25</java.version>

View File

@ -95,11 +95,12 @@ public class LLMToolLoop {
stopIfCancelled(iteration);
stopIfOutOfTime(deadline, iteration);
int promptChars = totalContentChars(messages) + toolsOverheadChars;
ToolChatResult turn = provider.chatWithTools(descriptor.model(), messages, toolbox.tools(),
credential, descriptor.parameters());
logEvent(eventLogger, ExecutionEventType.LLM_REQUEST,
"Called LLM model " + descriptor.model(),
eventDetailsFor(descriptor, iteration, turn));
eventDetailsFor(descriptor, iteration, turn, promptChars));
if (!turn.hasToolCalls()) {
if (turn.content().isBlank()) {
@ -267,12 +268,20 @@ public class LLMToolLoop {
return total;
}
private static Map<String, Object> eventDetailsFor(LLMDescriptor descriptor, int iteration, ToolChatResult turn) {
private static Map<String, Object> eventDetailsFor(LLMDescriptor descriptor, int iteration, ToolChatResult turn,
int promptChars) {
Map<String, Object> details = new LinkedHashMap<>();
details.put("provider", descriptor.provider());
details.put("model", descriptor.model());
details.put("iteration", iteration);
details.put("toolCalls", turn.toolCalls().size());
// What was sent, and what was allowed to come back: the two numbers that decide whether the
// model's window was ever going to hold this. Without them, a context failure can only be
// reconstructed by reading the node's configuration next to a warning about a budget.
details.put("promptChars", promptChars);
if (descriptor.parameters() != null && descriptor.parameters().maxTokens() != null) {
details.put("maxTokens", descriptor.parameters().maxTokens());
}
if (turn.finishReason() != null) {
details.put("finishReason", turn.finishReason());
}

View File

@ -42,7 +42,7 @@ app.security.key=${WFEDITOR_SECRET_KEY:dev-local-only-change-me-dev-local-only-c
app.security.default-key=dev-local-only-change-me-dev-local-only-change-me
app.security.require-explicit-key=${WFEDITOR_REQUIRE_EXPLICIT_SECRET:false}
app.ollama.internal.key=${OLLAMA_INTERNAL_KEY:ollama}
app.ollama.internal.url=${OLLAMA_INTERNAL_URL:https://ollama.internal/api}
app.ollama.internal.url=${OLLAMA_INTERNAL_URL:http://localhost:11434/api}
app.mcp.bridge.url=${MCP_BRIDGE_URL:http://localhost:8000}
app.mcp.bridge.open-timeout-seconds=${MCP_BRIDGE_OPEN_TIMEOUT_SECONDS:90}
app.mcp.bridge.query-timeout-seconds=${MCP_BRIDGE_QUERY_TIMEOUT_SECONDS:600}
@ -73,15 +73,21 @@ app.llm.tools.max-duration-seconds=${LLM_TOOLS_MAX_DURATION_SECONDS:900}
# this was tuned against, going over does not fail cleanly: it drops the earliest message and answers
# "no user query found in messages", or the model context-shifts mid-generation and answers nothing at
# all. Characters, not tokens: nothing here has a tokenizer for whatever model is configured, so this
# is a conservative proxy, sized with real headroom under a typical 32k-token context window for the
# model's own output and reasoning. 0 disables pruning.
app.llm.tools.context-budget-chars=${LLM_TOOLS_CONTEXT_BUDGET_CHARS:60000}
# is a proxy - and the ratio between the two is what makes the number hard to pick. Prose runs about
# four characters per token; a conversation of JSON, file paths and UUIDs runs closer to two and a
# half, and a single UUID costs a dozen tokens on its own. 60000 was sized for the first, and an
# orchestrator reading its own registry hit a 32k-token window at 66332 characters - roughly 26k
# tokens, with the answer still to generate. Sized for the second now, which is what a node with
# file tools actually reads. Raise it only against a model whose window you have checked. 0 disables
# pruning.
app.llm.tools.context-budget-chars=${LLM_TOOLS_CONTEXT_BUDGET_CHARS:40000}
app.mcp.client.request-timeout-seconds=${MCP_CLIENT_REQUEST_TIMEOUT_SECONDS:120}
# Kept well under the context budget above: a single call reading one large file at this cap already
# uses a fifth of the whole conversation's budget, and the protected "current iteration" tool results
# are the ones pruning can never shrink - a handful of large reads in the same turn is what emptied
# the budget in one step the first time this was tried against a real, 24-task plan.
app.mcp.client.max-tool-result-chars=${MCP_CLIENT_MAX_TOOL_RESULT_CHARS:12000}
# Kept well under the context budget above: the protected "current iteration" tool results are the
# ones pruning can never shrink, so what one turn reads is what decides whether the next call fits.
# A model that reads the same file twice in one turn - which happens - spends double this, and at
# 12000 that alone was over half the budget. The file is still there to read again if the model
# needs more of it; the conversation is not.
app.mcp.client.max-tool-result-chars=${MCP_CLIENT_MAX_TOOL_RESULT_CHARS:6000}
app.executions.cache.max-size=${APP_EXECUTIONS_CACHE_MAX_SIZE:1000}
app.executions.cache.final-ttl-ms=${APP_EXECUTIONS_CACHE_FINAL_TTL_MS:1800000}
app.executions.cache.cleanup-interval-ms=${APP_EXECUTIONS_CACHE_CLEANUP_INTERVAL_MS:60000}

View File

@ -162,6 +162,11 @@ class LLMToolLoopTest {
return new LLMDescriptor("Scripted", "test-model", null);
}
private static LLMDescriptor descriptorWithMaxTokens(int maxTokens) {
return new LLMDescriptor("Scripted", "test-model",
new it.cnr.isti.workflow.manager.llms.ModelParameters(null, null, null, maxTokens, null));
}
private static ObjectNode argumentsWithJsonLookingContent() {
ObjectNode arguments = JsonNodeFactory.instance.objectNode();
arguments.put("path", "state.json");
@ -442,6 +447,61 @@ class LLMToolLoopTest {
}
}
@Test
void everyCallRecordsHowMuchWasSentAndWhatWasAllowedBack() throws Exception {
// A context failure used to be reconstructable only by reading the node's configuration next
// to a warning about characters. These two numbers are the ones that decide whether the
// model's window was ever going to hold the call.
HttpServer server = startServerWithFixedResultSize(200);
ScriptedProvider provider = new ScriptedProvider(true);
provider.turns.add(bigToolCallTurn());
provider.turns.add(ToolChatResult.text("DONE"));
List<ExecutionEventType> loggedTypes = new ArrayList<>();
List<Map<String, Object>> loggedDetails = new ArrayList<>();
ExecutionEventLogger recordingLogger = (level, type, message, details) -> {
loggedTypes.add(type);
loggedDetails.add(details);
};
try {
new LLMToolLoop(factoryFor(server), 10, 60, NO_PRUNING)
.run(provider, descriptorWithMaxTokens(4096), null, "do it", bindings(), Map.of(),
recordingLogger);
int firstRequest = loggedTypes.indexOf(ExecutionEventType.LLM_REQUEST);
assertTrue(firstRequest >= 0, "expected an LLM_REQUEST event, got: " + loggedTypes);
Map<String, Object> details = loggedDetails.get(firstRequest);
assertEquals(4096, details.get("maxTokens"));
// The first call carries the prompt alone, so this is small but never zero.
assertTrue((int) details.get("promptChars") > 0, "promptChars: " + details.get("promptChars"));
int secondRequest = loggedTypes.subList(firstRequest + 1, loggedTypes.size())
.indexOf(ExecutionEventType.LLM_REQUEST) + firstRequest + 1;
// By the second call the tool result is in the conversation, so more is being sent.
assertTrue((int) loggedDetails.get(secondRequest).get("promptChars")
> (int) details.get("promptChars"));
} finally {
server.stop(0);
}
}
@Test
void omitsTheOutputCapWhenTheNodeDidNotSetOne() throws Exception {
HttpServer server = startServerWithFixedResultSize(200);
ScriptedProvider provider = new ScriptedProvider(true);
provider.turns.add(ToolChatResult.text("DONE"));
List<Map<String, Object>> loggedDetails = new ArrayList<>();
ExecutionEventLogger recordingLogger = (level, type, message, details) -> loggedDetails.add(details);
try {
new LLMToolLoop(factoryFor(server), 10, 60, NO_PRUNING)
.run(provider, descriptor(), null, "do it", bindings(), Map.of(), recordingLogger);
assertTrue(loggedDetails.stream().noneMatch(details -> details != null && details.containsKey("maxTokens")),
"an unset cap must not be reported as a value");
} finally {
server.stop(0);
}
}
@Test
void reportsWhatItPrunedAsAnEvent() throws Exception {
HttpServer server = startServerWithFixedResultSize(800);