diff --git a/.gitignore b/.gitignore index 4c176e5..a1f5f18 100644 --- a/.gitignore +++ b/.gitignore @@ -36,3 +36,10 @@ build/ ### macOS ### .DS_Store + +### Local environment and tooling ### +# Holds DB and provider credentials: never tracked. +local.env +.claude/ +# Working notes, not part of the published source. +docs/ diff --git a/CITATION.cff b/CITATION.cff index e410777..be4c673 100644 --- a/CITATION.cff +++ b/CITATION.cff @@ -13,7 +13,7 @@ authors: affiliation: >- Institute of Information Science and Technologies "A. Faedo" (ISTI-CNR), Pisa, Italy license: AGPL-3.0-or-later -repository-code: "https://github.com/luciolelii/humainflow-service" +repository-code: "https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service" keywords: - workflow - human-in-the-loop diff --git a/LICENSE-ADDENDUM b/LICENSE-ADDENDUM index 3cf23f2..8ec46d7 100644 --- a/LICENSE-ADDENDUM +++ b/LICENSE-ADDENDUM @@ -19,7 +19,7 @@ or unmodified, you must preserve the following attribution: Based on HumAIn Flow, originally developed by Lucio Lelii (Institute of Information Science and Technologies "A. Faedo" (ISTI-CNR), Pisa, Italy) - https://github.com/luciolelii/humainflow-service + https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service The attribution must be preserved in both of the following places: diff --git a/NOTICE b/NOTICE index d8da3a0..e7df1d0 100644 --- a/NOTICE +++ b/NOTICE @@ -23,4 +23,4 @@ one: Based on HumAIn Flow, originally developed by Lucio Lelii (Institute of Information Science and Technologies "A. Faedo" (ISTI-CNR), Pisa, Italy) - https://github.com/luciolelii/humainflow-service + https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service diff --git a/README.md b/README.md index 85e6608..f7897e7 100644 --- a/README.md +++ b/README.md @@ -6,11 +6,6 @@ docker buildx build --platform linux/amd64,linux/arm64 -t luciolelii/humainflow:latest . --push -## HOST - -lelii@build-host.internal - - ## License Copyright (C) 2025-2026 Lucio Lelii — ISTI-CNR. @@ -28,7 +23,7 @@ In practice this means: and where the interface shows its legal notices: > Based on HumAIn Flow, originally developed by Lucio Lelii (ISTI-CNR) - > — https://github.com/luciolelii/humainflow-service + > — https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service If you use HumAIn Flow in academic work, please cite it — see [CITATION.cff](CITATION.cff). diff --git a/pom.xml b/pom.xml index aad414b..6a757e3 100644 --- a/pom.xml +++ b/pom.xml @@ -15,7 +15,7 @@ jar workflow-manager Workflow server project - https://github.com/luciolelii/humainflow-service + https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service GNU Affero General Public License v3.0 or later @@ -34,10 +34,10 @@ - scm:git:https://github.com/luciolelii/humainflow-service.git - scm:git:git@github.com:luciolelii/humainflow-service.git + scm:git:https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service.git + scm:git:git@gitea-s2i2s.isti.cnr.it:lelii/humainflow-service.git HEAD - https://github.com/luciolelii/humainflow-service + https://gitea-s2i2s.isti.cnr.it/lelii/humainflow-service 25 diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMToolLoop.java b/src/main/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMToolLoop.java index 05e883c..1904016 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMToolLoop.java +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMToolLoop.java @@ -95,11 +95,12 @@ public class LLMToolLoop { stopIfCancelled(iteration); stopIfOutOfTime(deadline, iteration); + int promptChars = totalContentChars(messages) + toolsOverheadChars; ToolChatResult turn = provider.chatWithTools(descriptor.model(), messages, toolbox.tools(), credential, descriptor.parameters()); logEvent(eventLogger, ExecutionEventType.LLM_REQUEST, "Called LLM model " + descriptor.model(), - eventDetailsFor(descriptor, iteration, turn)); + eventDetailsFor(descriptor, iteration, turn, promptChars)); if (!turn.hasToolCalls()) { if (turn.content().isBlank()) { @@ -267,12 +268,20 @@ public class LLMToolLoop { return total; } - private static Map eventDetailsFor(LLMDescriptor descriptor, int iteration, ToolChatResult turn) { + private static Map eventDetailsFor(LLMDescriptor descriptor, int iteration, ToolChatResult turn, + int promptChars) { Map details = new LinkedHashMap<>(); details.put("provider", descriptor.provider()); details.put("model", descriptor.model()); details.put("iteration", iteration); details.put("toolCalls", turn.toolCalls().size()); + // What was sent, and what was allowed to come back: the two numbers that decide whether the + // model's window was ever going to hold this. Without them, a context failure can only be + // reconstructed by reading the node's configuration next to a warning about a budget. + details.put("promptChars", promptChars); + if (descriptor.parameters() != null && descriptor.parameters().maxTokens() != null) { + details.put("maxTokens", descriptor.parameters().maxTokens()); + } if (turn.finishReason() != null) { details.put("finishReason", turn.finishReason()); } diff --git a/src/main/resources/application.properties b/src/main/resources/application.properties index 39fb53b..f5808ec 100644 --- a/src/main/resources/application.properties +++ b/src/main/resources/application.properties @@ -42,7 +42,7 @@ app.security.key=${WFEDITOR_SECRET_KEY:dev-local-only-change-me-dev-local-only-c app.security.default-key=dev-local-only-change-me-dev-local-only-change-me app.security.require-explicit-key=${WFEDITOR_REQUIRE_EXPLICIT_SECRET:false} app.ollama.internal.key=${OLLAMA_INTERNAL_KEY:ollama} -app.ollama.internal.url=${OLLAMA_INTERNAL_URL:https://ollama.internal/api} +app.ollama.internal.url=${OLLAMA_INTERNAL_URL:http://localhost:11434/api} app.mcp.bridge.url=${MCP_BRIDGE_URL:http://localhost:8000} app.mcp.bridge.open-timeout-seconds=${MCP_BRIDGE_OPEN_TIMEOUT_SECONDS:90} app.mcp.bridge.query-timeout-seconds=${MCP_BRIDGE_QUERY_TIMEOUT_SECONDS:600} @@ -73,15 +73,21 @@ app.llm.tools.max-duration-seconds=${LLM_TOOLS_MAX_DURATION_SECONDS:900} # this was tuned against, going over does not fail cleanly: it drops the earliest message and answers # "no user query found in messages", or the model context-shifts mid-generation and answers nothing at # all. Characters, not tokens: nothing here has a tokenizer for whatever model is configured, so this -# is a conservative proxy, sized with real headroom under a typical 32k-token context window for the -# model's own output and reasoning. 0 disables pruning. -app.llm.tools.context-budget-chars=${LLM_TOOLS_CONTEXT_BUDGET_CHARS:60000} +# is a proxy - and the ratio between the two is what makes the number hard to pick. Prose runs about +# four characters per token; a conversation of JSON, file paths and UUIDs runs closer to two and a +# half, and a single UUID costs a dozen tokens on its own. 60000 was sized for the first, and an +# orchestrator reading its own registry hit a 32k-token window at 66332 characters - roughly 26k +# tokens, with the answer still to generate. Sized for the second now, which is what a node with +# file tools actually reads. Raise it only against a model whose window you have checked. 0 disables +# pruning. +app.llm.tools.context-budget-chars=${LLM_TOOLS_CONTEXT_BUDGET_CHARS:40000} app.mcp.client.request-timeout-seconds=${MCP_CLIENT_REQUEST_TIMEOUT_SECONDS:120} -# Kept well under the context budget above: a single call reading one large file at this cap already -# uses a fifth of the whole conversation's budget, and the protected "current iteration" tool results -# are the ones pruning can never shrink - a handful of large reads in the same turn is what emptied -# the budget in one step the first time this was tried against a real, 24-task plan. -app.mcp.client.max-tool-result-chars=${MCP_CLIENT_MAX_TOOL_RESULT_CHARS:12000} +# Kept well under the context budget above: the protected "current iteration" tool results are the +# ones pruning can never shrink, so what one turn reads is what decides whether the next call fits. +# A model that reads the same file twice in one turn - which happens - spends double this, and at +# 12000 that alone was over half the budget. The file is still there to read again if the model +# needs more of it; the conversation is not. +app.mcp.client.max-tool-result-chars=${MCP_CLIENT_MAX_TOOL_RESULT_CHARS:6000} app.executions.cache.max-size=${APP_EXECUTIONS_CACHE_MAX_SIZE:1000} app.executions.cache.final-ttl-ms=${APP_EXECUTIONS_CACHE_FINAL_TTL_MS:1800000} app.executions.cache.cleanup-interval-ms=${APP_EXECUTIONS_CACHE_CLEANUP_INTERVAL_MS:60000} diff --git a/src/test/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMToolLoopTest.java b/src/test/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMToolLoopTest.java index 4431f30..ced9c3b 100644 --- a/src/test/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMToolLoopTest.java +++ b/src/test/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMToolLoopTest.java @@ -162,6 +162,11 @@ class LLMToolLoopTest { return new LLMDescriptor("Scripted", "test-model", null); } + private static LLMDescriptor descriptorWithMaxTokens(int maxTokens) { + return new LLMDescriptor("Scripted", "test-model", + new it.cnr.isti.workflow.manager.llms.ModelParameters(null, null, null, maxTokens, null)); + } + private static ObjectNode argumentsWithJsonLookingContent() { ObjectNode arguments = JsonNodeFactory.instance.objectNode(); arguments.put("path", "state.json"); @@ -442,6 +447,61 @@ class LLMToolLoopTest { } } + @Test + void everyCallRecordsHowMuchWasSentAndWhatWasAllowedBack() throws Exception { + // A context failure used to be reconstructable only by reading the node's configuration next + // to a warning about characters. These two numbers are the ones that decide whether the + // model's window was ever going to hold the call. + HttpServer server = startServerWithFixedResultSize(200); + ScriptedProvider provider = new ScriptedProvider(true); + provider.turns.add(bigToolCallTurn()); + provider.turns.add(ToolChatResult.text("DONE")); + List loggedTypes = new ArrayList<>(); + List> loggedDetails = new ArrayList<>(); + ExecutionEventLogger recordingLogger = (level, type, message, details) -> { + loggedTypes.add(type); + loggedDetails.add(details); + }; + try { + new LLMToolLoop(factoryFor(server), 10, 60, NO_PRUNING) + .run(provider, descriptorWithMaxTokens(4096), null, "do it", bindings(), Map.of(), + recordingLogger); + + int firstRequest = loggedTypes.indexOf(ExecutionEventType.LLM_REQUEST); + assertTrue(firstRequest >= 0, "expected an LLM_REQUEST event, got: " + loggedTypes); + Map details = loggedDetails.get(firstRequest); + assertEquals(4096, details.get("maxTokens")); + // The first call carries the prompt alone, so this is small but never zero. + assertTrue((int) details.get("promptChars") > 0, "promptChars: " + details.get("promptChars")); + + int secondRequest = loggedTypes.subList(firstRequest + 1, loggedTypes.size()) + .indexOf(ExecutionEventType.LLM_REQUEST) + firstRequest + 1; + // By the second call the tool result is in the conversation, so more is being sent. + assertTrue((int) loggedDetails.get(secondRequest).get("promptChars") + > (int) details.get("promptChars")); + } finally { + server.stop(0); + } + } + + @Test + void omitsTheOutputCapWhenTheNodeDidNotSetOne() throws Exception { + HttpServer server = startServerWithFixedResultSize(200); + ScriptedProvider provider = new ScriptedProvider(true); + provider.turns.add(ToolChatResult.text("DONE")); + List> loggedDetails = new ArrayList<>(); + ExecutionEventLogger recordingLogger = (level, type, message, details) -> loggedDetails.add(details); + try { + new LLMToolLoop(factoryFor(server), 10, 60, NO_PRUNING) + .run(provider, descriptor(), null, "do it", bindings(), Map.of(), recordingLogger); + + assertTrue(loggedDetails.stream().noneMatch(details -> details != null && details.containsKey("maxTokens")), + "an unset cap must not be reported as a value"); + } finally { + server.stop(0); + } + } + @Test void reportsWhatItPrunedAsAnEvent() throws Exception { HttpServer server = startServerWithFixedResultSize(800);