Check a flow's configured models against their provider at start

A wrong model name only ever failed on the first call to it, which can be
minutes into a run for a step deep in a flow - by which point the endpoint
and credential that would have let it fail immediately were already known.

AuthorizationRequirementResolver.resolveAllDescriptors mirrors the existing
requirement-collecting walk (blocks, containers, Loop guard subflows) to
list every LLMDescriptor in a flow instead. ExecutionsService verifies each
one against its provider's own catalogue, for a provider whose
canListModels() is true, before starting - today that is only our own
Ollama, since every hosted provider declares canListModels() false
precisely because it cannot be asked without a credential the check does
not have. A model that is empty or still a template placeholder is
skipped, since its real value is only known at call time; a catalogue
that cannot be listed just now does not block the run either - this is a
defense in depth, not a gate a transient network failure should be able
to close.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
Lucio Lelii 2026-09-17 15:08:48 +02:00
parent ed128ac014
commit 11c70115c1
3 changed files with 177 additions and 1 deletions

View File

@ -127,6 +127,39 @@ final class AuthorizationRequirementResolver {
.addStepReference(block.getId(), block.getName());
}
/**
* Every {@link LLMDescriptor} in the flow, wherever one appears - the same walk
* {@link #collectRequirements} does, without the authorization bookkeeping. Used to check a
* model's availability against its provider's catalogue at execution start, before any step
* that would otherwise be the first to find out the model is wrong.
*/
static List<LLMDescriptorReference> resolveAllDescriptors(FlowData flow) {
List<LLMDescriptorReference> references = new ArrayList<>();
collectDescriptors(flow, references);
return references;
}
private static void collectDescriptors(FlowData flow, List<LLMDescriptorReference> references) {
if (flow == null || flow.getNodes().isEmpty()) {
return;
}
for (Block<?> block : flow.getBlocks() == null ? List.<Block<?>>of() : flow.getBlocks()) {
resolveDescriptors(block).forEach(descriptor ->
references.add(new LLMDescriptorReference(descriptor, block == null ? null : block.getName())));
}
for (Container<?> container : flow.getContainers() == null ? List.<Container<?>>of() : flow.getContainers()) {
if (container != null && container.getSpecificConfiguration() instanceof ContainerConfiguration<?> containerConfiguration) {
collectDescriptors(containerConfiguration.getSubFlow(), references);
if (containerConfiguration instanceof LoopContainerConfiguration loopConfiguration) {
collectDescriptors(loopConfiguration.getGuardSubFlow(), references);
}
}
}
}
record LLMDescriptorReference(LLMDescriptor descriptor, String blockName) {
}
static LLMProvider resolveProvider(Map<String, LLMProvider> llmProviders, String providerName) {
LLMProvider provider = llmProviders.get(providerName);
if (provider != null) {

View File

@ -1411,11 +1411,58 @@ public class ExecutionsService {
@Transactional
public ExecutionObject startExecution(String id) {
ExecutionObject eo = getExecution(id);
verifyModelAvailability(eo);
eo.start();
touchExecution(id);
return eo;
}
/**
* Best-effort: a model the flow names but its provider does not actually serve fails on the
* very first call, which can be minutes into a run. Endpoint and credential have just been
* supplied, and the machinery to enumerate every {@code (provider, model)} pair already exists
* for authorization requirements - this reuses the same walk to ask each provider that can
* answer, before the run starts, rather than after.
*
* <p>Only a provider whose {@link LLMProvider#canListModels()} is true is asked at all: every
* hosted provider says false there precisely because it cannot be asked without a credential
* the editor does not have, so this only ever fires for our own Ollama today. A provider that
* can be asked but fails to answer just now - the network, not the model - does not block the
* run either: this is a defense in depth, not a gate that a transient failure should be able to
* close.
*/
private void verifyModelAvailability(ExecutionObject eo) {
for (AuthorizationRequirementResolver.LLMDescriptorReference reference
: AuthorizationRequirementResolver.resolveAllDescriptors(eo.getFlow())) {
LLMDescriptor descriptor = reference.descriptor();
String model = descriptor.model() == null ? "" : descriptor.model().trim();
// A template placeholder resolves only at call time - global inputs cannot reach this
// field through a port, so writing one here is the only way to make the model depend on
// one, and its actual value is unknowable this early.
if (model.isEmpty() || model.contains("${{")) {
continue;
}
LLMProvider provider = AuthorizationRequirementResolver.resolveProvider(llmProviders, descriptor.provider());
if (provider == null || !provider.canListModels()) {
continue;
}
List<String> registered;
try {
registered = provider.getRegisteredModels();
} catch (RuntimeException exception) {
logger.warn("Could not verify model availability for provider {}: {}", provider.getName(),
exception.getMessage());
continue;
}
if (registered != null && !registered.isEmpty() && !registered.contains(model)) {
throw new ResponseStatusException(HttpStatus.BAD_REQUEST,
"Model \"" + model + "\" is not available from provider " + provider.getName()
+ (reference.blockName() == null || reference.blockName().isBlank()
? "" : " (used by " + reference.blockName() + ")"));
}
}
}
/**
* @param credentialId a vault secret id for the simulator's own provider, or null to fall back
* to whatever the flow's own steps already provided for that exact
@ -1436,6 +1483,7 @@ public class ExecutionsService {
"Simulation requires a simulator descriptor");
}
resolveSimulatorAuthorization(eo, simulatorDescriptor, credentialId);
verifyModelAvailability(eo);
eo.setInteractionSimulationDescriptor(simulatorDescriptor);
eo.startSimulation();
touchExecution(id);

View File

@ -164,7 +164,7 @@ public class ExecutionTest {
}
@Override
public List<String> getRegisteredModels() {
return List.of("testModel");
return List.of("testModel", "configuredModel");
}
@Override
@ -207,8 +207,35 @@ public class ExecutionTest {
}
};
}
/** Declares a real, non-empty catalogue - unlike every other provider stub in this file. */
@Bean
public LLMProvider modelCheckedProvider() {
return new LLMProvider() {
@Override
public String getName() {
return "modelCheckedProvider";
}
@Override
public List<String> getRegisteredModels() {
if (modelListingShouldFail.get()) {
throw new IllegalStateException("catalogue temporarily unreachable");
}
return List.of("known-model");
}
@Override
public String generate(String model, String prompt) {
return "answered by " + model;
}
};
}
}
private static final java.util.concurrent.atomic.AtomicBoolean modelListingShouldFail =
new java.util.concurrent.atomic.AtomicBoolean(false);
@Autowired
ExecutionsService executionsService;
@ -1079,6 +1106,74 @@ public class ExecutionTest {
.get(new FieldKey(block.getId(), LLMBlockFactory.OUTPUT_NAME)));
}
@Test
public void startExecutionFailsFastWhenTheConfiguredModelIsNotInTheProvidersCatalogue() {
// The gap this closes: before, a model the flow names but its provider does not actually
// serve failed only on the first real call, which can be minutes into a run for a step deep
// in a flow. modelCheckedProvider is the one stub in this file whose catalogue is both real
// (non-empty) and narrow, so a mismatch here is caught before the execution ever starts.
modelListingShouldFail.set(false);
Block<LLMBlockType> llmBlock = llmBlockFactory.create(LLMBlockConfiguration.builder()
.name("Ask the model")
.prompt("Which model answered?")
.llmDescriptor(LLMDescriptor.builder().provider("modelCheckedProvider").model("nonexistent-model").build())
.build());
ExecutionObject execObject = executionsService.createExecution(
Flow.builder().name("Unknown model flow").description("").block(llmBlock).build());
ResponseStatusException error = assertThrows(ResponseStatusException.class,
() -> executionsService.startExecution(execObject.getId()));
assertEquals(HttpStatus.BAD_REQUEST, error.getStatusCode());
assertTrue(error.getReason().contains("nonexistent-model"));
assertNotEquals(ExecutionStatus.RUNNING,
executionsService.getExecution(execObject.getId()).getContext().getStatus());
}
@Test
public void startExecutionProceedsWhenTheConfiguredModelIsInTheProvidersCatalogue() {
modelListingShouldFail.set(false);
Block<LLMBlockType> llmBlock = llmBlockFactory.create(LLMBlockConfiguration.builder()
.name("Ask the model")
.prompt("Which model answered?")
.llmDescriptor(LLMDescriptor.builder().provider("modelCheckedProvider").model("known-model").build())
.build());
ExecutionObject execObject = executionsService.createExecution(
Flow.builder().name("Known model flow").description("").block(llmBlock).build());
execObject = executionsService.startExecution(execObject.getId());
while (execObject.getContext().getStatus() == ExecutionStatus.RUNNING) {
execObject = executionsService.getExecution(execObject.getId());
}
assertEquals(ExecutionStatus.SUCCESS, execObject.getContext().getStatus());
}
@Test
public void startExecutionProceedsWhenTheProvidersCatalogueCannotBeListedJustNow() {
// Best-effort: a provider that can normally be asked, but fails to answer right now, must not
// turn a transient network problem into every execution refusing to start.
modelListingShouldFail.set(true);
try {
Block<LLMBlockType> llmBlock = llmBlockFactory.create(LLMBlockConfiguration.builder()
.name("Ask the model")
.prompt("Which model answered?")
.llmDescriptor(LLMDescriptor.builder().provider("modelCheckedProvider").model("known-model").build())
.build());
ExecutionObject execObject = executionsService.createExecution(
Flow.builder().name("Unreachable catalogue flow").description("").block(llmBlock).build());
execObject = executionsService.startExecution(execObject.getId());
while (execObject.getContext().getStatus() == ExecutionStatus.RUNNING) {
execObject = executionsService.getExecution(execObject.getId());
}
assertEquals(ExecutionStatus.SUCCESS, execObject.getContext().getStatus());
} finally {
modelListingShouldFail.set(false);
}
}
@Test
public void llmExecutionPrependsSelectedSkillsToPrompt() {
Block<LLMBlockType> llmBlock = llmBlockFactory.create(LLMBlockConfiguration.builder()