From d634343df504c1afb69341f835df82e384453452 Mon Sep 17 00:00:00 2001 From: Lucio Lelii Date: Wed, 9 Sep 2026 12:23:32 +0200 Subject: [PATCH] Pre-fill no temperature for an LLM assessment The assessment dialog opened with temperature 0, for a judgement that reads the same twice. But on the JSON path the provider already forces a low baseline of its own, and a 0 typed in here overrode it; the field that actually makes an assessment repeatable is the seed, which sits next to it. Every sampling box now starts empty, meaning "the provider decides", and the request no longer carries defaults for the shared picker to merge. Co-Authored-By: Claude Opus 5 (1M context) --- src/app/services/bias/bias-report-judge.spec.ts | 7 +++++-- src/app/services/bias/bias-report-judge.ts | 9 +++++---- .../llm-descriptor-settings/llm-descriptor-settings.ts | 10 +--------- 3 files changed, 11 insertions(+), 15 deletions(-) diff --git a/src/app/services/bias/bias-report-judge.spec.ts b/src/app/services/bias/bias-report-judge.spec.ts index 6c986e2..3251324 100644 --- a/src/app/services/bias/bias-report-judge.spec.ts +++ b/src/app/services/bias/bias-report-judge.spec.ts @@ -68,13 +68,16 @@ describe('BiasReportJudgeService', () => { }); }); - it('offers temperature 0 by default, so two readings of one comparison agree', async () => { + it('pre-fills no sampling, leaving the provider its own baseline', async () => { configure({ provider: 'InternalOllama', model: 'gemma:7b' }, job('COMPLETED', { report })); await service.assess('report-1'); expect(open.mock.calls[0][0].title).toBe('Evaluate impact with LLM'); - expect(open.mock.calls[0][0].initial.temperature).toBe('0'); + // Temperature 0 used to be offered here for repeatability, and overrode the low baseline the + // provider forces on its JSON path. The seed is the field that makes an assessment repeatable. + expect(open.mock.calls[0][0].initial.temperature).toBeUndefined(); + expect(open.mock.calls[0][0].initial.seed).toBeUndefined(); }); it('asks for nothing when the model picker is dismissed', async () => { diff --git a/src/app/services/bias/bias-report-judge.ts b/src/app/services/bias/bias-report-judge.ts index 9456d38..ca74f43 100644 --- a/src/app/services/bias/bias-report-judge.ts +++ b/src/app/services/bias/bias-report-judge.ts @@ -12,8 +12,10 @@ import { openLLMDescriptorSettings } from '@shared/llm-descriptor-settings/llm-d *

The report viewer is mounted by three different hosts, and all three offer the same action, so * the picking of the model, the queued job and the waiting live here rather than three times over. * - *

Temperature starts at 0: a judgement that reads differently every time it is asked for is - * worse than none, and the seed field next to it is there for the same reason. + *

No sampling is pre-filled. A judgement that reads differently every time it is asked for is + * worse than none, but temperature 0 was the wrong lever for it: on the JSON path the provider + * already forces a low baseline of its own - one the flow assistant depends on - and setting 0 here + * overrode it. The seed field is the one that makes an assessment repeatable, and it is right there. */ @Injectable({ providedIn: 'root' }) export class BiasReportJudgeService { @@ -27,8 +29,7 @@ export class BiasReportJudgeService { */ async assess(reportId: string): Promise { const judge = await openLLMDescriptorSettings(this.settingsDialog, this.fieldRetriever, { - title: 'Evaluate impact with LLM', - defaultParameters: { temperature: 0 } + title: 'Evaluate impact with LLM' }); if (!judge) return null; diff --git a/src/app/shared/llm-descriptor-settings/llm-descriptor-settings.ts b/src/app/shared/llm-descriptor-settings/llm-descriptor-settings.ts index 179a81c..eb5b384 100644 --- a/src/app/shared/llm-descriptor-settings/llm-descriptor-settings.ts +++ b/src/app/shared/llm-descriptor-settings/llm-descriptor-settings.ts @@ -1,5 +1,5 @@ import { firstValueFrom } from 'rxjs'; -import { LLMDescriptor, ModelParameters } from '@models/flow'; +import { LLMDescriptor } from '@models/flow'; import { FieldRetriever } from '@services/retriever/field-retriever'; import { NodeSettingField, NodeSettingsDialogService } from '@services/dialogs/node-settings-dialog'; import { readSimulatorParameters } from '@shared/task-execution-viewer/execution-viewer.utils'; @@ -60,8 +60,6 @@ const PARAMETER_FIELDS: NodeSettingField[] = [ export type LLMDescriptorSettingsRequest = { title: string; - /** Sampling defaults offered in the closed section, e.g. temperature 0 for a judge. */ - defaultParameters?: Partial>; /** * What to start from, when the caller already has a model worth repeating - the simulator of the * run this one is a rerun of. Preselecting it is the difference between a comparison of one @@ -126,10 +124,6 @@ export async function openLLMDescriptorSettings( : { ...field, group: undefined })) ]; - const defaults = Object.fromEntries( - Object.entries(request.defaultParameters ?? {}).map(([key, value]) => [key, String(value)]) - ); - const inheritedModel = inheritedProvider && inherited?.model && initialModelOptions.some((option) => option.value === inherited.model) ? inherited.model @@ -141,8 +135,6 @@ export async function openLLMDescriptorSettings( initial: { provider: defaultProvider, model: inheritedModel ?? initialModelOptions[0]?.value ?? '', - ...defaults, - // Last: what the run being repeated used wins over a generic default. ...inheritedParameters }, onValuesChange: async (draft) => {