diff --git a/src/app/services/bias/bias-report-judge.spec.ts b/src/app/services/bias/bias-report-judge.spec.ts index 6c986e2..3251324 100644 --- a/src/app/services/bias/bias-report-judge.spec.ts +++ b/src/app/services/bias/bias-report-judge.spec.ts @@ -68,13 +68,16 @@ describe('BiasReportJudgeService', () => { }); }); - it('offers temperature 0 by default, so two readings of one comparison agree', async () => { + it('pre-fills no sampling, leaving the provider its own baseline', async () => { configure({ provider: 'InternalOllama', model: 'gemma:7b' }, job('COMPLETED', { report })); await service.assess('report-1'); expect(open.mock.calls[0][0].title).toBe('Evaluate impact with LLM'); - expect(open.mock.calls[0][0].initial.temperature).toBe('0'); + // Temperature 0 used to be offered here for repeatability, and overrode the low baseline the + // provider forces on its JSON path. The seed is the field that makes an assessment repeatable. + expect(open.mock.calls[0][0].initial.temperature).toBeUndefined(); + expect(open.mock.calls[0][0].initial.seed).toBeUndefined(); }); it('asks for nothing when the model picker is dismissed', async () => { diff --git a/src/app/services/bias/bias-report-judge.ts b/src/app/services/bias/bias-report-judge.ts index 9456d38..ca74f43 100644 --- a/src/app/services/bias/bias-report-judge.ts +++ b/src/app/services/bias/bias-report-judge.ts @@ -12,8 +12,10 @@ import { openLLMDescriptorSettings } from '@shared/llm-descriptor-settings/llm-d *
The report viewer is mounted by three different hosts, and all three offer the same action, so * the picking of the model, the queued job and the waiting live here rather than three times over. * - *
Temperature starts at 0: a judgement that reads differently every time it is asked for is - * worse than none, and the seed field next to it is there for the same reason. + *
No sampling is pre-filled. A judgement that reads differently every time it is asked for is
+ * worse than none, but temperature 0 was the wrong lever for it: on the JSON path the provider
+ * already forces a low baseline of its own - one the flow assistant depends on - and setting 0 here
+ * overrode it. The seed field is the one that makes an assessment repeatable, and it is right there.
*/
@Injectable({ providedIn: 'root' })
export class BiasReportJudgeService {
@@ -27,8 +29,7 @@ export class BiasReportJudgeService {
*/
async assess(reportId: string): Promise