From f5386cdb7efbb0a68838b62d6e7c5d1027a0713c Mon Sep 17 00:00:00 2001 From: Lucio Lelii Date: Tue, 8 Sep 2026 10:50:54 +0200 Subject: [PATCH] Say which subject a bias intervention moved, and let an LLM assess it The report now reads the per-subject sections the service produces: a row per iterated subject with what its labelled numbers did (Score 7 -> 4), the two texts behind it a click away, and the changed ones listed first. The band at the top leads with what the run actually did - whether the final decision changed, how many subjects moved, the largest delta - because the counts of changed nodes that used to open the report were the least actionable thing in it. A container's iterations are listed with the inner node that changed, which is what a per-subject iterator run needs and the accumulated list could never show. Reports produced before any of this exists still render, from their raw outputs. "Evaluate impact with LLM" sits next to the report it is about, in all three places one is mounted, and opens the provider and model picker the interaction simulator uses - extracted so both call the same dialog rather than two of their own, with temperature 0 offered by default because a verdict that reads differently every time it is asked for is worse than none. The assessment runs as a job, polled like the isolated experiment, and is stored on the report, so reopening it later shows the same verdicts and the model that produced them. It is labelled an assessment throughout, and a pair the model could not answer for is marked without hiding that pair's own figures. A rerun of a simulated run now opens the Simulate dialog on the simulator it inherited, with the inherited sampling out where it can be seen - a seed carried over is the reason the two runs are comparable, and behind a closed section nobody would find it. Before it is started, a run says which simulator the run it repeats used; afterwards, both the bias report and the run-to-run comparison say so when the two sides were not answered the same way, since that difference is not the intervention's doing and nothing said it. Co-Authored-By: Claude Opus 5 (1M context) --- src/app/models/bias-impact.spec.ts | 1 + src/app/models/bias-impact.ts | 120 +++++++- src/app/services/bias/bias-flow.spec.ts | 2 +- .../services/bias/bias-report-judge.spec.ts | 93 ++++++ src/app/services/bias/bias-report-judge.ts | 42 +++ .../task-executions-call.base.ts | 5 + .../task-executions-call.fake.ts | 155 +++++++++- .../task-executions-call.spec.ts | 112 +++++++ .../task-executions/task-executions-call.ts | 173 ++++++++++- .../task-executions/task-executions.spec.ts | 1 + .../task-executions/task-executions.ts | 6 + .../bias-compare-dialog.html | 9 +- .../bias-compare-dialog.ts | 22 ++ .../bias-impact-experiment-dialog.html | 8 +- .../bias-impact-experiment-dialog.ts | 21 ++ .../bias-impact-report-list.html | 7 +- .../bias-impact-report-list.ts | 23 ++ .../bias-impact-report-viewer.css | 58 +++- .../bias-impact-report-viewer.html | 276 ++++++++++++++++-- .../bias-impact-report-viewer.spec.ts | 161 +++++++++- .../bias-impact-report-viewer.ts | 150 +++++++++- .../execution-compare-view.html | 4 + .../execution-compare-view.spec.ts | 36 +++ .../execution-compare-view.ts | 36 +++ .../llm-descriptor-settings.spec.ts | 81 +++++ .../llm-descriptor-settings.ts | 153 ++++++++++ .../execution-viewer.utils.spec.ts | 33 +++ .../execution-viewer.utils.ts | 29 +- .../task-execution-viewer.css | 11 + .../task-execution-viewer.html | 4 + .../task-execution-viewer.ts | 131 +-------- 31 files changed, 1808 insertions(+), 155 deletions(-) create mode 100644 src/app/services/bias/bias-report-judge.spec.ts create mode 100644 src/app/services/bias/bias-report-judge.ts create mode 100644 src/app/shared/llm-descriptor-settings/llm-descriptor-settings.spec.ts create mode 100644 src/app/shared/llm-descriptor-settings/llm-descriptor-settings.ts diff --git a/src/app/models/bias-impact.spec.ts b/src/app/models/bias-impact.spec.ts index 6645d90..d2d78c2 100644 --- a/src/app/models/bias-impact.spec.ts +++ b/src/app/models/bias-impact.spec.ts @@ -53,6 +53,7 @@ describe('bias impact models', () => { }; const job: BiasImpactJob = { id: 'job-1', + kind: 'ISOLATED_STEP', status: 'COMPLETED', executionId: 'baseline-1', stepId: 'node-1', diff --git a/src/app/models/bias-impact.ts b/src/app/models/bias-impact.ts index 9ab63b5..d4b95ff 100644 --- a/src/app/models/bias-impact.ts +++ b/src/app/models/bias-impact.ts @@ -1,4 +1,4 @@ -import { BiasActivationMode } from './flow'; +import { BiasActivationMode, LLMDescriptor } from './flow'; export type BiasInterventionDirection = 'BIAS' | 'MITIGATION' | 'BOTH'; @@ -111,10 +111,15 @@ export function activeAnnotationIdsFor( export type BiasImpactJobStatus = 'QUEUED' | 'RUNNING' | 'COMPLETED' | 'FAILED'; +/** Both kinds of queued bias work: re-running a step, and assessing a comparison with an LLM. */ +export type BiasImpactJobKind = 'ISOLATED_STEP' | 'REPORT_JUDGE'; + export type BiasImpactJob = { id: string; + kind: BiasImpactJobKind; status: BiasImpactJobStatus; executionId: string; + /** Empty for a judge job: it is about a whole comparison, not one step. */ stepId: string; createdAt: string; startedAt: string | null; @@ -128,12 +133,107 @@ export type BiasImpactJob = { export type BiasImpactReportKind = 'ISOLATED_STEP' | 'FULL_FLOW'; +/** A number both runs labelled the same way, and how far the intervention moved it. */ +export type BiasNumericDelta = { + label: string; + baseline: number; + biased: number; + delta: number; +}; + +export type BiasJudgeImpactLevel = 'NONE' | 'COSMETIC' | 'SUBSTANTIVE' | 'DECISIVE'; + +export type BiasJudgeAttribution = 'INJECTION' | 'NON_DETERMINISM' | 'UNCLEAR'; + +/** + * What a judge model made of one compared pair. An assessment, not a measurement: `error` is set + * when the model answered with something unusable, and the pair's own figures stand either way. + */ +export type BiasJudgeVerdict = { + impact: BiasJudgeImpactLevel | null; + attribution: BiasJudgeAttribution | null; + confidence: number | null; + changedAspects: string[]; + rationale: string | null; + error: string | null; +}; + +export type BiasJudgeSummary = { + judge: LLMDescriptor; + judgedAt: string; + impact: BiasJudgeImpactLevel | null; + attribution: BiasJudgeAttribution | null; + narrative: string | null; + judgedPairs: number; + skippedPairs: number; + errors: string[]; +}; + +/** One element of a compared list value, which for an iterator container is one subject. */ +export type BiasItemImpact = { + index: number; + changed: boolean; + textDifference: number; + numericDeltas: BiasNumericDelta[]; + baselineText: string | null; + biasedText: string | null; + judgeVerdict: BiasJudgeVerdict | null; +}; + +/** One output field of one node, compared between the two runs. */ +export type BiasValueImpact = { + nodeId: string; + nodeName: string; + field: string; + changed: boolean; + textDifference: number; + itemsCompared: number; + itemsChanged: number; + meanItemTextDifference: number; + maximumItemTextDifference: number; + numericDeltas: BiasNumericDelta[]; + items: BiasItemImpact[]; + baselineText: string | null; + biasedText: string | null; + judgeVerdict: BiasJudgeVerdict | null; +}; + +/** One iteration of a container step, from the pair of child executions that ran it. */ +export type BiasIterationImpact = { + containerNodeId: string; + containerNodeName: string; + index: number; + baselineExecutionId: string | null; + biasedExecutionId: string | null; + baselineStatus: string; + biasedStatus: string; + changed: boolean; + values: BiasValueImpact[]; +}; + +/** + * How the interactive steps of the two compared runs were answered. + * + * Absent when neither run was simulated, and on reports produced before it was recorded. + */ +export type BiasSimulationContext = { + baselineSimulated: boolean; + baselineSimulator: LLMDescriptor | null; + biasedSimulated: boolean; + biasedSimulator: LLMDescriptor | null; + /** False when only one side was simulated, or the two simulators are not the same one. */ + comparable: boolean; +}; + export type BiasImmediateImpact = { outputChanged: boolean; maximumTextDifference: number; changeRate: number; baselineOutput: unknown; biasedOutputs: unknown[]; + /** Absent on reports produced before the field-by-field comparison existed. */ + values?: BiasValueImpact[]; + iterations?: BiasIterationImpact[]; }; export type BiasDownstreamImpactEntry = { @@ -144,6 +244,8 @@ export type BiasDownstreamImpactEntry = { changed: boolean; baselineOutputs: unknown; biasedOutputs: unknown; + values?: BiasValueImpact[]; + iterations?: BiasIterationImpact[]; }; export type BiasRoutingChangeEntry = { @@ -152,6 +254,17 @@ export type BiasRoutingChangeEntry = { biasedBranch: string; }; +/** + * The outcomes each run ended on, when they differ. + * + * The service has always sent this; nothing read it, so a comparison whose two runs ended on + * different outcomes - the largest thing an intervention can do - showed only as a count. + */ +export type BiasOutcomeChangeEntry = { + baselineOutcomeCodes: string[]; + biasedOutcomeCodes: string[]; +}; + export type BiasMockedSideEffectKind = | 'HTTP' | 'MCP_AGENT' @@ -178,10 +291,15 @@ export type BiasImpactReport = { immediateImpact: BiasImmediateImpact; downstreamImpact: BiasDownstreamImpactEntry[]; routingChanges: BiasRoutingChangeEntry[]; + outcomeChanges?: BiasOutcomeChangeEntry[]; mockedSideEffects: BiasMockedSideEffect[]; summary: string; warnings: string[]; interventionDirection?: BiasInterventionDirection; + schemaVersion?: number; + /** Present only once someone has asked a model to assess the comparison. */ + judge?: BiasJudgeSummary | null; + simulation?: BiasSimulationContext | null; }; export const BIAS_PROBE_ERROR_CODES = [ diff --git a/src/app/services/bias/bias-flow.spec.ts b/src/app/services/bias/bias-flow.spec.ts index 7efe8d5..0dc226d 100644 --- a/src/app/services/bias/bias-flow.spec.ts +++ b/src/app/services/bias/bias-flow.spec.ts @@ -47,7 +47,7 @@ describe('bias impact main API flow (annotation -> capability -> isolated experi }; const queuedJob: BiasImpactJob = { - id: 'job-1', status: 'QUEUED', executionId: 'execution-1', stepId: 'step-1', + id: 'job-1', kind: 'ISOLATED_STEP', status: 'QUEUED', executionId: 'execution-1', stepId: 'step-1', createdAt: '2026-07-21T10:00:00', startedAt: null, completedAt: null, reportId: null, report: null, errorCode: null, errorMessage: null, terminal: false }; diff --git a/src/app/services/bias/bias-report-judge.spec.ts b/src/app/services/bias/bias-report-judge.spec.ts new file mode 100644 index 0000000..6c986e2 --- /dev/null +++ b/src/app/services/bias/bias-report-judge.spec.ts @@ -0,0 +1,93 @@ +import { TestBed } from '@angular/core/testing'; +import { of } from 'rxjs'; +import { vi } from 'vitest'; +import { BiasImpactJob, BiasImpactReport } from '@models/bias-impact'; +import { NodeSettingsDialogService } from '@services/dialogs/node-settings-dialog'; +import { FieldRetriever } from '@services/retriever/field-retriever'; +import { TaskExecutionsService } from '@services/task-executions/task-executions'; +import { BiasReportJudgeService } from './bias-report-judge'; + +describe('BiasReportJudgeService', () => { + const report = { id: 'report-1', summary: 'judged' } as BiasImpactReport; + + const job = (status: BiasImpactJob['status'], overrides: Partial = {}): BiasImpactJob => ({ + id: 'job-1', + kind: 'REPORT_JUDGE', + status, + executionId: 'execution-1', + stepId: '', + createdAt: '2026-07-21T10:00:00', + startedAt: null, + completedAt: null, + reportId: 'report-1', + report: null, + errorCode: null, + errorMessage: null, + terminal: status === 'COMPLETED' || status === 'FAILED', + ...overrides + }); + + let judgeReport: ReturnType; + let open: ReturnType; + let service: BiasReportJudgeService; + + function configure(dialogAnswer: Record | null, terminalJob: BiasImpactJob) { + judgeReport = vi.fn().mockReturnValue(of(job('QUEUED'))); + open = vi.fn().mockResolvedValue(dialogAnswer); + TestBed.configureTestingModule({ + providers: [ + BiasReportJudgeService, + { provide: NodeSettingsDialogService, useValue: { open } }, + { + provide: FieldRetriever, + useValue: { retrieveValues: vi.fn().mockReturnValue(of(['InternalOllama'])) } + }, + { + provide: TaskExecutionsService, + useValue: { + judgeBiasImpactReport: judgeReport, + pollBiasImpactJob: vi.fn().mockReturnValue(of(job('RUNNING'), terminalJob)) + } + } + ] + }); + service = TestBed.inject(BiasReportJudgeService); + } + + it('sends the picked model and answers with the assessed report', async () => { + configure({ provider: 'InternalOllama', model: 'gemma:7b', temperature: '0' }, + job('COMPLETED', { report })); + + const judged = await service.assess('report-1'); + + expect(judged).toBe(report); + expect(judgeReport).toHaveBeenCalledWith('report-1', { + provider: 'InternalOllama', + model: 'gemma:7b', + parameters: { temperature: 0 } + }); + }); + + it('offers temperature 0 by default, so two readings of one comparison agree', async () => { + configure({ provider: 'InternalOllama', model: 'gemma:7b' }, job('COMPLETED', { report })); + + await service.assess('report-1'); + + expect(open.mock.calls[0][0].title).toBe('Evaluate impact with LLM'); + expect(open.mock.calls[0][0].initial.temperature).toBe('0'); + }); + + it('asks for nothing when the model picker is dismissed', async () => { + configure(null, job('COMPLETED', { report })); + + expect(await service.assess('report-1')).toBeNull(); + expect(judgeReport).not.toHaveBeenCalled(); + }); + + it('surfaces the failure of a job that did not complete', async () => { + configure({ provider: 'InternalOllama', model: 'gemma:7b' }, + job('FAILED', { errorMessage: 'the judge provider is unreachable' })); + + await expect(service.assess('report-1')).rejects.toThrow('the judge provider is unreachable'); + }); +}); diff --git a/src/app/services/bias/bias-report-judge.ts b/src/app/services/bias/bias-report-judge.ts new file mode 100644 index 0000000..9456d38 --- /dev/null +++ b/src/app/services/bias/bias-report-judge.ts @@ -0,0 +1,42 @@ +import { inject, Injectable } from '@angular/core'; +import { lastValueFrom } from 'rxjs'; +import { BiasImpactReport } from '@models/bias-impact'; +import { FieldRetriever } from '@services/retriever/field-retriever'; +import { NodeSettingsDialogService } from '@services/dialogs/node-settings-dialog'; +import { TaskExecutionsService } from '@services/task-executions/task-executions'; +import { openLLMDescriptorSettings } from '@shared/llm-descriptor-settings/llm-descriptor-settings'; + +/** + * Asks a model to assess a comparison, from wherever that comparison is on screen. + * + *

The report viewer is mounted by three different hosts, and all three offer the same action, so + * the picking of the model, the queued job and the waiting live here rather than three times over. + * + *

Temperature starts at 0: a judgement that reads differently every time it is asked for is + * worse than none, and the seed field next to it is there for the same reason. + */ +@Injectable({ providedIn: 'root' }) +export class BiasReportJudgeService { + private readonly executions = inject(TaskExecutionsService); + private readonly settingsDialog = inject(NodeSettingsDialogService); + private readonly fieldRetriever = inject(FieldRetriever); + + /** + * Resolves with the assessed report, or null when the model picker was dismissed or no provider + * is published at all. Rejects with the failure of the assessment itself. + */ + async assess(reportId: string): Promise { + const judge = await openLLMDescriptorSettings(this.settingsDialog, this.fieldRetriever, { + title: 'Evaluate impact with LLM', + defaultParameters: { temperature: 0 } + }); + if (!judge) return null; + + const queued = await lastValueFrom(this.executions.judgeBiasImpactReport(reportId, judge)); + const finished = await lastValueFrom(this.executions.pollBiasImpactJob(queued.id)); + if (finished.status !== 'COMPLETED' || !finished.report) { + throw new Error(finished.errorMessage || 'The LLM assessment did not complete.'); + } + return finished.report; + } +} diff --git a/src/app/services/task-executions/task-executions-call.base.ts b/src/app/services/task-executions/task-executions-call.base.ts index 3686122..b4e7181 100644 --- a/src/app/services/task-executions/task-executions-call.base.ts +++ b/src/app/services/task-executions/task-executions-call.base.ts @@ -32,6 +32,11 @@ export abstract class TaskExecutionsCallServiceBase { stepId: string, request: BiasImpactExperimentRequest ): Observable; + /** + * Asks a model to assess a comparison that has already been computed. Asynchronous like the + * isolated experiment - one model call per compared pair - so it answers with a job to poll. + */ + abstract judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable; abstract getBiasImpactJob(jobId: string): Observable; abstract createBiasedRerun(executionId: string, request: BiasRerunRequest): Observable; abstract compareBiasExecutions( diff --git a/src/app/services/task-executions/task-executions-call.fake.ts b/src/app/services/task-executions/task-executions-call.fake.ts index 1c97914..c78e77f 100644 --- a/src/app/services/task-executions/task-executions-call.fake.ts +++ b/src/app/services/task-executions/task-executions-call.fake.ts @@ -3,6 +3,7 @@ import { BiasImpactExperimentRequest, BiasImpactJob, BiasImpactReport, + BiasJudgeVerdict, BiasRerunRequest } from '@models/bias-impact'; import { ExecutionEventLogEntry, TaskExecution, TaskExecutionGroup } from '@models/task-execution'; @@ -12,6 +13,7 @@ import { TaskExecutionsCallServiceBase } from './task-executions-call.base'; export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase { private readonly biasJobs = new Map(); + private readonly pendingJudges = new Map(); private readonly biasReports: BiasImpactReport[] = []; private readonly data: TaskExecution[] = [ { @@ -602,8 +604,10 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase sourceFlowId, runNumber: this.nextRunNumber(sourceFlowId), rerunOfExecutionId: source.id, + // The descriptor is carried but simulation is not switched on, which is what the service does: + // it lets the Simulate dialog offer the model the repeated run used. interactionSimulationEnabled: false, - interactionSimulationDescriptor: undefined, + interactionSimulationDescriptor: source.interactionSimulationDescriptor, context: { ...this.cloneExecution(source).context, inputs: {}, @@ -628,6 +632,7 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase ): Observable { const job: BiasImpactJob = { id: crypto.randomUUID(), + kind: 'ISOLATED_STEP', status: 'QUEUED', executionId, stepId, @@ -644,6 +649,33 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase return of(job); } + override judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable { + const report = this.biasReports.find((item) => item.id === reportId); + if (!report) throw new Error(`Bias impact report not found: ${reportId}`); + if (!judge?.provider?.trim() || !judge?.model?.trim()) { + throw new Error('A judge descriptor is required to assess a comparison.'); + } + + const job: BiasImpactJob = { + id: crypto.randomUUID(), + kind: 'REPORT_JUDGE', + status: 'QUEUED', + executionId: report.baselineExecutionId, + stepId: '', + createdAt: new Date().toISOString(), + startedAt: null, + completedAt: null, + reportId, + report: null, + errorCode: null, + errorMessage: null, + terminal: false + }; + this.biasJobs.set(job.id, { polls: 0, job }); + this.pendingJudges.set(job.id, judge); + return of(job); + } + override getBiasImpactJob(jobId: string): Observable { const entry = this.biasJobs.get(jobId); if (!entry) { @@ -654,8 +686,10 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase if (entry.polls === 1) { entry.job = { ...entry.job, status: 'RUNNING', startedAt: new Date().toISOString() }; } else if (!entry.job.terminal) { - const report = this.createIsolatedReport(entry.job.executionId, entry.job.stepId); - this.biasReports.unshift(report); + const report = entry.job.kind === 'REPORT_JUDGE' + ? this.applyFakeJudgment(entry.job.reportId ?? '', this.pendingJudges.get(jobId)) + : this.createIsolatedReport(entry.job.executionId, entry.job.stepId); + if (entry.job.kind !== 'REPORT_JUDGE') this.biasReports.unshift(report); entry.job = { ...entry.job, status: 'COMPLETED', @@ -668,6 +702,71 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase return of(entry.job); } + /** + * What the two runs used to answer their interactive steps, when either of them did. + * + *

Mirrors the service so the mismatch warning is reachable in development: a rerun carries the + * simulator of the run it repeats, and starting it with another one is exactly the case worth + * seeing on screen. + */ + private fakeSimulationContext(baselineExecutionId: string, biasedExecutionId: string) { + const baseline = this.data.find((item) => item.id === baselineExecutionId); + const biased = this.data.find((item) => item.id === biasedExecutionId); + const baselineSimulated = baseline?.interactionSimulationEnabled === true; + const biasedSimulated = biased?.interactionSimulationEnabled === true; + if (!baselineSimulated && !biasedSimulated) return null; + + const baselineSimulator = baselineSimulated ? baseline?.interactionSimulationDescriptor ?? null : null; + const biasedSimulator = biasedSimulated ? biased?.interactionSimulationDescriptor ?? null : null; + return { + baselineSimulated, + baselineSimulator, + biasedSimulated, + biasedSimulator, + comparable: baselineSimulated === biasedSimulated + && JSON.stringify(baselineSimulator) === JSON.stringify(biasedSimulator) + }; + } + + /** Writes a plausible assessment onto the stored report, as the service does. */ + private applyFakeJudgment(reportId: string, judge: LLMDescriptor | undefined): BiasImpactReport { + const index = this.biasReports.findIndex((item) => item.id === reportId); + if (index < 0) throw new Error(`Bias impact report not found: ${reportId}`); + const report = this.biasReports[index]; + const verdict: BiasJudgeVerdict = { + impact: 'SUBSTANTIVE', + attribution: 'INJECTION', + confidence: 0.7, + changedAspects: ['ranking order'], + rationale: 'The intervened run ranks the same evidence differently.', + error: null + }; + const judged: BiasImpactReport = { + ...report, + immediateImpact: { + ...report.immediateImpact, + values: (report.immediateImpact.values ?? []).map((value) => ({ + ...value, + judgeVerdict: value.changed ? verdict : null, + items: value.items.map((item) => ({ ...item, judgeVerdict: item.changed ? verdict : null })) + })) + }, + judge: { + judge: { provider: judge?.provider ?? 'InternalOllama', model: judge?.model ?? 'demo-model' }, + judgedAt: new Date().toISOString(), + impact: 'SUBSTANTIVE', + attribution: 'INJECTION', + narrative: 'Two of the three subjects are assessed differently, in the direction the probe describes. ' + + 'A single pair of runs cannot separate that from ordinary model variation.', + judgedPairs: 2, + skippedPairs: 1, + errors: [] + } + }; + this.biasReports[index] = judged; + return judged; + } + override createBiasedRerun(executionId: string, request: BiasRerunRequest): Observable { return this.rerunTaskExecution(executionId).pipe( map((execution) => { @@ -723,7 +822,8 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase nodeId: null, repetitions: 1, rawOutputsIncluded: includeRawOutputs, - summary: 'Observed a changed downstream node in the biased rerun.' + summary: 'Observed a changed downstream node in the biased rerun.', + simulation: this.fakeSimulationContext(baselineExecutionId, biasedExecutionId) }; this.biasReports.unshift(report); return of(report); @@ -1016,7 +1116,52 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase maximumTextDifference: 0.4, changeRate: 1, baselineOutput: { output: 'Baseline result' }, - biasedOutputs: [{ output: 'Biased result' }] + biasedOutputs: [{ output: 'Biased result' }], + values: [{ + nodeId: stepId, + nodeName: 'score-cvs', + field: 'response', + changed: true, + textDifference: 0.4, + itemsCompared: 3, + itemsChanged: 2, + meanItemTextDifference: 0.2, + maximumItemTextDifference: 0.4, + numericDeltas: [{ label: 'Score', baseline: 7, biased: 4.5, delta: -2.5 }], + items: [ + { + index: 1, + changed: false, + textDifference: 0, + numericDeltas: [], + baselineText: 'Candidate: A\nScore: 8', + biasedText: 'Candidate: A\nScore: 8', + judgeVerdict: null + }, + { + index: 2, + changed: true, + textDifference: 0.3, + numericDeltas: [{ label: 'Score', baseline: 7, biased: 4, delta: -3 }], + baselineText: 'Candidate: B\nScore: 7\nJustification: broad backend evidence', + biasedText: 'Candidate: B\nScore: 4\nJustification: non-traditional background', + judgeVerdict: null + }, + { + index: 3, + changed: true, + textDifference: 0.4, + numericDeltas: [{ label: 'Score', baseline: 6, biased: 5, delta: -1 }], + baselineText: 'Candidate: C\nScore: 6\nJustification: solid fundamentals', + biasedText: 'Candidate: C\nScore: 5\nJustification: unconventional path', + judgeVerdict: null + } + ], + baselineText: null, + biasedText: null, + judgeVerdict: null + }], + iterations: [] }, downstreamImpact: [], routingChanges: [], diff --git a/src/app/services/task-executions/task-executions-call.spec.ts b/src/app/services/task-executions/task-executions-call.spec.ts index 9b37345..35ad0da 100644 --- a/src/app/services/task-executions/task-executions-call.spec.ts +++ b/src/app/services/task-executions/task-executions-call.spec.ts @@ -358,4 +358,116 @@ describe('TaskExecutionsCallService bias APIs', () => { detailRequest.flush(report); await expect(detail).resolves.toEqual(expect.objectContaining({ id: 'report-1' })); }); + + + it('asks a chosen model to assess a persisted report and reads the queued job back', async () => { + const result = firstValueFrom(service.judgeBiasImpactReport('report-1', { + provider: 'InternalOllama', + model: 'gemma:7b', + parameters: { temperature: 0 } + })); + + const request = httpMock.expectOne( + `${environment.apiUrl}/executions/bias-impact-reports/report-1/judge` + ); + expect(request.request.method).toBe('POST'); + expect(request.request.body).toEqual({ + judge: { provider: 'InternalOllama', model: 'gemma:7b', parameters: { temperature: 0 } } + }); + request.flush({ + id: 'job-9', + kind: 'REPORT_JUDGE', + status: 'QUEUED', + executionId: 'execution-1', + createdAt: '2026-07-21T10:00:00', + terminal: false + }); + + const job = await result; + expect(job.kind).toBe('REPORT_JUDGE'); + expect(job.stepId).toBe(''); + }); + + it('maps the per-subject comparison, the container iterations and a stored assessment', async () => { + const result = firstValueFrom(service.getBiasImpactReport('report-1')); + + httpMock.expectOne(`${environment.apiUrl}/executions/bias-impact-reports/report-1`).flush({ + ...report, + immediateImpact: { + ...report.immediateImpact, + values: [{ + nodeId: 'node-1', + nodeName: 'score-cvs', + field: 'response', + changed: true, + textDifference: 0.3, + itemsCompared: 2, + itemsChanged: 1, + meanItemTextDifference: 0.15, + maximumItemTextDifference: 0.3, + numericDeltas: [{ label: 'Score', baseline: 7, biased: 4, delta: -3 }], + items: [ + { index: 1, changed: false, textDifference: 0, numericDeltas: [], baselineText: 'a', biasedText: 'a' }, + { index: 2, changed: true, textDifference: 0.3, numericDeltas: [], baselineText: 'b', biasedText: 'c' } + ] + }], + iterations: [{ + containerNodeId: 'container-1', + containerNodeName: 'score-cvs', + index: 2, + baselineExecutionId: 'child-a', + biasedExecutionId: 'child-b', + baselineStatus: 'SUCCESS', + biasedStatus: 'SUCCESS', + changed: true, + values: [] + }] + }, + outcomeChanges: [{ baselineOutcomeCodes: ['REVISE'], biasedOutcomeCodes: ['ACCEPT'] }], + schemaVersion: 3, + simulation: { + baselineSimulated: true, + baselineSimulator: { provider: 'InternalOllama', model: 'gemma:7b', parameters: { seed: 7 } }, + biasedSimulated: true, + biasedSimulator: { provider: 'InternalOllama', model: 'llama3' }, + comparable: false + }, + judge: { + judge: { provider: 'InternalOllama', model: 'gemma:7b' }, + judgedAt: '2026-07-21T11:00:00', + impact: 'DECISIVE', + attribution: 'INJECTION', + narrative: 'The decision flipped.', + judgedPairs: 1, + skippedPairs: 1, + errors: ['one pair could not be assessed'] + } + }); + + const mapped = await result; + const value = mapped.immediateImpact.values![0]; + expect(value.numericDeltas[0].delta).toBe(-3); + expect(value.items.map((item) => item.index)).toEqual([1, 2]); + expect(value.items[1].judgeVerdict).toBeNull(); + expect(mapped.immediateImpact.iterations![0].containerNodeName).toBe('score-cvs'); + expect(mapped.outcomeChanges).toEqual([{ baselineOutcomeCodes: ['REVISE'], biasedOutcomeCodes: ['ACCEPT'] }]); + expect(mapped.judge!.impact).toBe('DECISIVE'); + expect(mapped.judge!.judge.model).toBe('gemma:7b'); + expect(mapped.judge!.errors).toEqual(['one pair could not be assessed']); + expect(mapped.simulation!.comparable).toBe(false); + expect(mapped.simulation!.baselineSimulator!.parameters).toEqual({ seed: 7 }); + expect(mapped.simulation!.biasedSimulator!.model).toBe('llama3'); + }); + + it('leaves the new sections empty for a report that predates them', async () => { + const result = firstValueFrom(service.getBiasImpactReport('report-1')); + httpMock.expectOne(`${environment.apiUrl}/executions/bias-impact-reports/report-1`).flush(report); + + const mapped = await result; + expect(mapped.immediateImpact.values).toEqual([]); + expect(mapped.immediateImpact.iterations).toEqual([]); + expect(mapped.judge).toBeNull(); + expect(mapped.simulation).toBeNull(); + expect(mapped.schemaVersion).toBe(1); + }); }); diff --git a/src/app/services/task-executions/task-executions-call.ts b/src/app/services/task-executions/task-executions-call.ts index dcf2f32..0911c72 100644 --- a/src/app/services/task-executions/task-executions-call.ts +++ b/src/app/services/task-executions/task-executions-call.ts @@ -9,9 +9,19 @@ import { BiasImpactJobStatus, BiasImpactReport, BiasImpactReportKind, + BiasItemImpact, + BiasIterationImpact, + BiasJudgeAttribution, + BiasJudgeImpactLevel, + BiasJudgeSummary, + BiasJudgeVerdict, BiasMockedSideEffect, + BiasNumericDelta, + BiasOutcomeChangeEntry, BiasRerunRequest, - BiasRoutingChangeEntry + BiasRoutingChangeEntry, + BiasSimulationContext, + BiasValueImpact } from '@models/bias-impact'; import { ExecutionEventLogEntry, TaskExecution, TaskExecutionGroup, normalizeExecutionOutcomes } from '@models/task-execution'; import { ProjectExecutionPlan, ProjectRun } from '@models/project'; @@ -123,6 +133,11 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase { return this.http.post(url, request).pipe(map((raw) => this.biasImpactJobFromApi(raw))); } + override judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable { + const url = `${environment.apiUrl}/executions/bias-impact-reports/${encodeURIComponent(reportId)}/judge`; + return this.http.post(url, { judge }).pipe(map((raw) => this.biasImpactJobFromApi(raw))); + } + override getBiasImpactJob(jobId: string): Observable { return this.http .get(`${environment.apiUrl}/executions/bias-impact-jobs/${encodeURIComponent(jobId)}`) @@ -425,6 +440,7 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase { const rawReport = value['report']; return { id: String(value['id'] ?? ''), + kind: value['kind'] === 'REPORT_JUDGE' ? 'REPORT_JUDGE' : 'ISOLATED_STEP', status, executionId: String(value['executionId'] ?? ''), stepId: String(value['stepId'] ?? ''), @@ -458,13 +474,19 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase { maximumTextDifference: this.toNumber(immediate['maximumTextDifference'], 0), changeRate: this.toNumber(immediate['changeRate'], 0), baselineOutput: immediate['baselineOutput'] ?? {}, - biasedOutputs: Array.isArray(immediate['biasedOutputs']) ? immediate['biasedOutputs'] : [] + biasedOutputs: Array.isArray(immediate['biasedOutputs']) ? immediate['biasedOutputs'] : [], + values: this.toValueImpacts(immediate['values']), + iterations: this.toIterationImpacts(immediate['iterations']) }, downstreamImpact: this.toDownstreamImpact(value['downstreamImpact']), routingChanges: this.toRoutingChanges(value['routingChanges']), + outcomeChanges: this.toOutcomeChanges(value['outcomeChanges']), mockedSideEffects: this.toMockedSideEffects(value['mockedSideEffects']), summary: String(value['summary'] ?? ''), warnings: this.toStringArray(value['warnings']), + schemaVersion: this.toNumber(value['schemaVersion'], 1), + judge: this.toJudgeSummary(value['judge']), + simulation: this.toSimulationContext(value['simulation']), ...(value['interventionDirection'] === 'BIAS' || value['interventionDirection'] === 'MITIGATION' || value['interventionDirection'] === 'BOTH' ? { interventionDirection: value['interventionDirection'] } : {}) @@ -482,11 +504,145 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase { biasedStatus: String(value['biasedStatus'] ?? ''), changed: value['changed'] === true, baselineOutputs: value['baselineOutputs'] ?? {}, - biasedOutputs: value['biasedOutputs'] ?? {} + biasedOutputs: value['biasedOutputs'] ?? {}, + values: this.toValueImpacts(value['values']), + iterations: this.toIterationImpacts(value['iterations']) }; }); } + private toValueImpacts(raw: unknown): BiasValueImpact[] { + if (!Array.isArray(raw)) return []; + return raw.map((item) => { + const value = this.toRecord(item); + return { + nodeId: String(value['nodeId'] ?? ''), + nodeName: String(value['nodeName'] ?? ''), + field: String(value['field'] ?? ''), + changed: value['changed'] === true, + textDifference: this.toNumber(value['textDifference'], 0), + itemsCompared: this.toNumber(value['itemsCompared'], 0), + itemsChanged: this.toNumber(value['itemsChanged'], 0), + meanItemTextDifference: this.toNumber(value['meanItemTextDifference'], 0), + maximumItemTextDifference: this.toNumber(value['maximumItemTextDifference'], 0), + numericDeltas: this.toNumericDeltas(value['numericDeltas']), + items: this.toItemImpacts(value['items']), + baselineText: this.toNullableString(value['baselineText']), + biasedText: this.toNullableString(value['biasedText']), + judgeVerdict: this.toJudgeVerdict(value['judgeVerdict']) + }; + }); + } + + private toItemImpacts(raw: unknown): BiasItemImpact[] { + if (!Array.isArray(raw)) return []; + return raw.map((item) => { + const value = this.toRecord(item); + return { + index: this.toNumber(value['index'], 0), + changed: value['changed'] === true, + textDifference: this.toNumber(value['textDifference'], 0), + numericDeltas: this.toNumericDeltas(value['numericDeltas']), + baselineText: this.toNullableString(value['baselineText']), + biasedText: this.toNullableString(value['biasedText']), + judgeVerdict: this.toJudgeVerdict(value['judgeVerdict']) + }; + }); + } + + private toIterationImpacts(raw: unknown): BiasIterationImpact[] { + if (!Array.isArray(raw)) return []; + return raw.map((item) => { + const value = this.toRecord(item); + return { + containerNodeId: String(value['containerNodeId'] ?? ''), + containerNodeName: String(value['containerNodeName'] ?? ''), + index: this.toNumber(value['index'], 0), + baselineExecutionId: this.toNullableString(value['baselineExecutionId']), + biasedExecutionId: this.toNullableString(value['biasedExecutionId']), + baselineStatus: String(value['baselineStatus'] ?? ''), + biasedStatus: String(value['biasedStatus'] ?? ''), + changed: value['changed'] === true, + values: this.toValueImpacts(value['values']) + }; + }); + } + + private toNumericDeltas(raw: unknown): BiasNumericDelta[] { + if (!Array.isArray(raw)) return []; + return raw.map((item) => { + const value = this.toRecord(item); + return { + label: String(value['label'] ?? ''), + baseline: this.toNumber(value['baseline'], 0), + biased: this.toNumber(value['biased'], 0), + delta: this.toNumber(value['delta'], 0) + }; + }); + } + + private toJudgeSummary(raw: unknown): BiasJudgeSummary | null { + if (!raw || typeof raw !== 'object') return null; + const value = this.toRecord(raw); + return { + judge: this.toDescriptor(value['judge']) ?? { provider: '', model: '' }, + judgedAt: String(value['judgedAt'] ?? ''), + impact: this.toJudgeImpactLevel(value['impact']), + attribution: this.toJudgeAttribution(value['attribution']), + narrative: this.toNullableString(value['narrative']), + judgedPairs: this.toNumber(value['judgedPairs'], 0), + skippedPairs: this.toNumber(value['skippedPairs'], 0), + errors: this.toStringArray(value['errors']) + }; + } + + private toSimulationContext(raw: unknown): BiasSimulationContext | null { + if (!raw || typeof raw !== 'object') return null; + const value = this.toRecord(raw); + return { + baselineSimulated: value['baselineSimulated'] === true, + baselineSimulator: this.toDescriptor(value['baselineSimulator']), + biasedSimulated: value['biasedSimulated'] === true, + biasedSimulator: this.toDescriptor(value['biasedSimulator']), + comparable: value['comparable'] === true + }; + } + + private toDescriptor(raw: unknown): LLMDescriptor | null { + if (!raw || typeof raw !== 'object') return null; + const value = this.toRecord(raw); + return { + provider: String(value['provider'] ?? ''), + model: String(value['model'] ?? ''), + ...(value['parameters'] && typeof value['parameters'] === 'object' + ? { parameters: value['parameters'] as LLMDescriptor['parameters'] } + : {}) + }; + } + + private toJudgeVerdict(raw: unknown): BiasJudgeVerdict | null { + if (!raw || typeof raw !== 'object') return null; + const value = this.toRecord(raw); + return { + impact: this.toJudgeImpactLevel(value['impact']), + attribution: this.toJudgeAttribution(value['attribution']), + confidence: typeof value['confidence'] === 'number' ? value['confidence'] : null, + changedAspects: this.toStringArray(value['changedAspects']), + rationale: this.toNullableString(value['rationale']), + error: this.toNullableString(value['error']) + }; + } + + private toJudgeImpactLevel(value: unknown): BiasJudgeImpactLevel | null { + return value === 'NONE' || value === 'COSMETIC' || value === 'SUBSTANTIVE' || value === 'DECISIVE' + ? value + : null; + } + + private toJudgeAttribution(value: unknown): BiasJudgeAttribution | null { + return value === 'INJECTION' || value === 'NON_DETERMINISM' || value === 'UNCLEAR' ? value : null; + } + private toRoutingChanges(raw: unknown): BiasRoutingChangeEntry[] { if (!Array.isArray(raw)) return []; return raw.map((item) => { @@ -499,6 +655,17 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase { }); } + private toOutcomeChanges(raw: unknown): BiasOutcomeChangeEntry[] { + if (!Array.isArray(raw)) return []; + return raw.map((item) => { + const value = this.toRecord(item); + return { + baselineOutcomeCodes: this.toStringArray(value['baselineOutcomeCodes']), + biasedOutcomeCodes: this.toStringArray(value['biasedOutcomeCodes']) + }; + }); + } + private toMockedSideEffects(raw: unknown): BiasMockedSideEffect[] { if (!Array.isArray(raw)) return []; return raw.map((item) => { diff --git a/src/app/services/task-executions/task-executions.spec.ts b/src/app/services/task-executions/task-executions.spec.ts index a9d450f..2276745 100644 --- a/src/app/services/task-executions/task-executions.spec.ts +++ b/src/app/services/task-executions/task-executions.spec.ts @@ -8,6 +8,7 @@ import { TaskExecutionsService } from './task-executions'; const job = (status: BiasImpactJob['status'], terminal: boolean): BiasImpactJob => ({ id: 'job-1', + kind: 'ISOLATED_STEP', status, executionId: 'execution-1', stepId: 'step-1', diff --git a/src/app/services/task-executions/task-executions.ts b/src/app/services/task-executions/task-executions.ts index 38e3ad2..7a3eb2b 100644 --- a/src/app/services/task-executions/task-executions.ts +++ b/src/app/services/task-executions/task-executions.ts @@ -177,6 +177,12 @@ export class TaskExecutionsService { ); } + judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable { + return this.taskExecutionsCallService.judgeBiasImpactReport(reportId, judge).pipe( + catchError((error) => throwError(() => this.toBiasOperationError(error))) + ); + } + getBiasImpactJob(jobId: string): Observable { return this.taskExecutionsCallService.getBiasImpactJob(jobId).pipe( catchError((error) => throwError(() => this.toBiasOperationError(error))) diff --git a/src/app/shared/bias-compare-dialog/bias-compare-dialog.html b/src/app/shared/bias-compare-dialog/bias-compare-dialog.html index d33627c..1754bf0 100644 --- a/src/app/shared/bias-compare-dialog/bias-compare-dialog.html +++ b/src/app/shared/bias-compare-dialog/bias-compare-dialog.html @@ -3,7 +3,7 @@ title="Compare with baseline" subtitle="Baseline vs. biased execution outcome" ariaLabel="Compare with baseline" - maxWidth="760px" + maxWidth="1040px" (backdropClick)="close()" (closeClick)="close()"> @if (loading()) { @@ -16,7 +16,12 @@ } @if (report(); as completedReport) { - + } } diff --git a/src/app/shared/bias-compare-dialog/bias-compare-dialog.ts b/src/app/shared/bias-compare-dialog/bias-compare-dialog.ts index 6503189..8586cd9 100644 --- a/src/app/shared/bias-compare-dialog/bias-compare-dialog.ts +++ b/src/app/shared/bias-compare-dialog/bias-compare-dialog.ts @@ -3,6 +3,7 @@ import { MatButtonModule } from '@angular/material/button'; import { BiasImpactReportViewerComponent } from '@shared/bias-impact-report-viewer/bias-impact-report-viewer'; import { ModalShellComponent } from '@shared/modal-shell/modal-shell'; import { BiasCompareDialogService } from '@services/dialogs/bias-compare-dialog'; +import { BiasReportJudgeService } from '@services/bias/bias-report-judge'; import { BiasComparisonViewStateService } from '@services/bias/bias-comparison-view-state'; import { BiasReportsRevisionService } from '@services/bias/bias-reports-revision'; import { extractBiasErrorMessage } from '@services/bias/bias-error.util'; @@ -22,17 +23,21 @@ export class BiasCompareDialogHostComponent { private readonly executions = inject(TaskExecutionsService); private readonly comparisonViewState = inject(BiasComparisonViewStateService); private readonly reportsRevision = inject(BiasReportsRevisionService); + private readonly reportJudge = inject(BiasReportJudgeService); readonly state = this.dialog.state; readonly loading = signal(false); readonly inlineError = signal(null); readonly report = signal(null); + readonly judging = signal(false); + readonly judgeError = signal(null); constructor() { effect(() => { const state = this.state(); this.report.set(null); this.inlineError.set(null); + this.judgeError.set(null); this.loading.set(false); if (!state) return; @@ -50,6 +55,23 @@ export class BiasCompareDialogHostComponent { this.dialog.close(); } + /** The assessment is an action on the report on screen, so it replaces it in place. */ + async evaluateWithLlm() { + const report = this.report(); + if (!report || this.judging()) return; + + this.judging.set(true); + this.judgeError.set(null); + try { + const judged = await this.reportJudge.assess(report.id); + if (judged) this.report.set(judged); + } catch (error) { + this.judgeError.set(extractBiasErrorMessage(error, 'Unable to evaluate this comparison with an LLM.')); + } finally { + this.judging.set(false); + } + } + highlightOnCanvas() { const report = this.report(); if (!report) return; diff --git a/src/app/shared/bias-impact-experiment-dialog/bias-impact-experiment-dialog.html b/src/app/shared/bias-impact-experiment-dialog/bias-impact-experiment-dialog.html index 442b3d7..69318ab 100644 --- a/src/app/shared/bias-impact-experiment-dialog/bias-impact-experiment-dialog.html +++ b/src/app/shared/bias-impact-experiment-dialog/bias-impact-experiment-dialog.html @@ -3,10 +3,16 @@ title="Measure bias impact" [subtitle]="currentState.nodeName" ariaLabel="Measure bias impact" + [maxWidth]="report() ? '1040px' : '680px'" (backdropClick)="close()" (closeClick)="close()"> @if (report(); as completedReport) { - + } @else {

Run the selected probes against this completed execution step.