diff --git a/src/app/models/bias-impact.spec.ts b/src/app/models/bias-impact.spec.ts index 6645d90..d2d78c2 100644 --- a/src/app/models/bias-impact.spec.ts +++ b/src/app/models/bias-impact.spec.ts @@ -53,6 +53,7 @@ describe('bias impact models', () => { }; const job: BiasImpactJob = { id: 'job-1', + kind: 'ISOLATED_STEP', status: 'COMPLETED', executionId: 'baseline-1', stepId: 'node-1', diff --git a/src/app/models/bias-impact.ts b/src/app/models/bias-impact.ts index 9ab63b5..d4b95ff 100644 --- a/src/app/models/bias-impact.ts +++ b/src/app/models/bias-impact.ts @@ -1,4 +1,4 @@ -import { BiasActivationMode } from './flow'; +import { BiasActivationMode, LLMDescriptor } from './flow'; export type BiasInterventionDirection = 'BIAS' | 'MITIGATION' | 'BOTH'; @@ -111,10 +111,15 @@ export function activeAnnotationIdsFor( export type BiasImpactJobStatus = 'QUEUED' | 'RUNNING' | 'COMPLETED' | 'FAILED'; +/** Both kinds of queued bias work: re-running a step, and assessing a comparison with an LLM. */ +export type BiasImpactJobKind = 'ISOLATED_STEP' | 'REPORT_JUDGE'; + export type BiasImpactJob = { id: string; + kind: BiasImpactJobKind; status: BiasImpactJobStatus; executionId: string; + /** Empty for a judge job: it is about a whole comparison, not one step. */ stepId: string; createdAt: string; startedAt: string | null; @@ -128,12 +133,107 @@ export type BiasImpactJob = { export type BiasImpactReportKind = 'ISOLATED_STEP' | 'FULL_FLOW'; +/** A number both runs labelled the same way, and how far the intervention moved it. */ +export type BiasNumericDelta = { + label: string; + baseline: number; + biased: number; + delta: number; +}; + +export type BiasJudgeImpactLevel = 'NONE' | 'COSMETIC' | 'SUBSTANTIVE' | 'DECISIVE'; + +export type BiasJudgeAttribution = 'INJECTION' | 'NON_DETERMINISM' | 'UNCLEAR'; + +/** + * What a judge model made of one compared pair. An assessment, not a measurement: `error` is set + * when the model answered with something unusable, and the pair's own figures stand either way. + */ +export type BiasJudgeVerdict = { + impact: BiasJudgeImpactLevel | null; + attribution: BiasJudgeAttribution | null; + confidence: number | null; + changedAspects: string[]; + rationale: string | null; + error: string | null; +}; + +export type BiasJudgeSummary = { + judge: LLMDescriptor; + judgedAt: string; + impact: BiasJudgeImpactLevel | null; + attribution: BiasJudgeAttribution | null; + narrative: string | null; + judgedPairs: number; + skippedPairs: number; + errors: string[]; +}; + +/** One element of a compared list value, which for an iterator container is one subject. */ +export type BiasItemImpact = { + index: number; + changed: boolean; + textDifference: number; + numericDeltas: BiasNumericDelta[]; + baselineText: string | null; + biasedText: string | null; + judgeVerdict: BiasJudgeVerdict | null; +}; + +/** One output field of one node, compared between the two runs. */ +export type BiasValueImpact = { + nodeId: string; + nodeName: string; + field: string; + changed: boolean; + textDifference: number; + itemsCompared: number; + itemsChanged: number; + meanItemTextDifference: number; + maximumItemTextDifference: number; + numericDeltas: BiasNumericDelta[]; + items: BiasItemImpact[]; + baselineText: string | null; + biasedText: string | null; + judgeVerdict: BiasJudgeVerdict | null; +}; + +/** One iteration of a container step, from the pair of child executions that ran it. */ +export type BiasIterationImpact = { + containerNodeId: string; + containerNodeName: string; + index: number; + baselineExecutionId: string | null; + biasedExecutionId: string | null; + baselineStatus: string; + biasedStatus: string; + changed: boolean; + values: BiasValueImpact[]; +}; + +/** + * How the interactive steps of the two compared runs were answered. + * + * Absent when neither run was simulated, and on reports produced before it was recorded. + */ +export type BiasSimulationContext = { + baselineSimulated: boolean; + baselineSimulator: LLMDescriptor | null; + biasedSimulated: boolean; + biasedSimulator: LLMDescriptor | null; + /** False when only one side was simulated, or the two simulators are not the same one. */ + comparable: boolean; +}; + export type BiasImmediateImpact = { outputChanged: boolean; maximumTextDifference: number; changeRate: number; baselineOutput: unknown; biasedOutputs: unknown[]; + /** Absent on reports produced before the field-by-field comparison existed. */ + values?: BiasValueImpact[]; + iterations?: BiasIterationImpact[]; }; export type BiasDownstreamImpactEntry = { @@ -144,6 +244,8 @@ export type BiasDownstreamImpactEntry = { changed: boolean; baselineOutputs: unknown; biasedOutputs: unknown; + values?: BiasValueImpact[]; + iterations?: BiasIterationImpact[]; }; export type BiasRoutingChangeEntry = { @@ -152,6 +254,17 @@ export type BiasRoutingChangeEntry = { biasedBranch: string; }; +/** + * The outcomes each run ended on, when they differ. + * + * The service has always sent this; nothing read it, so a comparison whose two runs ended on + * different outcomes - the largest thing an intervention can do - showed only as a count. + */ +export type BiasOutcomeChangeEntry = { + baselineOutcomeCodes: string[]; + biasedOutcomeCodes: string[]; +}; + export type BiasMockedSideEffectKind = | 'HTTP' | 'MCP_AGENT' @@ -178,10 +291,15 @@ export type BiasImpactReport = { immediateImpact: BiasImmediateImpact; downstreamImpact: BiasDownstreamImpactEntry[]; routingChanges: BiasRoutingChangeEntry[]; + outcomeChanges?: BiasOutcomeChangeEntry[]; mockedSideEffects: BiasMockedSideEffect[]; summary: string; warnings: string[]; interventionDirection?: BiasInterventionDirection; + schemaVersion?: number; + /** Present only once someone has asked a model to assess the comparison. */ + judge?: BiasJudgeSummary | null; + simulation?: BiasSimulationContext | null; }; export const BIAS_PROBE_ERROR_CODES = [ diff --git a/src/app/services/bias/bias-flow.spec.ts b/src/app/services/bias/bias-flow.spec.ts index 7efe8d5..0dc226d 100644 --- a/src/app/services/bias/bias-flow.spec.ts +++ b/src/app/services/bias/bias-flow.spec.ts @@ -47,7 +47,7 @@ describe('bias impact main API flow (annotation -> capability -> isolated experi }; const queuedJob: BiasImpactJob = { - id: 'job-1', status: 'QUEUED', executionId: 'execution-1', stepId: 'step-1', + id: 'job-1', kind: 'ISOLATED_STEP', status: 'QUEUED', executionId: 'execution-1', stepId: 'step-1', createdAt: '2026-07-21T10:00:00', startedAt: null, completedAt: null, reportId: null, report: null, errorCode: null, errorMessage: null, terminal: false }; diff --git a/src/app/services/bias/bias-report-judge.spec.ts b/src/app/services/bias/bias-report-judge.spec.ts new file mode 100644 index 0000000..6c986e2 --- /dev/null +++ b/src/app/services/bias/bias-report-judge.spec.ts @@ -0,0 +1,93 @@ +import { TestBed } from '@angular/core/testing'; +import { of } from 'rxjs'; +import { vi } from 'vitest'; +import { BiasImpactJob, BiasImpactReport } from '@models/bias-impact'; +import { NodeSettingsDialogService } from '@services/dialogs/node-settings-dialog'; +import { FieldRetriever } from '@services/retriever/field-retriever'; +import { TaskExecutionsService } from '@services/task-executions/task-executions'; +import { BiasReportJudgeService } from './bias-report-judge'; + +describe('BiasReportJudgeService', () => { + const report = { id: 'report-1', summary: 'judged' } as BiasImpactReport; + + const job = (status: BiasImpactJob['status'], overrides: Partial = {}): BiasImpactJob => ({ + id: 'job-1', + kind: 'REPORT_JUDGE', + status, + executionId: 'execution-1', + stepId: '', + createdAt: '2026-07-21T10:00:00', + startedAt: null, + completedAt: null, + reportId: 'report-1', + report: null, + errorCode: null, + errorMessage: null, + terminal: status === 'COMPLETED' || status === 'FAILED', + ...overrides + }); + + let judgeReport: ReturnType; + let open: ReturnType; + let service: BiasReportJudgeService; + + function configure(dialogAnswer: Record | null, terminalJob: BiasImpactJob) { + judgeReport = vi.fn().mockReturnValue(of(job('QUEUED'))); + open = vi.fn().mockResolvedValue(dialogAnswer); + TestBed.configureTestingModule({ + providers: [ + BiasReportJudgeService, + { provide: NodeSettingsDialogService, useValue: { open } }, + { + provide: FieldRetriever, + useValue: { retrieveValues: vi.fn().mockReturnValue(of(['InternalOllama'])) } + }, + { + provide: TaskExecutionsService, + useValue: { + judgeBiasImpactReport: judgeReport, + pollBiasImpactJob: vi.fn().mockReturnValue(of(job('RUNNING'), terminalJob)) + } + } + ] + }); + service = TestBed.inject(BiasReportJudgeService); + } + + it('sends the picked model and answers with the assessed report', async () => { + configure({ provider: 'InternalOllama', model: 'gemma:7b', temperature: '0' }, + job('COMPLETED', { report })); + + const judged = await service.assess('report-1'); + + expect(judged).toBe(report); + expect(judgeReport).toHaveBeenCalledWith('report-1', { + provider: 'InternalOllama', + model: 'gemma:7b', + parameters: { temperature: 0 } + }); + }); + + it('offers temperature 0 by default, so two readings of one comparison agree', async () => { + configure({ provider: 'InternalOllama', model: 'gemma:7b' }, job('COMPLETED', { report })); + + await service.assess('report-1'); + + expect(open.mock.calls[0][0].title).toBe('Evaluate impact with LLM'); + expect(open.mock.calls[0][0].initial.temperature).toBe('0'); + }); + + it('asks for nothing when the model picker is dismissed', async () => { + configure(null, job('COMPLETED', { report })); + + expect(await service.assess('report-1')).toBeNull(); + expect(judgeReport).not.toHaveBeenCalled(); + }); + + it('surfaces the failure of a job that did not complete', async () => { + configure({ provider: 'InternalOllama', model: 'gemma:7b' }, + job('FAILED', { errorMessage: 'the judge provider is unreachable' })); + + await expect(service.assess('report-1')).rejects.toThrow('the judge provider is unreachable'); + }); +}); diff --git a/src/app/services/bias/bias-report-judge.ts b/src/app/services/bias/bias-report-judge.ts new file mode 100644 index 0000000..9456d38 --- /dev/null +++ b/src/app/services/bias/bias-report-judge.ts @@ -0,0 +1,42 @@ +import { inject, Injectable } from '@angular/core'; +import { lastValueFrom } from 'rxjs'; +import { BiasImpactReport } from '@models/bias-impact'; +import { FieldRetriever } from '@services/retriever/field-retriever'; +import { NodeSettingsDialogService } from '@services/dialogs/node-settings-dialog'; +import { TaskExecutionsService } from '@services/task-executions/task-executions'; +import { openLLMDescriptorSettings } from '@shared/llm-descriptor-settings/llm-descriptor-settings'; + +/** + * Asks a model to assess a comparison, from wherever that comparison is on screen. + * + *

The report viewer is mounted by three different hosts, and all three offer the same action, so + * the picking of the model, the queued job and the waiting live here rather than three times over. + * + *

Temperature starts at 0: a judgement that reads differently every time it is asked for is + * worse than none, and the seed field next to it is there for the same reason. + */ +@Injectable({ providedIn: 'root' }) +export class BiasReportJudgeService { + private readonly executions = inject(TaskExecutionsService); + private readonly settingsDialog = inject(NodeSettingsDialogService); + private readonly fieldRetriever = inject(FieldRetriever); + + /** + * Resolves with the assessed report, or null when the model picker was dismissed or no provider + * is published at all. Rejects with the failure of the assessment itself. + */ + async assess(reportId: string): Promise { + const judge = await openLLMDescriptorSettings(this.settingsDialog, this.fieldRetriever, { + title: 'Evaluate impact with LLM', + defaultParameters: { temperature: 0 } + }); + if (!judge) return null; + + const queued = await lastValueFrom(this.executions.judgeBiasImpactReport(reportId, judge)); + const finished = await lastValueFrom(this.executions.pollBiasImpactJob(queued.id)); + if (finished.status !== 'COMPLETED' || !finished.report) { + throw new Error(finished.errorMessage || 'The LLM assessment did not complete.'); + } + return finished.report; + } +} diff --git a/src/app/services/task-executions/task-executions-call.base.ts b/src/app/services/task-executions/task-executions-call.base.ts index 3686122..b4e7181 100644 --- a/src/app/services/task-executions/task-executions-call.base.ts +++ b/src/app/services/task-executions/task-executions-call.base.ts @@ -32,6 +32,11 @@ export abstract class TaskExecutionsCallServiceBase { stepId: string, request: BiasImpactExperimentRequest ): Observable; + /** + * Asks a model to assess a comparison that has already been computed. Asynchronous like the + * isolated experiment - one model call per compared pair - so it answers with a job to poll. + */ + abstract judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable; abstract getBiasImpactJob(jobId: string): Observable; abstract createBiasedRerun(executionId: string, request: BiasRerunRequest): Observable; abstract compareBiasExecutions( diff --git a/src/app/services/task-executions/task-executions-call.fake.ts b/src/app/services/task-executions/task-executions-call.fake.ts index 1c97914..c78e77f 100644 --- a/src/app/services/task-executions/task-executions-call.fake.ts +++ b/src/app/services/task-executions/task-executions-call.fake.ts @@ -3,6 +3,7 @@ import { BiasImpactExperimentRequest, BiasImpactJob, BiasImpactReport, + BiasJudgeVerdict, BiasRerunRequest } from '@models/bias-impact'; import { ExecutionEventLogEntry, TaskExecution, TaskExecutionGroup } from '@models/task-execution'; @@ -12,6 +13,7 @@ import { TaskExecutionsCallServiceBase } from './task-executions-call.base'; export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase { private readonly biasJobs = new Map(); + private readonly pendingJudges = new Map(); private readonly biasReports: BiasImpactReport[] = []; private readonly data: TaskExecution[] = [ { @@ -602,8 +604,10 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase sourceFlowId, runNumber: this.nextRunNumber(sourceFlowId), rerunOfExecutionId: source.id, + // The descriptor is carried but simulation is not switched on, which is what the service does: + // it lets the Simulate dialog offer the model the repeated run used. interactionSimulationEnabled: false, - interactionSimulationDescriptor: undefined, + interactionSimulationDescriptor: source.interactionSimulationDescriptor, context: { ...this.cloneExecution(source).context, inputs: {}, @@ -628,6 +632,7 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase ): Observable { const job: BiasImpactJob = { id: crypto.randomUUID(), + kind: 'ISOLATED_STEP', status: 'QUEUED', executionId, stepId, @@ -644,6 +649,33 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase return of(job); } + override judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable { + const report = this.biasReports.find((item) => item.id === reportId); + if (!report) throw new Error(`Bias impact report not found: ${reportId}`); + if (!judge?.provider?.trim() || !judge?.model?.trim()) { + throw new Error('A judge descriptor is required to assess a comparison.'); + } + + const job: BiasImpactJob = { + id: crypto.randomUUID(), + kind: 'REPORT_JUDGE', + status: 'QUEUED', + executionId: report.baselineExecutionId, + stepId: '', + createdAt: new Date().toISOString(), + startedAt: null, + completedAt: null, + reportId, + report: null, + errorCode: null, + errorMessage: null, + terminal: false + }; + this.biasJobs.set(job.id, { polls: 0, job }); + this.pendingJudges.set(job.id, judge); + return of(job); + } + override getBiasImpactJob(jobId: string): Observable { const entry = this.biasJobs.get(jobId); if (!entry) { @@ -654,8 +686,10 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase if (entry.polls === 1) { entry.job = { ...entry.job, status: 'RUNNING', startedAt: new Date().toISOString() }; } else if (!entry.job.terminal) { - const report = this.createIsolatedReport(entry.job.executionId, entry.job.stepId); - this.biasReports.unshift(report); + const report = entry.job.kind === 'REPORT_JUDGE' + ? this.applyFakeJudgment(entry.job.reportId ?? '', this.pendingJudges.get(jobId)) + : this.createIsolatedReport(entry.job.executionId, entry.job.stepId); + if (entry.job.kind !== 'REPORT_JUDGE') this.biasReports.unshift(report); entry.job = { ...entry.job, status: 'COMPLETED', @@ -668,6 +702,71 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase return of(entry.job); } + /** + * What the two runs used to answer their interactive steps, when either of them did. + * + *

Mirrors the service so the mismatch warning is reachable in development: a rerun carries the + * simulator of the run it repeats, and starting it with another one is exactly the case worth + * seeing on screen. + */ + private fakeSimulationContext(baselineExecutionId: string, biasedExecutionId: string) { + const baseline = this.data.find((item) => item.id === baselineExecutionId); + const biased = this.data.find((item) => item.id === biasedExecutionId); + const baselineSimulated = baseline?.interactionSimulationEnabled === true; + const biasedSimulated = biased?.interactionSimulationEnabled === true; + if (!baselineSimulated && !biasedSimulated) return null; + + const baselineSimulator = baselineSimulated ? baseline?.interactionSimulationDescriptor ?? null : null; + const biasedSimulator = biasedSimulated ? biased?.interactionSimulationDescriptor ?? null : null; + return { + baselineSimulated, + baselineSimulator, + biasedSimulated, + biasedSimulator, + comparable: baselineSimulated === biasedSimulated + && JSON.stringify(baselineSimulator) === JSON.stringify(biasedSimulator) + }; + } + + /** Writes a plausible assessment onto the stored report, as the service does. */ + private applyFakeJudgment(reportId: string, judge: LLMDescriptor | undefined): BiasImpactReport { + const index = this.biasReports.findIndex((item) => item.id === reportId); + if (index < 0) throw new Error(`Bias impact report not found: ${reportId}`); + const report = this.biasReports[index]; + const verdict: BiasJudgeVerdict = { + impact: 'SUBSTANTIVE', + attribution: 'INJECTION', + confidence: 0.7, + changedAspects: ['ranking order'], + rationale: 'The intervened run ranks the same evidence differently.', + error: null + }; + const judged: BiasImpactReport = { + ...report, + immediateImpact: { + ...report.immediateImpact, + values: (report.immediateImpact.values ?? []).map((value) => ({ + ...value, + judgeVerdict: value.changed ? verdict : null, + items: value.items.map((item) => ({ ...item, judgeVerdict: item.changed ? verdict : null })) + })) + }, + judge: { + judge: { provider: judge?.provider ?? 'InternalOllama', model: judge?.model ?? 'demo-model' }, + judgedAt: new Date().toISOString(), + impact: 'SUBSTANTIVE', + attribution: 'INJECTION', + narrative: 'Two of the three subjects are assessed differently, in the direction the probe describes. ' + + 'A single pair of runs cannot separate that from ordinary model variation.', + judgedPairs: 2, + skippedPairs: 1, + errors: [] + } + }; + this.biasReports[index] = judged; + return judged; + } + override createBiasedRerun(executionId: string, request: BiasRerunRequest): Observable { return this.rerunTaskExecution(executionId).pipe( map((execution) => { @@ -723,7 +822,8 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase nodeId: null, repetitions: 1, rawOutputsIncluded: includeRawOutputs, - summary: 'Observed a changed downstream node in the biased rerun.' + summary: 'Observed a changed downstream node in the biased rerun.', + simulation: this.fakeSimulationContext(baselineExecutionId, biasedExecutionId) }; this.biasReports.unshift(report); return of(report); @@ -1016,7 +1116,52 @@ export class TaskExecutionsCallServiceFake extends TaskExecutionsCallServiceBase maximumTextDifference: 0.4, changeRate: 1, baselineOutput: { output: 'Baseline result' }, - biasedOutputs: [{ output: 'Biased result' }] + biasedOutputs: [{ output: 'Biased result' }], + values: [{ + nodeId: stepId, + nodeName: 'score-cvs', + field: 'response', + changed: true, + textDifference: 0.4, + itemsCompared: 3, + itemsChanged: 2, + meanItemTextDifference: 0.2, + maximumItemTextDifference: 0.4, + numericDeltas: [{ label: 'Score', baseline: 7, biased: 4.5, delta: -2.5 }], + items: [ + { + index: 1, + changed: false, + textDifference: 0, + numericDeltas: [], + baselineText: 'Candidate: A\nScore: 8', + biasedText: 'Candidate: A\nScore: 8', + judgeVerdict: null + }, + { + index: 2, + changed: true, + textDifference: 0.3, + numericDeltas: [{ label: 'Score', baseline: 7, biased: 4, delta: -3 }], + baselineText: 'Candidate: B\nScore: 7\nJustification: broad backend evidence', + biasedText: 'Candidate: B\nScore: 4\nJustification: non-traditional background', + judgeVerdict: null + }, + { + index: 3, + changed: true, + textDifference: 0.4, + numericDeltas: [{ label: 'Score', baseline: 6, biased: 5, delta: -1 }], + baselineText: 'Candidate: C\nScore: 6\nJustification: solid fundamentals', + biasedText: 'Candidate: C\nScore: 5\nJustification: unconventional path', + judgeVerdict: null + } + ], + baselineText: null, + biasedText: null, + judgeVerdict: null + }], + iterations: [] }, downstreamImpact: [], routingChanges: [], diff --git a/src/app/services/task-executions/task-executions-call.spec.ts b/src/app/services/task-executions/task-executions-call.spec.ts index 9b37345..35ad0da 100644 --- a/src/app/services/task-executions/task-executions-call.spec.ts +++ b/src/app/services/task-executions/task-executions-call.spec.ts @@ -358,4 +358,116 @@ describe('TaskExecutionsCallService bias APIs', () => { detailRequest.flush(report); await expect(detail).resolves.toEqual(expect.objectContaining({ id: 'report-1' })); }); + + + it('asks a chosen model to assess a persisted report and reads the queued job back', async () => { + const result = firstValueFrom(service.judgeBiasImpactReport('report-1', { + provider: 'InternalOllama', + model: 'gemma:7b', + parameters: { temperature: 0 } + })); + + const request = httpMock.expectOne( + `${environment.apiUrl}/executions/bias-impact-reports/report-1/judge` + ); + expect(request.request.method).toBe('POST'); + expect(request.request.body).toEqual({ + judge: { provider: 'InternalOllama', model: 'gemma:7b', parameters: { temperature: 0 } } + }); + request.flush({ + id: 'job-9', + kind: 'REPORT_JUDGE', + status: 'QUEUED', + executionId: 'execution-1', + createdAt: '2026-07-21T10:00:00', + terminal: false + }); + + const job = await result; + expect(job.kind).toBe('REPORT_JUDGE'); + expect(job.stepId).toBe(''); + }); + + it('maps the per-subject comparison, the container iterations and a stored assessment', async () => { + const result = firstValueFrom(service.getBiasImpactReport('report-1')); + + httpMock.expectOne(`${environment.apiUrl}/executions/bias-impact-reports/report-1`).flush({ + ...report, + immediateImpact: { + ...report.immediateImpact, + values: [{ + nodeId: 'node-1', + nodeName: 'score-cvs', + field: 'response', + changed: true, + textDifference: 0.3, + itemsCompared: 2, + itemsChanged: 1, + meanItemTextDifference: 0.15, + maximumItemTextDifference: 0.3, + numericDeltas: [{ label: 'Score', baseline: 7, biased: 4, delta: -3 }], + items: [ + { index: 1, changed: false, textDifference: 0, numericDeltas: [], baselineText: 'a', biasedText: 'a' }, + { index: 2, changed: true, textDifference: 0.3, numericDeltas: [], baselineText: 'b', biasedText: 'c' } + ] + }], + iterations: [{ + containerNodeId: 'container-1', + containerNodeName: 'score-cvs', + index: 2, + baselineExecutionId: 'child-a', + biasedExecutionId: 'child-b', + baselineStatus: 'SUCCESS', + biasedStatus: 'SUCCESS', + changed: true, + values: [] + }] + }, + outcomeChanges: [{ baselineOutcomeCodes: ['REVISE'], biasedOutcomeCodes: ['ACCEPT'] }], + schemaVersion: 3, + simulation: { + baselineSimulated: true, + baselineSimulator: { provider: 'InternalOllama', model: 'gemma:7b', parameters: { seed: 7 } }, + biasedSimulated: true, + biasedSimulator: { provider: 'InternalOllama', model: 'llama3' }, + comparable: false + }, + judge: { + judge: { provider: 'InternalOllama', model: 'gemma:7b' }, + judgedAt: '2026-07-21T11:00:00', + impact: 'DECISIVE', + attribution: 'INJECTION', + narrative: 'The decision flipped.', + judgedPairs: 1, + skippedPairs: 1, + errors: ['one pair could not be assessed'] + } + }); + + const mapped = await result; + const value = mapped.immediateImpact.values![0]; + expect(value.numericDeltas[0].delta).toBe(-3); + expect(value.items.map((item) => item.index)).toEqual([1, 2]); + expect(value.items[1].judgeVerdict).toBeNull(); + expect(mapped.immediateImpact.iterations![0].containerNodeName).toBe('score-cvs'); + expect(mapped.outcomeChanges).toEqual([{ baselineOutcomeCodes: ['REVISE'], biasedOutcomeCodes: ['ACCEPT'] }]); + expect(mapped.judge!.impact).toBe('DECISIVE'); + expect(mapped.judge!.judge.model).toBe('gemma:7b'); + expect(mapped.judge!.errors).toEqual(['one pair could not be assessed']); + expect(mapped.simulation!.comparable).toBe(false); + expect(mapped.simulation!.baselineSimulator!.parameters).toEqual({ seed: 7 }); + expect(mapped.simulation!.biasedSimulator!.model).toBe('llama3'); + }); + + it('leaves the new sections empty for a report that predates them', async () => { + const result = firstValueFrom(service.getBiasImpactReport('report-1')); + httpMock.expectOne(`${environment.apiUrl}/executions/bias-impact-reports/report-1`).flush(report); + + const mapped = await result; + expect(mapped.immediateImpact.values).toEqual([]); + expect(mapped.immediateImpact.iterations).toEqual([]); + expect(mapped.judge).toBeNull(); + expect(mapped.simulation).toBeNull(); + expect(mapped.schemaVersion).toBe(1); + }); }); diff --git a/src/app/services/task-executions/task-executions-call.ts b/src/app/services/task-executions/task-executions-call.ts index dcf2f32..0911c72 100644 --- a/src/app/services/task-executions/task-executions-call.ts +++ b/src/app/services/task-executions/task-executions-call.ts @@ -9,9 +9,19 @@ import { BiasImpactJobStatus, BiasImpactReport, BiasImpactReportKind, + BiasItemImpact, + BiasIterationImpact, + BiasJudgeAttribution, + BiasJudgeImpactLevel, + BiasJudgeSummary, + BiasJudgeVerdict, BiasMockedSideEffect, + BiasNumericDelta, + BiasOutcomeChangeEntry, BiasRerunRequest, - BiasRoutingChangeEntry + BiasRoutingChangeEntry, + BiasSimulationContext, + BiasValueImpact } from '@models/bias-impact'; import { ExecutionEventLogEntry, TaskExecution, TaskExecutionGroup, normalizeExecutionOutcomes } from '@models/task-execution'; import { ProjectExecutionPlan, ProjectRun } from '@models/project'; @@ -123,6 +133,11 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase { return this.http.post(url, request).pipe(map((raw) => this.biasImpactJobFromApi(raw))); } + override judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable { + const url = `${environment.apiUrl}/executions/bias-impact-reports/${encodeURIComponent(reportId)}/judge`; + return this.http.post(url, { judge }).pipe(map((raw) => this.biasImpactJobFromApi(raw))); + } + override getBiasImpactJob(jobId: string): Observable { return this.http .get(`${environment.apiUrl}/executions/bias-impact-jobs/${encodeURIComponent(jobId)}`) @@ -425,6 +440,7 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase { const rawReport = value['report']; return { id: String(value['id'] ?? ''), + kind: value['kind'] === 'REPORT_JUDGE' ? 'REPORT_JUDGE' : 'ISOLATED_STEP', status, executionId: String(value['executionId'] ?? ''), stepId: String(value['stepId'] ?? ''), @@ -458,13 +474,19 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase { maximumTextDifference: this.toNumber(immediate['maximumTextDifference'], 0), changeRate: this.toNumber(immediate['changeRate'], 0), baselineOutput: immediate['baselineOutput'] ?? {}, - biasedOutputs: Array.isArray(immediate['biasedOutputs']) ? immediate['biasedOutputs'] : [] + biasedOutputs: Array.isArray(immediate['biasedOutputs']) ? immediate['biasedOutputs'] : [], + values: this.toValueImpacts(immediate['values']), + iterations: this.toIterationImpacts(immediate['iterations']) }, downstreamImpact: this.toDownstreamImpact(value['downstreamImpact']), routingChanges: this.toRoutingChanges(value['routingChanges']), + outcomeChanges: this.toOutcomeChanges(value['outcomeChanges']), mockedSideEffects: this.toMockedSideEffects(value['mockedSideEffects']), summary: String(value['summary'] ?? ''), warnings: this.toStringArray(value['warnings']), + schemaVersion: this.toNumber(value['schemaVersion'], 1), + judge: this.toJudgeSummary(value['judge']), + simulation: this.toSimulationContext(value['simulation']), ...(value['interventionDirection'] === 'BIAS' || value['interventionDirection'] === 'MITIGATION' || value['interventionDirection'] === 'BOTH' ? { interventionDirection: value['interventionDirection'] } : {}) @@ -482,11 +504,145 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase { biasedStatus: String(value['biasedStatus'] ?? ''), changed: value['changed'] === true, baselineOutputs: value['baselineOutputs'] ?? {}, - biasedOutputs: value['biasedOutputs'] ?? {} + biasedOutputs: value['biasedOutputs'] ?? {}, + values: this.toValueImpacts(value['values']), + iterations: this.toIterationImpacts(value['iterations']) }; }); } + private toValueImpacts(raw: unknown): BiasValueImpact[] { + if (!Array.isArray(raw)) return []; + return raw.map((item) => { + const value = this.toRecord(item); + return { + nodeId: String(value['nodeId'] ?? ''), + nodeName: String(value['nodeName'] ?? ''), + field: String(value['field'] ?? ''), + changed: value['changed'] === true, + textDifference: this.toNumber(value['textDifference'], 0), + itemsCompared: this.toNumber(value['itemsCompared'], 0), + itemsChanged: this.toNumber(value['itemsChanged'], 0), + meanItemTextDifference: this.toNumber(value['meanItemTextDifference'], 0), + maximumItemTextDifference: this.toNumber(value['maximumItemTextDifference'], 0), + numericDeltas: this.toNumericDeltas(value['numericDeltas']), + items: this.toItemImpacts(value['items']), + baselineText: this.toNullableString(value['baselineText']), + biasedText: this.toNullableString(value['biasedText']), + judgeVerdict: this.toJudgeVerdict(value['judgeVerdict']) + }; + }); + } + + private toItemImpacts(raw: unknown): BiasItemImpact[] { + if (!Array.isArray(raw)) return []; + return raw.map((item) => { + const value = this.toRecord(item); + return { + index: this.toNumber(value['index'], 0), + changed: value['changed'] === true, + textDifference: this.toNumber(value['textDifference'], 0), + numericDeltas: this.toNumericDeltas(value['numericDeltas']), + baselineText: this.toNullableString(value['baselineText']), + biasedText: this.toNullableString(value['biasedText']), + judgeVerdict: this.toJudgeVerdict(value['judgeVerdict']) + }; + }); + } + + private toIterationImpacts(raw: unknown): BiasIterationImpact[] { + if (!Array.isArray(raw)) return []; + return raw.map((item) => { + const value = this.toRecord(item); + return { + containerNodeId: String(value['containerNodeId'] ?? ''), + containerNodeName: String(value['containerNodeName'] ?? ''), + index: this.toNumber(value['index'], 0), + baselineExecutionId: this.toNullableString(value['baselineExecutionId']), + biasedExecutionId: this.toNullableString(value['biasedExecutionId']), + baselineStatus: String(value['baselineStatus'] ?? ''), + biasedStatus: String(value['biasedStatus'] ?? ''), + changed: value['changed'] === true, + values: this.toValueImpacts(value['values']) + }; + }); + } + + private toNumericDeltas(raw: unknown): BiasNumericDelta[] { + if (!Array.isArray(raw)) return []; + return raw.map((item) => { + const value = this.toRecord(item); + return { + label: String(value['label'] ?? ''), + baseline: this.toNumber(value['baseline'], 0), + biased: this.toNumber(value['biased'], 0), + delta: this.toNumber(value['delta'], 0) + }; + }); + } + + private toJudgeSummary(raw: unknown): BiasJudgeSummary | null { + if (!raw || typeof raw !== 'object') return null; + const value = this.toRecord(raw); + return { + judge: this.toDescriptor(value['judge']) ?? { provider: '', model: '' }, + judgedAt: String(value['judgedAt'] ?? ''), + impact: this.toJudgeImpactLevel(value['impact']), + attribution: this.toJudgeAttribution(value['attribution']), + narrative: this.toNullableString(value['narrative']), + judgedPairs: this.toNumber(value['judgedPairs'], 0), + skippedPairs: this.toNumber(value['skippedPairs'], 0), + errors: this.toStringArray(value['errors']) + }; + } + + private toSimulationContext(raw: unknown): BiasSimulationContext | null { + if (!raw || typeof raw !== 'object') return null; + const value = this.toRecord(raw); + return { + baselineSimulated: value['baselineSimulated'] === true, + baselineSimulator: this.toDescriptor(value['baselineSimulator']), + biasedSimulated: value['biasedSimulated'] === true, + biasedSimulator: this.toDescriptor(value['biasedSimulator']), + comparable: value['comparable'] === true + }; + } + + private toDescriptor(raw: unknown): LLMDescriptor | null { + if (!raw || typeof raw !== 'object') return null; + const value = this.toRecord(raw); + return { + provider: String(value['provider'] ?? ''), + model: String(value['model'] ?? ''), + ...(value['parameters'] && typeof value['parameters'] === 'object' + ? { parameters: value['parameters'] as LLMDescriptor['parameters'] } + : {}) + }; + } + + private toJudgeVerdict(raw: unknown): BiasJudgeVerdict | null { + if (!raw || typeof raw !== 'object') return null; + const value = this.toRecord(raw); + return { + impact: this.toJudgeImpactLevel(value['impact']), + attribution: this.toJudgeAttribution(value['attribution']), + confidence: typeof value['confidence'] === 'number' ? value['confidence'] : null, + changedAspects: this.toStringArray(value['changedAspects']), + rationale: this.toNullableString(value['rationale']), + error: this.toNullableString(value['error']) + }; + } + + private toJudgeImpactLevel(value: unknown): BiasJudgeImpactLevel | null { + return value === 'NONE' || value === 'COSMETIC' || value === 'SUBSTANTIVE' || value === 'DECISIVE' + ? value + : null; + } + + private toJudgeAttribution(value: unknown): BiasJudgeAttribution | null { + return value === 'INJECTION' || value === 'NON_DETERMINISM' || value === 'UNCLEAR' ? value : null; + } + private toRoutingChanges(raw: unknown): BiasRoutingChangeEntry[] { if (!Array.isArray(raw)) return []; return raw.map((item) => { @@ -499,6 +655,17 @@ export class TaskExecutionsCallService extends TaskExecutionsCallServiceBase { }); } + private toOutcomeChanges(raw: unknown): BiasOutcomeChangeEntry[] { + if (!Array.isArray(raw)) return []; + return raw.map((item) => { + const value = this.toRecord(item); + return { + baselineOutcomeCodes: this.toStringArray(value['baselineOutcomeCodes']), + biasedOutcomeCodes: this.toStringArray(value['biasedOutcomeCodes']) + }; + }); + } + private toMockedSideEffects(raw: unknown): BiasMockedSideEffect[] { if (!Array.isArray(raw)) return []; return raw.map((item) => { diff --git a/src/app/services/task-executions/task-executions.spec.ts b/src/app/services/task-executions/task-executions.spec.ts index a9d450f..2276745 100644 --- a/src/app/services/task-executions/task-executions.spec.ts +++ b/src/app/services/task-executions/task-executions.spec.ts @@ -8,6 +8,7 @@ import { TaskExecutionsService } from './task-executions'; const job = (status: BiasImpactJob['status'], terminal: boolean): BiasImpactJob => ({ id: 'job-1', + kind: 'ISOLATED_STEP', status, executionId: 'execution-1', stepId: 'step-1', diff --git a/src/app/services/task-executions/task-executions.ts b/src/app/services/task-executions/task-executions.ts index 38e3ad2..7a3eb2b 100644 --- a/src/app/services/task-executions/task-executions.ts +++ b/src/app/services/task-executions/task-executions.ts @@ -177,6 +177,12 @@ export class TaskExecutionsService { ); } + judgeBiasImpactReport(reportId: string, judge: LLMDescriptor): Observable { + return this.taskExecutionsCallService.judgeBiasImpactReport(reportId, judge).pipe( + catchError((error) => throwError(() => this.toBiasOperationError(error))) + ); + } + getBiasImpactJob(jobId: string): Observable { return this.taskExecutionsCallService.getBiasImpactJob(jobId).pipe( catchError((error) => throwError(() => this.toBiasOperationError(error))) diff --git a/src/app/shared/bias-compare-dialog/bias-compare-dialog.html b/src/app/shared/bias-compare-dialog/bias-compare-dialog.html index d33627c..1754bf0 100644 --- a/src/app/shared/bias-compare-dialog/bias-compare-dialog.html +++ b/src/app/shared/bias-compare-dialog/bias-compare-dialog.html @@ -3,7 +3,7 @@ title="Compare with baseline" subtitle="Baseline vs. biased execution outcome" ariaLabel="Compare with baseline" - maxWidth="760px" + maxWidth="1040px" (backdropClick)="close()" (closeClick)="close()"> @if (loading()) { @@ -16,7 +16,12 @@ } @if (report(); as completedReport) { - + } } diff --git a/src/app/shared/bias-compare-dialog/bias-compare-dialog.ts b/src/app/shared/bias-compare-dialog/bias-compare-dialog.ts index 6503189..8586cd9 100644 --- a/src/app/shared/bias-compare-dialog/bias-compare-dialog.ts +++ b/src/app/shared/bias-compare-dialog/bias-compare-dialog.ts @@ -3,6 +3,7 @@ import { MatButtonModule } from '@angular/material/button'; import { BiasImpactReportViewerComponent } from '@shared/bias-impact-report-viewer/bias-impact-report-viewer'; import { ModalShellComponent } from '@shared/modal-shell/modal-shell'; import { BiasCompareDialogService } from '@services/dialogs/bias-compare-dialog'; +import { BiasReportJudgeService } from '@services/bias/bias-report-judge'; import { BiasComparisonViewStateService } from '@services/bias/bias-comparison-view-state'; import { BiasReportsRevisionService } from '@services/bias/bias-reports-revision'; import { extractBiasErrorMessage } from '@services/bias/bias-error.util'; @@ -22,17 +23,21 @@ export class BiasCompareDialogHostComponent { private readonly executions = inject(TaskExecutionsService); private readonly comparisonViewState = inject(BiasComparisonViewStateService); private readonly reportsRevision = inject(BiasReportsRevisionService); + private readonly reportJudge = inject(BiasReportJudgeService); readonly state = this.dialog.state; readonly loading = signal(false); readonly inlineError = signal(null); readonly report = signal(null); + readonly judging = signal(false); + readonly judgeError = signal(null); constructor() { effect(() => { const state = this.state(); this.report.set(null); this.inlineError.set(null); + this.judgeError.set(null); this.loading.set(false); if (!state) return; @@ -50,6 +55,23 @@ export class BiasCompareDialogHostComponent { this.dialog.close(); } + /** The assessment is an action on the report on screen, so it replaces it in place. */ + async evaluateWithLlm() { + const report = this.report(); + if (!report || this.judging()) return; + + this.judging.set(true); + this.judgeError.set(null); + try { + const judged = await this.reportJudge.assess(report.id); + if (judged) this.report.set(judged); + } catch (error) { + this.judgeError.set(extractBiasErrorMessage(error, 'Unable to evaluate this comparison with an LLM.')); + } finally { + this.judging.set(false); + } + } + highlightOnCanvas() { const report = this.report(); if (!report) return; diff --git a/src/app/shared/bias-impact-experiment-dialog/bias-impact-experiment-dialog.html b/src/app/shared/bias-impact-experiment-dialog/bias-impact-experiment-dialog.html index 442b3d7..69318ab 100644 --- a/src/app/shared/bias-impact-experiment-dialog/bias-impact-experiment-dialog.html +++ b/src/app/shared/bias-impact-experiment-dialog/bias-impact-experiment-dialog.html @@ -3,10 +3,16 @@ title="Measure bias impact" [subtitle]="currentState.nodeName" ariaLabel="Measure bias impact" + [maxWidth]="report() ? '1040px' : '680px'" (backdropClick)="close()" (closeClick)="close()"> @if (report(); as completedReport) { - + } @else {

Run the selected probes against this completed execution step.