From e8ee6ae614e4163e5199f6442c2cb0957e5d825d Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 1 Oct 2026 05:55:02 +0000 Subject: [PATCH 1/2] Let add-models re-judge a model a run already has A judging outage can leave a model's scores in a published run mostly failed, which scores it near zero. With rejudge, the add-models job keeps that model's saved responses, drops its saved scores and judges only it again, publishing a new run. Other models keep their scores untouched. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_015roAwvpcszBvjX1Ce5NFuj --- .github/workflows/add-models-to-runs.yml | 7 ++++- .../__tests__/route.test.ts | 11 +++++++ .../api/internal/add-models-to-runs/route.ts | 11 ++++--- .../add-models-to-run-service.test.ts | 30 ++++++++++++++++++ src/cli/services/add-models-to-run-service.ts | 31 ++++++++++++++----- 5 files changed, 78 insertions(+), 12 deletions(-) diff --git a/.github/workflows/add-models-to-runs.yml b/.github/workflows/add-models-to-runs.yml index e10545d..c25b61f 100644 --- a/.github/workflows/add-models-to-runs.yml +++ b/.github/workflows/add-models-to-runs.yml @@ -12,6 +12,10 @@ on: description: 'Rebuild homepage, leaderboards and model summaries at the end' type: boolean default: true + rejudge: + description: 'Judge the models again from their saved responses where a run already has them' + type: boolean + default: false jobs: start: @@ -22,8 +26,9 @@ jobs: MODELS: ${{ github.event.inputs.models }} CONFIG_IDS: ${{ github.event.inputs.config_ids }} REBUILD: ${{ github.event.inputs.rebuild_summaries }} + REJUDGE: ${{ github.event.inputs.rejudge }} run: | - BODY=$(python3 -c 'import json, os; split = lambda v: [x.strip() for x in v.split(",") if x.strip()]; print(json.dumps({"models": split(os.environ["MODELS"]), "configIds": split(os.environ["CONFIG_IDS"]), "rebuildSummaries": os.environ["REBUILD"] == "true"}))') + BODY=$(python3 -c 'import json, os; split = lambda v: [x.strip() for x in v.split(",") if x.strip()]; print(json.dumps({"models": split(os.environ["MODELS"]), "configIds": split(os.environ["CONFIG_IDS"]), "rebuildSummaries": os.environ["REBUILD"] == "true", "rejudge": os.environ.get("REJUDGE") == "true"}))') # RAILWAY_APP_URL may redirect (http -> https, www -> apex); follow it as a POST. STATUS=$(curl -sS -L --post301 --post302 --post303 --max-time 60 -o response.txt -w '%{http_code}' -X POST \ -H "Content-Type: application/json" \ diff --git a/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts b/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts index 56e487b..b349e67 100644 --- a/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts +++ b/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts @@ -33,9 +33,20 @@ describe('POST /api/internal/add-models-to-runs', () => { expect(vi.mocked(addModelsToLatestRun).mock.calls.map(c => c[0])).toEqual(['a', 'b', 'c__d']); expect(vi.mocked(addModelsToLatestRun).mock.calls[0][1]).toEqual([APERTUS]); + expect(vi.mocked(addModelsToLatestRun).mock.calls[0][3]).toEqual({ rejudge: false }); expect(actionBackfillSummary).toHaveBeenCalledTimes(1); }); + it('passes rejudge through to each blueprint', async () => { + vi.mocked(addModelsToLatestRun).mockImplementation(async (configId) => ({ configId, status: 'added' })); + + const res = await POST(request({ models: [APERTUS], configIds: ['a'], rejudge: true })); + expect(res.status).toBe(202); + await flush(); + + expect(vi.mocked(addModelsToLatestRun).mock.calls[0][3]).toEqual({ rejudge: true }); + }); + it('keeps going after a blueprint throws, and skips the rebuild when nothing was added', async () => { vi.mocked(addModelsToLatestRun) .mockRejectedValueOnce(new Error('boom')) diff --git a/src/app/api/internal/add-models-to-runs/route.ts b/src/app/api/internal/add-models-to-runs/route.ts index c6633a3..6d825fe 100644 --- a/src/app/api/internal/add-models-to-runs/route.ts +++ b/src/app/api/internal/add-models-to-runs/route.ts @@ -18,6 +18,8 @@ const MAX_CONFIGS = 300; * Returns 202 immediately and works through the blueprints one at a time in * the background, logging under "add-models:". With rebuildSummaries, it * rebuilds the homepage, leaderboards and model summaries once at the end. + * With rejudge, listed models a run already has are judged again from their + * saved responses instead of being skipped. */ export async function POST(req: NextRequest) { const authError = checkBackgroundAuth(req); @@ -32,6 +34,7 @@ export async function POST(req: NextRequest) { const models: unknown = body?.models; const configIds: unknown = body?.configIds; const rebuildSummaries = body?.rebuildSummaries === true; + const rejudge = body?.rejudge === true; if (!Array.isArray(models) || models.length === 0 || !models.every(m => typeof m === 'string' && MODEL_ID_RE.test(m))) { return NextResponse.json({ error: "'models' must be a non-empty list of provider:model IDs." }, { status: 400 }); @@ -54,20 +57,20 @@ export async function POST(req: NextRequest) { }, }); - void runJob(models as string[], configIds as string[], rebuildSummaries, logger); + void runJob(models as string[], configIds as string[], rebuildSummaries, rejudge, logger); return NextResponse.json( - { message: 'Accepted. Progress is logged under "add-models:".', models, configs: configIds.length, rebuildSummaries }, + { message: 'Accepted. Progress is logged under "add-models:".', models, configs: configIds.length, rebuildSummaries, rejudge }, { status: 202 }, ); } -async function runJob(models: string[], configIds: string[], rebuildSummaries: boolean, logger: any): Promise { +async function runJob(models: string[], configIds: string[], rebuildSummaries: boolean, rejudge: boolean, logger: any): Promise { const results: AddModelsResult[] = []; for (const [i, configId] of configIds.entries()) { logger.info(`[AddModels] (${i + 1}/${configIds.length}) ${configId}...`); try { - const result = await addModelsToLatestRun(configId, models, logger); + const result = await addModelsToLatestRun(configId, models, logger, { rejudge }); results.push(result); logger.info(`[AddModels] (${i + 1}/${configIds.length}) ${configId}: ${result.status}${result.reason ? ` (${result.reason})` : ''}`); } catch (error: any) { diff --git a/src/cli/services/__tests__/add-models-to-run-service.test.ts b/src/cli/services/__tests__/add-models-to-run-service.test.ts index 4224d13..4f4fd80 100644 --- a/src/cli/services/__tests__/add-models-to-run-service.test.ts +++ b/src/cli/services/__tests__/add-models-to-run-service.test.ts @@ -105,6 +105,36 @@ describe('addModelsToLatestRun', () => { expect(executeComparisonPipeline).not.toHaveBeenCalled(); }); + it('with rejudge, judges a model the run already has again from its saved responses', async () => { + const withApertus = sourceRun({ + config: { ...sourceRun().config, models: ['openai:gpt-4o', APERTUS] }, + effectiveModels: ['openai:gpt-4o[temp:0]', `${APERTUS}[temp:0]`], + allFinalAssistantResponses: { + p1: { 'openai:gpt-4o[temp:0]': 'A1', [`${APERTUS}[temp:0]`]: 'Apertus 1' }, + p2: { 'openai:gpt-4o[temp:0]': 'A2', [`${APERTUS}[temp:0]`]: 'Apertus 2' }, + }, + errors: {}, + evaluationResults: { + llmCoverageScores: { + p1: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.8 }, [`${APERTUS}[temp:0]`]: { avgCoverageExtent: 0 } }, + p2: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.6 }, [`${APERTUS}[temp:0]`]: { avgCoverageExtent: 0 } }, + }, + }, + }); + vi.mocked(storage.getResultByFileName).mockImplementation(async (_c, f) => (f.startsWith('old') ? withApertus : { configId: 'bp' }) as any); + + const result = await addModelsToLatestRun('bp', [APERTUS], logger, { rejudge: true }); + + expect(result).toMatchObject({ status: 'added', generated: 0, rejudged: 2 }); + expect(generateResponseForPair).not.toHaveBeenCalled(); + const [config, , , , responseMap, , , , , , , , prefilled] = vi.mocked(executeComparisonPipeline).mock.calls[0] as any[]; + expect(config.models).toEqual(['openai:gpt-4o', APERTUS]); + expect(responseMap.get('p1').modelResponses[`${APERTUS}[temp:0]`]).toMatchObject({ finalAssistantResponseText: 'Apertus 1', hasError: false }); + // Apertus loses its saved scores so it is judged again; the others keep theirs. + expect(Object.keys(prefilled.p1)).toEqual(['openai:gpt-4o[temp:0]']); + expect(Object.keys(prefilled.p2)).toEqual(['openai:gpt-4o[temp:0]']); + }); + it('publishes nothing when every request to the added model fails', async () => { vi.mocked(generateResponseForPair).mockResolvedValue({ text: '<>x<>', history: [], hasError: true, errorMessage: 'x' }); diff --git a/src/cli/services/add-models-to-run-service.ts b/src/cli/services/add-models-to-run-service.ts index 83fe23c..6d66504 100644 --- a/src/cli/services/add-models-to-run-service.ts +++ b/src/cli/services/add-models-to-run-service.ts @@ -27,6 +27,7 @@ export interface AddModelsResult { reused?: number; generated?: number; generationErrors?: number; + rejudged?: number; } /** @@ -58,12 +59,16 @@ function parseSuffix(suffix: string): { temperature?: number; spIdx?: number } { * are asked, once per variant the run used (temperatures, system prompts), and * only they are judged. The prompts come from the run's own embedded config, * so later blueprint or model-collection edits can't trigger a full re-run. + * + * With `rejudge`, listed models that the latest run already has keep their + * responses but lose their saved scores, so only they are judged again (for + * example after a judging outage left their scores incomplete). */ export async function addModelsToLatestRun( configId: string, modelsToAdd: string[], logger: Logger, - options: { concurrency?: number } = {}, + options: { concurrency?: number; rejudge?: boolean } = {}, ): Promise { const runs = await listRunsForConfig(configId); if (runs.length === 0) return { configId, status: 'skipped', reason: 'no published runs' }; @@ -78,7 +83,8 @@ export async function addModelsToLatestRun( .map((m: any) => (typeof m === 'string' ? m : m?.id)) .filter(Boolean); const newModels = modelsToAdd.filter(m => !sourceModels.includes(m)); - if (newModels.length === 0) { + const rejudgeModels = options.rejudge ? modelsToAdd.filter(m => sourceModels.includes(m)) : []; + if (newModels.length === 0 && rejudgeModels.length === 0) { return { configId, status: 'skipped', reason: 'latest run already includes the model(s)' }; } @@ -101,6 +107,7 @@ export async function addModelsToLatestRun( let reused = 0; let generated = 0; let generationErrors = 0; + let rejudged = 0; for (const prompt of targetConfig.prompts) { const promptData: PromptResponseData = { @@ -114,7 +121,8 @@ export async function addModelsToLatestRun( // Existing models: copy the saved response and score as they are. for (const effectiveId of sourceEffectiveIds) { - if (!splitEffectiveId(effectiveId, sourceModels)) continue; + const split = splitEffectiveId(effectiveId, sourceModels); + if (!split) continue; const text = source.allFinalAssistantResponses?.[prompt.id]?.[effectiveId]; const sourceError = source.errors?.[prompt.id]?.[effectiveId]; const ok = typeof text === 'string' && !sourceError && !text.startsWith('<>'); @@ -127,6 +135,12 @@ export async function addModelsToLatestRun( } as any; if (ok) reused++; + // Models being re-judged keep the response but not the score. + if (rejudgeModels.includes(split.base)) { + if (ok) rejudged++; + continue; + } + const inline = inlineCoverage?.[prompt.id]?.[effectiveId]; if (inline) { (prefilledCoverage[prompt.id] ||= {})[effectiveId] = inline; @@ -172,10 +186,13 @@ export async function addModelsToLatestRun( } await Promise.all(tasks); - logger.info(`[AddModels] ${configId}: reused ${reused} responses, asked ${generated} (${generationErrors} failed).`); + logger.info(`[AddModels] ${configId}: reused ${reused} responses, asked ${generated} (${generationErrors} failed), re-judging ${rejudged}.`); + if (newModels.length === 0 && rejudged === 0) { + return { configId, status: 'skipped', reason: 'no saved responses to re-judge', reused, generated, generationErrors, rejudged }; + } if (generated > 0 && generationErrors === generated) { - return { configId, status: 'failed', reason: 'every request to the added model(s) failed', reused, generated, generationErrors }; + return { configId, status: 'failed', reason: 'every request to the added model(s) failed', reused, generated, generationErrors, rejudged }; } const evalMethods: EvaluationMethod[] = Array.isArray(source.evalMethodsUsed) && source.evalMethodsUsed.length > 0 @@ -199,7 +216,7 @@ export async function addModelsToLatestRun( prefilledCoverage, ); if (!fileName) { - return { configId, status: 'failed', reason: 'the pipeline saved nothing', reused, generated, generationErrors }; + return { configId, status: 'failed', reason: 'the pipeline saved nothing', reused, generated, generationErrors, rejudged }; } const newRun = await getResultByFileName(configId, fileName); @@ -211,5 +228,5 @@ export async function addModelsToLatestRun( logger.warn(`[AddModels] ${configId}: saved ${fileName} but could not read it back to update the page summary.`); } - return { configId, status: 'added', fileName, reused, generated, generationErrors }; + return { configId, status: 'added', fileName, reused, generated, generationErrors, rejudged }; } From a99732dc788ec1630810961adb100ecedfe3eee6 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 1 Oct 2026 15:07:27 +0000 Subject: [PATCH 2/2] Let add-models retry a model's failed answers Some providers reject a share of requests, which leaves a model with missing answers in an otherwise complete run. With retryFailed, the add-models job asks a model the run already has again only on the prompts where its answer failed, judges those new answers, and keeps every other answer and score. The asking code is shared with the path that adds new models. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_015roAwvpcszBvjX1Ce5NFuj --- .github/workflows/add-models-to-runs.yml | 7 +- .../__tests__/route.test.ts | 8 +- .../api/internal/add-models-to-runs/route.ts | 18 ++-- .../add-models-to-run-service.test.ts | 51 +++++++++++ src/cli/services/add-models-to-run-service.ts | 88 +++++++++++-------- 5 files changed, 127 insertions(+), 45 deletions(-) diff --git a/.github/workflows/add-models-to-runs.yml b/.github/workflows/add-models-to-runs.yml index c25b61f..cc07b43 100644 --- a/.github/workflows/add-models-to-runs.yml +++ b/.github/workflows/add-models-to-runs.yml @@ -16,6 +16,10 @@ on: description: 'Judge the models again from their saved responses where a run already has them' type: boolean default: false + retry_failed: + description: 'Ask the models again where a run already has them but their answer failed' + type: boolean + default: false jobs: start: @@ -27,8 +31,9 @@ jobs: CONFIG_IDS: ${{ github.event.inputs.config_ids }} REBUILD: ${{ github.event.inputs.rebuild_summaries }} REJUDGE: ${{ github.event.inputs.rejudge }} + RETRY_FAILED: ${{ github.event.inputs.retry_failed }} run: | - BODY=$(python3 -c 'import json, os; split = lambda v: [x.strip() for x in v.split(",") if x.strip()]; print(json.dumps({"models": split(os.environ["MODELS"]), "configIds": split(os.environ["CONFIG_IDS"]), "rebuildSummaries": os.environ["REBUILD"] == "true", "rejudge": os.environ.get("REJUDGE") == "true"}))') + BODY=$(python3 -c 'import json, os; split = lambda v: [x.strip() for x in v.split(",") if x.strip()]; print(json.dumps({"models": split(os.environ["MODELS"]), "configIds": split(os.environ["CONFIG_IDS"]), "rebuildSummaries": os.environ["REBUILD"] == "true", "rejudge": os.environ.get("REJUDGE") == "true", "retryFailed": os.environ.get("RETRY_FAILED") == "true"}))') # RAILWAY_APP_URL may redirect (http -> https, www -> apex); follow it as a POST. STATUS=$(curl -sS -L --post301 --post302 --post303 --max-time 60 -o response.txt -w '%{http_code}' -X POST \ -H "Content-Type: application/json" \ diff --git a/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts b/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts index b349e67..4cdf57c 100644 --- a/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts +++ b/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts @@ -33,18 +33,18 @@ describe('POST /api/internal/add-models-to-runs', () => { expect(vi.mocked(addModelsToLatestRun).mock.calls.map(c => c[0])).toEqual(['a', 'b', 'c__d']); expect(vi.mocked(addModelsToLatestRun).mock.calls[0][1]).toEqual([APERTUS]); - expect(vi.mocked(addModelsToLatestRun).mock.calls[0][3]).toEqual({ rejudge: false }); + expect(vi.mocked(addModelsToLatestRun).mock.calls[0][3]).toEqual({ rejudge: false, retryFailed: false }); expect(actionBackfillSummary).toHaveBeenCalledTimes(1); }); - it('passes rejudge through to each blueprint', async () => { + it('passes rejudge and retryFailed through to each blueprint', async () => { vi.mocked(addModelsToLatestRun).mockImplementation(async (configId) => ({ configId, status: 'added' })); - const res = await POST(request({ models: [APERTUS], configIds: ['a'], rejudge: true })); + const res = await POST(request({ models: [APERTUS], configIds: ['a'], rejudge: true, retryFailed: true })); expect(res.status).toBe(202); await flush(); - expect(vi.mocked(addModelsToLatestRun).mock.calls[0][3]).toEqual({ rejudge: true }); + expect(vi.mocked(addModelsToLatestRun).mock.calls[0][3]).toEqual({ rejudge: true, retryFailed: true }); }); it('keeps going after a blueprint throws, and skips the rebuild when nothing was added', async () => { diff --git a/src/app/api/internal/add-models-to-runs/route.ts b/src/app/api/internal/add-models-to-runs/route.ts index 6d825fe..5cb2569 100644 --- a/src/app/api/internal/add-models-to-runs/route.ts +++ b/src/app/api/internal/add-models-to-runs/route.ts @@ -19,7 +19,8 @@ const MAX_CONFIGS = 300; * the background, logging under "add-models:". With rebuildSummaries, it * rebuilds the homepage, leaderboards and model summaries once at the end. * With rejudge, listed models a run already has are judged again from their - * saved responses instead of being skipped. + * saved responses instead of being skipped. With retryFailed, they are asked + * again on the prompts where their answer failed. */ export async function POST(req: NextRequest) { const authError = checkBackgroundAuth(req); @@ -35,6 +36,7 @@ export async function POST(req: NextRequest) { const configIds: unknown = body?.configIds; const rebuildSummaries = body?.rebuildSummaries === true; const rejudge = body?.rejudge === true; + const retryFailed = body?.retryFailed === true; if (!Array.isArray(models) || models.length === 0 || !models.every(m => typeof m === 'string' && MODEL_ID_RE.test(m))) { return NextResponse.json({ error: "'models' must be a non-empty list of provider:model IDs." }, { status: 400 }); @@ -57,20 +59,26 @@ export async function POST(req: NextRequest) { }, }); - void runJob(models as string[], configIds as string[], rebuildSummaries, rejudge, logger); + void runJob(models as string[], configIds as string[], rebuildSummaries, { rejudge, retryFailed }, logger); return NextResponse.json( - { message: 'Accepted. Progress is logged under "add-models:".', models, configs: configIds.length, rebuildSummaries, rejudge }, + { message: 'Accepted. Progress is logged under "add-models:".', models, configs: configIds.length, rebuildSummaries, rejudge, retryFailed }, { status: 202 }, ); } -async function runJob(models: string[], configIds: string[], rebuildSummaries: boolean, rejudge: boolean, logger: any): Promise { +async function runJob( + models: string[], + configIds: string[], + rebuildSummaries: boolean, + options: { rejudge: boolean; retryFailed: boolean }, + logger: any, +): Promise { const results: AddModelsResult[] = []; for (const [i, configId] of configIds.entries()) { logger.info(`[AddModels] (${i + 1}/${configIds.length}) ${configId}...`); try { - const result = await addModelsToLatestRun(configId, models, logger, { rejudge }); + const result = await addModelsToLatestRun(configId, models, logger, options); results.push(result); logger.info(`[AddModels] (${i + 1}/${configIds.length}) ${configId}: ${result.status}${result.reason ? ` (${result.reason})` : ''}`); } catch (error: any) { diff --git a/src/cli/services/__tests__/add-models-to-run-service.test.ts b/src/cli/services/__tests__/add-models-to-run-service.test.ts index 4f4fd80..30e6ee0 100644 --- a/src/cli/services/__tests__/add-models-to-run-service.test.ts +++ b/src/cli/services/__tests__/add-models-to-run-service.test.ts @@ -135,6 +135,57 @@ describe('addModelsToLatestRun', () => { expect(Object.keys(prefilled.p2)).toEqual(['openai:gpt-4o[temp:0]']); }); + it('with retryFailed, asks a model the run already has again only where its answer failed', async () => { + const withApertus = sourceRun({ + config: { ...sourceRun().config, models: ['openai:gpt-4o', APERTUS] }, + effectiveModels: ['openai:gpt-4o[temp:0]', `${APERTUS}[temp:0]`], + allFinalAssistantResponses: { + p1: { 'openai:gpt-4o[temp:0]': 'A1', [`${APERTUS}[temp:0]`]: 'Apertus 1' }, + p2: { 'openai:gpt-4o[temp:0]': 'A2', [`${APERTUS}[temp:0]`]: '<>rejected as invalid<>' }, + }, + errors: { p2: { [`${APERTUS}[temp:0]`]: 'rejected as invalid' } }, + evaluationResults: { + llmCoverageScores: { + p1: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.8 }, [`${APERTUS}[temp:0]`]: { avgCoverageExtent: 0.7 } }, + p2: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.6 }, [`${APERTUS}[temp:0]`]: { error: 'Generation failed' } }, + }, + }, + }); + vi.mocked(storage.getResultByFileName).mockImplementation(async (_c, f) => (f.startsWith('old') ? withApertus : { configId: 'bp' }) as any); + + const result = await addModelsToLatestRun('bp', [APERTUS], logger, { retryFailed: true }); + + expect(result).toMatchObject({ status: 'added', generated: 1, retried: 1, rejudged: 0 }); + // Only the failed pair is asked again, in the run's own variant. + expect(generateResponseForPair).toHaveBeenCalledTimes(1); + expect(vi.mocked(generateResponseForPair).mock.calls[0][0]).toMatchObject({ modelId: APERTUS, temperature: 0, messages: [{ role: 'user', content: 'Q2' }] }); + const [, , , , responseMap, , , , , , , , prefilled] = vi.mocked(executeComparisonPipeline).mock.calls[0] as any[]; + expect(responseMap.get('p2').modelResponses[`${APERTUS}[temp:0]`]).toMatchObject({ finalAssistantResponseText: 'Apertus answer', hasError: false }); + expect(responseMap.get('p1').modelResponses[`${APERTUS}[temp:0]`]).toMatchObject({ finalAssistantResponseText: 'Apertus 1' }); + // The good answer keeps its score; the retried one is judged fresh. + expect(prefilled.p1[`${APERTUS}[temp:0]`]).toEqual({ avgCoverageExtent: 0.7 }); + expect(prefilled.p2[`${APERTUS}[temp:0]`]).toBeUndefined(); + }); + + it('with retryFailed, skips a blueprint where the model has no failed answers', async () => { + const clean = sourceRun({ + config: { ...sourceRun().config, models: ['openai:gpt-4o', APERTUS] }, + effectiveModels: ['openai:gpt-4o[temp:0]', `${APERTUS}[temp:0]`], + allFinalAssistantResponses: { + p1: { 'openai:gpt-4o[temp:0]': 'A1', [`${APERTUS}[temp:0]`]: 'Apertus 1' }, + p2: { 'openai:gpt-4o[temp:0]': 'A2', [`${APERTUS}[temp:0]`]: 'Apertus 2' }, + }, + errors: {}, + }); + vi.mocked(storage.getResultByFileName).mockResolvedValue(clean as any); + + const result = await addModelsToLatestRun('bp', [APERTUS], logger, { retryFailed: true }); + + expect(result.status).toBe('skipped'); + expect(generateResponseForPair).not.toHaveBeenCalled(); + expect(executeComparisonPipeline).not.toHaveBeenCalled(); + }); + it('publishes nothing when every request to the added model fails', async () => { vi.mocked(generateResponseForPair).mockResolvedValue({ text: '<>x<>', history: [], hasError: true, errorMessage: 'x' }); diff --git a/src/cli/services/add-models-to-run-service.ts b/src/cli/services/add-models-to-run-service.ts index 6d66504..45daada 100644 --- a/src/cli/services/add-models-to-run-service.ts +++ b/src/cli/services/add-models-to-run-service.ts @@ -28,6 +28,7 @@ export interface AddModelsResult { generated?: number; generationErrors?: number; rejudged?: number; + retried?: number; } /** @@ -63,12 +64,17 @@ function parseSuffix(suffix: string): { temperature?: number; spIdx?: number } { * With `rejudge`, listed models that the latest run already has keep their * responses but lose their saved scores, so only they are judged again (for * example after a judging outage left their scores incomplete). + * + * With `retryFailed`, listed models that the latest run already has are asked + * again only on the prompts where their answer failed (for example when a + * provider rejected the request), and those new answers are judged. Their + * other answers and scores are kept. */ export async function addModelsToLatestRun( configId: string, modelsToAdd: string[], logger: Logger, - options: { concurrency?: number; rejudge?: boolean } = {}, + options: { concurrency?: number; rejudge?: boolean; retryFailed?: boolean } = {}, ): Promise { const runs = await listRunsForConfig(configId); if (runs.length === 0) return { configId, status: 'skipped', reason: 'no published runs' }; @@ -83,8 +89,10 @@ export async function addModelsToLatestRun( .map((m: any) => (typeof m === 'string' ? m : m?.id)) .filter(Boolean); const newModels = modelsToAdd.filter(m => !sourceModels.includes(m)); - const rejudgeModels = options.rejudge ? modelsToAdd.filter(m => sourceModels.includes(m)) : []; - if (newModels.length === 0 && rejudgeModels.length === 0) { + const existingModels = modelsToAdd.filter(m => sourceModels.includes(m)); + const rejudgeModels = options.rejudge ? existingModels : []; + const retryModels = options.retryFailed ? existingModels : []; + if (newModels.length === 0 && rejudgeModels.length === 0 && retryModels.length === 0) { return { configId, status: 'skipped', reason: 'latest run already includes the model(s)' }; } @@ -108,6 +116,33 @@ export async function addModelsToLatestRun( let generated = 0; let generationErrors = 0; let rejudged = 0; + let retried = 0; + + // Asks `model` for `prompt` in one of the run's variants and stores the answer under its effective ID. + const ask = (model: string, suffix: string, prompt: (typeof targetConfig.prompts)[number], promptData: PromptResponseData) => { + const effectiveId = `${model}${suffix}`; + const { temperature, spIdx } = parseSuffix(suffix); + const systemFromArray = Array.isArray(targetConfig.systems) ? targetConfig.systems[spIdx ?? 0] : undefined; + const systemPrompt = resolveSystemForPrompt(targetConfig, prompt, systemFromArray); + tasks.push(limit(async () => { + const res = await generateResponseForPair({ + modelId: model, + temperature, + systemPrompt, + messages: prompt.messages as ConversationMessage[], + useCache: false, + }); + promptData.modelResponses[effectiveId] = { + finalAssistantResponseText: res.text, + fullConversationHistory: res.history, + hasError: res.hasError, + errorMessage: res.errorMessage, + systemPromptUsed: systemPrompt, + } as any; + generated++; + if (res.hasError) generationErrors++; + })); + }; for (const prompt of targetConfig.prompts) { const promptData: PromptResponseData = { @@ -135,9 +170,15 @@ export async function addModelsToLatestRun( } as any; if (ok) reused++; + // Models being retried are asked again where their answer failed; the new answer is judged. + if (!ok && retryModels.includes(split.base)) { + retried++; + ask(split.base, split.suffix, prompt, promptData); + continue; + } // Models being re-judged keep the response but not the score. - if (rejudgeModels.includes(split.base)) { - if (ok) rejudged++; + if (ok && rejudgeModels.includes(split.base)) { + rejudged++; continue; } @@ -158,41 +199,18 @@ export async function addModelsToLatestRun( // Added models: ask once per variant the run used. for (const model of newModels) { - for (const suffix of suffixes) { - const effectiveId = `${model}${suffix}`; - const { temperature, spIdx } = parseSuffix(suffix); - const systemFromArray = Array.isArray(targetConfig.systems) ? targetConfig.systems[spIdx ?? 0] : undefined; - const systemPrompt = resolveSystemForPrompt(targetConfig, prompt, systemFromArray); - tasks.push(limit(async () => { - const res = await generateResponseForPair({ - modelId: model, - temperature, - systemPrompt, - messages: prompt.messages as ConversationMessage[], - useCache: false, - }); - promptData.modelResponses[effectiveId] = { - finalAssistantResponseText: res.text, - fullConversationHistory: res.history, - hasError: res.hasError, - errorMessage: res.errorMessage, - systemPromptUsed: systemPrompt, - } as any; - generated++; - if (res.hasError) generationErrors++; - })); - } + for (const suffix of suffixes) ask(model, suffix, prompt, promptData); } } await Promise.all(tasks); - logger.info(`[AddModels] ${configId}: reused ${reused} responses, asked ${generated} (${generationErrors} failed), re-judging ${rejudged}.`); + logger.info(`[AddModels] ${configId}: reused ${reused} responses, asked ${generated} (${generationErrors} failed; ${retried} were retries), re-judging ${rejudged}.`); - if (newModels.length === 0 && rejudged === 0) { - return { configId, status: 'skipped', reason: 'no saved responses to re-judge', reused, generated, generationErrors, rejudged }; + if (newModels.length === 0 && rejudged === 0 && retried === 0) { + return { configId, status: 'skipped', reason: 'nothing to re-judge or retry', reused, generated, generationErrors, rejudged, retried }; } if (generated > 0 && generationErrors === generated) { - return { configId, status: 'failed', reason: 'every request to the added model(s) failed', reused, generated, generationErrors, rejudged }; + return { configId, status: 'failed', reason: 'every request to the added model(s) failed', reused, generated, generationErrors, rejudged, retried }; } const evalMethods: EvaluationMethod[] = Array.isArray(source.evalMethodsUsed) && source.evalMethodsUsed.length > 0 @@ -216,7 +234,7 @@ export async function addModelsToLatestRun( prefilledCoverage, ); if (!fileName) { - return { configId, status: 'failed', reason: 'the pipeline saved nothing', reused, generated, generationErrors, rejudged }; + return { configId, status: 'failed', reason: 'the pipeline saved nothing', reused, generated, generationErrors, rejudged, retried }; } const newRun = await getResultByFileName(configId, fileName); @@ -228,5 +246,5 @@ export async function addModelsToLatestRun( logger.warn(`[AddModels] ${configId}: saved ${fileName} but could not read it back to update the page summary.`); } - return { configId, status: 'added', fileName, reused, generated, generationErrors, rejudged }; + return { configId, status: 'added', fileName, reused, generated, generationErrors, rejudged, retried }; }