diff --git a/.github/workflows/add-models-to-runs.yml b/.github/workflows/add-models-to-runs.yml index e10545d..cc07b43 100644 --- a/.github/workflows/add-models-to-runs.yml +++ b/.github/workflows/add-models-to-runs.yml @@ -12,6 +12,14 @@ on: description: 'Rebuild homepage, leaderboards and model summaries at the end' type: boolean default: true + rejudge: + description: 'Judge the models again from their saved responses where a run already has them' + type: boolean + default: false + retry_failed: + description: 'Ask the models again where a run already has them but their answer failed' + type: boolean + default: false jobs: start: @@ -22,8 +30,10 @@ jobs: MODELS: ${{ github.event.inputs.models }} CONFIG_IDS: ${{ github.event.inputs.config_ids }} REBUILD: ${{ github.event.inputs.rebuild_summaries }} + REJUDGE: ${{ github.event.inputs.rejudge }} + RETRY_FAILED: ${{ github.event.inputs.retry_failed }} run: | - BODY=$(python3 -c 'import json, os; split = lambda v: [x.strip() for x in v.split(",") if x.strip()]; print(json.dumps({"models": split(os.environ["MODELS"]), "configIds": split(os.environ["CONFIG_IDS"]), "rebuildSummaries": os.environ["REBUILD"] == "true"}))') + BODY=$(python3 -c 'import json, os; split = lambda v: [x.strip() for x in v.split(",") if x.strip()]; print(json.dumps({"models": split(os.environ["MODELS"]), "configIds": split(os.environ["CONFIG_IDS"]), "rebuildSummaries": os.environ["REBUILD"] == "true", "rejudge": os.environ.get("REJUDGE") == "true", "retryFailed": os.environ.get("RETRY_FAILED") == "true"}))') # RAILWAY_APP_URL may redirect (http -> https, www -> apex); follow it as a POST. STATUS=$(curl -sS -L --post301 --post302 --post303 --max-time 60 -o response.txt -w '%{http_code}' -X POST \ -H "Content-Type: application/json" \ diff --git a/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts b/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts index 56e487b..4cdf57c 100644 --- a/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts +++ b/src/app/api/internal/add-models-to-runs/__tests__/route.test.ts @@ -33,9 +33,20 @@ describe('POST /api/internal/add-models-to-runs', () => { expect(vi.mocked(addModelsToLatestRun).mock.calls.map(c => c[0])).toEqual(['a', 'b', 'c__d']); expect(vi.mocked(addModelsToLatestRun).mock.calls[0][1]).toEqual([APERTUS]); + expect(vi.mocked(addModelsToLatestRun).mock.calls[0][3]).toEqual({ rejudge: false, retryFailed: false }); expect(actionBackfillSummary).toHaveBeenCalledTimes(1); }); + it('passes rejudge and retryFailed through to each blueprint', async () => { + vi.mocked(addModelsToLatestRun).mockImplementation(async (configId) => ({ configId, status: 'added' })); + + const res = await POST(request({ models: [APERTUS], configIds: ['a'], rejudge: true, retryFailed: true })); + expect(res.status).toBe(202); + await flush(); + + expect(vi.mocked(addModelsToLatestRun).mock.calls[0][3]).toEqual({ rejudge: true, retryFailed: true }); + }); + it('keeps going after a blueprint throws, and skips the rebuild when nothing was added', async () => { vi.mocked(addModelsToLatestRun) .mockRejectedValueOnce(new Error('boom')) diff --git a/src/app/api/internal/add-models-to-runs/route.ts b/src/app/api/internal/add-models-to-runs/route.ts index c6633a3..5cb2569 100644 --- a/src/app/api/internal/add-models-to-runs/route.ts +++ b/src/app/api/internal/add-models-to-runs/route.ts @@ -18,6 +18,9 @@ const MAX_CONFIGS = 300; * Returns 202 immediately and works through the blueprints one at a time in * the background, logging under "add-models:". With rebuildSummaries, it * rebuilds the homepage, leaderboards and model summaries once at the end. + * With rejudge, listed models a run already has are judged again from their + * saved responses instead of being skipped. With retryFailed, they are asked + * again on the prompts where their answer failed. */ export async function POST(req: NextRequest) { const authError = checkBackgroundAuth(req); @@ -32,6 +35,8 @@ export async function POST(req: NextRequest) { const models: unknown = body?.models; const configIds: unknown = body?.configIds; const rebuildSummaries = body?.rebuildSummaries === true; + const rejudge = body?.rejudge === true; + const retryFailed = body?.retryFailed === true; if (!Array.isArray(models) || models.length === 0 || !models.every(m => typeof m === 'string' && MODEL_ID_RE.test(m))) { return NextResponse.json({ error: "'models' must be a non-empty list of provider:model IDs." }, { status: 400 }); @@ -54,20 +59,26 @@ export async function POST(req: NextRequest) { }, }); - void runJob(models as string[], configIds as string[], rebuildSummaries, logger); + void runJob(models as string[], configIds as string[], rebuildSummaries, { rejudge, retryFailed }, logger); return NextResponse.json( - { message: 'Accepted. Progress is logged under "add-models:".', models, configs: configIds.length, rebuildSummaries }, + { message: 'Accepted. Progress is logged under "add-models:".', models, configs: configIds.length, rebuildSummaries, rejudge, retryFailed }, { status: 202 }, ); } -async function runJob(models: string[], configIds: string[], rebuildSummaries: boolean, logger: any): Promise { +async function runJob( + models: string[], + configIds: string[], + rebuildSummaries: boolean, + options: { rejudge: boolean; retryFailed: boolean }, + logger: any, +): Promise { const results: AddModelsResult[] = []; for (const [i, configId] of configIds.entries()) { logger.info(`[AddModels] (${i + 1}/${configIds.length}) ${configId}...`); try { - const result = await addModelsToLatestRun(configId, models, logger); + const result = await addModelsToLatestRun(configId, models, logger, options); results.push(result); logger.info(`[AddModels] (${i + 1}/${configIds.length}) ${configId}: ${result.status}${result.reason ? ` (${result.reason})` : ''}`); } catch (error: any) { diff --git a/src/cli/services/__tests__/add-models-to-run-service.test.ts b/src/cli/services/__tests__/add-models-to-run-service.test.ts index 4224d13..30e6ee0 100644 --- a/src/cli/services/__tests__/add-models-to-run-service.test.ts +++ b/src/cli/services/__tests__/add-models-to-run-service.test.ts @@ -105,6 +105,87 @@ describe('addModelsToLatestRun', () => { expect(executeComparisonPipeline).not.toHaveBeenCalled(); }); + it('with rejudge, judges a model the run already has again from its saved responses', async () => { + const withApertus = sourceRun({ + config: { ...sourceRun().config, models: ['openai:gpt-4o', APERTUS] }, + effectiveModels: ['openai:gpt-4o[temp:0]', `${APERTUS}[temp:0]`], + allFinalAssistantResponses: { + p1: { 'openai:gpt-4o[temp:0]': 'A1', [`${APERTUS}[temp:0]`]: 'Apertus 1' }, + p2: { 'openai:gpt-4o[temp:0]': 'A2', [`${APERTUS}[temp:0]`]: 'Apertus 2' }, + }, + errors: {}, + evaluationResults: { + llmCoverageScores: { + p1: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.8 }, [`${APERTUS}[temp:0]`]: { avgCoverageExtent: 0 } }, + p2: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.6 }, [`${APERTUS}[temp:0]`]: { avgCoverageExtent: 0 } }, + }, + }, + }); + vi.mocked(storage.getResultByFileName).mockImplementation(async (_c, f) => (f.startsWith('old') ? withApertus : { configId: 'bp' }) as any); + + const result = await addModelsToLatestRun('bp', [APERTUS], logger, { rejudge: true }); + + expect(result).toMatchObject({ status: 'added', generated: 0, rejudged: 2 }); + expect(generateResponseForPair).not.toHaveBeenCalled(); + const [config, , , , responseMap, , , , , , , , prefilled] = vi.mocked(executeComparisonPipeline).mock.calls[0] as any[]; + expect(config.models).toEqual(['openai:gpt-4o', APERTUS]); + expect(responseMap.get('p1').modelResponses[`${APERTUS}[temp:0]`]).toMatchObject({ finalAssistantResponseText: 'Apertus 1', hasError: false }); + // Apertus loses its saved scores so it is judged again; the others keep theirs. + expect(Object.keys(prefilled.p1)).toEqual(['openai:gpt-4o[temp:0]']); + expect(Object.keys(prefilled.p2)).toEqual(['openai:gpt-4o[temp:0]']); + }); + + it('with retryFailed, asks a model the run already has again only where its answer failed', async () => { + const withApertus = sourceRun({ + config: { ...sourceRun().config, models: ['openai:gpt-4o', APERTUS] }, + effectiveModels: ['openai:gpt-4o[temp:0]', `${APERTUS}[temp:0]`], + allFinalAssistantResponses: { + p1: { 'openai:gpt-4o[temp:0]': 'A1', [`${APERTUS}[temp:0]`]: 'Apertus 1' }, + p2: { 'openai:gpt-4o[temp:0]': 'A2', [`${APERTUS}[temp:0]`]: '<>rejected as invalid<>' }, + }, + errors: { p2: { [`${APERTUS}[temp:0]`]: 'rejected as invalid' } }, + evaluationResults: { + llmCoverageScores: { + p1: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.8 }, [`${APERTUS}[temp:0]`]: { avgCoverageExtent: 0.7 } }, + p2: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.6 }, [`${APERTUS}[temp:0]`]: { error: 'Generation failed' } }, + }, + }, + }); + vi.mocked(storage.getResultByFileName).mockImplementation(async (_c, f) => (f.startsWith('old') ? withApertus : { configId: 'bp' }) as any); + + const result = await addModelsToLatestRun('bp', [APERTUS], logger, { retryFailed: true }); + + expect(result).toMatchObject({ status: 'added', generated: 1, retried: 1, rejudged: 0 }); + // Only the failed pair is asked again, in the run's own variant. + expect(generateResponseForPair).toHaveBeenCalledTimes(1); + expect(vi.mocked(generateResponseForPair).mock.calls[0][0]).toMatchObject({ modelId: APERTUS, temperature: 0, messages: [{ role: 'user', content: 'Q2' }] }); + const [, , , , responseMap, , , , , , , , prefilled] = vi.mocked(executeComparisonPipeline).mock.calls[0] as any[]; + expect(responseMap.get('p2').modelResponses[`${APERTUS}[temp:0]`]).toMatchObject({ finalAssistantResponseText: 'Apertus answer', hasError: false }); + expect(responseMap.get('p1').modelResponses[`${APERTUS}[temp:0]`]).toMatchObject({ finalAssistantResponseText: 'Apertus 1' }); + // The good answer keeps its score; the retried one is judged fresh. + expect(prefilled.p1[`${APERTUS}[temp:0]`]).toEqual({ avgCoverageExtent: 0.7 }); + expect(prefilled.p2[`${APERTUS}[temp:0]`]).toBeUndefined(); + }); + + it('with retryFailed, skips a blueprint where the model has no failed answers', async () => { + const clean = sourceRun({ + config: { ...sourceRun().config, models: ['openai:gpt-4o', APERTUS] }, + effectiveModels: ['openai:gpt-4o[temp:0]', `${APERTUS}[temp:0]`], + allFinalAssistantResponses: { + p1: { 'openai:gpt-4o[temp:0]': 'A1', [`${APERTUS}[temp:0]`]: 'Apertus 1' }, + p2: { 'openai:gpt-4o[temp:0]': 'A2', [`${APERTUS}[temp:0]`]: 'Apertus 2' }, + }, + errors: {}, + }); + vi.mocked(storage.getResultByFileName).mockResolvedValue(clean as any); + + const result = await addModelsToLatestRun('bp', [APERTUS], logger, { retryFailed: true }); + + expect(result.status).toBe('skipped'); + expect(generateResponseForPair).not.toHaveBeenCalled(); + expect(executeComparisonPipeline).not.toHaveBeenCalled(); + }); + it('publishes nothing when every request to the added model fails', async () => { vi.mocked(generateResponseForPair).mockResolvedValue({ text: '<>x<>', history: [], hasError: true, errorMessage: 'x' }); diff --git a/src/cli/services/add-models-to-run-service.ts b/src/cli/services/add-models-to-run-service.ts index 83fe23c..45daada 100644 --- a/src/cli/services/add-models-to-run-service.ts +++ b/src/cli/services/add-models-to-run-service.ts @@ -27,6 +27,8 @@ export interface AddModelsResult { reused?: number; generated?: number; generationErrors?: number; + rejudged?: number; + retried?: number; } /** @@ -58,12 +60,21 @@ function parseSuffix(suffix: string): { temperature?: number; spIdx?: number } { * are asked, once per variant the run used (temperatures, system prompts), and * only they are judged. The prompts come from the run's own embedded config, * so later blueprint or model-collection edits can't trigger a full re-run. + * + * With `rejudge`, listed models that the latest run already has keep their + * responses but lose their saved scores, so only they are judged again (for + * example after a judging outage left their scores incomplete). + * + * With `retryFailed`, listed models that the latest run already has are asked + * again only on the prompts where their answer failed (for example when a + * provider rejected the request), and those new answers are judged. Their + * other answers and scores are kept. */ export async function addModelsToLatestRun( configId: string, modelsToAdd: string[], logger: Logger, - options: { concurrency?: number } = {}, + options: { concurrency?: number; rejudge?: boolean; retryFailed?: boolean } = {}, ): Promise { const runs = await listRunsForConfig(configId); if (runs.length === 0) return { configId, status: 'skipped', reason: 'no published runs' }; @@ -78,7 +89,10 @@ export async function addModelsToLatestRun( .map((m: any) => (typeof m === 'string' ? m : m?.id)) .filter(Boolean); const newModels = modelsToAdd.filter(m => !sourceModels.includes(m)); - if (newModels.length === 0) { + const existingModels = modelsToAdd.filter(m => sourceModels.includes(m)); + const rejudgeModels = options.rejudge ? existingModels : []; + const retryModels = options.retryFailed ? existingModels : []; + if (newModels.length === 0 && rejudgeModels.length === 0 && retryModels.length === 0) { return { configId, status: 'skipped', reason: 'latest run already includes the model(s)' }; } @@ -101,6 +115,34 @@ export async function addModelsToLatestRun( let reused = 0; let generated = 0; let generationErrors = 0; + let rejudged = 0; + let retried = 0; + + // Asks `model` for `prompt` in one of the run's variants and stores the answer under its effective ID. + const ask = (model: string, suffix: string, prompt: (typeof targetConfig.prompts)[number], promptData: PromptResponseData) => { + const effectiveId = `${model}${suffix}`; + const { temperature, spIdx } = parseSuffix(suffix); + const systemFromArray = Array.isArray(targetConfig.systems) ? targetConfig.systems[spIdx ?? 0] : undefined; + const systemPrompt = resolveSystemForPrompt(targetConfig, prompt, systemFromArray); + tasks.push(limit(async () => { + const res = await generateResponseForPair({ + modelId: model, + temperature, + systemPrompt, + messages: prompt.messages as ConversationMessage[], + useCache: false, + }); + promptData.modelResponses[effectiveId] = { + finalAssistantResponseText: res.text, + fullConversationHistory: res.history, + hasError: res.hasError, + errorMessage: res.errorMessage, + systemPromptUsed: systemPrompt, + } as any; + generated++; + if (res.hasError) generationErrors++; + })); + }; for (const prompt of targetConfig.prompts) { const promptData: PromptResponseData = { @@ -114,7 +156,8 @@ export async function addModelsToLatestRun( // Existing models: copy the saved response and score as they are. for (const effectiveId of sourceEffectiveIds) { - if (!splitEffectiveId(effectiveId, sourceModels)) continue; + const split = splitEffectiveId(effectiveId, sourceModels); + if (!split) continue; const text = source.allFinalAssistantResponses?.[prompt.id]?.[effectiveId]; const sourceError = source.errors?.[prompt.id]?.[effectiveId]; const ok = typeof text === 'string' && !sourceError && !text.startsWith('<>'); @@ -127,6 +170,18 @@ export async function addModelsToLatestRun( } as any; if (ok) reused++; + // Models being retried are asked again where their answer failed; the new answer is judged. + if (!ok && retryModels.includes(split.base)) { + retried++; + ask(split.base, split.suffix, prompt, promptData); + continue; + } + // Models being re-judged keep the response but not the score. + if (ok && rejudgeModels.includes(split.base)) { + rejudged++; + continue; + } + const inline = inlineCoverage?.[prompt.id]?.[effectiveId]; if (inline) { (prefilledCoverage[prompt.id] ||= {})[effectiveId] = inline; @@ -144,38 +199,18 @@ export async function addModelsToLatestRun( // Added models: ask once per variant the run used. for (const model of newModels) { - for (const suffix of suffixes) { - const effectiveId = `${model}${suffix}`; - const { temperature, spIdx } = parseSuffix(suffix); - const systemFromArray = Array.isArray(targetConfig.systems) ? targetConfig.systems[spIdx ?? 0] : undefined; - const systemPrompt = resolveSystemForPrompt(targetConfig, prompt, systemFromArray); - tasks.push(limit(async () => { - const res = await generateResponseForPair({ - modelId: model, - temperature, - systemPrompt, - messages: prompt.messages as ConversationMessage[], - useCache: false, - }); - promptData.modelResponses[effectiveId] = { - finalAssistantResponseText: res.text, - fullConversationHistory: res.history, - hasError: res.hasError, - errorMessage: res.errorMessage, - systemPromptUsed: systemPrompt, - } as any; - generated++; - if (res.hasError) generationErrors++; - })); - } + for (const suffix of suffixes) ask(model, suffix, prompt, promptData); } } await Promise.all(tasks); - logger.info(`[AddModels] ${configId}: reused ${reused} responses, asked ${generated} (${generationErrors} failed).`); + logger.info(`[AddModels] ${configId}: reused ${reused} responses, asked ${generated} (${generationErrors} failed; ${retried} were retries), re-judging ${rejudged}.`); + if (newModels.length === 0 && rejudged === 0 && retried === 0) { + return { configId, status: 'skipped', reason: 'nothing to re-judge or retry', reused, generated, generationErrors, rejudged, retried }; + } if (generated > 0 && generationErrors === generated) { - return { configId, status: 'failed', reason: 'every request to the added model(s) failed', reused, generated, generationErrors }; + return { configId, status: 'failed', reason: 'every request to the added model(s) failed', reused, generated, generationErrors, rejudged, retried }; } const evalMethods: EvaluationMethod[] = Array.isArray(source.evalMethodsUsed) && source.evalMethodsUsed.length > 0 @@ -199,7 +234,7 @@ export async function addModelsToLatestRun( prefilledCoverage, ); if (!fileName) { - return { configId, status: 'failed', reason: 'the pipeline saved nothing', reused, generated, generationErrors }; + return { configId, status: 'failed', reason: 'the pipeline saved nothing', reused, generated, generationErrors, rejudged, retried }; } const newRun = await getResultByFileName(configId, fileName); @@ -211,5 +246,5 @@ export async function addModelsToLatestRun( logger.warn(`[AddModels] ${configId}: saved ${fileName} but could not read it back to update the page summary.`); } - return { configId, status: 'added', fileName, reused, generated, generationErrors }; + return { configId, status: 'added', fileName, reused, generated, generationErrors, rejudged, retried }; }