Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 11 additions & 1 deletion .github/workflows/add-models-to-runs.yml
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,14 @@ on:
description: 'Rebuild homepage, leaderboards and model summaries at the end'
type: boolean
default: true
rejudge:
description: 'Judge the models again from their saved responses where a run already has them'
type: boolean
default: false
retry_failed:
description: 'Ask the models again where a run already has them but their answer failed'
type: boolean
default: false

jobs:
start:
Expand All @@ -22,8 +30,10 @@ jobs:
MODELS: ${{ github.event.inputs.models }}
CONFIG_IDS: ${{ github.event.inputs.config_ids }}
REBUILD: ${{ github.event.inputs.rebuild_summaries }}
REJUDGE: ${{ github.event.inputs.rejudge }}
RETRY_FAILED: ${{ github.event.inputs.retry_failed }}
run: |
BODY=$(python3 -c 'import json, os; split = lambda v: [x.strip() for x in v.split(",") if x.strip()]; print(json.dumps({"models": split(os.environ["MODELS"]), "configIds": split(os.environ["CONFIG_IDS"]), "rebuildSummaries": os.environ["REBUILD"] == "true"}))')
BODY=$(python3 -c 'import json, os; split = lambda v: [x.strip() for x in v.split(",") if x.strip()]; print(json.dumps({"models": split(os.environ["MODELS"]), "configIds": split(os.environ["CONFIG_IDS"]), "rebuildSummaries": os.environ["REBUILD"] == "true", "rejudge": os.environ.get("REJUDGE") == "true", "retryFailed": os.environ.get("RETRY_FAILED") == "true"}))')
# RAILWAY_APP_URL may redirect (http -> https, www -> apex); follow it as a POST.
STATUS=$(curl -sS -L --post301 --post302 --post303 --max-time 60 -o response.txt -w '%{http_code}' -X POST \
-H "Content-Type: application/json" \
Expand Down
11 changes: 11 additions & 0 deletions src/app/api/internal/add-models-to-runs/__tests__/route.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -33,9 +33,20 @@ describe('POST /api/internal/add-models-to-runs', () => {

expect(vi.mocked(addModelsToLatestRun).mock.calls.map(c => c[0])).toEqual(['a', 'b', 'c__d']);
expect(vi.mocked(addModelsToLatestRun).mock.calls[0][1]).toEqual([APERTUS]);
expect(vi.mocked(addModelsToLatestRun).mock.calls[0][3]).toEqual({ rejudge: false, retryFailed: false });
expect(actionBackfillSummary).toHaveBeenCalledTimes(1);
});

it('passes rejudge and retryFailed through to each blueprint', async () => {
vi.mocked(addModelsToLatestRun).mockImplementation(async (configId) => ({ configId, status: 'added' }));

const res = await POST(request({ models: [APERTUS], configIds: ['a'], rejudge: true, retryFailed: true }));
expect(res.status).toBe(202);
await flush();

expect(vi.mocked(addModelsToLatestRun).mock.calls[0][3]).toEqual({ rejudge: true, retryFailed: true });
});

it('keeps going after a blueprint throws, and skips the rebuild when nothing was added', async () => {
vi.mocked(addModelsToLatestRun)
.mockRejectedValueOnce(new Error('boom'))
Expand Down
19 changes: 15 additions & 4 deletions src/app/api/internal/add-models-to-runs/route.ts
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,9 @@ const MAX_CONFIGS = 300;
* Returns 202 immediately and works through the blueprints one at a time in
* the background, logging under "add-models:". With rebuildSummaries, it
* rebuilds the homepage, leaderboards and model summaries once at the end.
* With rejudge, listed models a run already has are judged again from their
* saved responses instead of being skipped. With retryFailed, they are asked
* again on the prompts where their answer failed.
*/
export async function POST(req: NextRequest) {
const authError = checkBackgroundAuth(req);
Expand All @@ -32,6 +35,8 @@ export async function POST(req: NextRequest) {
const models: unknown = body?.models;
const configIds: unknown = body?.configIds;
const rebuildSummaries = body?.rebuildSummaries === true;
const rejudge = body?.rejudge === true;
const retryFailed = body?.retryFailed === true;

if (!Array.isArray(models) || models.length === 0 || !models.every(m => typeof m === 'string' && MODEL_ID_RE.test(m))) {
return NextResponse.json({ error: "'models' must be a non-empty list of provider:model IDs." }, { status: 400 });
Expand All @@ -54,20 +59,26 @@ export async function POST(req: NextRequest) {
},
});

void runJob(models as string[], configIds as string[], rebuildSummaries, logger);
void runJob(models as string[], configIds as string[], rebuildSummaries, { rejudge, retryFailed }, logger);

return NextResponse.json(
{ message: 'Accepted. Progress is logged under "add-models:".', models, configs: configIds.length, rebuildSummaries },
{ message: 'Accepted. Progress is logged under "add-models:".', models, configs: configIds.length, rebuildSummaries, rejudge, retryFailed },
{ status: 202 },
);
}

async function runJob(models: string[], configIds: string[], rebuildSummaries: boolean, logger: any): Promise<void> {
async function runJob(
models: string[],
configIds: string[],
rebuildSummaries: boolean,
options: { rejudge: boolean; retryFailed: boolean },
logger: any,
): Promise<void> {
const results: AddModelsResult[] = [];
for (const [i, configId] of configIds.entries()) {
logger.info(`[AddModels] (${i + 1}/${configIds.length}) ${configId}...`);
try {
const result = await addModelsToLatestRun(configId, models, logger);
const result = await addModelsToLatestRun(configId, models, logger, options);
results.push(result);
logger.info(`[AddModels] (${i + 1}/${configIds.length}) ${configId}: ${result.status}${result.reason ? ` (${result.reason})` : ''}`);
} catch (error: any) {
Expand Down
81 changes: 81 additions & 0 deletions src/cli/services/__tests__/add-models-to-run-service.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -105,6 +105,87 @@ describe('addModelsToLatestRun', () => {
expect(executeComparisonPipeline).not.toHaveBeenCalled();
});

it('with rejudge, judges a model the run already has again from its saved responses', async () => {
const withApertus = sourceRun({
config: { ...sourceRun().config, models: ['openai:gpt-4o', APERTUS] },
effectiveModels: ['openai:gpt-4o[temp:0]', `${APERTUS}[temp:0]`],
allFinalAssistantResponses: {
p1: { 'openai:gpt-4o[temp:0]': 'A1', [`${APERTUS}[temp:0]`]: 'Apertus 1' },
p2: { 'openai:gpt-4o[temp:0]': 'A2', [`${APERTUS}[temp:0]`]: 'Apertus 2' },
},
errors: {},
evaluationResults: {
llmCoverageScores: {
p1: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.8 }, [`${APERTUS}[temp:0]`]: { avgCoverageExtent: 0 } },
p2: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.6 }, [`${APERTUS}[temp:0]`]: { avgCoverageExtent: 0 } },
},
},
});
vi.mocked(storage.getResultByFileName).mockImplementation(async (_c, f) => (f.startsWith('old') ? withApertus : { configId: 'bp' }) as any);

const result = await addModelsToLatestRun('bp', [APERTUS], logger, { rejudge: true });

expect(result).toMatchObject({ status: 'added', generated: 0, rejudged: 2 });
expect(generateResponseForPair).not.toHaveBeenCalled();
const [config, , , , responseMap, , , , , , , , prefilled] = vi.mocked(executeComparisonPipeline).mock.calls[0] as any[];
expect(config.models).toEqual(['openai:gpt-4o', APERTUS]);
expect(responseMap.get('p1').modelResponses[`${APERTUS}[temp:0]`]).toMatchObject({ finalAssistantResponseText: 'Apertus 1', hasError: false });
// Apertus loses its saved scores so it is judged again; the others keep theirs.
expect(Object.keys(prefilled.p1)).toEqual(['openai:gpt-4o[temp:0]']);
expect(Object.keys(prefilled.p2)).toEqual(['openai:gpt-4o[temp:0]']);
});

it('with retryFailed, asks a model the run already has again only where its answer failed', async () => {
const withApertus = sourceRun({
config: { ...sourceRun().config, models: ['openai:gpt-4o', APERTUS] },
effectiveModels: ['openai:gpt-4o[temp:0]', `${APERTUS}[temp:0]`],
allFinalAssistantResponses: {
p1: { 'openai:gpt-4o[temp:0]': 'A1', [`${APERTUS}[temp:0]`]: 'Apertus 1' },
p2: { 'openai:gpt-4o[temp:0]': 'A2', [`${APERTUS}[temp:0]`]: '<<error>>rejected as invalid<</error>>' },
},
errors: { p2: { [`${APERTUS}[temp:0]`]: 'rejected as invalid' } },
evaluationResults: {
llmCoverageScores: {
p1: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.8 }, [`${APERTUS}[temp:0]`]: { avgCoverageExtent: 0.7 } },
p2: { 'openai:gpt-4o[temp:0]': { avgCoverageExtent: 0.6 }, [`${APERTUS}[temp:0]`]: { error: 'Generation failed' } },
},
},
});
vi.mocked(storage.getResultByFileName).mockImplementation(async (_c, f) => (f.startsWith('old') ? withApertus : { configId: 'bp' }) as any);

const result = await addModelsToLatestRun('bp', [APERTUS], logger, { retryFailed: true });

expect(result).toMatchObject({ status: 'added', generated: 1, retried: 1, rejudged: 0 });
// Only the failed pair is asked again, in the run's own variant.
expect(generateResponseForPair).toHaveBeenCalledTimes(1);
expect(vi.mocked(generateResponseForPair).mock.calls[0][0]).toMatchObject({ modelId: APERTUS, temperature: 0, messages: [{ role: 'user', content: 'Q2' }] });
const [, , , , responseMap, , , , , , , , prefilled] = vi.mocked(executeComparisonPipeline).mock.calls[0] as any[];
expect(responseMap.get('p2').modelResponses[`${APERTUS}[temp:0]`]).toMatchObject({ finalAssistantResponseText: 'Apertus answer', hasError: false });
expect(responseMap.get('p1').modelResponses[`${APERTUS}[temp:0]`]).toMatchObject({ finalAssistantResponseText: 'Apertus 1' });
// The good answer keeps its score; the retried one is judged fresh.
expect(prefilled.p1[`${APERTUS}[temp:0]`]).toEqual({ avgCoverageExtent: 0.7 });
expect(prefilled.p2[`${APERTUS}[temp:0]`]).toBeUndefined();
});

it('with retryFailed, skips a blueprint where the model has no failed answers', async () => {
const clean = sourceRun({
config: { ...sourceRun().config, models: ['openai:gpt-4o', APERTUS] },
effectiveModels: ['openai:gpt-4o[temp:0]', `${APERTUS}[temp:0]`],
allFinalAssistantResponses: {
p1: { 'openai:gpt-4o[temp:0]': 'A1', [`${APERTUS}[temp:0]`]: 'Apertus 1' },
p2: { 'openai:gpt-4o[temp:0]': 'A2', [`${APERTUS}[temp:0]`]: 'Apertus 2' },
},
errors: {},
});
vi.mocked(storage.getResultByFileName).mockResolvedValue(clean as any);

const result = await addModelsToLatestRun('bp', [APERTUS], logger, { retryFailed: true });

expect(result.status).toBe('skipped');
expect(generateResponseForPair).not.toHaveBeenCalled();
expect(executeComparisonPipeline).not.toHaveBeenCalled();
});

it('publishes nothing when every request to the added model fails', async () => {
vi.mocked(generateResponseForPair).mockResolvedValue({ text: '<<error>>x<</error>>', history: [], hasError: true, errorMessage: 'x' });

Expand Down
97 changes: 66 additions & 31 deletions src/cli/services/add-models-to-run-service.ts
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,8 @@ export interface AddModelsResult {
reused?: number;
generated?: number;
generationErrors?: number;
rejudged?: number;
retried?: number;
}

/**
Expand Down Expand Up @@ -58,12 +60,21 @@ function parseSuffix(suffix: string): { temperature?: number; spIdx?: number } {
* are asked, once per variant the run used (temperatures, system prompts), and
* only they are judged. The prompts come from the run's own embedded config,
* so later blueprint or model-collection edits can't trigger a full re-run.
*
* With `rejudge`, listed models that the latest run already has keep their
* responses but lose their saved scores, so only they are judged again (for
* example after a judging outage left their scores incomplete).
*
* With `retryFailed`, listed models that the latest run already has are asked
* again only on the prompts where their answer failed (for example when a
* provider rejected the request), and those new answers are judged. Their
* other answers and scores are kept.
*/
export async function addModelsToLatestRun(
configId: string,
modelsToAdd: string[],
logger: Logger,
options: { concurrency?: number } = {},
options: { concurrency?: number; rejudge?: boolean; retryFailed?: boolean } = {},
): Promise<AddModelsResult> {
const runs = await listRunsForConfig(configId);
if (runs.length === 0) return { configId, status: 'skipped', reason: 'no published runs' };
Expand All @@ -78,7 +89,10 @@ export async function addModelsToLatestRun(
.map((m: any) => (typeof m === 'string' ? m : m?.id))
.filter(Boolean);
const newModels = modelsToAdd.filter(m => !sourceModels.includes(m));
if (newModels.length === 0) {
const existingModels = modelsToAdd.filter(m => sourceModels.includes(m));
const rejudgeModels = options.rejudge ? existingModels : [];
const retryModels = options.retryFailed ? existingModels : [];
if (newModels.length === 0 && rejudgeModels.length === 0 && retryModels.length === 0) {
return { configId, status: 'skipped', reason: 'latest run already includes the model(s)' };
}

Expand All @@ -101,6 +115,34 @@ export async function addModelsToLatestRun(
let reused = 0;
let generated = 0;
let generationErrors = 0;
let rejudged = 0;
let retried = 0;

// Asks `model` for `prompt` in one of the run's variants and stores the answer under its effective ID.
const ask = (model: string, suffix: string, prompt: (typeof targetConfig.prompts)[number], promptData: PromptResponseData) => {
const effectiveId = `${model}${suffix}`;
const { temperature, spIdx } = parseSuffix(suffix);
const systemFromArray = Array.isArray(targetConfig.systems) ? targetConfig.systems[spIdx ?? 0] : undefined;
const systemPrompt = resolveSystemForPrompt(targetConfig, prompt, systemFromArray);
tasks.push(limit(async () => {
const res = await generateResponseForPair({
modelId: model,
temperature,
systemPrompt,
messages: prompt.messages as ConversationMessage[],
useCache: false,
});
promptData.modelResponses[effectiveId] = {
finalAssistantResponseText: res.text,
fullConversationHistory: res.history,
hasError: res.hasError,
errorMessage: res.errorMessage,
systemPromptUsed: systemPrompt,
} as any;
generated++;
if (res.hasError) generationErrors++;
}));
};

for (const prompt of targetConfig.prompts) {
const promptData: PromptResponseData = {
Expand All @@ -114,7 +156,8 @@ export async function addModelsToLatestRun(

// Existing models: copy the saved response and score as they are.
for (const effectiveId of sourceEffectiveIds) {
if (!splitEffectiveId(effectiveId, sourceModels)) continue;
const split = splitEffectiveId(effectiveId, sourceModels);
if (!split) continue;
const text = source.allFinalAssistantResponses?.[prompt.id]?.[effectiveId];
const sourceError = source.errors?.[prompt.id]?.[effectiveId];
const ok = typeof text === 'string' && !sourceError && !text.startsWith('<<error>>');
Expand All @@ -127,6 +170,18 @@ export async function addModelsToLatestRun(
} as any;
if (ok) reused++;

// Models being retried are asked again where their answer failed; the new answer is judged.
if (!ok && retryModels.includes(split.base)) {
retried++;
ask(split.base, split.suffix, prompt, promptData);
continue;
}
// Models being re-judged keep the response but not the score.
if (ok && rejudgeModels.includes(split.base)) {
rejudged++;
continue;
}

const inline = inlineCoverage?.[prompt.id]?.[effectiveId];
if (inline) {
(prefilledCoverage[prompt.id] ||= {})[effectiveId] = inline;
Expand All @@ -144,38 +199,18 @@ export async function addModelsToLatestRun(

// Added models: ask once per variant the run used.
for (const model of newModels) {
for (const suffix of suffixes) {
const effectiveId = `${model}${suffix}`;
const { temperature, spIdx } = parseSuffix(suffix);
const systemFromArray = Array.isArray(targetConfig.systems) ? targetConfig.systems[spIdx ?? 0] : undefined;
const systemPrompt = resolveSystemForPrompt(targetConfig, prompt, systemFromArray);
tasks.push(limit(async () => {
const res = await generateResponseForPair({
modelId: model,
temperature,
systemPrompt,
messages: prompt.messages as ConversationMessage[],
useCache: false,
});
promptData.modelResponses[effectiveId] = {
finalAssistantResponseText: res.text,
fullConversationHistory: res.history,
hasError: res.hasError,
errorMessage: res.errorMessage,
systemPromptUsed: systemPrompt,
} as any;
generated++;
if (res.hasError) generationErrors++;
}));
}
for (const suffix of suffixes) ask(model, suffix, prompt, promptData);
}
}

await Promise.all(tasks);
logger.info(`[AddModels] ${configId}: reused ${reused} responses, asked ${generated} (${generationErrors} failed).`);
logger.info(`[AddModels] ${configId}: reused ${reused} responses, asked ${generated} (${generationErrors} failed; ${retried} were retries), re-judging ${rejudged}.`);

if (newModels.length === 0 && rejudged === 0 && retried === 0) {
return { configId, status: 'skipped', reason: 'nothing to re-judge or retry', reused, generated, generationErrors, rejudged, retried };
}
if (generated > 0 && generationErrors === generated) {
return { configId, status: 'failed', reason: 'every request to the added model(s) failed', reused, generated, generationErrors };
return { configId, status: 'failed', reason: 'every request to the added model(s) failed', reused, generated, generationErrors, rejudged, retried };
}

const evalMethods: EvaluationMethod[] = Array.isArray(source.evalMethodsUsed) && source.evalMethodsUsed.length > 0
Expand All @@ -199,7 +234,7 @@ export async function addModelsToLatestRun(
prefilledCoverage,
);
if (!fileName) {
return { configId, status: 'failed', reason: 'the pipeline saved nothing', reused, generated, generationErrors };
return { configId, status: 'failed', reason: 'the pipeline saved nothing', reused, generated, generationErrors, rejudged, retried };
}

const newRun = await getResultByFileName(configId, fileName);
Expand All @@ -211,5 +246,5 @@ export async function addModelsToLatestRun(
logger.warn(`[AddModels] ${configId}: saved ${fileName} but could not read it back to update the page summary.`);
}

return { configId, status: 'added', fileName, reused, generated, generationErrors };
return { configId, status: 'added', fileName, reused, generated, generationErrors, rejudged, retried };
}
Loading