From 0c79536d68c713cbc54522a8cfb737e693bb8ebf Mon Sep 17 00:00:00 2001 From: Bertan Ari Date: Wed, 23 Sep 2026 13:20:24 -0700 Subject: [PATCH 1/2] fix(vscode-lm): use measured Opus 5.5 and Astra input budgets --- .../types/src/__tests__/vscode-llm.spec.ts | 50 ++++++++++++++++--- packages/types/src/providers/vscode-llm.ts | 23 +++++++-- 2 files changed, 62 insertions(+), 11 deletions(-) diff --git a/packages/types/src/__tests__/vscode-llm.spec.ts b/packages/types/src/__tests__/vscode-llm.spec.ts index 71c0f7f5db8..ac51abb367a 100644 --- a/packages/types/src/__tests__/vscode-llm.spec.ts +++ b/packages/types/src/__tests__/vscode-llm.spec.ts @@ -2,9 +2,9 @@ import { describe, it, expect } from "vitest" import { vscodeLlmModels, vscodeLlmDefaultModelId } from "../providers/vscode-llm.js" -// The five families added and the two refreshed on 2026-09-10 (VS Code 1.137.0). +// The five families added and the two refreshed on 2026-09-10 (VS Code 1.137.0), with Astra remeasured 2026-09-23. const SCOPED_ROWS = { - "gpt-6-astra": { contextWindow: 871793, maxInputTokens: 271783, supportsImages: true }, + "gpt-6-astra": { contextWindow: 921793, maxInputTokens: 271789, supportsImages: true }, "grok-4.5": { contextWindow: 424794, maxInputTokens: 199783, supportsImages: false }, "grok-4.6": { contextWindow: 424794, maxInputTokens: 199784, supportsImages: false }, "gemini-3.7-flash": { contextWindow: 935793, maxInputTokens: 935783, supportsImages: true }, @@ -71,12 +71,13 @@ describe("vscodeLlmModels", () => { expect(model.contextWindow, `${family}: contextWindow`).toBe(expected.contextWindow) expect(model.supportsImages, `${family}: supportsImages`).toBe(expected.supportsImages) - // Table-wide conventions: prices are 0 because these rows carry no per-token accounting, - // and supportsPromptCache is false because no cache is modelled here (not a claim about - // the backend). Tool calling is required for Roo to function. + // Caching false is a schema default, not a backend claim. Keep legacy zero prices except + // Astra's unmeasured prices, whose omission is checked separately below. expect(model.supportsPromptCache, `${family}: supportsPromptCache`).toBe(false) - expect(model.inputPrice, `${family}: inputPrice`).toBe(0) - expect(model.outputPrice, `${family}: outputPrice`).toBe(0) + if (family !== "gpt-6-astra") { + expect(model, `${family}: inputPrice`).toHaveProperty("inputPrice", 0) + expect(model, `${family}: outputPrice`).toHaveProperty("outputPrice", 0) + } expect(model.supportsToolCalling, `${family}: supportsToolCalling`).toBe(true) // Provider and gauge look rows up by the live client's family string, so the key, family @@ -86,6 +87,41 @@ describe("vscodeLlmModels", () => { } }) + it("records the 2026-09-23 Opus 5.5 row at its verified accepted lower bound", () => { + // 677108 is the largest ACCEPTED request over 13 trials (VS Code 1.137.0, copilot), not the + // exact ceiling — 695778 was rejected and model-declined errors left the bracket open. + // Images and tool calling were observed; prompt caching remains unverified. + expect(vscodeLlmModels).toHaveProperty("claude-opus-5.5") + expect(vscodeLlmModels["claude-opus-5.5"].maxInputTokens).toBe(677108) + expect(vscodeLlmModels["claude-opus-5.5"].contextWindow).toBe(871793) + expect(vscodeLlmModels["claude-opus-5.5"].family).toBe("claude-opus-5.5") + expect(vscodeLlmModels["claude-opus-5.5"].version).toBe("claude-opus-5.5") + expect(vscodeLlmModels["claude-opus-5.5"].name).toBe("Claude Opus 5.5") + expect(vscodeLlmModels["claude-opus-5.5"].supportsImages).toBe(true) + expect(vscodeLlmModels["claude-opus-5.5"].supportsToolCalling).toBe(true) + expect(vscodeLlmModels["claude-opus-5.5"].supportsPromptCache).toBe(false) + expect(vscodeLlmModels["claude-opus-5.5"].inputPrice).toBe(0) + expect(vscodeLlmModels["claude-opus-5.5"].outputPrice).toBe(0) + }) + + it("records the 2026-09-23 measured GPT-6 Astra row without unmeasured pricing or caching claims", () => { + // 271789 accepted with an adjacent rejection at 271790 over 20 trials; images and tool calling + // observed. supportsPromptCache `false` is the required-schema default, NOT a verified result. + const astra = vscodeLlmModels["gpt-6-astra"] + expect(astra.maxInputTokens).toBe(271789) + expect(astra.contextWindow).toBe(921793) + expect(astra.family).toBe("gpt-6-astra") + expect(astra.version).toBe("gpt-6-astra") + expect(astra.name).toBe("GPT-6 Astra") + expect(astra.supportsImages).toBe(true) + expect(astra.supportsToolCalling).toBe(true) + expect(astra.supportsPromptCache).toBe(false) + // Pricing was never measured, and ModelInfo makes those fields optional, so they are omitted + // rather than filled with zeros the way older hand-authored rows were. + expect(astra).not.toHaveProperty("inputPrice") + expect(astra).not.toHaveProperty("outputPrice") + }) + it("keeps both window fields populated and positive for every row", () => { // The two fields are ALLOWED to differ (claude-opus-4.8: 679560 vs 197897), so equality is // deliberately NOT asserted. The invariant is positive integers on both fields; a missing or diff --git a/packages/types/src/providers/vscode-llm.ts b/packages/types/src/providers/vscode-llm.ts index 99a737efe0f..32a967f7b01 100644 --- a/packages/types/src/providers/vscode-llm.ts +++ b/packages/types/src/providers/vscode-llm.ts @@ -10,6 +10,21 @@ export const vscodeLlmDefaultModelId: VscodeLlmModelId = "claude-sonnet-4.5" // sibling row. Per-row evidence, lower-bound and vendor caveats: // myplans/vscode-lm-model-table-integrity/vscode-lm-model-table-integrity-design.md export const vscodeLlmModels = { + // Measured 2026-09-23, VS Code 1.137.0/copilot, 13 trials: accepted 677108, rejected 695778. + // Refusals prevent an exact ceiling; keep the accepted lower bound rather than a sibling limit. + // Image/tool inputs succeeded; caching `false` is an unverified schema-required default. + "claude-opus-5.5": { + contextWindow: 871793, + supportsImages: true, + supportsPromptCache: false, + inputPrice: 0, + outputPrice: 0, + family: "claude-opus-5.5", + version: "claude-opus-5.5", + name: "Claude Opus 5.5", + supportsToolCalling: true, + maxInputTokens: 677108, + }, "claude-opus-5": { contextWindow: 935793, supportsImages: true, @@ -118,17 +133,17 @@ export const vscodeLlmModels = { supportsToolCalling: true, maxInputTokens: 135790, }, + // Measured 2026-09-23, VS Code 1.137.0/copilot, 20-trial binary search: 271789 accepted, 271790 rejected. + // Image/tool inputs succeeded; caching `false` is an unverified schema-required default. "gpt-6-astra": { - contextWindow: 871793, + contextWindow: 921793, supportsImages: true, supportsPromptCache: false, - inputPrice: 0, - outputPrice: 0, family: "gpt-6-astra", version: "gpt-6-astra", name: "GPT-6 Astra", supportsToolCalling: true, - maxInputTokens: 271783, + maxInputTokens: 271789, }, "gpt-5.6-luna": { contextWindow: 199753, From 4b819b4e7b0a19e4db4184650b41d0f12120d1cf Mon Sep 17 00:00:00 2001 From: Bertan Ari Date: Wed, 23 Sep 2026 13:36:51 -0700 Subject: [PATCH 2/2] test(vscode-lm): align astra budget assertions with measured 271789 --- src/api/providers/__tests__/vscode-lm.spec.ts | 4 ++-- .../components/ui/hooks/__tests__/useSelectedModel.spec.ts | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/src/api/providers/__tests__/vscode-lm.spec.ts b/src/api/providers/__tests__/vscode-lm.spec.ts index 8fc38ef8d1d..0467b420c86 100644 --- a/src/api/providers/__tests__/vscode-lm.spec.ts +++ b/src/api/providers/__tests__/vscode-lm.spec.ts @@ -142,7 +142,7 @@ describe("VsCodeLmHandler", () => { }) it.each([ - ["gpt-6-astra", 271783], + ["gpt-6-astra", 271789], ["grok-4.5", 199783], ["grok-4.6", 199784], ["gemini-3.7-flash", 935783], @@ -192,7 +192,7 @@ describe("VsCodeLmHandler", () => { vsCodeLmModelSelector: { vendor: "some-other-vendor", family: "gpt-6-astra" }, }) - expect(copilotHandler.getCondenseContextWindow()).toBe(271783) + expect(copilotHandler.getCondenseContextWindow()).toBe(271789) expect(otherVendorHandler.getCondenseContextWindow()).toBe(copilotHandler.getCondenseContextWindow()) copilotHandler.dispose() diff --git a/webview-ui/src/components/ui/hooks/__tests__/useSelectedModel.spec.ts b/webview-ui/src/components/ui/hooks/__tests__/useSelectedModel.spec.ts index dab5a31bbf2..80013a0119b 100644 --- a/webview-ui/src/components/ui/hooks/__tests__/useSelectedModel.spec.ts +++ b/webview-ui/src/components/ui/hooks/__tests__/useSelectedModel.spec.ts @@ -810,7 +810,7 @@ describe("useSelectedModel", () => { }) it.each([ - ["gpt-6-astra", 271783, true], + ["gpt-6-astra", 271789, true], ["grok-4.5", 199783, false], ["grok-4.6", 199784, false], ["gemini-3.7-flash", 935783, true],