Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
50 changes: 43 additions & 7 deletions packages/types/src/__tests__/vscode-llm.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2,9 +2,9 @@ import { describe, it, expect } from "vitest"

import { vscodeLlmModels, vscodeLlmDefaultModelId } from "../providers/vscode-llm.js"

// The five families added and the two refreshed on 2026-09-10 (VS Code 1.137.0).
// The five families added and the two refreshed on 2026-09-10 (VS Code 1.137.0), with Astra remeasured 2026-09-23.
const SCOPED_ROWS = {
"gpt-6-astra": { contextWindow: 871793, maxInputTokens: 271783, supportsImages: true },
"gpt-6-astra": { contextWindow: 921793, maxInputTokens: 271789, supportsImages: true },
"grok-4.5": { contextWindow: 424794, maxInputTokens: 199783, supportsImages: false },
"grok-4.6": { contextWindow: 424794, maxInputTokens: 199784, supportsImages: false },
"gemini-3.7-flash": { contextWindow: 935793, maxInputTokens: 935783, supportsImages: true },
Expand Down Expand Up @@ -71,12 +71,13 @@ describe("vscodeLlmModels", () => {
expect(model.contextWindow, `${family}: contextWindow`).toBe(expected.contextWindow)
expect(model.supportsImages, `${family}: supportsImages`).toBe(expected.supportsImages)

// Table-wide conventions: prices are 0 because these rows carry no per-token accounting,
// and supportsPromptCache is false because no cache is modelled here (not a claim about
// the backend). Tool calling is required for Roo to function.
// Caching false is a schema default, not a backend claim. Keep legacy zero prices except
// Astra's unmeasured prices, whose omission is checked separately below.
expect(model.supportsPromptCache, `${family}: supportsPromptCache`).toBe(false)
expect(model.inputPrice, `${family}: inputPrice`).toBe(0)
expect(model.outputPrice, `${family}: outputPrice`).toBe(0)
if (family !== "gpt-6-astra") {
expect(model, `${family}: inputPrice`).toHaveProperty("inputPrice", 0)
expect(model, `${family}: outputPrice`).toHaveProperty("outputPrice", 0)
}
expect(model.supportsToolCalling, `${family}: supportsToolCalling`).toBe(true)

// Provider and gauge look rows up by the live client's family string, so the key, family
Expand All @@ -86,6 +87,41 @@ describe("vscodeLlmModels", () => {
}
})

it("records the 2026-09-23 Opus 5.5 row at its verified accepted lower bound", () => {
// 677108 is the largest ACCEPTED request over 13 trials (VS Code 1.137.0, copilot), not the
// exact ceiling — 695778 was rejected and model-declined errors left the bracket open.
// Images and tool calling were observed; prompt caching remains unverified.
expect(vscodeLlmModels).toHaveProperty("claude-opus-5.5")
expect(vscodeLlmModels["claude-opus-5.5"].maxInputTokens).toBe(677108)
expect(vscodeLlmModels["claude-opus-5.5"].contextWindow).toBe(871793)
expect(vscodeLlmModels["claude-opus-5.5"].family).toBe("claude-opus-5.5")
expect(vscodeLlmModels["claude-opus-5.5"].version).toBe("claude-opus-5.5")
expect(vscodeLlmModels["claude-opus-5.5"].name).toBe("Claude Opus 5.5")
expect(vscodeLlmModels["claude-opus-5.5"].supportsImages).toBe(true)
expect(vscodeLlmModels["claude-opus-5.5"].supportsToolCalling).toBe(true)
expect(vscodeLlmModels["claude-opus-5.5"].supportsPromptCache).toBe(false)
expect(vscodeLlmModels["claude-opus-5.5"].inputPrice).toBe(0)
expect(vscodeLlmModels["claude-opus-5.5"].outputPrice).toBe(0)
})

it("records the 2026-09-23 measured GPT-6 Astra row without unmeasured pricing or caching claims", () => {
// 271789 accepted with an adjacent rejection at 271790 over 20 trials; images and tool calling
// observed. supportsPromptCache `false` is the required-schema default, NOT a verified result.
const astra = vscodeLlmModels["gpt-6-astra"]
expect(astra.maxInputTokens).toBe(271789)
expect(astra.contextWindow).toBe(921793)
expect(astra.family).toBe("gpt-6-astra")
expect(astra.version).toBe("gpt-6-astra")
expect(astra.name).toBe("GPT-6 Astra")
expect(astra.supportsImages).toBe(true)
expect(astra.supportsToolCalling).toBe(true)
expect(astra.supportsPromptCache).toBe(false)
// Pricing was never measured, and ModelInfo makes those fields optional, so they are omitted
// rather than filled with zeros the way older hand-authored rows were.
expect(astra).not.toHaveProperty("inputPrice")
expect(astra).not.toHaveProperty("outputPrice")
})

it("keeps both window fields populated and positive for every row", () => {
// The two fields are ALLOWED to differ (claude-opus-4.8: 679560 vs 197897), so equality is
// deliberately NOT asserted. The invariant is positive integers on both fields; a missing or
Expand Down
23 changes: 19 additions & 4 deletions packages/types/src/providers/vscode-llm.ts
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,21 @@ export const vscodeLlmDefaultModelId: VscodeLlmModelId = "claude-sonnet-4.5"
// sibling row. Per-row evidence, lower-bound and vendor caveats:
// myplans/vscode-lm-model-table-integrity/vscode-lm-model-table-integrity-design.md
export const vscodeLlmModels = {
// Measured 2026-09-23, VS Code 1.137.0/copilot, 13 trials: accepted 677108, rejected 695778.
// Refusals prevent an exact ceiling; keep the accepted lower bound rather than a sibling limit.
// Image/tool inputs succeeded; caching `false` is an unverified schema-required default.
"claude-opus-5.5": {
contextWindow: 871793,
supportsImages: true,
supportsPromptCache: false,
inputPrice: 0,
outputPrice: 0,
family: "claude-opus-5.5",
version: "claude-opus-5.5",
name: "Claude Opus 5.5",
supportsToolCalling: true,
maxInputTokens: 677108,
},
"claude-opus-5": {
contextWindow: 935793,
supportsImages: true,
Expand Down Expand Up @@ -118,17 +133,17 @@ export const vscodeLlmModels = {
supportsToolCalling: true,
maxInputTokens: 135790,
},
// Measured 2026-09-23, VS Code 1.137.0/copilot, 20-trial binary search: 271789 accepted, 271790 rejected.
// Image/tool inputs succeeded; caching `false` is an unverified schema-required default.
"gpt-6-astra": {
contextWindow: 871793,
contextWindow: 921793,
supportsImages: true,
supportsPromptCache: false,
inputPrice: 0,
outputPrice: 0,
family: "gpt-6-astra",
version: "gpt-6-astra",
name: "GPT-6 Astra",
supportsToolCalling: true,
maxInputTokens: 271783,
maxInputTokens: 271789,
},
"gpt-5.6-luna": {
contextWindow: 199753,
Expand Down
4 changes: 2 additions & 2 deletions src/api/providers/__tests__/vscode-lm.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -142,7 +142,7 @@ describe("VsCodeLmHandler", () => {
})

it.each([
["gpt-6-astra", 271783],
["gpt-6-astra", 271789],
["grok-4.5", 199783],
["grok-4.6", 199784],
["gemini-3.7-flash", 935783],
Expand Down Expand Up @@ -192,7 +192,7 @@ describe("VsCodeLmHandler", () => {
vsCodeLmModelSelector: { vendor: "some-other-vendor", family: "gpt-6-astra" },
})

expect(copilotHandler.getCondenseContextWindow()).toBe(271783)
expect(copilotHandler.getCondenseContextWindow()).toBe(271789)
expect(otherVendorHandler.getCondenseContextWindow()).toBe(copilotHandler.getCondenseContextWindow())

copilotHandler.dispose()
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -810,7 +810,7 @@ describe("useSelectedModel", () => {
})

it.each([
["gpt-6-astra", 271783, true],
["gpt-6-astra", 271789, true],
["grok-4.5", 199783, false],
["grok-4.6", 199784, false],
["gemini-3.7-flash", 935783, true],
Expand Down
Loading