diff --git a/packages/test/src/test/ai-provider-api/LongContextPricing.test.ts b/packages/test/src/test/ai-provider-api/LongContextPricing.test.ts index 1c787ee8d..630cd2e55 100644 --- a/packages/test/src/test/ai-provider-api/LongContextPricing.test.ts +++ b/packages/test/src/test/ai-provider-api/LongContextPricing.test.ts @@ -75,6 +75,9 @@ describe("Gemini long-context pricing", () => { describe("OpenAI long-context pricing", () => { const longContext = [ { id: "gpt-6-astra", short: 10, long: 20, longOutput: 75 }, + { id: "gpt-6.1-sol", short: 2, long: 4, longOutput: 15 }, + { id: "gpt-6-sol", short: 2, long: 4, longOutput: 15 }, + { id: "gpt-6-luna", short: 0.1, long: 0.2, longOutput: 0.75 }, { id: "gpt-5.6-sol", short: 4, long: 8, longOutput: 30 }, { id: "gpt-5.6-terra", short: 2, long: 4, longOutput: 18 }, { id: "gpt-5.6-luna", short: 0.2, long: 0.4, longOutput: 1.8 }, diff --git a/packages/test/src/test/ai-provider-api/ModelSearchPricing.test.ts b/packages/test/src/test/ai-provider-api/ModelSearchPricing.test.ts index 3b59a29e1..1b0cf9258 100644 --- a/packages/test/src/test/ai-provider-api/ModelSearchPricing.test.ts +++ b/packages/test/src/test/ai-provider-api/ModelSearchPricing.test.ts @@ -247,6 +247,7 @@ describe("a rate card must match the model's billing unit", () => { */ it.each([ ["gpt-6-astra", 10, 12.5], + ["gpt-6.1-sol", 2, 2.5], ["gpt-5.6-sol", 4, 5], ["gpt-5.6-terra", 2, 2.5], ["gpt-5.6-luna", 0.2, 0.25], diff --git a/packages/test/src/test/ai-provider-api/OpenAiEffortPolicy.test.ts b/packages/test/src/test/ai-provider-api/OpenAiEffortPolicy.test.ts index bd37b5d74..00091cf31 100644 --- a/packages/test/src/test/ai-provider-api/OpenAiEffortPolicy.test.ts +++ b/packages/test/src/test/ai-provider-api/OpenAiEffortPolicy.test.ts @@ -40,6 +40,10 @@ describe("openaiEffortPolicy", () => { }); }); + it("treats gpt-6.1-sol like gpt-6", () => { + expect(openaiEffortPolicy(cfg("gpt-6.1-sol"))).toEqual(openaiEffortPolicy(cfg("gpt-6-astra"))); + }); + it("returns no levels for embeddings, image, and gpt-4o", () => { expect(openaiEffortPolicy(cfg("text-embedding-3-small"))?.supported).toEqual([]); expect(openaiEffortPolicy(cfg("gpt-image-2"))?.supported).toEqual([]); diff --git a/packages/test/src/test/ai-provider/OpenAI_ReasoningTemperature.test.ts b/packages/test/src/test/ai-provider/OpenAI_ReasoningTemperature.test.ts index 49a7f1721..23ed0e122 100644 --- a/packages/test/src/test/ai-provider/OpenAI_ReasoningTemperature.test.ts +++ b/packages/test/src/test/ai-provider/OpenAI_ReasoningTemperature.test.ts @@ -5,8 +5,30 @@ */ import { _testOnly } from "@workglow/openai/ai"; -import { describe, expect, it } from "vitest"; -const { finalizeResponsesRequest } = _testOnly; +import { setLogger } from "@workglow/util"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +const { finalizeResponsesRequest, _resetOpenAIResponsesWarnings } = _testOnly; + +const warn = vi.fn(); +const logger = { + debug: () => {}, + info: () => {}, + warn, + error: () => {}, + fatal: () => {}, + child: () => logger, + time: () => {}, + timeEnd: () => {}, + group: () => {}, + groupEnd: () => {}, +}; + +beforeEach(() => { + warn.mockClear(); + _resetOpenAIResponsesWarnings(); + setLogger(logger as never); +}); +afterEach(() => setLogger(undefined as never)); /** * `temperature` and `reasoning` are not independently selectable on the OpenAI @@ -30,10 +52,20 @@ describe("finalizeResponsesRequest on a reasoning model", () => { expect(params.reasoning).toEqual({ effort: "medium" }); }); - it("sends the default and drops a pinned temperature rather than turning reasoning off", () => { - const params = finalizeResponsesRequest(luna(), { model: "gpt-5.6-luna", temperature: 0.7 }); + it("turns reasoning off to honour a pinned temperature when no effort is configured", () => { + const params = finalizeResponsesRequest(luna(), { model: "gpt-5.6-luna", temperature: 0 }); + expect(params).toMatchObject({ reasoning: { effort: "none" }, temperature: 0 }); + expect(warn).not.toHaveBeenCalled(); + }); + + it("keeps the default effort when effort_options leave none out, and warns on the drop", () => { + const params = finalizeResponsesRequest(luna({ effort_options: ["medium", "high"] }), { + model: "gpt-5.6-luna", + temperature: 0.7, + }); expect(params.reasoning).toEqual({ effort: "medium" }); expect(params.temperature).toBeUndefined(); + expect(warn).toHaveBeenCalledTimes(1); }); it("maps model.effort over the class default", () => { @@ -48,6 +80,8 @@ describe("finalizeResponsesRequest on a reasoning model", () => { }); expect(params.reasoning).toEqual({ effort: "high" }); expect(params.temperature).toBeUndefined(); + expect(warn).toHaveBeenCalledTimes(1); + expect(warn.mock.calls[0][0]).toContain("temperature"); }); it("keeps the temperature alongside an effort of none", () => { @@ -80,6 +114,7 @@ describe("finalizeResponsesRequest on a model that cannot turn reasoning off", ( const params = finalizeResponsesRequest(astra(), { model: "gpt-6-astra", temperature: 0.4 }); expect(params.reasoning).toEqual({ effort: "medium" }); expect(params.temperature).toBeUndefined(); + expect(warn).toHaveBeenCalledTimes(1); }); it("does not map model.effort none onto the request", () => { @@ -129,3 +164,34 @@ describe("finalizeResponsesRequest on a model that takes no reasoning", () => { expect(params.temperature).toBe(0.4); }); }); + +describe("finalizeResponsesRequest across model classes with a pinned temperature", () => { + it("drops it and warns on gpt-6.1-sol, which cannot turn reasoning off", () => { + const params = finalizeResponsesRequest( + { provider_config: { model_name: "gpt-6.1-sol" } } as never, + { model: "gpt-6.1-sol", temperature: 0 } + ); + expect(params.reasoning).toEqual({ effort: "medium" }); + expect(params.temperature).toBeUndefined(); + expect(warn).toHaveBeenCalledTimes(1); + }); + + it("drops it and warns on an o-series model rather than sending effort none", () => { + const params = finalizeResponsesRequest({ provider_config: { model_name: "o3" } } as never, { + model: "o3", + temperature: 0, + }); + expect(params.reasoning).toEqual({ effort: "medium" }); + expect(params.temperature).toBeUndefined(); + expect(warn).toHaveBeenCalledTimes(1); + }); + + it("drops it and warns on gpt-5.5 rather than sending effort none", () => { + const params = finalizeResponsesRequest( + { provider_config: { model_name: "gpt-5.5" } } as never, + { model: "gpt-5.5", temperature: 0 } + ); + expect(params.reasoning).toEqual({ effort: "medium" }); + expect(params.temperature).toBeUndefined(); + }); +}); diff --git a/providers/openai/src/ai/common/OpenAI_Client.ts b/providers/openai/src/ai/common/OpenAI_Client.ts index 29374f225..0eff56064 100644 --- a/providers/openai/src/ai/common/OpenAI_Client.ts +++ b/providers/openai/src/ai/common/OpenAI_Client.ts @@ -8,6 +8,7 @@ import { isBrowserLike, resolveApiKey, validateProviderBaseUrl } from "@workglow import { resolveEnabledEffort, type ModelEffort } from "@workglow/ai/worker"; import { openaiEffortPolicy } from "./OpenAI_EffortPolicy"; import type { OpenAiModelConfig } from "./OpenAI_ModelSchema"; +import { warnTemperatureDroppedOnce } from "./OpenAI_ResponsesWarnings"; /** Maps coarse {@link ModelEffort} to OpenAI Responses `reasoning.effort`. */ const EFFORT_TO_OPENAI: Record = { @@ -177,6 +178,10 @@ export function resolvePromptCacheKey( return `wg-${fnv1aHex(material)}`; } +function getModelId(model: OpenAiModelConfig | undefined): string { + return model?.provider_config?.model_name ?? ""; +} + /** * Applies the per-request Responses fields common to every OpenAI text run-fn: * the model's `reasoning` config and a stable `prompt_cache_key`. Mutates and @@ -193,8 +198,10 @@ export function resolvePromptCacheKey( * * `temperature` is rejected alongside any reasoning effort but `"none"` — * verified live against `gpt-5.6-luna` and `gpt-6-astra` — so it is dropped - * whenever reasoning is on rather than failing the request. To sample at a - * pinned temperature, set the effort to `none` on a model that allows it. + * whenever reasoning is on rather than failing the request, with a warning. + * A pinned temperature with no effort configured keeps reasoning off only on + * the GPT-5.6 family, where `none` plus a temperature is accepted; any other + * model is not guessed at, so its temperature is dropped. */ export function finalizeResponsesRequest( model: OpenAiModelConfig | undefined, @@ -202,12 +209,28 @@ export function finalizeResponsesRequest( ): Record { const policy = openaiEffortPolicy(model); const fallback = resolveEnabledEffort({ ...model, effort: policy.default }, policy); + const configured = getReasoningConfig(model); + const canDisable = + /^gpt-5\.6/i.test(getModelId(model)) && + policy.supported.includes("none") && + resolveEnabledEffort({ ...model, effort: "none" }, policy) === "none"; const reasoning = - getReasoningConfig(model) ?? - (fallback !== undefined ? { effort: EFFORT_TO_OPENAI[fallback] } : undefined); + configured ?? + (params.temperature !== undefined && canDisable + ? { effort: EFFORT_TO_OPENAI.none } + : fallback !== undefined + ? { effort: EFFORT_TO_OPENAI[fallback] } + : undefined); if (reasoning !== undefined) { params.reasoning = reasoning; - if (reasoning.effort !== "none") delete params.temperature; + if (reasoning.effort !== "none" && params.temperature !== undefined) { + const requested = params.model; + warnTemperatureDroppedOnce( + typeof requested === "string" ? requested : getModelId(model), + reasoning.effort + ); + delete params.temperature; + } } params.prompt_cache_key = resolvePromptCacheKey(model, params); return params; diff --git a/providers/openai/src/ai/common/OpenAI_ModelSearch.ts b/providers/openai/src/ai/common/OpenAI_ModelSearch.ts index 0cda41a2e..68a4ef05f 100644 --- a/providers/openai/src/ai/common/OpenAI_ModelSearch.ts +++ b/providers/openai/src/ai/common/OpenAI_ModelSearch.ts @@ -24,6 +24,7 @@ interface OpenAiModelListItem { const OPENAI_FALLBACK: Array<{ label: string; value: string }> = [ { label: "gpt-6-astra", value: "gpt-6-astra" }, + { label: "gpt-6.1-sol", value: "gpt-6.1-sol" }, { label: "gpt-6-sol", value: "gpt-6-sol" }, { label: "gpt-6-luna", value: "gpt-6-luna" }, { label: "gpt-image-2.5-sunburst", value: "gpt-image-2.5-sunburst" }, diff --git a/providers/openai/src/ai/common/OpenAI_Pricing.ts b/providers/openai/src/ai/common/OpenAI_Pricing.ts index d78979630..61ce419ca 100644 --- a/providers/openai/src/ai/common/OpenAI_Pricing.ts +++ b/providers/openai/src/ai/common/OpenAI_Pricing.ts @@ -100,6 +100,7 @@ function gptImageCard(): ModelPricing { */ export const OPENAI_PRICING: Record = { "gpt-6-astra": flagshipCard({ input: 10, output: 50, cached: 1, cacheWrite: 12.5 }), + "gpt-6.1-sol": flagshipCard({ input: 2, output: 10, cached: 0.1, cacheWrite: 2.5 }), "gpt-6-sol": flagshipCard({ input: 2, output: 10, cached: 0.2, cacheWrite: 2.5 }), "gpt-6-luna": flagshipCard({ input: 0.1, output: 0.5, cached: 0.01, cacheWrite: 0.125 }), "gpt-5.6-sol": flagshipCard({ input: 4, output: 20, cached: 0.4, cacheWrite: 5 }), diff --git a/providers/openai/src/ai/common/OpenAI_ResponsesWarnings.ts b/providers/openai/src/ai/common/OpenAI_ResponsesWarnings.ts index f5c557984..ddf3411ba 100644 --- a/providers/openai/src/ai/common/OpenAI_ResponsesWarnings.ts +++ b/providers/openai/src/ai/common/OpenAI_ResponsesWarnings.ts @@ -50,8 +50,22 @@ export function warnStrictDowngradedOnce(model: string, reason: string): void { ); } -/** @internal test helper — clear both dedupe sets. */ +const warnedTemperatureDrops = new Set(); + +export function warnTemperatureDroppedOnce(model: string, effort: string | undefined): void { + const key = `${model}::${effort ?? ""}`; + if (warnedTemperatureDrops.has(key)) return; + warnedTemperatureDrops.add(key); + getLogger().warn( + `OpenAI model "${model}" rejects temperature alongside reasoning` + + (effort ? ` effort "${effort}"` : "") + + `; the pinned temperature has been dropped.` + ); +} + +/** @internal test helper — clear all dedupe sets. */ export function _resetOpenAIResponsesWarnings(): void { + warnedTemperatureDrops.clear(); warnedPenaltyDrops.clear(); warnedStrictDownshifts.clear(); } diff --git a/providers/openai/src/ai/index.ts b/providers/openai/src/ai/index.ts index f17ff4f8a..6d8ab3755 100644 --- a/providers/openai/src/ai/index.ts +++ b/providers/openai/src/ai/index.ts @@ -27,6 +27,7 @@ import { _resetOpenAIResponsesWarnings, warnPenaltyDroppedOnce, warnStrictDowngradedOnce, + warnTemperatureDroppedOnce, } from "./common/OpenAI_ResponsesWarnings"; import { isStrictCompatibleSchema } from "./common/OpenAI_StructuredGeneration"; import { OpenAiQueuedProvider } from "./OpenAiQueuedProvider"; @@ -44,6 +45,7 @@ export const _testOnly = { isStrictCompatibleSchema, warnPenaltyDroppedOnce, warnStrictDowngradedOnce, + warnTemperatureDroppedOnce, _resetOpenAIResponsesWarnings, setOpenAIClientForTests: clientTestOnly.setOpenAIClientForTests, } as const;