From 347204ff33e42bba7781e7b057443ea9808e9d06 Mon Sep 17 00:00:00 2001 From: luvs01 <27862058+luvs01@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:24:19 +0900 Subject: [PATCH] fix(spend): enforce ceilings on native chat sends --- src/server/chat-completions.ts | 11 +++++++++- src/server/chat-native.ts | 15 +++++++++++++ structure/transports/responses.md | 4 +++- .../chat-completions-endpoint.test.ts | 22 +++++++++++++++++++ 4 files changed, 50 insertions(+), 2 deletions(-) diff --git a/src/server/chat-completions.ts b/src/server/chat-completions.ts index 3cb6c9a3798..c21c6e66a3a 100644 --- a/src/server/chat-completions.ts +++ b/src/server/chat-completions.ts @@ -199,7 +199,16 @@ async function handleChatCompletionsWithBudget( } // Combos must enter the Responses routing path so child selection, forced default // effort, failover, and per-attempt telemetry run before any native Chat send. - if (!route.combo && !effortRow && isNativeChatRouteEligible(route, chatBody, config)) chatNativeRoute = route; + if (!route.combo && !effortRow && isNativeChatRouteEligible(route, chatBody, config)) { + chatNativeRoute = route; + if (logCtx.usageLogInputTokens === undefined) { + logCtx.usageLogInputTokens = Math.max(1, estimateTokens(JSON.stringify(chatBody.messages ?? []), requestedModel)); + } + const outputCeiling = chatBody.max_completion_tokens ?? chatBody.max_tokens; + if (typeof outputCeiling === "number" && outputCeiling > 0) { + logCtx.spendOutputCeilingTokens = Math.trunc(outputCeiling); + } + } } catch (err) { if (err instanceof AdmissionModelDeniedError) { logCtx.requestedModel = requestedModel; diff --git a/src/server/chat-native.ts b/src/server/chat-native.ts index 71ef323ebf5..4d759fbef5d 100644 --- a/src/server/chat-native.ts +++ b/src/server/chat-native.ts @@ -48,6 +48,7 @@ import { transientRetryPolicyFor, } from "../providers/key-failover"; import { fastPolicyForModel } from "../providers/service-tier"; +import { stampApiKeyAccountLabel } from "../providers/label"; import { providerApiKeySelectionIsCurrent, resolveCurrentProviderApiKeyTransport } from "../providers/api-key-selection"; import { enrichOpenCodeZenFreeTierMessage } from "../providers/opencode-zen-rate-limit"; import type { OcxProviderTransport } from "../providers/xai-transport"; @@ -68,12 +69,16 @@ import { } from "./request-log"; import { jsonCompletionSse, nativeChatSse, structuredError, usageFromChat } from "./chat-native-sse"; import { registerTurn, unregisterTurn } from "./lifecycle"; +import { attachRequestSpendTracker } from "./responses/request-spend"; +import { workflowRefusalResponse } from "./workflow-refusal"; type Rec = Record; const MAX_NATIVE_CHAT_JSON_BYTES = 32 * 1024 * 1024; const MAX_NATIVE_CHAT_ERROR_BYTES = 64 * 1024; +class NativeChatSpendRefusal extends Error {} + const chatEffortSnapshots = new WeakMap {}); } catch { /* already closed */ } activeProvider = rotated; + stampApiKeyAccountLabel(logCtx, route.providerName, activeProvider); activeAdapter = createOpenAIChatAdapter(activeProvider); releaseRetainedRequest(); activeRequest = buildActiveRequest(); @@ -443,6 +453,11 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio cleanupAbort(); upstream.abort(); if (req.signal.aborted) return fail(499, "Client cancelled request", "client_cancelled"); + if (error instanceof NativeChatSpendRefusal) { + const refusal = workflowRefusalResponse("workflow-spend-exhausted", logCtx); + finishLog(429); + return refusal; + } if (isTranslatorBudgetExceededError(error)) { return fail(413, "request translation buffer exceeded the safe limit", "request_too_large", "translation_buffer_limit"); } diff --git a/structure/transports/responses.md b/structure/transports/responses.md index a82e8f8cb08..e6fac914d34 100644 --- a/structure/transports/responses.md +++ b/structure/transports/responses.md @@ -1229,7 +1229,9 @@ target's own recovery decision, while the physical-send total is what binds ever The request's send budget bounds how many times it may reach upstream; the spend ledger bounds what those sends may cost, and it is the only bound here that survives a restart. Its production caller is `request-spend.ts`, installed on the execution budget at genuine ingress in `core.ts` -and parked on the log context so `addFinalRequestLog` can settle it. +and parked on the log context so `addFinalRequestLog` can settle it. Native Chat installs the +same tracker before its independent physical-send ladder and charges it immediately before each +dispatch, so taking that fast path cannot bypass root, identity, or provider-pool ceilings. It books by observing the budget's own send counter rather than by being called from each dispatch site. That counter moves exactly once per physical send — a reservation increments it, a diff --git a/tests/responses/chat-completions-endpoint.test.ts b/tests/responses/chat-completions-endpoint.test.ts index b87b3177779..034230a2109 100644 --- a/tests/responses/chat-completions-endpoint.test.ts +++ b/tests/responses/chat-completions-endpoint.test.ts @@ -170,6 +170,28 @@ function mockConfig(baseUrl: string, providerOverrides: Partial { + takeSpendHome(); + const upstream = mockChatUpstreamCapturing(); + const config = mockConfig(`${upstream.server.url.toString().replace(/\/$/, "")}/v1`); + config.spend = { pool: { maxTokens: 1 } }; + saveConfig(config); + const server = startServer(0); + try { + const response = await fetch(new URL("/v1/chat/completions", server.url), { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ model: "mock/test-model", messages: [{ role: "user", content: "hello" }] }), + }); + expect(response.status).toBe(429); + expect(response.headers.get("x-opencodex-local-refusal")).toBe("workflow_spend_exhausted"); + expect(upstream.captured).toHaveLength(0); + } finally { + await server.stop(true); + upstream.server.stop(true); + } +}); + type StreamedToolCall = { index?: number; id?: string;