diff --git a/docs-site/src/content/docs/ja/reference/configuration/server.md b/docs-site/src/content/docs/ja/reference/configuration/server.md index cce5a94257b..227ded775dd 100644 --- a/docs-site/src/content/docs/ja/reference/configuration/server.md +++ b/docs-site/src/content/docs/ja/reference/configuration/server.md @@ -199,3 +199,7 @@ This takes effect on the **next routed `ocx claude` launch**, injecting `CLAUDE_ Claude Code **2.1.257 or newer** is required for FORCE. Plugin and built-in agents (including Explore/Plan) and per-call model arguments are overridden. Forks and subagent skills with `model: inherit` keep the main conversation model. The main loop and Haiku/small-fast sidecars are unaffected. Existing roster files remain available. The dashboard warns about old or unknown CLI versions, unavailable targets, and either variable already present in `settings.json` → `env` (which overrides launch env). Detection is read-only and server-local: it cannot inspect another launch shell, another machine, or project-local settings. An unknown result is not proof of force support. + +## トークン予約と上限 + +`spend.root.maxTokens`、`spend.identity.maxTokens`、`spend.pool.maxTokens` のいずれかがリクエストに適用される場合、追跡容量の不足などでトークン予約を記録できなければ送信を拒否します。適用される上限がないリクエストは観測のみを続けます。 diff --git a/docs-site/src/content/docs/ko/reference/configuration/server.md b/docs-site/src/content/docs/ko/reference/configuration/server.md index ee3a2fe8729..14c89eaa329 100644 --- a/docs-site/src/content/docs/ko/reference/configuration/server.md +++ b/docs-site/src/content/docs/ko/reference/configuration/server.md @@ -256,3 +256,7 @@ This takes effect on the **next routed `ocx claude` launch**, injecting `CLAUDE_ Claude Code **2.1.257 or newer** is required for FORCE. Plugin and built-in agents (including Explore/Plan) and per-call model arguments are overridden. Forks and subagent skills with `model: inherit` keep the main conversation model. The main loop and Haiku/small-fast sidecars are unaffected. Existing roster files remain available. The dashboard warns about old or unknown CLI versions, unavailable targets, and either variable already present in `settings.json` → `env` (which overrides launch env). Detection is read-only and server-local: it cannot inspect another launch shell, another machine, or project-local settings. An unknown result is not proof of force support. + +## 토큰 예약과 한도 + +요청에 `spend.root.maxTokens`, `spend.identity.maxTokens`, `spend.pool.maxTokens` 중 하나가 적용되면 추적 용량 부족 등으로 토큰 예약을 기록할 수 없을 때 전송을 거부합니다. 적용되는 한도가 없는 요청은 계속 관측만 합니다. diff --git a/docs-site/src/content/docs/reference/configuration/server.md b/docs-site/src/content/docs/reference/configuration/server.md index 527c9ce6c10..e5ba4e2dc33 100644 --- a/docs-site/src/content/docs/reference/configuration/server.md +++ b/docs-site/src/content/docs/reference/configuration/server.md @@ -29,7 +29,7 @@ runs helper features around provider requests. | `usageLedgerMaxBytes?` | `number` | unset | Opt-in ceiling in bytes for `usage.jsonl`. Absent means the request history grows without limit, which stays the default. See [usage history size](#usage-history-size). | | `appOwnedMemoryBudgetMb?` | `number` | `256` | Cap in MiB for evictable app-owned logs, caches, blobs, and continuation payloads. Range 64–4096; not an RSS cap. | | `metricsExport.enabled?` | `boolean` | `false` | Enable process-local aggregate request metrics at authenticated `GET /api/metrics`. Restart required; disabled mode returns 404 and starts no exporter activity. | -| `spend?` | `{ root?: { maxTokens?: number }; identity?: { maxTokens?: number }; pool?: { maxTokens?: number }; retentionDays?: number }` | unset | Durable token ceilings, off unless you write one. Each scope bounds settled spend plus in-flight reservations plus unresolved spend: `root` is one task including its whole fan-out, `identity` is one account across every task it serves, and `pool` is one provider pool. They intersect, so a request is admitted only when all three have room — which is what holds a ceiling against a client that mints a new task id per request. A reservation is the request's whole input plus its enforceable output ceiling, counted as if every cached prefix misses. Observe-only mode still journals, so every server owns the state directory's single-writer lease; an explicit sibling must use a separate `OPENCODEX_HOME`. Spend survives an ordinary process restart when its writes reached the filesystem, but the journal does not promise survival across host power loss because each append is not fsynced. Raising or removing the value is what grants more. `maxTokens` must be a positive integer (0 would refuse everything), `retentionDays` is 1–365 and defaults to 7, and an unknown key in this section is rejected rather than ignored. A refusal is a local HTTP 429 carrying `x-opencodex-local-refusal: workflow_spend_exhausted`, and its message names the scope and the ceiling; no provider is contacted. | +| `spend?` | `{ root?: { maxTokens?: number }; identity?: { maxTokens?: number }; pool?: { maxTokens?: number }; retentionDays?: number }` | unset | Durable token ceilings, off unless you write one. Each scope bounds settled spend plus in-flight reservations plus unresolved spend: `root` is one task including its whole fan-out, `identity` is one account across every task it serves, and `pool` is one provider pool. They intersect, so a request is admitted only when all three have room — which is what holds a ceiling against a client that mints a new task id per request. A reservation is the request's whole input plus its enforceable output ceiling, counted as if every cached prefix misses. Observe-only mode still journals, so every server owns the state directory's single-writer lease; an explicit sibling must use a separate `OPENCODEX_HOME`. Spend survives an ordinary process restart when its writes reached the filesystem, but the journal does not promise survival across host power loss because each append is not fsynced. Raising or removing the value is what grants more. `maxTokens` must be a positive integer (0 would refuse everything), `retentionDays` is 1–365 and defaults to 7, and an unknown key in this section is rejected rather than ignored. A refusal is a local HTTP 429 carrying `x-opencodex-local-refusal: workflow_spend_exhausted`, and its message names the scope and the ceiling; no provider is contacted. With an applicable ceiling, dispatch is also refused if its token reservation cannot be booked, including full tracking capacity. Requests without an applicable ceiling remain observe-only. | | `codexAutoStart?` | `boolean` | `true` | Let the Codex shim run `ocx ensure` before launching Codex. False makes ensure a no-op. | | `codexShimAutoRestore?` | `boolean` | `true` | Restore an installed shim after a completed external Codex update replaces it. Environment opt-out: `OPENCODEX_CODEX_SHIM_AUTO_RESTORE=0`. | diff --git a/docs-site/src/content/docs/ru/reference/configuration/server.md b/docs-site/src/content/docs/ru/reference/configuration/server.md index 3c859f5a5d6..6ca105b9dc9 100644 --- a/docs-site/src/content/docs/ru/reference/configuration/server.md +++ b/docs-site/src/content/docs/ru/reference/configuration/server.md @@ -247,3 +247,7 @@ This takes effect on the **next routed `ocx claude` launch**, injecting `CLAUDE_ Claude Code **2.1.257 or newer** is required for FORCE. Plugin and built-in agents (including Explore/Plan) and per-call model arguments are overridden. Forks and subagent skills with `model: inherit` keep the main conversation model. The main loop and Haiku/small-fast sidecars are unaffected. Existing roster files remain available. The dashboard warns about old or unknown CLI versions, unavailable targets, and either variable already present in `settings.json` → `env` (which overrides launch env). Detection is read-only and server-local: it cannot inspect another launch shell, another machine, or project-local settings. An unknown result is not proof of force support. + +## Резервирование токенов и лимиты + +Если к запросу применяется `spend.root.maxTokens`, `spend.identity.maxTokens` или `spend.pool.maxTokens`, отправка отклоняется, когда резерв токенов нельзя учесть, в том числе из-за заполнения хранилища отслеживаемых записей. Запросы без применимого лимита остаются в режиме наблюдения. diff --git a/docs-site/src/content/docs/zh-cn/reference/configuration/server.md b/docs-site/src/content/docs/zh-cn/reference/configuration/server.md index f261601a65f..9a2d8ceba78 100644 --- a/docs-site/src/content/docs/zh-cn/reference/configuration/server.md +++ b/docs-site/src/content/docs/zh-cn/reference/configuration/server.md @@ -213,3 +213,7 @@ This takes effect on the **next routed `ocx claude` launch**, injecting `CLAUDE_ Claude Code **2.1.257 or newer** is required for FORCE. Plugin and built-in agents (including Explore/Plan) and per-call model arguments are overridden. Forks and subagent skills with `model: inherit` keep the main conversation model. The main loop and Haiku/small-fast sidecars are unaffected. Existing roster files remain available. The dashboard warns about old or unknown CLI versions, unavailable targets, and either variable already present in `settings.json` → `env` (which overrides launch env). Detection is read-only and server-local: it cannot inspect another launch shell, another machine, or project-local settings. An unknown result is not proof of force support. + +## 令牌预留与额度限制 + +如果请求受 `spend.root.maxTokens`、`spend.identity.maxTokens` 或 `spend.pool.maxTokens` 限制,无法记录令牌预留时会拒绝发送,包括跟踪容量已满的情况。没有适用额度限制的请求仍保持仅观察模式。 diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index a6d5af795fa..29885b87eaf 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -1552,6 +1552,7 @@ "responses-snapshot-repair.test.ts": "responses", "responses-sparse-terminal-tool-scope.test.ts": "responses", "responses-spend-ledger-wiring.test.ts": "responses", + "responses-spend-capacity-guard.test.ts": "responses", "responses-spill-acl-recovery.test.ts": "responses", "responses-spill-inspection.test.ts": "responses", "responses-spill-orphan-sweep.test.ts": "responses", diff --git a/src/server/responses/request-spend.ts b/src/server/responses/request-spend.ts index 8485b3d79dd..fd0d4d0c212 100644 --- a/src/server/responses/request-spend.ts +++ b/src/server/responses/request-spend.ts @@ -1,9 +1,9 @@ import { randomUUID } from "node:crypto"; import type { RequestSendObserver } from "../../lib/request-execution-budget"; -import { sharedSpendLedger, type SpendReservationLedger } from "../../lib/spend-reservation-ledger"; +import { sharedSpendLedger, type SpendReservationLedger, type SpendScopes } from "../../lib/spend-reservation-ledger"; import { SpendLedgerOwnerError } from "../../lib/spend-ledger-owner"; import { markLocalRequestLogRefusal, type RequestLogContext } from "../request-log"; -import { recordWorkflowRefusalEvent, workflowDenialSummary } from "../../lib/workflow-budget"; +import { recordWorkflowRefusalEvent, workflowDenialSummary, type WorkflowDenial } from "../../lib/workflow-budget"; /** The terminal usage a request reported, in the only two fields the ledger books. */ export interface TerminalSpendUsage { @@ -81,44 +81,44 @@ export function createRequestSpendTracker( // short of its limit forever and refuse nothing. const alreadySent = options?.alreadySent === true; const sendId = randomUUID(); - const decision = ledger().reserve({ + const bookedLedger = ledger(); + const scopes: SpendScopes = { + ...(rootId !== undefined ? { rootId } : {}), + // The existing privacy-safe log label is aliased again by the ledger. + ...(logCtx.accountLogLabel !== undefined ? { identityId: logCtx.accountLogLabel } : {}), + ...(logCtx.provider !== undefined ? { poolId: logCtx.provider } : {}), + }; + const policy = bookedLedger.policy; + const enforced = (scopes.rootId !== undefined && policy.root.maxTokens !== undefined) + || (scopes.identityId !== undefined && policy.identity.maxTokens !== undefined) + || (scopes.poolId !== undefined && policy.pool.maxTokens !== undefined); + const decision = bookedLedger.reserve({ sendId, - scopes: { - ...(rootId !== undefined ? { rootId } : {}), - // Already the privacy-safe label the request log uses, and the ledger aliases it - // again on the way to disk. A raw credential never reaches either. - ...(logCtx.accountLogLabel !== undefined ? { identityId: logCtx.accountLogLabel } : {}), - ...(logCtx.provider !== undefined ? { poolId: logCtx.provider } : {}), - }, + scopes, inputTokens: logCtx.spendInputEstimateTokens ?? logCtx.usageLogInputTokens ?? 0, outputCeilingTokens: logCtx.spendOutputCeilingTokens ?? 0, ...(alreadySent ? { alreadySent: true } : {}), }); if (!decision.reserved) { + // An applicable ceiling cannot authorize an unbooked send, including when tracking + // capacity is full. Unconfigured/nonapplicable requests remain observe-only, and a + // physical send reported after dispatch cannot be refused retroactively. + if (alreadySent || !enforced) return true; refusals += 1; const denial = decision.denial; - // A ceiling refuses, and so does a ledger that cannot make the reservation durable - // under one. That second case is the whole reason this store is on disk: admitting a - // send whose record a restart would forget is how an exhausted budget comes back with - // a fresh allowance, and the ledger raises those two denials ONLY when a limit is - // configured -- so an install that configured nothing is still never refused here. - // Capacity and a duplicate send id stay permissive: they say the ledger cannot account - // for this send, which is a degradation to report, not an outage to cause. - if (denial.reason === "reserve-not-durable" || denial.reason === "journal-corrupt") { - return alreadySent; - } - if (denial.reason !== "spend-limit-exceeded") return true; - // This send is refused, and the dispatch path that asked will report an exhausted send - // budget -- from there, that is all it can see. The row is where an operator actually - // looks, so the ceiling is named on it here: a locally assigned code wins in - // addFinalRequestLog, so the request that CROSSED the ceiling reads as a spend refusal - // rather than as the ordinary budget exhaustion it would otherwise be indistinguishable - // from. The event ring gets the same pair so /api/workflow-budget agrees with the row. - const detail = { scope: denial.scope, limit: denial.limit, projected: denial.projected }; - const summary = workflowDenialSummary("workflow-spend-exhausted", detail); + const reason: WorkflowDenial = denial.reason === "duplicate-send-id" + ? "workflow-send-replayed" + : denial.reason === "reserve-not-durable" || denial.reason === "journal-corrupt" + ? "workflow-spend-undurable" + : denial.reason === "tracking-capacity-exhausted" + ? "workflow-tracking-exhausted" + : "workflow-spend-exhausted"; + const detail = denial.reason === "spend-limit-exceeded" + ? { scope: denial.scope, limit: denial.limit, projected: denial.projected } : undefined; + const summary = workflowDenialSummary(reason, detail); markLocalRequestLogRefusal(logCtx, summary.code); logCtx.errorCode = summary.code; - recordWorkflowRefusalEvent(rootId, "workflow-spend-exhausted", Date.now(), detail); + recordWorkflowRefusalEvent(rootId, reason, Date.now(), detail); return false; } live.push(sendId); diff --git a/structure/transports/responses-spend.md b/structure/transports/responses-spend.md index 4e1f2777658..63b04dbf076 100644 --- a/structure/transports/responses-spend.md +++ b/structure/transports/responses-spend.md @@ -130,6 +130,14 @@ so a second restart has nothing to redo. booking, the settlement split, the refund, a ceiling that refuses a dispatch rather than describing it afterwards, and the restart. +The request tracker checks the current policy against the exact root, identity and pool scopes +on each charge. With an applicable ceiling, any refused booking prevents a new dispatch, +including full tracking capacity or a duplicate send id; the log uses the existing specific +workflow refusal reason. Requests without an applicable ceiling remain observe-only. Reports +of sends that already left stay permissive and do not count as local refusals; this guard does +not promise complete accounting when those post-dispatch bookings fail. +`tests/responses/responses-spend-capacity-guard.test.ts` covers these boundaries. + The default policy still sets no token ceiling on any scope, so an unconfigured install accounts and reports without refusing. An operator turns enforcement on with the `spend` section in config.json, which `src/lib/spend-reservation-ledger.ts` resolves through diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index b08962894fa..6d7377fa4f3 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -1564,6 +1564,7 @@ "responses-snapshot-repair.test.ts": "responses", "responses-sparse-terminal-tool-scope.test.ts": "responses", "responses-spend-ledger-wiring.test.ts": "responses", + "responses-spend-capacity-guard.test.ts": "responses", "responses-spill-acl-recovery.test.ts": "responses", "responses-spill-inspection.test.ts": "responses", "responses-spill-orphan-sweep.test.ts": "responses", diff --git a/tests/responses/responses-spend-capacity-guard.test.ts b/tests/responses/responses-spend-capacity-guard.test.ts new file mode 100644 index 00000000000..1cafcfdfa30 --- /dev/null +++ b/tests/responses/responses-spend-capacity-guard.test.ts @@ -0,0 +1,114 @@ +import { describe, expect, test } from "bun:test"; +import { createSpendReservationLedger, DEFAULT_SPEND_RESERVATION_POLICY, type SpendDenial, type SpendReservationPolicy } from "../../src/lib/spend-reservation-ledger"; +import { createRequestExecutionBudget } from "../../src/lib/request-execution-budget"; +import { createRequestSpendTracker } from "../../src/server/responses/request-spend"; +import type { RequestLogContext } from "../../src/server/request-log"; + +function fixture(policy: Partial, rootId: string | undefined = "fixture-root", identityId: string | undefined = "fixture-account") { + const ledger = createSpendReservationLedger({ policy: { ...DEFAULT_SPEND_RESERVATION_POLICY, ...policy }, salt: "fixture-only" }); + const context: RequestLogContext = { model: "fixture-model", provider: "fixture-pool", accountLogLabel: identityId, + usageLogInputTokens: 10, spendOutputCeilingTokens: 20 }; + const tracker = createRequestSpendTracker(context, rootId, ledger); + const budget = createRequestExecutionBudget(undefined, "fixture-request", tracker); + let dispatched = 0; + const attempt = () => { + const decision = budget.reserveDispatch({ sendClass: "initial", targetKey: "fixture-pool/model" }); + if (decision.allowed && decision.permit.use()) dispatched++; + return decision; + }; + return { ledger, context, tracker, budget, attempt, dispatched: () => dispatched }; +} + +describe("configured spend cannot dispatch without a capacity booking", () => { + for (const limit of ["root", "identity", "pool"] as const) { + for (const capacity of ["scopes", "sends"] as const) { + test(`${limit} ceiling refuses exhausted ${capacity} tracking`, () => { + const f = fixture({ [limit]: { maxTokens: 1000 }, ...(capacity === "scopes" ? { maxTrackedScopes: 1 } : { maxTrackedSends: 1 }) }); + if (capacity === "sends") expect(f.ledger.reserve({ sendId: "busy", scopes: { poolId: "fixture-pool" }, inputTokens: 1, outputCeilingTokens: 0 }).reserved).toBe(true); + expect(f.attempt().allowed).toBe(false); + expect(f.dispatched()).toBe(0); + expect(f.budget.used).toBe(0); + expect(f.tracker.refusals).toBe(1); + expect(f.context.errorCode).toBe("workflow_tracking_exhausted"); + expect(f.context.terminalSource).toBe("synthetic"); + expect(f.ledger.snapshot("pool", "fixture-pool")?.reserved).toBe(capacity === "sends" ? 1 : undefined); + }); + } + } + + for (const mode of ["unconfigured", "rootless", "identityless"] as const) { + test(`${mode} capacity failure stays permissive with no local refusal`, () => { + // Passing explicit undefined to the factory would use its default, so remove the + // irrelevant scope in the tracker itself while retaining real ledger capacity pressure. + const ledger = createSpendReservationLedger({ policy: { ...DEFAULT_SPEND_RESERVATION_POLICY, maxTrackedSends: 1, + ...(mode === "rootless" ? { root: { maxTokens: 1000 } } : mode === "identityless" ? { identity: { maxTokens: 1000 } } : {}), + } }); + ledger.reserve({ sendId: "busy", scopes: { poolId: "pool" }, inputTokens: 1, outputCeilingTokens: 0 }); + const context: RequestLogContext = { model: "fixture", provider: "pool", spendOutputCeilingTokens: 30 }; + const tracker = createRequestSpendTracker(context, undefined, ledger); + const budget = createRequestExecutionBudget(undefined, undefined, tracker); + const decision = budget.reserveDispatch({ sendClass: "initial", targetKey: "pool/model" }); + expect(decision.allowed).toBe(true); + if (!decision.allowed) throw new Error("expected observe-only dispatch"); + expect(decision.permit.use()).toBe(true); + expect(budget.used).toBe(1); + expect(tracker.refusals).toBe(0); + expect(context.localTerminalReason).toBeUndefined(); + expect(context.errorCode).toBeUndefined(); + expect(ledger.snapshot("pool", "pool")?.reserved).toBe(1); + }); + } + + test("an already-sent capacity failure preserves physical count without a false refusal", () => { + const f = fixture({ pool: { maxTokens: 1000 }, maxTrackedSends: 1 }); + f.ledger.reserve({ sendId: "busy", scopes: { poolId: "fixture-pool" }, inputTokens: 1, outputCeilingTokens: 0 }); + f.budget.used += 1; + expect(f.budget.used).toBe(1); + expect(f.tracker.refusals).toBe(0); + expect(f.context.errorCode).toBeUndefined(); + expect(f.context.localTerminalReason).toBeUndefined(); + expect(f.ledger.snapshot("pool", "fixture-pool")?.reserved).toBe(1); + }); + + test("enabling a ceiling affects the existing tracker on its next charge", () => { + const f = fixture({ maxTrackedScopes: 1 }); + expect(f.attempt().allowed).toBe(true); + f.ledger.reconfigure({ ...f.ledger.policy, pool: { maxTokens: 1000 } }); + expect(f.attempt().allowed).toBe(false); + expect(f.dispatched()).toBe(1); + expect(f.budget.used).toBe(1); + expect(f.tracker.refusals).toBe(1); + }); + + test("newly resolved identity applies on the next charge without reconstructing the tracker", () => { + const f = fixture({ identity: { maxTokens: 1000 }, maxTrackedScopes: 1 }); + delete f.context.accountLogLabel; + expect(f.attempt().allowed).toBe(true); + f.context.accountLogLabel = "newly-resolved-account"; + expect(f.attempt().allowed).toBe(false); + expect(f.dispatched()).toBe(1); + expect(f.budget.used).toBe(1); + }); +}); + +for (const [denial, code] of [ + [{ reason: "duplicate-send-id", sendId: "duplicate" }, "workflow_send_replayed"], + [{ reason: "reserve-not-durable", sendId: "failed-write" }, "workflow_spend_undurable"], + [{ reason: "journal-corrupt", corruptRecords: 1 }, "workflow_spend_undurable"], +] as const satisfies ReadonlyArray) { + test(`${denial.reason} refuses only an applicable new dispatch`, () => { + for (const enforced of [false, true]) { + for (const alreadySent of [false, true]) { + const f = fixture(enforced ? { pool: { maxTokens: 1000 } } : {}); + // Force the exact refusal to test the consumer; real capacity failures are above, + // and actual journal failures/corruption are covered by the ledger regression suite. + const ledger = Object.create(f.ledger) as typeof f.ledger; + ledger.reserve = () => ({ reserved: false, denial }); + const tracker = createRequestSpendTracker(f.context, undefined, ledger); + expect(tracker.charge({ alreadySent })).toBe(alreadySent || !enforced); + expect(tracker.refusals).toBe(enforced && !alreadySent ? 1 : 0); + expect(f.context.errorCode).toBe(enforced && !alreadySent ? code : undefined); + } + } + }); +}