From 97ad520bf4a5f1804c21f0df56bdcc028f125dd7 Mon Sep 17 00:00:00 2001 From: shawnkim Date: Sun, 27 Sep 2026 02:01:28 +0900 Subject: [PATCH 01/11] fix(responses): preserve Meta Muse tool-choice semantics (#5969) Carried from #5969 as one squashed commit. Co-authored-by: shawnkim --- design-debt.md | 30 +++ .../docs/ko/reference/platform-support.md | 2 + .../docs/reference/platform-support.md | 6 + scripts/test-layout/layout.json | 1 + .../openai-responses/muse-tool-choice.ts | 31 +++ src/adapters/openai-responses/passthrough.ts | 7 +- src/server/responses/passthrough-dispatch.ts | 2 + structure/providers-and-adapters.md | 12 ++ structure/transports/responses.md | 2 +- tests/fixtures/test-layout-expected.json | 1 + tests/providers/muse-tool-name-alias.test.ts | 6 +- .../responses-muse-tool-choice.test.ts | 195 ++++++++++++++++++ .../responses-muse-tool-name-alias.test.ts | 144 ++++++------- 13 files changed, 362 insertions(+), 77 deletions(-) create mode 100644 design-debt.md create mode 100644 src/adapters/openai-responses/muse-tool-choice.ts create mode 100644 tests/responses/responses-muse-tool-choice.test.ts diff --git a/design-debt.md b/design-debt.md new file mode 100644 index 00000000000..b6caadf9dd6 --- /dev/null +++ b/design-debt.md @@ -0,0 +1,30 @@ +# Scoped design debt + +This audit covers the Meta Muse Responses tool-choice change only. All three modules below were swept against the design-review flags. It records confirmed findings only. + +## Inventory + +| Module | Role | Review | +| --- | --- | --- | +| `src/adapters/openai-responses/muse-tool-choice.ts` | Validates the caller's original and effective selector, then normalizes Muse's final request body. | Swept against the design-review flags. No confirmed findings. | +| `src/adapters/openai-responses/passthrough.ts` | Applies the normalizer at the final Meta Responses body boundary. | Full module swept. No confirmed findings. | +| `src/server/responses/passthrough-dispatch.ts` | Maps the compatibility error to an initial HTTP 400 before dispatch. | Full module swept. No confirmed findings. | + +Provider registry and configuration modules are excluded. The selected design adds no provider +flag or persisted setting. Other adapters, transports, and request-selection paths are outside this +behavior's scope. + +## Confirmed findings + +| ID | Severity | Flag | Evidence | Smallest redesign | Status | +| --- | --- | --- | --- | --- | --- | +| - | - | - | No confirmed findings in the three-module sweep. | - | - | + +Severity counts are S1: 0, S2: 0, S3: 0. No findings were refuted because none were raised. No +security findings are recorded here. + +Audit date: 2026-09-27. Modules swept: 3. Modules inconclusive: 0. + +The nose comparison covered 60 files in Responses normalization and provider registry code. +It retained 109 duplication families, with no new family. One recheck contained two unchanged +streaming accumulator regions whose line locations moved. Both regions were compared with `origin/dev`. diff --git a/docs-site/src/content/docs/ko/reference/platform-support.md b/docs-site/src/content/docs/ko/reference/platform-support.md index 5c74818a3e7..375f47368e0 100644 --- a/docs-site/src/content/docs/ko/reference/platform-support.md +++ b/docs-site/src/content/docs/ko/reference/platform-support.md @@ -38,6 +38,8 @@ macOS에서는 `muse login` 뒤 Muse Code CLI가 이미 저장한 API 키를 ope 다른 플랫폼에서는 키를 붙여넣도록 요청합니다. Meta는 네이티브 Windows CLI를 제공하지 않습니다. Linux에는 CLI가 있지만 자격 증명을 저장하는 위치가 검증되지 않아 opencodex가 저장소를 추측하지 않습니다. 같은 키는 [Meta 개발자 콘솔](https://dev.meta.ai)에서도 볼 수 있습니다. 붙여넣은 키도 가져온 키와 똑같은 형식 검사와 Model API에 대한 실시간 검증을 거칩니다. +Meta로 보내는 Responses 요청은 tool 선택을 생략하거나 `auto`로 지정할 수 있습니다. 명시적인 `none`은 `tools: []`로 보내고 `input` 안의 `additional_tools` 항목을 제거합니다. 강제 선택, 함수 이름 지정, `allowed_tools` 선택은 Muse가 `auto`만 지원하므로 Meta로 보내기 전에 HTTP 400으로 거부합니다. 다른 Responses 대상의 tool 선택 동작은 바뀌지 않습니다. + ## Windows 참고 사항 Windows 서비스는 Task Scheduler 또는 네이티브 WinSW 서비스로 실행할 수 있으며 둘을 동시에 사용할 수는 없습니다. `ocx service repair`가 두 방식의 상태를 모두 발견하면 진행을 거부합니다. 어느 쪽을 원하는지 추측하면 한 컴퓨터에서 두 프록시가 같은 포트를 놓고 충돌할 수 있기 때문입니다. diff --git a/docs-site/src/content/docs/reference/platform-support.md b/docs-site/src/content/docs/reference/platform-support.md index 29beee473a9..5309bf7badc 100644 --- a/docs-site/src/content/docs/reference/platform-support.md +++ b/docs-site/src/content/docs/reference/platform-support.md @@ -59,6 +59,12 @@ offers an input surface. Pasted keys use the same format and Model API validatio as imported ones. Management login requires a dashboard session before either credential-acquisition path; see the [provider guide](/guides/providers/). +For Responses requests to Meta, omitted or `auto` tool selection is supported. +Explicit `none` sends `tools: []` and removes `additional_tools` items from the +input. Forced, named, and `allowed_tools` selections return HTTP 400 before the +request reaches Meta because Muse accepts only `auto`. Other Responses +destinations keep their existing tool-selection behavior. + ## Windows notes The Windows service can run under Task Scheduler or as a native WinSW service, diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index cc5dd7842a5..86759e4820f 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -1562,6 +1562,7 @@ "responses-json-events.test.ts": "responses", "responses-legacy-dotted-tool-name-repair.test.ts": "responses", "responses-muse-tool-name-alias.test.ts": "responses", + "responses-muse-tool-choice.test.ts": "responses", "responses-native-main-refresh.test.ts": "responses", "responses-opaque-blob-recovery.test.ts": "responses", "responses-parser-agent-message.test.ts": "responses", diff --git a/src/adapters/openai-responses/muse-tool-choice.ts b/src/adapters/openai-responses/muse-tool-choice.ts new file mode 100644 index 00000000000..7cdd8230210 --- /dev/null +++ b/src/adapters/openai-responses/muse-tool-choice.ts @@ -0,0 +1,31 @@ +import { isPlainObject } from "./internal"; + +export class MuseToolChoiceCompatibilityError extends Error { + constructor() { + super("This tool_choice cannot be preserved for Meta Responses; use auto or none."); + this.name = "MuseToolChoiceCompatibilityError"; + } +} + +function selectorKind(choice: unknown): "auto" | "none" { + if (choice === undefined || choice === "auto") return "auto"; + if (choice === "none") return "none"; + throw new MuseToolChoiceCompatibilityError(); +} + +export function normalizeMuseToolChoice(body: unknown, originalChoice: unknown): unknown { + const originalKind = selectorKind(originalChoice); + const effectiveKind = selectorKind(isPlainObject(body) ? body.tool_choice : undefined); + if (originalKind !== "none" && effectiveKind !== "none") return body; + if (!isPlainObject(body)) return body; + + const input = Array.isArray(body.input) + ? body.input.filter(item => !isPlainObject(item) || item.type !== "additional_tools") + : body.input; + const { + tool_choice: _toolChoice, + parallel_tool_calls: _parallelToolCalls, + ...rest + } = body; + return { ...rest, tools: [], ...(input !== body.input ? { input } : {}) }; +} diff --git a/src/adapters/openai-responses/passthrough.ts b/src/adapters/openai-responses/passthrough.ts index 147a96419b1..7540c3ac876 100644 --- a/src/adapters/openai-responses/passthrough.ts +++ b/src/adapters/openai-responses/passthrough.ts @@ -44,6 +44,7 @@ import { applyTierDecisionToResponsesBody, normalizeCanonicalForwardContinuation import { normalizeImageGenClientTools, preferConfiguredHostedTools } from "./image-gen"; import { stripMuseSparkUnsupportedWebSearchFields, stripOpenAiOnlyWebSearchFields } from "./web-search"; import { observeOutbound } from "../../usage/cache-diagnostic"; +import { normalizeMuseToolChoice } from "./muse-tool-choice"; /** * Identifies DeepSeek's strict Responses replay contract: tool-bearing continuations need @@ -495,7 +496,7 @@ export function createResponsesPassthroughAdapter(provider: OcxProviderConfig): parsed.modelId, ); // Normalize the wire model before deriving model-dependent transport metadata. - const finalBody = + let finalBody = provider.modelSuffixBracketStrip && unnormalizedBody !== null && typeof unnormalizedBody === "object" @@ -503,6 +504,10 @@ export function createResponsesPassthroughAdapter(provider: OcxProviderConfig): && typeof (unnormalizedBody as { model?: unknown }).model === "string" ? { ...(unnormalizedBody as Record), model: stripBracketedModelSuffix((unnormalizedBody as { model: string }).model) } : unnormalizedBody; + if (isMetaAiResponsesDestination(url)) { + const originalChoice = isPlainObject(parsed._rawBody) ? parsed._rawBody.tool_choice : undefined; + finalBody = normalizeMuseToolChoice(finalBody, originalChoice); + } if (isCanonicalOpenAiForwardProvider(provider)) { const routingHeaders = new Headers(headers); applyCodexRoutingHint(routingHeaders, finalBody); diff --git a/src/server/responses/passthrough-dispatch.ts b/src/server/responses/passthrough-dispatch.ts index 41bdbfe931d..1b953bc0f8f 100644 --- a/src/server/responses/passthrough-dispatch.ts +++ b/src/server/responses/passthrough-dispatch.ts @@ -44,6 +44,7 @@ import { } from "../../responses/namespace-tool-compat"; import { restoreRoutedCustomCalls, RoutedCustomToolCompatError } from "../../responses/custom-tool-compat"; import { XaiToolSchemaCompatibilityError } from "../../adapters/xai-tool-schema"; +import { MuseToolChoiceCompatibilityError } from "../../adapters/openai-responses/muse-tool-choice"; import { formatErrorResponse } from "../../bridge"; import { redactSecretString } from "../../lib/redact"; import { @@ -337,6 +338,7 @@ export async function preparePassthroughExchange( error instanceof NamespaceToolCollisionError || error instanceof XaiToolSchemaCompatibilityError || error instanceof RoutedCustomToolCompatError + || error instanceof MuseToolChoiceCompatibilityError ) { return formatErrorResponse(400, "invalid_request_error", redactSecretString(error.message)); } diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index 94dc5987050..923aedc3bdd 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -395,6 +395,18 @@ this field in the tree, so a value that survives file load still cannot be spent silently by design; config-time is where the operator is told why. The planner requires the provider name for that assessment, so `planPassthroughWebSearchBridge` takes it explicitly. +## Meta Responses tool selection + +At the final request boundary for `api.meta.ai`, `src/adapters/openai-responses/passthrough.ts` +uses `src/adapters/openai-responses/muse-tool-choice.ts` to normalize `tool_choice`. Omitted or +`auto` selection keeps its meaning. Explicit `none` sets `tools` to an empty list, removes +`additional_tools` items from `input`, and omits `tool_choice` and `parallel_tool_calls` from the +outgoing body. Forced, named, and `allowed_tools` selections fail with HTTP 400 before the +upstream send because Muse supports only `auto`. Filtering a required tool never changes the +caller's obligation into `none` or `auto`. This rule applies only to the Meta Responses +destination. The input body, historical tool calls and results, and non-Meta requests keep their +existing meaning. + ## Shared type declarations `src/types/` holds the declarations every layer imports: config types (`src/types/config.ts`), diff --git a/structure/transports/responses.md b/structure/transports/responses.md index f47a747511c..db9c51b2914 100644 --- a/structure/transports/responses.md +++ b/structure/transports/responses.md @@ -118,7 +118,7 @@ echoed bare name to its namespaced identity before authorizing anything echo is a guess rather than a nomination. The bridges check the declared set before consulting `toolNsMap`, so there a bare helper echo is refused either way. A genuine namespace-free declaration is untouched throughout: that is the caller declaring the tool, not a namespace being -discarded to manufacture a bare name. +discarded to manufacture a bare name. Meta Responses also applies [tool-selection compatibility](../providers-and-adapters.md#meta-responses-tool-selection). Function-call wrappers around freeform bodies are restored by `src/responses/apply-patch-envelope.ts`. The declared `input` field is authoritative. For bare diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index a419094b1b4..3a649b4c214 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -1388,6 +1388,7 @@ "responses-json-events.test.ts": "responses", "responses-legacy-dotted-tool-name-repair.test.ts": "responses", "responses-muse-tool-name-alias.test.ts": "responses", + "responses-muse-tool-choice.test.ts": "responses", "responses-native-main-refresh.test.ts": "responses", "responses-opaque-blob-recovery.test.ts": "responses", "responses-parser-agent-message.test.ts": "responses", diff --git a/tests/providers/muse-tool-name-alias.test.ts b/tests/providers/muse-tool-name-alias.test.ts index 52228c20bde..a59da21a2aa 100644 --- a/tests/providers/muse-tool-name-alias.test.ts +++ b/tests/providers/muse-tool-name-alias.test.ts @@ -124,7 +124,7 @@ describe("#4410 Meta Muse 64-char tool-name aliasing", () => { expect(aliases?.size).toBe(20); }); - test("history function_call and tool_choice are aliased on api.meta.ai", () => { + test("history function_call is aliased while auto tool_choice remains unchanged", () => { const longName = LONG_ISSUE_NAMES[0]!; const wire = hashedName(longName); const { body, aliases } = buildForProvider(META_PROVIDER, "muse-spark-1.3-contributor", { @@ -133,7 +133,7 @@ describe("#4410 Meta Muse 64-char tool-name aliasing", () => { { type: "function_call", name: longName, call_id: "c1", arguments: "{\"q\":\"hub\"}" }, { type: "function_call_output", call_id: "c1", output: "ok" }, ], - tool_choice: { type: "function", name: longName }, + tool_choice: "auto", }); expect((body.tools as Array<{ name: string }>)[0]!.name).toBe(wire); expect((body.input as Array>)[0]).toMatchObject({ @@ -141,7 +141,7 @@ describe("#4410 Meta Muse 64-char tool-name aliasing", () => { name: wire, arguments: "{\"q\":\"hub\"}", }); - expect((body.tool_choice as { name: string }).name).toBe(wire); + expect(body.tool_choice).toBe("auto"); expect(aliases?.get(wire)).toBe(longName); }); diff --git a/tests/responses/responses-muse-tool-choice.test.ts b/tests/responses/responses-muse-tool-choice.test.ts new file mode 100644 index 00000000000..d5f7e95d6c1 --- /dev/null +++ b/tests/responses/responses-muse-tool-choice.test.ts @@ -0,0 +1,195 @@ +import { afterEach, describe, expect, test } from "bun:test"; +import { createResponsesPassthroughAdapter as createResponsesPassthroughAdapterProduction } from "../../src/adapters/openai-responses"; +import { handleResponses, handleResponsesCompact } from "../../src/server/responses"; +import type { OcxConfig, OcxProviderConfig } from "../../src/types"; +import { acquireOwnedSpendHome } from "../helpers/owned-spend-home"; +import { withTestTranslatorBudget } from "../helpers/translator-budget"; + +const createAdapter = (...args: Parameters) => + withTestTranslatorBudget(createResponsesPassthroughAdapterProduction(...args)); + +const meta = { adapter: "openai-responses", baseUrl: "https://api.meta.ai/v1", apiKey: "test-key" } as OcxProviderConfig; +const xai = { ...meta, baseUrl: "https://api.x.ai/v1" } as OcxProviderConfig; +const longName = "mcp__plugin_huggingface-skills_huggingface-skills__hub_repo_search"; +const longWireName = "mcp__plugin_huggingface-skills_huggingface-skills__hub__dec57ce4"; +const tool = { type: "function", name: longName, parameters: { type: "object", properties: {} } }; +let releaseSpendHome: (() => void) | undefined; + +afterEach(() => { + releaseSpendHome?.(); + releaseSpendHome = undefined; +}); + +function build(provider: OcxProviderConfig, rawBody: Record) { + return createAdapter(provider).buildRequest({ + modelId: "muse-spark-1.3", + context: { messages: [] }, + stream: false, + options: {}, + _rawBody: { model: "muse-spark-1.3", input: "continue", ...rawBody }, + }, { headers: new Headers() }); +} + +describe("Meta Muse tool choice", () => { + test("none removes declaration carriers while preserving history and caller input", () => { + const raw = { + input: [ + { type: "function_call", name: longName, call_id: "c1", arguments: "{}" }, + { type: "function_call_output", call_id: "c1", output: "done" }, + { type: "additional_tools", tools: [tool] }, + ], + tools: [tool], + tool_choice: "none", + parallel_tool_calls: true, + }; + const original = structuredClone(raw); + const body = JSON.parse(build(meta, raw).body) as Record; + + expect(body.tools).toEqual([]); + expect(body.input).toEqual([ + { ...original.input[0], name: longWireName }, + original.input[1], + ]); + expect(body).not.toHaveProperty("tool_choice"); + expect(body).not.toHaveProperty("parallel_tool_calls"); + expect(raw).toEqual(original); + }); + + test("auto and omitted choice remain unchanged", () => { + for (const choice of [undefined, "auto"] as const) { + const raw = { tools: [tool], ...(choice ? { tool_choice: choice } : {}) }; + const original = structuredClone(raw); + const body = JSON.parse(build(meta, raw).body) as Record; + expect(body.tools).toHaveLength(1); + expect((body.tools as Array<{ name: string }>)[0]!.name).toHaveLength(64); + expect(body.tool_choice).toBe(choice); + expect(raw).toEqual(original); + } + }); + + test("raw forced hosted choice fails if provider filtering removes its only tool", () => { + const provider = { ...meta, unsupportedHostedTools: ["web_search"] }; + for (const choice of ["required", { type: "web_search" }]) { + expect(() => build(provider, { + tools: [{ type: "web_search" }], + tool_choice: choice, + })).toThrow("This tool_choice cannot be preserved for Meta Responses; use auto or none."); + } + }); + + test("allowed_tools, null and unknown selectors fail with the compatibility error", () => { + for (const choice of [ + { type: "allowed_tools", mode: "auto", tools: [tool] }, + null, + { type: "unknown" }, + ]) { + expect(() => build(meta, { tools: [tool], tool_choice: choice })) + .toThrow("This tool_choice cannot be preserved for Meta Responses; use auto or none."); + } + }); + + test("non-Meta Responses destinations retain tool choices", () => { + const raw = { tools: [tool], tool_choice: { type: "function", name: longName } }; + const body = JSON.parse(build(xai, raw).body) as Record; + expect(body.tool_choice).toEqual(raw.tool_choice); + expect(body.tools).toEqual(raw.tools); + }); + + test("compaction does not bypass forced-choice rejection for a namespaced tool", async () => { + const config = { + port: 0, + defaultProvider: "fixture", + providers: { fixture: { ...meta, authMode: "key" } }, + } as OcxConfig; + const savedFetch = globalThis.fetch; + let sends = 0; + globalThis.fetch = (async () => { + sends += 1; + return Response.json({ id: "unexpected", status: "completed", output: [] }); + }) as typeof fetch; + try { + releaseSpendHome ??= acquireOwnedSpendHome(); + const response = await handleResponsesCompact(new Request("http://localhost/v1/responses/compact", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + model: "fixture/muse-spark-1.3", + input: [{ type: "function_call", name: longName, call_id: "c1", arguments: "{}" }], + tools: [tool], + tool_choice: { type: "function", name: longName }, + }), + }), config, { model: "", provider: "" }); + expect(response.status).toBe(400); + const error = await response.json() as { error: { message: string } }; + expect(error.error.message).toBe("This tool_choice cannot be preserved for Meta Responses; use auto or none."); + expect(sends).toBe(0); + } finally { + globalThis.fetch = savedFetch; + } + }); + + test("handleResponses returns a client 400 before any upstream send", async () => { + const config = { + port: 0, + defaultProvider: "fixture", + providers: { fixture: { ...meta, authMode: "key" } }, + } as OcxConfig; + const savedFetch = globalThis.fetch; + let sends = 0; + globalThis.fetch = (async () => { + sends += 1; + return new Response("unexpected upstream send", { status: 200 }); + }) as typeof fetch; + try { + releaseSpendHome ??= acquireOwnedSpendHome(); + for (const toolChoice of ["required", { type: "function", name: longName }]) { + const response = await handleResponses(new Request("http://localhost/v1/responses", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + model: "fixture/muse-spark-1.3", + input: "use the selected tool", + tools: [tool], + tool_choice: toolChoice, + }), + }), config, { model: "", provider: "" }); + expect(response.status).toBe(400); + const error = await response.json() as { error: { type: string; code: string; message: string } }; + expect(error).toMatchObject({ error: { type: "invalid_request_error", code: "invalid_request_error" } }); + expect(error.error.message).toBe("This tool_choice cannot be preserved for Meta Responses; use auto or none."); + } + expect(sends).toBe(0); + } finally { + globalThis.fetch = savedFetch; + } + }); + + test("none succeeds through handleResponses with no declared tools", async () => { + const config = { + port: 0, + defaultProvider: "fixture", + providers: { fixture: { ...meta, authMode: "key" } }, + } as OcxConfig; + const savedFetch = globalThis.fetch; + let outbound: Record | undefined; + globalThis.fetch = (async (_input, init) => { + outbound = JSON.parse(String(init?.body)) as Record; + return new Response(JSON.stringify({ id: "resp_none", status: "completed", output: [] }), { + headers: { "content-type": "application/json" }, + }); + }) as typeof fetch; + try { + releaseSpendHome ??= acquireOwnedSpendHome(); + const response = await handleResponses(new Request("http://localhost/v1/responses", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ model: "fixture/muse-spark-1.3", input: "no tools", tools: [tool], tool_choice: "none" }), + }), config, { model: "", provider: "" }); + expect(response.status).toBe(200); + expect(outbound?.tools).toEqual([]); + expect(outbound).not.toHaveProperty("tool_choice"); + } finally { + globalThis.fetch = savedFetch; + } + }); +}); diff --git a/tests/responses/responses-muse-tool-name-alias.test.ts b/tests/responses/responses-muse-tool-name-alias.test.ts index 30fc04cfee3..a378417d152 100644 --- a/tests/responses/responses-muse-tool-name-alias.test.ts +++ b/tests/responses/responses-muse-tool-name-alias.test.ts @@ -281,7 +281,7 @@ describe("muse tool-name inbound restore through handleResponses", () => { return [JSON.parse(data) as Record]; }); - test("non-stream function_call and tool_choice restore the original MCP name", async () => { + test("non-stream function_call restores the original MCP name", async () => { const savedFetch = globalThis.fetch; let outbound: Record | undefined; globalThis.fetch = (async (_input, init) => { @@ -289,7 +289,7 @@ describe("muse tool-name inbound restore through handleResponses", () => { return new Response(JSON.stringify({ id: "resp_json", status: "completed", - tool_choice: { type: "function", name: wire }, + tool_choice: "auto", output: [{ type: "function_call", name: wire, call_id: "c1", arguments: "{\"q\":\"x\"}" }], }), { headers: { "content-type": "application/json" } }); }) as typeof fetch; @@ -303,14 +303,14 @@ describe("muse tool-name inbound restore through handleResponses", () => { stream: false, input: "search", tools: [{ type: "function", name: original, parameters: { type: "object" } }], - tool_choice: { type: "function", name: original }, + tool_choice: "auto", }), }), config, { model: "", provider: "" }); - const json = await response.json() as { output: Array>; tool_choice?: { name: string } }; + const json = await response.json() as { output: Array>; tool_choice?: string }; expect((outbound?.tools as Array<{ name: string }>)[0]!.name).toBe(wire); - expect((outbound?.tool_choice as { name: string }).name).toBe(wire); + expect(outbound?.tool_choice).toBe("auto"); expect(json.output[0]).toMatchObject({ type: "function_call", name: original }); - expect(json.tool_choice?.name).toBe(original); + expect(json.tool_choice).toBe("auto"); } finally { globalThis.fetch = savedFetch; } @@ -348,75 +348,75 @@ describe("muse tool-name inbound restore through handleResponses", () => { } }); - for (const selectorKind of ["named", "allowed_tools"] as const) { - test(`sparse terminal reconstruction keeps the restored identity for ${selectorKind}`, async () => { - const savedFetch = globalThis.fetch; - let outbound: Record | undefined; - const itemId = "fc_sparse"; - const callId = "call_sparse"; - globalThis.fetch = (async (_input, init) => { - outbound = JSON.parse(String(init?.body)) as Record; - const choice = outbound.tool_choice as Record; - const wireName = selectorKind === "named" - ? choice.name as string - : ((choice.tools as Array<{ name: string }>)[0]!.name); - const item = { - type: "function_call", - id: itemId, - call_id: callId, - name: wireName, - arguments: "{}", - status: "completed", - }; - const upstream = [ - frame("response.output_item.done", { output_index: 0, item }), - frame("response.completed", { response: { id: "resp_sparse", status: "completed", output: [] } }), - "data: [DONE]", - ].join("\n\n") + "\n\n"; - return new Response(upstream, { headers: { "content-type": "text/event-stream" } }); - }) as typeof fetch; - try { - takeSpendHome(); - const toolChoice = selectorKind === "named" - ? { type: "function", name: original } - : { type: "allowed_tools", mode: "required", tools: [{ type: "function", name: original }] }; - const response = await handleResponses(new Request("http://localhost/v1/responses", { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "fixture/muse-spark-1.3", - stream: true, - input: "search", - tools: [{ type: "function", name: original, parameters: { type: "object" } }], - tool_choice: toolChoice, - }), - }), config, { model: "", provider: "", surface: "grok" }); - const outboundChoice = outbound?.tool_choice as Record; - const wireName = selectorKind === "named" - ? outboundChoice.name as string - : ((outboundChoice.tools as Array<{ name: string }>)[0]!.name); - expect(wireName).not.toBe(original); - expect((outbound?.tools as Array<{ name: string }>)[0]!.name).toBe(wireName); - - const terminal = ssePayloads(await response.text()) - .find(payload => payload.type === "response.completed"); - expect(terminal).toBeDefined(); - expect((terminal!.response as { output: unknown[] }).output).toEqual([expect.objectContaining({ - type: "function_call", - id: itemId, - call_id: callId, - name: original, - })]); - } finally { - globalThis.fetch = savedFetch; - } - }); - } + test("sparse terminal reconstruction keeps the restored identity with auto selection", async () => { + const savedFetch = globalThis.fetch; + let outbound: Record | undefined; + const itemId = "fc_sparse"; + const callId = "call_sparse"; + globalThis.fetch = (async (_input, init) => { + outbound = JSON.parse(String(init?.body)) as Record; + const wireName = (outbound.tools as Array<{ name: string }>)[0]!.name; + const item = { + type: "function_call", + id: itemId, + call_id: callId, + name: wireName, + arguments: "{}", + status: "completed", + }; + const upstream = [ + frame("response.output_item.done", { output_index: 0, item }), + frame("response.completed", { response: { id: "resp_sparse", status: "completed", output: [] } }), + "data: [DONE]", + ].join("\n\n") + "\n\n"; + return new Response(upstream, { headers: { "content-type": "text/event-stream" } }); + }) as typeof fetch; + try { + takeSpendHome(); + const response = await handleResponses(new Request("http://localhost/v1/responses", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + model: "fixture/muse-spark-1.3", + stream: true, + input: "search", + tools: [{ type: "function", name: original, parameters: { type: "object" } }], + tool_choice: "auto", + }), + }), config, { model: "", provider: "", surface: "grok" }); + const wireName = (outbound?.tools as Array<{ name: string }>)[0]!.name; + expect(wireName).not.toBe(original); + expect((outbound?.tools as Array<{ name: string }>)[0]!.name).toBe(wireName); - test("sparse terminal reconstruction still refuses an unselected aliased tool", async () => { + const terminal = ssePayloads(await response.text()) + .find(payload => payload.type === "response.completed"); + expect(terminal).toBeDefined(); + expect((terminal!.response as { output: unknown[] }).output).toEqual([expect.objectContaining({ + type: "function_call", + id: itemId, + call_id: callId, + name: original, + })]); + } finally { + globalThis.fetch = savedFetch; + } + }); + + test("non-Meta named selection still refuses an unselected tool", async () => { const savedFetch = globalThis.fetch; const selected = original; const unselected = "mcp__plugin_android-emulator_android-emulator__android_install_app"; + const nonMetaConfig = { + ...config, + providers: { + fixture: { + adapter: "openai-responses", + baseUrl: "https://api.x.ai/v1", + authMode: "key", + apiKey: "test-key", + }, + }, + } as OcxConfig; globalThis.fetch = (async (_input, init) => { const outbound = JSON.parse(String(init?.body)) as Record; const selectedWire = (outbound.tool_choice as { name: string }).name; @@ -452,7 +452,7 @@ describe("muse tool-name inbound restore through handleResponses", () => { ], tool_choice: { type: "function", name: selected }, }), - }), config, { model: "", provider: "", surface: "grok" }); + }), nonMetaConfig, { model: "", provider: "", surface: "grok" }); const terminal = ssePayloads(await response.text()) .find(payload => payload.type === GROK_REFUSED_TERMINAL_EVENT_TYPE); expect(terminal).toBeDefined(); From 86bf8c6efedc536e509fd160892c4a4953d04834 Mon Sep 17 00:00:00 2001 From: codingbo Date: Sun, 27 Sep 2026 02:01:56 +0900 Subject: [PATCH 02/11] fix(responses): strip unsupported hosted web_search on Xiaomi MiMo destinations (#5944) Carried from #5944 as one squashed commit. Co-authored-by: codingbo --- .../docs/reference/configuration/providers.md | 2 +- src/responses/hosted-tool-policy.ts | 20 ++++++-- structure/providers/chat-compat.md | 4 +- .../responses-hosted-tool-declaration.test.ts | 50 ++++++++++++++++++- 4 files changed, 67 insertions(+), 9 deletions(-) diff --git a/docs-site/src/content/docs/reference/configuration/providers.md b/docs-site/src/content/docs/reference/configuration/providers.md index e45d68b4bbb..e946dfebad7 100644 --- a/docs-site/src/content/docs/reference/configuration/providers.md +++ b/docs-site/src/content/docs/reference/configuration/providers.md @@ -249,7 +249,7 @@ Providers can expose a built-in shorthand, such as `agy` for `google-antigravity | `xaiResponsesXSearch?` | `boolean` | Disabled by default. On an xAI Responses destination, append the provider-hosted `x_search` declaration only when a live `web_search` tool survives final request normalization. Existing declarations are not duplicated, caller `tool_choice`/`allowed_tools` selectors are never widened, and this is separate from the web-search sidecar's `search.xSearch` options. | | `modelPreferHostedTools?` | `Record` | Exact-model opt-in for non-forward Responses gateways that reserve a hosted-tool namespace. Currently accepts only `["image_generation"]`; a matching model must use the `openai-responses` wire and support that hosted tool. It removes colliding client `image_gen` declarations and rewrites their selectors to preserve caller tool choice. For OpenAI API virtual `-pro` models, the selected public ID is matched first and the resolved base wire-model ID is a fallback. `modelAdapters` resolves the public ID first, then the base ID; the second resolution determines the final wire. Other models retain normal alias behavior. | | `annotateEmptyToolOutputs?` | `boolean` | Replace a present-but-empty tool result with a short marker before it reaches the model, so a blank result is not read as a missing one. Applies to blank strings and text-only part arrays; image, file, and encrypted parts are never touched. Defaults to `true` for DeepSeek from the built-in registry and is otherwise unset. Set `false` to opt a provider out — an explicit `false` is preserved across later edits that omit the field. `PATCH /api/providers?name=` accepts `true`, `false`, or `null` to clear the override and return to registry-default behavior. | -| `unsupportedHostedTools?` | `string[]` | Hosted tool declarations this Responses destination rejects, so they are stripped from `tools`, from client-loaded `additional_tools`, and from `tool_choice` instead of being forwarded and rejected upstream. Use it for an OpenAI-compatible gateway with a narrower capability set — one that accepts plain Responses requests and `function` tools but returns HTTP 400 for hosted `web_search` — so a text-only prompt is not failed by a capability it never needed. Accepts only hosted tool type names (`web_search`, `web_search_preview`, `file_search`, `computer_use_preview`, `computer_use`, `code_interpreter`, `image_generation`, `image_gen`, `mcp`, `tool_search`, `local_shell`, `x_search`); an unrecognized name is rejected rather than silently ignored. Spelling variants of one capability are aliased, so `["web_search"]` also denies `web_search_preview`. A provider cannot both deny a hosted tool here and prefer it in `modelPreferHostedTools`. This is independent of `supportsResponsesCustomTools`; set both for a gateway that also rejects native custom tools. `PATCH /api/providers?name=` accepts an array or `null` to clear it. | +| `unsupportedHostedTools?` | `string[]` | Hosted tool declarations this Responses destination rejects, so they are stripped from `tools`, from client-loaded `additional_tools`, and from `tool_choice` instead of being forwarded and rejected upstream. Use it for an OpenAI-compatible gateway with a narrower capability set — one that accepts plain Responses requests and `function` tools but returns HTTP 400 for hosted `web_search` — so a text-only prompt is not failed by a capability it never needed. Accepts only hosted tool type names (`web_search`, `web_search_preview`, `file_search`, `computer_use_preview`, `computer_use`, `code_interpreter`, `image_generation`, `image_gen`, `mcp`, `tool_search`, `local_shell`, `x_search`); an unrecognized name is rejected rather than silently ignored. Spelling variants of one capability are aliased, so `["web_search"]` also denies `web_search_preview`. A provider cannot both deny a hosted tool here and prefer it in `modelPreferHostedTools`. Xiaomi MiMo Responses destinations (`xiaomimimo.com` and its subdomains, including `api.xiaomimimo.com` and `token-plan-cn.xiaomimimo.com`) automatically strip `web_search` and `web_search_preview` while preserving function tools; no declaration is needed for these hosts. This is independent of `supportsResponsesCustomTools`; set both for a gateway that also rejects native custom tools. `PATCH /api/providers?name=` accepts an array or `null` to clear it. | | `reasoningEffortMap?` | `Record` | Provider-wide wire aliases for reasoning labels. Map a label to `"__omit__"` to drop the reasoning field from the upstream request entirely: `reasoning_effort` on an OpenAI-compatible wire, and Ollama's native `think` field on the Ollama native adapter (#2356). | | `modelReasoningEffortMap?` | `Record>` | Per-model wire aliases for reasoning labels. Map a label to `"__omit__"` to drop the reasoning field from the upstream request entirely. | | `reasoningWireFormat?` | `"gateway-object"` | For OpenAI-compatible gateways that accept `reasoning: { enabled, effort }` instead of `reasoning_effort`. The ClinePass preset sets this automatically. A provider save that keeps the destination keeps it; see [What a provider save keeps](#what-a-provider-save-keeps). `PATCH` accepts `"gateway-object"` or `null` to clear it. | diff --git a/src/responses/hosted-tool-policy.ts b/src/responses/hosted-tool-policy.ts index 9bd60fef720..0182c099c47 100644 --- a/src/responses/hosted-tool-policy.ts +++ b/src/responses/hosted-tool-policy.ts @@ -1,15 +1,25 @@ /** * Hosted tools rejected by specific native model slugs or exact provider destinations. * - * Currently empty. Its only row removed hosted web search for grok-4.6 on OpenCode Go, but that - * destination rejects two OpenAI-private fields rather than the tool: the xAI web-search - * normalizer in `src/adapters/xai-web-search.ts` now removes those fields for every Grok model - * there, and search works. Add a row only for a destination that refuses the tool itself. + * Add a row only for a destination that refuses the tool itself. */ const UNSUPPORTED_HOSTED_TOOLS: ReadonlyArray<{ match: (model: string, baseUrl?: string) => boolean; tools: ReadonlySet; -}> = []; +}> = [ + { + // MiMo rejects hosted search with "tool type 'web_search' is not supported by this gateway phase" (#5501). + match: (_model, baseUrl) => { + if (!baseUrl) return false; + try { + return /(?:^|\.)xiaomimimo\.com$/i.test(new URL(baseUrl).hostname); + } catch { + return false; + } + }, + tools: new Set(["web_search", "web_search_preview"]), + }, +]; /** * Hosted-tool declaration names an operator may list in `unsupportedHostedTools`. diff --git a/structure/providers/chat-compat.md b/structure/providers/chat-compat.md index ac4e96a7007..9aaeceababe 100644 --- a/structure/providers/chat-compat.md +++ b/structure/providers/chat-compat.md @@ -280,8 +280,8 @@ prompt failed before the model answered, because Codex's hosted declaration trav provider nobody has classified can now describe itself in config. The declaration is additive to that table, not a replacement for it. The table is for destinations that reject a tool regardless of configuration, so an operator who never heard of the field stays protected; -a declaration can only deny more, never re-enable a known-broken pairing. The table is currently empty: its only row removed hosted search for grok-4.6 on OpenCode Go, -but that destination refuses two OpenAI-private fields rather than the tool, and `src/adapters/xai-web-search.ts` now normalizes those fields for every Grok model there. +a declaration can only deny more, never re-enable a known-broken pairing. The table denies `web_search` and `web_search_preview` for `xiaomimimo.com` and its subdomains, including the public API and token-plan hosts (#5501), independent of model name. Matching uses the parsed URL hostname, so unrelated hosts with MiMo names in paths or queries are unaffected. +OpenCode Go keeps hosted search: it refuses two OpenAI-private fields rather than the tool, and `src/adapters/xai-web-search.ts` normalizes those fields for every Grok model there. Two properties are deliberate. Spelling variants of one capability are aliased, so declaring `web_search` also denies `web_search_preview` — the rest of the proxy already folds that pair into diff --git a/tests/responses/responses-hosted-tool-declaration.test.ts b/tests/responses/responses-hosted-tool-declaration.test.ts index 91b21e7633b..bc47121c636 100644 --- a/tests/responses/responses-hosted-tool-declaration.test.ts +++ b/tests/responses/responses-hosted-tool-declaration.test.ts @@ -128,7 +128,7 @@ describe("provider-declared unsupported hosted tools", () => { }); test("the declaration denies only what it names", () => { - // The built-in table is empty: OpenCode Go Grok, its former only row, accepts hosted + // OpenCode Go Grok accepts hosted // web_search once the xAI-refused fields are normalized away (xai-web-search.ts), so // neither an unrelated declaration nor the table may remove it there. const declaredImageOnly = declaredUnsupportedHostedTools({ unsupportedHostedTools: ["image_generation"] }); @@ -196,3 +196,51 @@ describe("unsupportedHostedTools configuration", () => { expect(error).toContain("unsupportedHostedTools"); }); }); + +// #5501: MiMo rejects hosted search before processing even a plain text prompt. +describe("MiMo destination hosted-tool policy", () => { + for (const baseUrl of [ + "https://api.xiaomimimo.com/v1", + "https://token-plan-cn.xiaomimimo.com/v1", + "https://xiaomimimo.com/v1", + "https://API.XIAOMIMIMO.COM:8443/v1", + ]) { + for (const type of ["web_search", "web_search_preview"]) { + test(`${baseUrl} strips ${type} on the serialized Responses request`, () => { + const body = build({ tools: [{ type }, functionTool] }, gateway({ baseUrl })); + expect(body.tools).toEqual([functionTool]); + }); + test(`${baseUrl} reconciles ${type} selectors and loaded tools`, () => { + const provider = gateway({ baseUrl }); + expect(stripUnsupportedHostedTools({ + tools: [{ type }], tool_choice: { type }, + }, provider)).toEqual({ tools: [], tool_choice: "none" }); + expect(stripUnsupportedHostedTools({ + input: [{ type: "additional_tools", tools: [{ type }, functionTool] }], + tool_choice: { type: "allowed_tools", mode: "auto", tools: [{ type }, functionTool] }, + }, provider)).toEqual({ + input: [{ type: "additional_tools", tools: [functionTool] }], + tool_choice: { type: "allowed_tools", mode: "auto", tools: [functionTool] }, + }); + }); + } + } + + for (const baseUrl of [ + "https://api.openai.com/v1", GATEWAY_BASE_URL, + "https://xiaomimimo.com.example/v1", "https://notxiaomimimo.com/v1", + "https://gateway.example/api.xiaomimimo.com/v1", + "https://gateway.example/?next=api.xiaomimimo.com/v1", + ]) { + test(`${baseUrl} preserves hosted search even for a MiMo model`, () => { + const tools = [{ type: "web_search" }, { type: "web_search_preview" }, functionTool]; + expect(build({ model: "mimo-v2-pro", tools }, gateway({ baseUrl })).tools).toEqual(tools); + }); + } + + test("missing or invalid URLs do not classify a destination as MiMo", () => { + for (const baseUrl of [undefined, "", "not a URL"]) { + expect(isHostedToolUnsupportedForModel("mimo-v2-pro", "web_search", baseUrl)).toBe(false); + } + }); +}); From fb665d88dce11fb56152ef2e16d27917ec24cbb5 Mon Sep 17 00:00:00 2001 From: codingbo Date: Sun, 27 Sep 2026 02:02:07 +0900 Subject: [PATCH 03/11] fix(service): restart Windows service wrapper on unexpected bun termination (#5938) Carried from #5938 as one squashed commit. Co-authored-by: codingbo --- .../content/docs/reference/cli/lifecycle.md | 6 ++ src/cli/dispatch.ts | 2 +- src/cli/index.ts | 16 ++--- src/service/windows-taskxml.ts | 18 +++--- src/service/windows-wrapper-exit.ts | 8 +++ structure/ops/docs-and-release.md | 8 +++ structure/runtime.md | 4 +- tests/cli/cli-dispatch.test.ts | 4 +- tests/cli/cli-ready.test.ts | 10 ++-- .../windows/windows-service-wrappers.test.ts | 60 +++++++++++++++++++ 10 files changed, 109 insertions(+), 27 deletions(-) create mode 100644 src/service/windows-wrapper-exit.ts diff --git a/docs-site/src/content/docs/reference/cli/lifecycle.md b/docs-site/src/content/docs/reference/cli/lifecycle.md index 24a63d5c7ad..21dbee3153a 100644 --- a/docs-site/src/content/docs/reference/cli/lifecycle.md +++ b/docs-site/src/content/docs/reference/cli/lifecycle.md @@ -400,6 +400,12 @@ previous catalog from memory. ### `ocx service [install|repair|restart|start|stop|status|uninstall|remove]` +On Windows Task Scheduler, the service wrapper restarts the proxy after five seconds +even when an external tool terminates it with exit code 0. If another opencodex proxy +already owns the port, the wrapper exits deliberately. Use `ocx stop` or +`ocx service stop` to stop the service and its restart loop. After upgrading an +existing installation, run `ocx service repair` to refresh the generated wrapper. + Run opencodex as a login-managed background service (macOS **launchd**, Linux **systemd user unit**, Windows **Task Scheduler**) that auto-starts on login and auto-restarts on crash. Service runs set `OCX_SERVICE=1` so a restart does not churn the Codex config. diff --git a/src/cli/dispatch.ts b/src/cli/dispatch.ts index 9df115fa24b..d45030d0be4 100644 --- a/src/cli/dispatch.ts +++ b/src/cli/dispatch.ts @@ -1059,7 +1059,7 @@ export type BusyPreferredPortDecision = * * Service-wrapper context keeps the semantics `decideStartWithLiveOwner` gives it: a * healthy proxy on the port means the port is served, and the wrapper's - * `if %ERRORLEVEL% NEQ 0` loop must see a zero exit rather than respawn every 5 seconds. + * retry loop must receive the intentional stay-out signal rather than respawn every 5 seconds. */ export function decideBusyPreferredPort(input: { preferredPort: number; diff --git a/src/cli/index.ts b/src/cli/index.ts index 827dbd6701b..1f257e8631b 100755 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -1,4 +1,5 @@ #!/usr/bin/env bun +import { serviceStayOutExitCode } from "../service/windows-wrapper-exit"; import { spawn } from "node:child_process"; import { homedir } from "node:os"; import { join } from "node:path"; @@ -330,10 +331,9 @@ async function chooseListenPort( ocxService: process.env.OCX_SERVICE, }); if (decision === "service-stay-out") { - // Same contract as the pre-bind owner check: the wrapper's retry loop terminates - // on a zero exit, and the port it was asked to serve is already served. + // Signal intentional stay-out to wrappers that support the exit protocol. console.log(`Proxy already running (PID ${holder?.pid ?? "unknown"}, port ${preferred}); service wrapper staying out of the way.`); - throw new StartCommandExit(0); + throw new StartCommandExit(serviceStayOutExitCode()); } if (decision === "refuse-live-proxy") { console.error(`⚠️ Proxy already running (PID ${holder?.pid ?? "unknown"}, port ${preferred}). Use 'ocx stop' first.`); @@ -444,13 +444,9 @@ async function handleStart(options: { block?: boolean } = {}) { ocxService: process.env.OCX_SERVICE, }); if (decision === "service-stay-out") { - // Service-wrapper context (opencodex-service.cmd `:loop`): a healthy proxy from - // ANY source means the requested port is already served. Exit 0 so the wrapper's - // `if %ERRORLEVEL% NEQ 0` retry loop terminates instead of respawning every 5s - // against a listener it can never claim (observed as an endless - // "Proxy already running" service.log loop). + // A live owner is an intentional stay-out, not an unexpected child exit. console.log(`Proxy already running (PID ${owner.live.pid ?? owner.pidSnapshot ?? "unknown"}, port ${owner.live.port}); service wrapper staying out of the way.`); - process.exit(0); + process.exit(serviceStayOutExitCode()); } if (decision === "refuse") { console.error(`⚠️ Proxy already running (PID ${owner.live.pid ?? owner.pidSnapshot ?? "unknown"}, port ${owner.live.port}). Use 'ocx stop' first.`); @@ -513,7 +509,7 @@ async function handleStart(options: { block?: boolean } = {}) { }); if (decision === "service-stay-out") { console.log(`Proxy already running (PID ${fencedLive.pid ?? "unknown"}, port ${fencedLive.port}); service wrapper staying out of the way.`); - throw new StartCommandExit(0); + throw new StartCommandExit(serviceStayOutExitCode()); } if (decision === "refuse") { console.error(`⚠️ Proxy appeared before bind (PID ${fencedLive.pid ?? "unknown"}, port ${fencedLive.port}). Use 'ocx stop' first.`); diff --git a/src/service/windows-taskxml.ts b/src/service/windows-taskxml.ts index 38d8286d5b9..b3f1baa1917 100644 --- a/src/service/windows-taskxml.ts +++ b/src/service/windows-taskxml.ts @@ -1,3 +1,4 @@ +import { WINDOWS_WRAPPER_PROTOCOL_ENV, WINDOWS_WRAPPER_STAY_OUT_EXIT_CODE } from "./windows-wrapper-exit"; import { readFileSync } from "node:fs"; import { TASK, windowsServiceScriptPath, windowsLauncherVbsPath, windowsTaskXmlPath } from "./state"; import { windowsWscript } from "./windows-scheduler"; @@ -64,11 +65,13 @@ export function buildWindowsServiceScript( const path = process.env.PATH ?? ""; const lines = [ "@echo off", - "setlocal", + "setlocal EnableExtensions DisableDelayedExpansion", + 'set "ERRORLEVEL="', // The wrapper console is hidden by the wscript launcher (window style 0), so switching // it to UTF-8 is safe (no leak into user shells) and lets cmd parse UTF-8 remnants. "chcp 65001 >nul", windowsBatchSet("OCX_SERVICE", "1"), + windowsBatchSet(WINDOWS_WRAPPER_PROTOCOL_ENV, "1"), windowsBatchSet(BUN_RUNTIME_SOURCE_ENV, bunRuntimeSource), windowsBatchSet(BUN_RUNTIME_PATH_ENV, bun, "path"), windowsBatchSet("PATH", path, "pathList"), @@ -110,15 +113,16 @@ export function buildWindowsServiceScript( cli ? " exit /b 3" : null, cli ? ")" : null, cli ? `"%OCX_BUN%" "%OCX_CLI%" start --port ${port} >>"%OCX_SERVICE_LOG%" 2>&1` : `"%OCX_BUN%" start --port ${port} >>"%OCX_SERVICE_LOG%" 2>&1`, - "if %ERRORLEVEL% NEQ 0 (", - ' >>"%OCX_SERVICE_LOG%" echo [%DATE% %TIME%] child exited with code %ERRORLEVEL%; restarting in 5s', + // Stop commands kill the wrapper; a zero child exit alone is not a stop request. + `if "%ERRORLEVEL%"=="${WINDOWS_WRAPPER_STAY_OUT_EXIT_CODE}" goto stopped`, + '>>"%OCX_SERVICE_LOG%" echo [%DATE% %TIME%] child exited with code %ERRORLEVEL%; restarting in 5s', // `timeout` needs console stdin and dies with "Input redirection is not supported" // under Task Scheduler, turning the 5s cooldown into a hot restart loop; ping doesn't. - " ping -n 6 127.0.0.1 >nul", - " goto loop", - ")", + "ping -n 6 127.0.0.1 >nul", + "goto loop", + ":stopped", "endlocal", - "goto :eof", + "exit /b 0", "", // #1942/#1849: a power loss mid-swap leaves the live package dir missing/broken and // a sibling .ocx-backup-* holding the previous version. This wrapper lives OUTSIDE diff --git a/src/service/windows-wrapper-exit.ts b/src/service/windows-wrapper-exit.ts new file mode 100644 index 00000000000..c4892e18fc4 --- /dev/null +++ b/src/service/windows-wrapper-exit.ts @@ -0,0 +1,8 @@ +// Only wrappers advertising this protocol understand a nonzero intentional exit. +export const WINDOWS_WRAPPER_PROTOCOL_ENV = "OCX_WINDOWS_WRAPPER_PROTOCOL"; +export const WINDOWS_WRAPPER_STAY_OUT_EXIT_CODE = 42; + +export function serviceStayOutExitCode(env: NodeJS.ProcessEnv = process.env): number { + return env.OCX_SERVICE === "1" && env[WINDOWS_WRAPPER_PROTOCOL_ENV] === "1" + ? WINDOWS_WRAPPER_STAY_OUT_EXIT_CODE : 0; +} diff --git a/structure/ops/docs-and-release.md b/structure/ops/docs-and-release.md index ac414690dff..b2f84c4a57c 100644 --- a/structure/ops/docs-and-release.md +++ b/structure/ops/docs-and-release.md @@ -180,6 +180,14 @@ Those controls still have no owner, so there is no image-publish workflow or off ## Windows service wrapper and incomplete updates +The scheduler wrapper retries child exits, including zero, after five seconds. Only the +opt-in CLI stay-out code ends it successfully; missing Bun/CLI paths still exit with +installation error 3. Explicit service stop terminates the wrapper itself. +`src/service/windows-wrapper-exit.ts` defines the opt-in contract: new wrappers set +`OCX_WINDOWS_WRAPPER_PROTOCOL=1`, and all three CLI live-owner exits return 42 in that +service context. The wrapper translates 42 into a successful exit; legacy service +contexts retain exit 0. + > Decision record: [ADR-0082](../decisions/ADR-0082-windows-service-wrapper-and-incomplete-updates.md) ## GitHub workflow map diff --git a/structure/runtime.md b/structure/runtime.md index 84c3c0d7095..6d4fb758ffa 100644 --- a/structure/runtime.md +++ b/structure/runtime.md @@ -198,8 +198,8 @@ lost probe deletes this home's pid record and then binds a second listener that records and re-points Codex at itself. `probePortOwner` in `src/server/proxy-liveness.ts` asks the busy port directly, on both loopback families, independent of the pid and runtime records; the outcome is the pure decision `decideBusyPreferredPort` in `src/cli/dispatch.ts`. An opencodex -holder is refused with the same message the owner check prints (exit 0 instead under -`OCX_SERVICE=1`, so the wrapper loop terminates), and a holder that does not identify as opencodex +holder is refused with the same message the owner check prints (intentional stay-out under +`OCX_SERVICE=1`, using the [Windows wrapper protocol](ops/docs-and-release.md#windows-service-wrapper-and-incomplete-updates)), and a holder that does not identify as opencodex is reported as such rather than called foreign, because an identity probe cannot distinguish a foreign server from an unreachable one. An explicit `--port` still never hops — it waits for the pin through `src/server/port-reclaim.ts` — and a configured `port: 0` still means "ask the OS". diff --git a/tests/cli/cli-dispatch.test.ts b/tests/cli/cli-dispatch.test.ts index 2f474e054c1..9be4069cab9 100644 --- a/tests/cli/cli-dispatch.test.ts +++ b/tests/cli/cli-dispatch.test.ts @@ -436,8 +436,8 @@ describe("a busy preferred port never becomes a second proxy (#5004)", () => { expect(fn).toMatch(/decision === "refuse-live-proxy"[\s\S]{0,400}?StartCommandExit\(1\)/); expect(fn).toContain("Use 'ocx stop' first."); expect(fn).toMatch(/decision === "refuse-unidentified-holder"[\s\S]{0,700}?StartCommandExit\(1\)/); - // The wrapper's `if %ERRORLEVEL% NEQ 0` loop still terminates on a served port. - expect(fn).toMatch(/decision === "service-stay-out"[\s\S]{0,500}?StartCommandExit\(0\)/); + // The wrapper receives an explicit stay-out signal for a served port. + expect(fn).toMatch(/decision === "service-stay-out"[\s\S]{0,500}?StartCommandExit\(serviceStayOutExitCode\(\)\)/); }); test("the pre-bind owner probe spends the same budget before it deletes state", () => { diff --git a/tests/cli/cli-ready.test.ts b/tests/cli/cli-ready.test.ts index c214ecde156..e8ddf5eb769 100644 --- a/tests/cli/cli-ready.test.ts +++ b/tests/cli/cli-ready.test.ts @@ -844,8 +844,8 @@ describe("runReady production findLiveProxy deadline wiring (source-level)", () // ── handleStart service-wrapper exit guard (source-level) ───────────────────── // #764 follow-up: in OCX_SERVICE context a healthy proxy from ANY source must -// end handleStart with exit 0, so the opencodex-service.cmd `:loop` wrapper -// (retry on non-zero) does not respawn every 5s against a listener it can never +// end handleStart with the intentional stay-out code, so the service wrapper +// does not respawn every 5s against a listener it can never // claim. Source-level pin so a future edit cannot drop the guard silently. describe("handleStart OCX_SERVICE exit guard (source-level)", () => { const cliSource = readFileSync(repoPath("src/cli/index.ts"), "utf8"); @@ -854,14 +854,14 @@ describe("handleStart OCX_SERVICE exit guard (source-level)", () => { // The `OCX_SERVICE === "1"` comparison moved into `decideStartWithLiveOwner` // (src/cli/dispatch.ts), where the sentinel semantics are asserted at runtime // across the whole matrix (tests/cli/cli-dispatch.test.ts). This oracle pins the - // typed exits that the decision routes to: stay-out returns 0, the conflict returns 1. + // typed exits: stay-out uses the wrapper protocol, the conflict returns 1. expect(cliSource).toMatch(/decideStartWithLiveOwner\(\{/); // Anchor after the lease transaction begins. The earlier preflight has the same decision // pair but does not need a typed exit because it owns no lease yet. const transaction = cliSource.slice(cliSource.indexOf("bindAndPublishStartOwnership({")); const ownerBranch = transaction.slice(transaction.indexOf("decideStartWithLiveOwner({")); - const stayOut = ownerBranch.match(/decision === "service-stay-out"[\s\S]{0,800}?StartCommandExit\(0\)/); - expect(stayOut, "the service stay-out decision must return 0 when the port is already served").not.toBeNull(); + const stayOut = ownerBranch.match(/decision === "service-stay-out"[\s\S]{0,800}?StartCommandExit\(serviceStayOutExitCode\(\)\)/); + expect(stayOut, "the service stay-out decision must signal the wrapper when the port is already served").not.toBeNull(); const nonService = ownerBranch.match(/decision === "refuse"[\s\S]{0,500}?StartCommandExit\(1\)/); expect(nonService, "non-service refusal keeps the exit 1 conflict error").not.toBeNull(); }); diff --git a/tests/windows/windows-service-wrappers.test.ts b/tests/windows/windows-service-wrappers.test.ts index 4453e845a66..a86471a0b1b 100644 --- a/tests/windows/windows-service-wrappers.test.ts +++ b/tests/windows/windows-service-wrappers.test.ts @@ -124,3 +124,63 @@ describe("both teardown paths use the shared killer", () => { } }); }); + + +describe("scheduler child exit contract", () => { + test("zero and failure exits reach the cooldown; only explicit stay-out terminates", async () => { + const { buildWindowsServiceScript } = await import("../../src/service/windows-taskxml"); + for (const cli of ["C:\\ocx\\cli.ts", null]) { + const batch = buildWindowsServiceScript({ bun: "C:\\ocx\\bun.exe", bunRuntimeSource: "bundled", cli }, 10100, []); + const tail = batch.slice(batch.indexOf(' start --port 10100')).split("\r\n").slice(1); + expect(tail.slice(0, 6)).toEqual([ + 'if "%ERRORLEVEL%"=="42" goto stopped', + '>>"%OCX_SERVICE_LOG%" echo [%DATE% %TIME%] child exited with code %ERRORLEVEL%; restarting in 5s', + 'ping -n 6 127.0.0.1 >nul', + 'goto loop', + ':stopped', + 'endlocal', + ]); + expect(batch).toContain('set "OCX_WINDOWS_WRAPPER_PROTOCOL=1"'); + expect(batch).toContain('set "ERRORLEVEL="'); + expect(batch).toContain('exit /b 0'); + } + }); +}); + + +test("stay-out exit code is opt-in for new wrappers, preserving legacy services", async () => { + const { serviceStayOutExitCode } = await import("../../src/service/windows-wrapper-exit"); + expect(serviceStayOutExitCode({})).toBe(0); + expect(serviceStayOutExitCode({ OCX_SERVICE: "1" })).toBe(0); + expect(serviceStayOutExitCode({ OCX_WINDOWS_WRAPPER_PROTOCOL: "1" })).toBe(0); + expect(serviceStayOutExitCode({ OCX_SERVICE: "1", OCX_WINDOWS_WRAPPER_PROTOCOL: "1" })).toBe(42); + expect(serviceStayOutExitCode({ OCX_SERVICE: "1", OCX_WINDOWS_WRAPPER_PROTOCOL: "2" })).toBe(0); + const cli = read("src/cli/index.ts"); + const branches = [...cli.matchAll(/if \(decision === "service-stay-out"\) \{([\s\S]*?)\n\s*\}/g)]; + expect(branches).toHaveLength(3); + for (const branch of branches) expect(branch[1]).toContain("serviceStayOutExitCode()"); +}); + +test.skipIf(process.platform !== "win32")("cmd restarts zero/crash exits and stops on explicit stay-out", async () => { + const { mkdtempSync, writeFileSync, rmSync } = await import("node:fs"); + const { tmpdir } = await import("node:os"); + const { spawnSync } = await import("node:child_process"); + const { buildWindowsServiceScript } = await import("../../src/service/windows-taskxml"); + const dir = mkdtempSync(join(tmpdir(), "ocx-wrapper-exit-")); + try { + const batch = buildWindowsServiceScript({ bun: "bun.exe", bunRuntimeSource: "bundled", cli: null }, 10100, []); + const tail = batch.slice(batch.indexOf(' start --port 10100')).split("\r\n").slice(1).join("\r\n").split(":restore_backup")[0]; + for (const code of [0, 1, 42, 43, -1073741510]) { + const file = join(dir, "exit.cmd"); + // Exercise the generated control flow; replace only the cooldown to keep this fast. + writeFileSync(file, '@echo off\r\nsetlocal EnableExtensions DisableDelayedExpansion\r\nset "ERRORLEVEL="\r\nset "OCX_SERVICE_LOG=NUL"\r\n' + + `cmd /d /c exit ${code}\r\n` + tail.replace("ping -n 6 127.0.0.1 >nul", "rem skip cooldown") + + '\r\n:loop\r\nexit /b 99\r\n'); + const result = spawnSync("cmd.exe", ["/d", "/c", file], { timeout: 5000, env: { ...process.env, ERRORLEVEL: "42" } }); + expect(result.error).toBeUndefined(); + expect(result.status).toBe(code === 42 ? 0 : 99); + } + } finally { + rmSync(dir, { recursive: true, force: true }); + } +}); From e2565f5e6b2ecd72669a618747dfb93a75dee90e Mon Sep 17 00:00:00 2001 From: kaladinhonor <266145786+kaladinhonor@users.noreply.github.com> Date: Sun, 27 Sep 2026 02:02:27 +0900 Subject: [PATCH 04/11] fix(claude): decline message threads on translated Messages routes (#5935) Carried from #5935 as one squashed commit. Co-authored-by: kaladinhonor <266145786+kaladinhonor@users.noreply.github.com> --- .../src/content/docs/guides/claude-code.md | 2 + scripts/test-layout/layout.json | 1 + src/claude/message-threads.ts | 28 +++ src/server/claude-messages.ts | 9 + structure/data-planes/inbound-compat.md | 14 ++ .../claude-messages-thread.test.ts | 173 ++++++++++++++++++ tests/fixtures/test-layout-expected.json | 1 + 7 files changed, 228 insertions(+) create mode 100644 src/claude/message-threads.ts create mode 100644 tests/claude-integration/claude-messages-thread.test.ts diff --git a/docs-site/src/content/docs/guides/claude-code.md b/docs-site/src/content/docs/guides/claude-code.md index fd4056f329c..0c76acdef0d 100644 --- a/docs-site/src/content/docs/guides/claude-code.md +++ b/docs-site/src/content/docs/guides/claude-code.md @@ -184,6 +184,8 @@ working. OpenCodex only writes two variables into the `env` block of `~/.claude/ Claude Desktop first-party routes its Code tab and subagents through OpenCodex. The standalone Claude Code CLI has a separate first-party switch. Both clients read the same `~/.claude/settings.json` proxy and CA settings: if only one switch is on, the other client still transits the local proxy, where TLS terminates, but its Messages requests relay to Anthropic unchanged. Other Anthropic paths relay unchanged and unrelated hosts remain blind tunnels. +Subagents on routed (non-Claude) models do not use Claude Code's server-side message threads, because only Anthropic stores that state. OpenCodex declines a threaded request for such a model, and Claude Code resends that turn, and the turns after it, with the full conversation. + Mode is persisted as `claudeCode.desktopMode`. Installs that already applied either mode retain it, including first-party installs from before the mode was persisted. An explicit mode takes priority; otherwise an owned selected gateway row, an applied gateway fingerprint, or owned first-party diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 86759e4820f..4a817d43e5e 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -481,6 +481,7 @@ "claude-management-api.test.ts": "claude-integration", "claude-manual-env.test.ts": "gui", "claude-messages-endpoint.test.ts": "claude-integration", + "claude-messages-thread.test.ts": "claude-integration", "messages-native.test.ts": "claude-integration", "messages-native-decline-trace.test.ts": "claude-integration", "messages-native-oauth.test.ts": "claude-integration", diff --git a/src/claude/message-threads.ts b/src/claude/message-threads.ts new file mode 100644 index 00000000000..3afd1df0e9c --- /dev/null +++ b/src/claude/message-threads.ts @@ -0,0 +1,28 @@ +/** + * Claude Code's message-threads beta, which it only enables against first-party Anthropic, sends + * a `thread` object on subagent turns. A `continue` carries just the messages after + * `previous_message_id` and may omit `system` and `tools`, because Anthropic replays them from the + * stored thread. A translated route has no such store, so translating that delta silently drops the + * task, the instructions and the earlier turns. + * + * Claude Code reads this error code as "threads are unsupported for this model": it resends the + * same turn with the full conversation and keeps that model stateless for the rest of the session. + */ +export const MESSAGE_THREAD_UNSUPPORTED_ERROR_CODE = "thread_unsupported_request"; + +export function carriesMessageThread(body: unknown): boolean { + if (body === null || typeof body !== "object" || Array.isArray(body)) return false; + const thread = (body as Record).thread; + return thread !== null && typeof thread === "object" && !Array.isArray(thread); +} + +export function messageThreadUnsupportedResponse(): Response { + return new Response(JSON.stringify({ + type: "error", + error: { + type: "invalid_request_error", + message: "message threads are not supported on translated routes", + details: { error_code: MESSAGE_THREAD_UNSUPPORTED_ERROR_CODE }, + }, + }), { status: 400, headers: { "Content-Type": "application/json" } }); +} diff --git a/src/server/claude-messages.ts b/src/server/claude-messages.ts index 4bcb558cde3..09388c0c4db 100644 --- a/src/server/claude-messages.ts +++ b/src/server/claude-messages.ts @@ -27,6 +27,7 @@ import { stripOneMillionMarker } from "../claude/context-windows"; import { captureClaudeInbound } from "../claude/inbound-debug"; import { claudeCodeForIngress } from "../claude/intercept/model-bindings"; import { analyzeClaudeCompatibility, isClaudeCompatibilityMode } from "../claude/compatibility"; +import { carriesMessageThread, messageThreadUnsupportedResponse } from "../claude/message-threads"; import { applyReplayRefusalClientHeaders, carryReplayRefusal, @@ -837,6 +838,12 @@ async function handleClaudeMessagesWithBudget( recordProtocolShadowPlan(logCtx, config, { inbound: "messages", model: requestedModel }); return await anthropicNativePassthrough(req, config, logCtx, logIds, anthropicBody, "/v1/messages"); } + // Only Anthropic holds message-thread state; the error makes Claude Code resend the full turn. + if (carriesMessageThread(anthropicBody)) { + logCtx.errorCode = "claude_thread_unsupported"; + if (logIds) addFinalRequestLog(logIds.requestId, logIds.start, logCtx, 400, { closeReason: "non_stream" }); + return messageThreadUnsupportedResponse(); + } // Capture source semantics before effort rewriting or translation drops fields. // This policy is uniform across translated targets, including later fallback attempts. const compatibilityMode: unknown = config.claudeCode?.compatibility; @@ -1431,6 +1438,8 @@ export async function handleClaudeCountTokens( if (wantsNativePassthrough(req, config, requestPolicy, model, cc)) { return await anthropicNativePassthrough(req, config, { model, provider: "anthropic-native", surface: "claude" }, undefined, raw, "/v1/messages/count_tokens"); } + // A thread delta would undercount; refuse it exactly as the translated Messages path does. + if (carriesMessageThread(raw)) return messageThreadUnsupportedResponse(); // PF-08: an eligible managed-key route counts the body the native lane would send. const nativeCountBody = resolveProtocolSettings(config).rollout.managedMessagesNative ? (await import("./messages-native")).nativeMessagesCountBody(config, cc, raw, { fastRow: countFastRow !== null }) diff --git a/structure/data-planes/inbound-compat.md b/structure/data-planes/inbound-compat.md index 80a0c2b1a09..9aa6fee0410 100644 --- a/structure/data-planes/inbound-compat.md +++ b/structure/data-planes/inbound-compat.md @@ -310,6 +310,20 @@ The [explicit model-capability contract](../config.md#explicit-per-model-capabil Provider-scoped approval reviewer settings are projected by the [catalog owner](../catalog.md#provider-scoped-approval-reviewer); this surface retains its existing routing, transport and account-selection behavior. +## Claude message threads on translated routes + +Claude Code enables its message-threads beta only against first-party Anthropic, so it reaches +OpenCodex through the first-party intercept. A threaded request carries a `thread` object; a +`continue` sends only the messages after `previous_message_id` and may leave `system` and `tools` +to the thread Anthropic stores. `src/server/claude-messages.ts` forwards the request unchanged on +native passthrough. On the translated path, before compatibility analysis or inference, it +answers any `thread` object with the 400 from `src/claude/message-threads.ts`, whose +`error.details.error_code` is `thread_unsupported_request` and whose request-log error code is +`claude_thread_unsupported`. A translated `count_tokens` request with a `thread` object gets the +same 400, because counting the delta would undercount the conversation. Claude Code then resends the turn with the full conversation and keeps +that model stateless for the session. Translating the delta instead would drop the task, +instructions and earlier turns without an error. + ## Shared inbound Chat image recognition `src/chat/image-parts.ts` owns which `messages[].content[]` shapes count as an image diff --git a/tests/claude-integration/claude-messages-thread.test.ts b/tests/claude-integration/claude-messages-thread.test.ts new file mode 100644 index 00000000000..5dd6ec87afa --- /dev/null +++ b/tests/claude-integration/claude-messages-thread.test.ts @@ -0,0 +1,173 @@ +import { afterEach, beforeEach, expect, test } from "bun:test"; +import { mkdtempSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { saveConfig } from "../../src/config"; +import { startServer } from "../../src/server"; +import { clearRequestLogsForTests, getRequestLogEntries } from "../../src/server/request-log"; +import type { OcxConfig } from "../../src/types"; +import { installIsolatedCodexHome, type IsolatedCodexHome } from "../helpers/isolated-codex-home"; +import { removeTreeWithRetry } from "../helpers/remove-tree"; + +let testDir = ""; +let previousHome: string | undefined; +let isolatedCodexHome: IsolatedCodexHome | null = null; + +beforeEach(() => { + previousHome = process.env.OPENCODEX_HOME; + isolatedCodexHome = installIsolatedCodexHome("ocx-claude-thread-"); + testDir = mkdtempSync(join(tmpdir(), "ocx-claude-thread-")); + process.env.OPENCODEX_HOME = testDir; +}); + +afterEach(() => { + if (previousHome === undefined) delete process.env.OPENCODEX_HOME; + else process.env.OPENCODEX_HOME = previousHome; + isolatedCodexHome?.restore(); + isolatedCodexHome = null; + if (testDir) removeTreeWithRetry(testDir); +}); + +function mockChatUpstream() { + const captured: Array> = []; + const server = Bun.serve({ + port: 0, + async fetch(req) { + captured.push(await req.json() as Record); + const frames = [ + `data: ${JSON.stringify({ choices: [{ index: 0, delta: { role: "assistant", content: "ok" }, finish_reason: "stop" }] })}\n\n`, + "data: [DONE]\n\n", + ]; + return new Response(frames.join(""), { headers: { "Content-Type": "text/event-stream" } }); + }, + }); + return { server, captured }; +} + +function routedConfig(baseUrl: string): OcxConfig { + return { + port: 0, + defaultProvider: "mock", + providers: { mock: { adapter: "openai-chat", baseUrl, apiKey: "k", allowPrivateNetwork: true } }, + } as OcxConfig; +} + +const headers = { + "content-type": "application/json", + "x-api-key": "placeholder", + "anthropic-beta": "message-threads-2026-08-12", +}; + +// Turn 2 of a Claude Code subagent as the message-threads beta sends it: only the delta after +// the anchor, with system and tools left to the server-side thread. +const continueTurn = { + model: "mock/test-model", + max_tokens: 64, + thread: { type: "continue", previous_message_id: "msg_turn_one" }, + messages: [{ role: "user", content: [{ type: "tool_result", tool_use_id: "toolu_1", content: "file body" }] }], +}; + +// The stateless resend Claude Code makes after the unsupported error. +const statelessTurn = { + model: "mock/test-model", + max_tokens: 64, + system: [{ type: "text", text: "You are a subagent." }], + tools: [{ name: "Read", input_schema: { type: "object" } }], + messages: [ + { role: "user", content: "Read the file and report the secret word." }, + { role: "assistant", content: [{ type: "tool_use", id: "toolu_1", name: "Read", input: { path: "a.txt" } }] }, + { role: "user", content: [{ type: "tool_result", tool_use_id: "toolu_1", content: "file body" }] }, + ], +}; + +test("a translated route refuses message threads with Claude Code's unsupported code before inference", async () => { + const upstream = mockChatUpstream(); + saveConfig(routedConfig(new URL("/v1", upstream.server.url).href)); + const server = startServer(0); + try { + clearRequestLogsForTests(); + for (const [index, thread] of [ + { type: "continue", previous_message_id: "msg_turn_one" }, + { type: "create" }, + ].entries()) { + const response = await fetch(new URL("/v1/messages?beta=true", server.url), { + method: "POST", + headers, + body: JSON.stringify({ ...continueTurn, stream: index === 0, thread }), + }); + expect(response.status).toBe(400); + expect(await response.json()).toEqual({ + type: "error", + error: { + type: "invalid_request_error", + message: "message threads are not supported on translated routes", + details: { error_code: "thread_unsupported_request" }, + }, + }); + expect(getRequestLogEntries().at(-1)?.errorCode).toBe("claude_thread_unsupported"); + } + expect(upstream.captured).toHaveLength(0); + + // A thread delta would undercount, so count_tokens refuses it the same way. + const counted = await fetch(new URL("/v1/messages/count_tokens?beta=true", server.url), { + method: "POST", + headers, + body: JSON.stringify(continueTurn), + }); + expect(counted.status).toBe(400); + expect(await counted.json()).toMatchObject({ error: { details: { error_code: "thread_unsupported_request" } } }); + + const resent = await fetch(new URL("/v1/messages?beta=true", server.url), { + method: "POST", + headers, + body: JSON.stringify(statelessTurn), + }); + expect(resent.status).toBe(200); + await resent.text(); + expect(upstream.captured).toHaveLength(1); + const sent = JSON.stringify(upstream.captured[0]); + expect(sent).toContain("You are a subagent."); + expect(sent).toContain("Read the file and report the secret word."); + } finally { + await server.stop(true); + upstream.server.stop(true); + } +}); + +test("native Anthropic passthrough still forwards the thread unchanged", async () => { + let captured: Record | null = null; + const upstream = Bun.serve({ + port: 0, + async fetch(req) { + captured = await req.json() as Record; + return Response.json({ + id: "msg_turn_two", + type: "message", + role: "assistant", + model: "claude-haiku-4-5", + content: [{ type: "text", text: "ok" }], + stop_reason: "end_turn", + stop_sequence: null, + usage: { input_tokens: 1, output_tokens: 1 }, + }); + }, + }); + saveConfig({ + ...routedConfig("http://127.0.0.1:1/v1"), + claudeCode: { anthropicBaseUrl: upstream.url.toString().replace(/\/$/, "") }, + } as OcxConfig); + const server = startServer(0); + try { + const response = await fetch(new URL("/v1/messages", server.url), { + method: "POST", + headers: { ...headers, "x-api-key": "sk-ant-test" }, + body: JSON.stringify({ ...continueTurn, model: "claude-haiku-4-5" }), + }); + expect(response.status).toBe(200); + await response.text(); + expect(captured).toMatchObject({ thread: { type: "continue", previous_message_id: "msg_turn_one" } }); + } finally { + await server.stop(true); + upstream.stop(true); + } +}); diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 3a649b4c214..c1445e8b1af 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -307,6 +307,7 @@ "claude-management-api.test.ts": "claude-integration", "claude-manual-env.test.ts": "gui", "claude-messages-endpoint.test.ts": "claude-integration", + "claude-messages-thread.test.ts": "claude-integration", "messages-native.test.ts": "claude-integration", "messages-native-decline-trace.test.ts": "claude-integration", "messages-native-oauth.test.ts": "claude-integration", From 645b805436371d24df2b040fa57e1704d3a4b20d Mon Sep 17 00:00:00 2001 From: boblob6969 Date: Sun, 27 Sep 2026 02:02:40 +0900 Subject: [PATCH 05/11] fix(responses): spell \0 as \x00 in tool-schema patterns (#5939) Carried from #5939 as one squashed commit. Co-authored-by: boblob6969 --- src/adapters/responses-tool-schema.ts | 34 +++++++++++++++++-- .../openai/openai-chat-hardening.test.ts | 23 +++++++++++++ 2 files changed, 54 insertions(+), 3 deletions(-) diff --git a/src/adapters/responses-tool-schema.ts b/src/adapters/responses-tool-schema.ts index 6b28cca6eae..ed679464fdf 100644 --- a/src/adapters/responses-tool-schema.ts +++ b/src/adapters/responses-tool-schema.ts @@ -96,9 +96,32 @@ function usesUnicodePropertyEscape(pattern: string): boolean { return false; } +/** + * Rewrite each unescaped `\0` not followed by a digit to the equivalent `\x00`. Both spell NUL + * in ECMAScript and Python `re`, but some destinations' schema validators refuse `\0` inside a + * character class: Meta's Responses API answers 400 "is not a \"regex\"" to Claude Code's + * Artifact tool, whose file-path parameters carry `^[^\0]*$`. `\0` followed by a digit is an + * octal escape in Python, so it is left alone. Returns the input when nothing changed. + */ +function rewriteNulEscapes(pattern: string): string { + let out: string | undefined; + let copied = 0; + for (let i = 0; i < pattern.length; i++) { + if (pattern[i] !== "\\") continue; + const next = pattern[i + 1]; + if (next === "0" && !/[0-9]/.test(pattern[i + 2] ?? "")) { + out = (out ?? "") + pattern.slice(copied, i) + "\\x00"; + copied = i + 2; + } + i++; + } + return out === undefined ? pattern : out + pattern.slice(copied); +} + /** * Remove unsupported Unicode property escapes from scalar `pattern` constraints in ordinary - * positive schema positions. This keeps built-in Artifact tools usable on Python-re backends; + * positive schema positions, and spell `\0` as `\x00` (rewriteNulEscapes). This keeps built-in + * Artifact tools usable on Python-re backends and on Meta's Responses API; * the omitted constraint is not enforced by this proxy and tools must validate their inputs. * * Regex-keyed objects are preserved. Removing a matcher can lose evaluated-property annotations @@ -181,8 +204,13 @@ export function stripUnicodePropertyPatterns(node: unknown, inNameBag = false): continue; } const [key, value] = next.value; - if (!frame.inNameBag && key === "pattern" && typeof value === "string" && usesUnicodePropertyEscape(value)) { - delete (cloneContainer(frame) as Record)[key]; + if (!frame.inNameBag && key === "pattern" && typeof value === "string") { + if (usesUnicodePropertyEscape(value)) { + delete (cloneContainer(frame) as Record)[key]; + continue; + } + const rewritten = rewriteNulEscapes(value); + if (rewritten !== value) (cloneContainer(frame) as Record)[key] = rewritten; continue; } if (!frame.inNameBag && (PRESERVED_PATTERN_SUBTREES.has(key) || SCHEMA_LITERAL_VALUE_KEYS.has(key))) { diff --git a/tests/adapters/openai/openai-chat-hardening.test.ts b/tests/adapters/openai/openai-chat-hardening.test.ts index 492dca51c08..109ec05aea6 100644 --- a/tests/adapters/openai/openai-chat-hardening.test.ts +++ b/tests/adapters/openai/openai-chat-hardening.test.ts @@ -362,6 +362,29 @@ describe("unicode property-escape pattern stripping", () => { expect(stripUnicodePropertyPatterns(before)).toBe(before); }); + test("`\\0` is rewritten to the equivalent `\\x00`, octal and escaped backslashes untouched", () => { + // Claude Code's Artifact tool ships `^[^\0]*$` on its file-path parameters; Meta's + // Responses API rejects `\0` inside a character class but accepts `\x00`. + const artifactPathPattern = "^[^\\0]*$"; + const before = { + type: "object", + properties: { + path: { type: "string", pattern: artifactPathPattern, maxLength: 1024 }, + octal: { type: "string", pattern: "^\\012$" }, + literal: { type: "string", pattern: "^\\\\0$" }, + }, + }; + const out = stripUnicodePropertyPatterns(before) as typeof before; + expect(out.properties.path.pattern).toBe("^[^\\x00]*$"); + expect(out.properties.path.maxLength).toBe(1024); + expect(out.properties.octal).toBe(before.properties.octal); + expect(out.properties.literal).toBe(before.properties.literal); + expect(before.properties.path.pattern).toBe(artifactPathPattern); + for (const s of ["\u0000", "a", "\u0000/b"]) { + expect(new RegExp(out.properties.path.pattern).test(s)).toBe(new RegExp(artifactPathPattern).test(s)); + } + }); + test("`\\P{…}` is dropped as well as `\\p{…}`", () => { const stripped = stripUnicodePropertyPatterns({ type: "string", pattern: "^\\P{L}+$" }) as Record; expect(stripped.pattern).toBeUndefined(); From c979f1c31a1fec3c959fa4b1624bab4ef0f4153e Mon Sep 17 00:00:00 2001 From: codingbo Date: Sun, 27 Sep 2026 02:04:16 +0900 Subject: [PATCH 06/11] fix(providers): parse Kiro meteringEvent credits and preserve in usage ledger (#5951) Carried from #5951 as one squashed commit. Co-authored-by: codingbo --- .../src/content/docs/guides/providers.md | 8 + scripts/test-layout/layout.json | 2 + src/adapters/kiro-events.ts | 27 ++- src/adapters/kiro/stream.ts | 11 +- src/server/request-log.ts | 4 +- .../responses/empty-completion-guard.ts | 2 + src/server/responses/terminal-guard.ts | 2 + src/types/request.ts | 2 + src/usage/log.ts | 2 + structure/dashboard-and-usage.md | 5 + structure/providers-and-adapters.md | 3 + structure/providers/kiro.md | 13 +- tests/fixtures/test-layout-expected.json | 2 + .../kiro/kiro-metering-events.test.ts | 173 ++++++++++++++++++ .../kiro/kiro-metering-usage.test.ts | 64 +++++++ .../responses/empty-completion-guard.test.ts | 5 +- .../server/server-kiro-completion-e2e.test.ts | 12 +- tests/server/terminal-guard.test.ts | 6 +- tests/usage/key-attribution.test.ts | 9 +- 19 files changed, 337 insertions(+), 15 deletions(-) create mode 100644 tests/providers/kiro/kiro-metering-events.test.ts create mode 100644 tests/providers/kiro/kiro-metering-usage.test.ts diff --git a/docs-site/src/content/docs/guides/providers.md b/docs-site/src/content/docs/guides/providers.md index 54372cce2f9..fa02f36a8db 100644 --- a/docs-site/src/content/docs/guides/providers.md +++ b/docs-site/src/content/docs/guides/providers.md @@ -402,6 +402,14 @@ those providers, but `ocx login codex --reauth` routes to their account-pool rea the dashboard Codex account pool also performs. See [`ocx status` / `ocx doctor`](/reference/cli/) in the CLI reference. +### Kiro request credits + +When Kiro emits credit metering, request logs preserve the reported spend as +`usage.providerCredits`, including in the persisted usage ledger. These are Kiro credits; +token counts may still be estimated, and the credit value does not replace USD cost estimates. +Completion fallback requests add their reported credits. An absent value means Kiro did not +report credit usage; an explicit zero means it reported no spend. + ### Kiro credential import Kiro login expects the Kiro CLI: on Unix, install it with `curl -fsSL https://cli.kiro.dev/install | bash`; diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 4a817d43e5e..97af5d79fe9 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -1080,6 +1080,8 @@ "kiro-remote-image.test.ts": "providers/kiro", "kiro-retry.test.ts": "providers/kiro", "kiro-review-regressions.test.ts": "providers/kiro", + "kiro-metering-events.test.ts": "providers/kiro", + "kiro-metering-usage.test.ts": "providers/kiro", "kiro-stream.test.ts": "providers/kiro", "kiro-transport-parity.test.ts": "providers/kiro", "kiro-usage-quota.test.ts": "providers/kiro", diff --git a/src/adapters/kiro-events.ts b/src/adapters/kiro-events.ts index 730ad2a2c4e..1ef837d9f62 100644 --- a/src/adapters/kiro-events.ts +++ b/src/adapters/kiro-events.ts @@ -1,10 +1,12 @@ import type { OcxUsage } from "../types"; +import { debugProviderDiagnostic } from "../lib/debug"; import { kiroTruncationReason } from "./kiro-truncation"; export type ParsedKiroEvent = | { type: "content"; data?: string; modelId?: string } | { type: "reasoning"; data?: string; signature?: string; redactedContent?: string } | { type: "context_usage"; contextUsagePercentage: number } + | { type: "metering"; unit: string; usage: number; unitPlural?: string } | { type: "tool"; name?: string; toolUseId?: string; input?: string; stop?: boolean } | { type: "truncation"; data: string } | { type: "metadata"; usage?: OcxUsage; contextUsagePercentage?: number; stopReason?: string } @@ -17,10 +19,12 @@ const KNOWN_EVENT_TYPES = new Set([ "reasoningContentEvent", "toolUseEvent", "messageMetadataEvent", + "initial-response", "metadataEvent", // Authoritative context pressure. Every capture (kiro-cli 2.14.1 and 2.16.0) put the percentage // HERE and left `metadataEvent` carrying only `stopReason`; metadataEvent's own // contextUsagePercentage stays supported as a fallback rather than being dropped. + "meteringEvent", "contextUsageEvent", "invalidStateEvent", "error", @@ -111,7 +115,10 @@ function parseTokenUsage(eventType: string, value: unknown): OcxUsage | undefine /** Decode a known Kiro event using its Smithy `:event-type` header. */ export function parseKiroEvent(eventType: string, payload: Uint8Array): ParsedKiroEvent | null { // Unknown event types are intentionally ignored without parsing or logging their payload. - if (!KNOWN_EVENT_TYPES.has(eventType)) return null; + if (!KNOWN_EVENT_TYPES.has(eventType)) { + debugProviderDiagnostic("kiro", "unknown_event", { eventType }); + return null; + } const parsed = parseObject(eventType, payload); // A metadataEvent's `stopReason` is Kiro's own terminal verdict and must reach the parser // intact. The generic truncation sniffer matches substrings ("max_tokens", "length", @@ -174,7 +181,25 @@ export function parseKiroEvent(eventType: string, payload: Uint8Array): ParsedKi ? { stop: optionalBoolean(eventType, parsed, "stop") } : {}), }; + case "meteringEvent": { + const unit = optionalString(eventType, parsed, "unit"); + if (unit === undefined) { + return malformed(eventType, "unit must be a string"); + } + const unitPlural = optionalString(eventType, parsed, "unitPlural"); + const rawUsage = parsed.usage !== undefined ? parsed.usage : parsed.amount; + if (typeof rawUsage !== "number" || !Number.isFinite(rawUsage) || rawUsage < 0) { + return malformed(eventType, "usage must be a finite non-negative number"); + } + return { + type: "metering", + unit, + usage: rawUsage, + ...(unitPlural !== undefined ? { unitPlural } : {}), + }; + } case "messageMetadataEvent": + case "initial-response": return { type: "message_metadata", conversationId: diff --git a/src/adapters/kiro/stream.ts b/src/adapters/kiro/stream.ts index 80a127afaa7..f10cc6fce99 100644 --- a/src/adapters/kiro/stream.ts +++ b/src/adapters/kiro/stream.ts @@ -176,6 +176,7 @@ function mergeKiroUsage( ...(sumOptional("cachedInputTokens") !== undefined ? { cachedInputTokens: sumOptional("cachedInputTokens") } : {}), ...(sumOptional("cacheReadInputTokens") !== undefined ? { cacheReadInputTokens: sumOptional("cacheReadInputTokens") } : {}), ...(sumOptional("cacheCreationInputTokens") !== undefined ? { cacheCreationInputTokens: sumOptional("cacheCreationInputTokens") } : {}), + ...(sumOptional("providerCredits") !== undefined ? { providerCredits: sumOptional("providerCredits") } : {}), ...(sumOptional("reasoningOutputTokens") !== undefined ? { reasoningOutputTokens: sumOptional("reasoningOutputTokens") } : {}), ...(first.estimated || second.estimated ? { estimated: true } : {}), }; @@ -320,6 +321,7 @@ async function* parseKiroAttemptEvents( let completionAnswer: string | undefined; let completionCalls = 0; let authoritativeUsage: OcxUsage | undefined; + let providerCredits: number | undefined; let stopReason: string | undefined; const fallbackEvents: AdapterEvent[] = []; const thinking = new InlineThinkTagParser(budget); @@ -380,7 +382,11 @@ async function* parseKiroAttemptEvents( contextUsageTotalFloor() ?? 0, authoritativeTurnTotal, ); - return contextTotal > 0 ? { ...base, contextTotalTokens: contextTotal } : base; + return { + ...base, + ...(contextTotal > 0 ? { contextTotalTokens: contextTotal } : {}), + ...(providerCredits !== undefined ? { providerCredits } : {}), + }; }; const classifiedTerminal = (failure: KiroErrorClassification): AdapterEvent => { @@ -600,6 +606,9 @@ async function* parseKiroAttemptEvents( const ev = parseKiroEvent(eventType, msg.payload); if (!ev) continue; switch (ev.type) { + case "metering": + if (ev.unit === "credit" || ev.unit === "credits") providerCredits = ev.usage; + break; case "metadata": if (ev.usage) authoritativeUsage = ev.usage; if (ev.contextUsagePercentage !== undefined && ev.contextUsagePercentage > 0) { diff --git a/src/server/request-log.ts b/src/server/request-log.ts index e565f546b06..2bc6cb6308e 100644 --- a/src/server/request-log.ts +++ b/src/server/request-log.ts @@ -1905,7 +1905,7 @@ export function aggregateAttemptUsage( const sumOptional = ( key: "cachedInputTokens" | "cacheReadInputTokens" | "cacheCreationInputTokens" - | "reasoningOutputTokens", + | "reasoningOutputTokens" | "providerCredits", ): number | undefined => { const present = usages.flatMap(usage => ( typeof usage[key] === "number" ? [usage[key] as number] : [] @@ -1916,6 +1916,7 @@ export function aggregateAttemptUsage( const cacheReadInputTokens = sumOptional("cacheReadInputTokens"); const cacheCreationInputTokens = sumOptional("cacheCreationInputTokens"); const reasoningOutputTokens = sumOptional("reasoningOutputTokens"); + const providerCredits = sumOptional("providerCredits"); const totalTokens = usages.reduce( (sum, usage) => sum + (usageTotalTokens(usage) ?? 0), 0, @@ -1928,6 +1929,7 @@ export function aggregateAttemptUsage( ...(cacheReadInputTokens !== undefined ? { cacheReadInputTokens } : {}), ...(cacheCreationInputTokens !== undefined ? { cacheCreationInputTokens } : {}), ...(reasoningOutputTokens !== undefined ? { reasoningOutputTokens } : {}), + ...(providerCredits !== undefined ? { providerCredits } : {}), ...(status === "estimated" ? { estimated: true } : {}), }; return { usage: aggregate, status, totalTokens }; diff --git a/src/server/responses/empty-completion-guard.ts b/src/server/responses/empty-completion-guard.ts index 92243523e1a..10a1b563b72 100644 --- a/src/server/responses/empty-completion-guard.ts +++ b/src/server/responses/empty-completion-guard.ts @@ -173,6 +173,7 @@ export function mergeUsage( const cacheReadInputTokens = sumOptional("cacheReadInputTokens"); const cacheCreationInputTokens = sumOptional("cacheCreationInputTokens"); const reasoningOutputTokens = sumOptional("reasoningOutputTokens"); + const providerCredits = sumOptional("providerCredits"); const contextTotalTokens = second.contextTotalTokens ?? first.contextTotalTokens; const inputTokens = first.inputTokens + second.inputTokens; const outputTokens = first.outputTokens + second.outputTokens; @@ -188,6 +189,7 @@ export function mergeUsage( ...(cacheReadInputTokens !== undefined ? { cacheReadInputTokens } : {}), ...(cacheCreationInputTokens !== undefined ? { cacheCreationInputTokens } : {}), ...(reasoningOutputTokens !== undefined ? { reasoningOutputTokens } : {}), + ...(providerCredits !== undefined ? { providerCredits } : {}), ...(first.estimated || second.estimated ? { estimated: true } : {}), ...(rawUsage !== undefined ? { rawUsage } : {}), }; diff --git a/src/server/responses/terminal-guard.ts b/src/server/responses/terminal-guard.ts index aa4a870645d..7f4859640bd 100644 --- a/src/server/responses/terminal-guard.ts +++ b/src/server/responses/terminal-guard.ts @@ -184,6 +184,7 @@ function mergeUsage(first: OcxUsage | undefined, second: OcxUsage | undefined): const cacheReadInputTokens = sumOptional("cacheReadInputTokens"); const cacheCreationInputTokens = sumOptional("cacheCreationInputTokens"); const reasoningOutputTokens = sumOptional("reasoningOutputTokens"); + const providerCredits = sumOptional("providerCredits"); const inputTokens = first.inputTokens + second.inputTokens; const outputTokens = first.outputTokens + second.outputTokens; return { @@ -194,6 +195,7 @@ function mergeUsage(first: OcxUsage | undefined, second: OcxUsage | undefined): ...(cacheReadInputTokens !== undefined ? { cacheReadInputTokens } : {}), ...(cacheCreationInputTokens !== undefined ? { cacheCreationInputTokens } : {}), ...(reasoningOutputTokens !== undefined ? { reasoningOutputTokens } : {}), + ...(providerCredits !== undefined ? { providerCredits } : {}), ...(first.estimated || second.estimated ? { estimated: true } : {}), }; } diff --git a/src/types/request.ts b/src/types/request.ts index c365826de94..459c12d19cf 100644 --- a/src/types/request.ts +++ b/src/types/request.ts @@ -438,6 +438,8 @@ export interface OcxUrlCitation { * - `totalTokens` = inputTokens + outputTokens. Never re-add cache detail on top. */ export interface OcxUsage { + /** Provider-reported credit spend, independent of token estimates and USD pricing. */ + providerCredits?: number; inputTokens: number; outputTokens: number; /** diff --git a/src/usage/log.ts b/src/usage/log.ts index d382f202b42..0c3b37e5b89 100644 --- a/src/usage/log.ts +++ b/src/usage/log.ts @@ -579,6 +579,7 @@ function normalizeUsageValue(usage: OcxUsage | undefined): OcxUsage | undefined ...(typeof usage.cacheReadInputTokens === "number" ? { cacheReadInputTokens: usage.cacheReadInputTokens } : {}), ...(typeof usage.cacheCreationInputTokens === "number" ? { cacheCreationInputTokens: usage.cacheCreationInputTokens } : {}), ...(typeof usage.reasoningOutputTokens === "number" ? { reasoningOutputTokens: usage.reasoningOutputTokens } : {}), + ...(isNonNegativeFiniteNumber(usage.providerCredits) ? { providerCredits: usage.providerCredits } : {}), ...(usage.estimated ? { estimated: true } : {}), }; } @@ -622,6 +623,7 @@ function normalizeAttemptUsage(raw: unknown): OcxUsage | null { "cacheReadInputTokens", "cacheCreationInputTokens", "reasoningOutputTokens", + "providerCredits", ] as const) { if (key in usage && !isNonNegativeFiniteNumber(usage[key])) return null; } diff --git a/structure/dashboard-and-usage.md b/structure/dashboard-and-usage.md index 93fb16fdca2..d5d747ab24e 100644 --- a/structure/dashboard-and-usage.md +++ b/structure/dashboard-and-usage.md @@ -163,6 +163,11 @@ keeps the saved state and renders fixed `ocx sync` guidance without server/accou ## Usage accounting +`OcxUsage.providerCredits` preserves provider-reported credit spend in request and attempt rows +through `src/usage/log.ts` normalization and ledger reloads. Missing readings stay absent, and zero +is a measured value. Separate attempts add credits when usage is merged. The field is independent +of token estimation (`estimated` describes tokens) and is never treated as USD or token usage. + ### Upstream key account attribution API-key attempts in `src/usage/log.ts` carry `accountLogLabel` as `k` plus 32 lowercase diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index 923aedc3bdd..b649d6704c5 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -113,6 +113,9 @@ Inline document admission shares one encoding predicate between its scanner and `src/responses/inline-document.ts`: malformed base64 quantum/padding lengths are refused, and valid padded or unpadded payloads pass unchanged without a decoding allocation. +Kiro metering uses the [provider credit contract](providers/kiro.md#kiro-reasoning-round-trip-signature); +`src/types/request.ts` keeps reported credits separate from estimated token usage. + Adapter output must stay in internal `AdapterEvent` form until `src/bridge/sse.ts` converts it back to Responses SSE or WebSocket frames, or `src/bridge/response-json.ts` buffers it into a JSON response. `src/bridge.ts` is the compatibility facade that re-exports both. diff --git a/structure/providers/kiro.md b/structure/providers/kiro.md index 9c5aa799942..1745968844e 100644 --- a/structure/providers/kiro.md +++ b/structure/providers/kiro.md @@ -132,8 +132,17 @@ from `metadataEvent` is legitimate rather than impossible. Both feed the same fi positive value overwrites an earlier one. Spend arrives in `meteringEvent` as **credits, not tokens**. No captured response carried -`tokenUsage` on any event, which is why Kiro usage stays estimated; `meteringEvent` is currently -ignored because a credit is not a token count. +`tokenUsage` on any event, which is why Kiro token usage stays estimated. The parser preserves +`meteringEvent` unit/usage (`amount` is an alias) and optional `unitPlural`; credit readings populate +`OcxUsage.providerCredits` independently of token metadata. The latest reading within a response +is a snapshot; separate completion-fallback responses add their credits. Missing metering stays +absent and measured zero stays zero. `initial-response` carries `conversationId` through the same +validated provider-state path as `messageMetadataEvent`. Unknown event types produce opt-in +`debugProviderDiagnostic` entries containing only the event type, never the payload. +Coverage: `tests/providers/kiro/kiro-metering-events.test.ts`, +`tests/providers/kiro/kiro-metering-usage.test.ts`, and +`tests/server/server-kiro-completion-e2e.test.ts`. + ## Remote image references Kiro's wire inlines base64 bytes only, so a remote `https` image reference cannot be diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index c1445e8b1af..c16c5386ed4 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -901,6 +901,8 @@ "kiro-remote-image.test.ts": "providers/kiro", "kiro-retry.test.ts": "providers/kiro", "kiro-review-regressions.test.ts": "providers/kiro", + "kiro-metering-events.test.ts": "providers/kiro", + "kiro-metering-usage.test.ts": "providers/kiro", "kiro-stream.test.ts": "providers/kiro", "kiro-transport-parity.test.ts": "providers/kiro", "kiro-usage-quota.test.ts": "providers/kiro", diff --git a/tests/providers/kiro/kiro-metering-events.test.ts b/tests/providers/kiro/kiro-metering-events.test.ts new file mode 100644 index 00000000000..4a1dfdccfed --- /dev/null +++ b/tests/providers/kiro/kiro-metering-events.test.ts @@ -0,0 +1,173 @@ +import { afterEach, beforeEach, describe, expect, spyOn, test } from "bun:test"; +import { parseKiroEvent } from "../../../src/adapters/kiro-events"; +import { getDebugLogEntries, resetDebugLogBufferForTests } from "../../../src/lib/debug-log-buffer"; +import { clearDebugSetting, getDebugSettings, setDebugSettings } from "../../../src/lib/debug-settings"; + +const enc = new TextEncoder(); + +describe("parseKiroEvent - meteringEvent", () => { + test("parses real precise sample metering event with unit and usage", () => { + const raw = enc.encode(JSON.stringify({ unit: "credit", usage: 0.04582331509121062 })); + expect(parseKiroEvent("meteringEvent", raw)).toEqual({ + type: "metering", + unit: "credit", + usage: 0.04582331509121062, + }); + + const withPlural = enc.encode( + JSON.stringify({ unit: "credit", unitPlural: "credits", usage: 0.04582331509121062 }), + ); + expect(parseKiroEvent("meteringEvent", withPlural)).toEqual({ + type: "metering", + unit: "credit", + usage: 0.04582331509121062, + unitPlural: "credits", + }); + }); + + test("parses zero usage", () => { + const raw = enc.encode(JSON.stringify({ unit: "credit", usage: 0 })); + expect(parseKiroEvent("meteringEvent", raw)).toEqual({ + type: "metering", + unit: "credit", + usage: 0, + }); + }); + + test("parses amount alias and prefers usage over amount when both present", () => { + const aliasOnly = enc.encode(JSON.stringify({ unit: "credit", amount: 0.01 })); + expect(parseKiroEvent("meteringEvent", aliasOnly)).toEqual({ + type: "metering", + unit: "credit", + usage: 0.01, + }); + + const both = enc.encode(JSON.stringify({ unit: "credit", usage: 0.05, amount: 0.01 })); + expect(parseKiroEvent("meteringEvent", both)).toEqual({ + type: "metering", + unit: "credit", + usage: 0.05, + }); + }); + + test("rejects invalid values", () => { + const invalidCases = [ + { payload: { unit: "credit", usage: -1 }, desc: "negative usage" }, + { payload: { unit: "credit", amount: -0.01 }, desc: "negative amount" }, + { payload: { unit: "credit", usage: "0.5" }, desc: "string usage" }, + { payload: { unit: "credit", amount: "0.5" }, desc: "string amount" }, + { payload: { unit: "credit", usage: null }, desc: "null usage" }, + { payload: { unit: "credit" }, desc: "missing usage and amount" }, + { payload: { usage: 1 }, desc: "missing unit" }, + { payload: { unit: 123, usage: 1 }, desc: "non-string unit" }, + { payload: { unit: null, usage: 1 }, desc: "null unit" }, + { payload: { unit: "credit", unitPlural: 123, usage: 1 }, desc: "non-string unitPlural" }, + ]; + + for (const { payload, desc } of invalidCases) { + const raw = enc.encode(JSON.stringify(payload)); + expect(() => parseKiroEvent("meteringEvent", raw), desc).toThrow( + /invalid Kiro meteringEvent payload/, + ); + } + + // Non-finite number + expect(() => + parseKiroEvent("meteringEvent", enc.encode('{"unit":"credit","usage":Infinity}')), + ).toThrow(/invalid Kiro meteringEvent payload/); + + // Malformed JSON / non-object + expect(() => parseKiroEvent("meteringEvent", enc.encode("not-json"))).toThrow( + /invalid Kiro meteringEvent payload/, + ); + expect(() => parseKiroEvent("meteringEvent", enc.encode("123"))).toThrow( + /invalid Kiro meteringEvent payload/, + ); + }); +}); + +describe("parseKiroEvent - initial-response", () => { + test("aliases message_metadata conversationId parsing", () => { + const withConv = enc.encode(JSON.stringify({ conversationId: "conv-12345" })); + expect(parseKiroEvent("initial-response", withConv)).toEqual({ + type: "message_metadata", + conversationId: "conv-12345", + }); + + const withUtt = enc.encode(JSON.stringify({ utteranceId: "utt-67890" })); + expect(parseKiroEvent("initial-response", withUtt)).toEqual({ + type: "message_metadata", + conversationId: "utt-67890", + }); + + const empty = enc.encode(JSON.stringify({})); + expect(parseKiroEvent("initial-response", empty)).toEqual({ + type: "message_metadata", + conversationId: undefined, + }); + }); +}); + +describe("parseKiroEvent - unknown event diagnostics", () => { + let origDebug: string | undefined; + let origDebugFrames: string | undefined; + let origDebugOverride: boolean | undefined; + + beforeEach(() => { + origDebug = process.env.OCX_DEBUG; + origDebugFrames = process.env.OCX_DEBUG_FRAMES; + origDebugOverride = getDebugSettings().runtimeOverride.debug; + delete process.env.OCX_DEBUG; + delete process.env.OCX_DEBUG_FRAMES; + clearDebugSetting("debug"); + resetDebugLogBufferForTests(); + }); + + afterEach(() => { + if (origDebug === undefined) delete process.env.OCX_DEBUG; else process.env.OCX_DEBUG = origDebug; + if (origDebugFrames === undefined) delete process.env.OCX_DEBUG_FRAMES; else process.env.OCX_DEBUG_FRAMES = origDebugFrames; + if (origDebugOverride === undefined) clearDebugSetting("debug"); + else setDebugSettings({ debug: origDebugOverride }); + resetDebugLogBufferForTests(); + }); + + test("unknown event type calls debugProviderDiagnostic with { eventType } and does not log/parse payload when debug enabled", () => { + setDebugSettings({ debug: true }); + const error = spyOn(console, "error").mockImplementation(() => {}); + + try { + const payload = enc.encode("sensitive-payload-that-must-not-be-parsed-or-logged"); + const result = parseKiroEvent("someUnknownFutureEvent", payload); + + expect(result).toBeNull(); + expect(error).toHaveBeenCalledTimes(1); + + const line = String(error.mock.calls[0]?.[0] ?? ""); + expect(line).toContain("[ocx:kiro:unknown_event]"); + expect(line).toContain('"eventType":"someUnknownFutureEvent"'); + expect(line).not.toContain("sensitive-payload"); + + const logEntries = getDebugLogEntries(); + expect(logEntries.some((entry) => entry.line.includes("[ocx:kiro:unknown_event]"))).toBe(true); + expect(logEntries.some((entry) => entry.line.includes("sensitive-payload"))).toBe(false); + } finally { + error.mockRestore(); + } + }); + + test("unknown event stays quiet when debug is disabled and does not log or parse payload", () => { + setDebugSettings({ debug: false }); + const error = spyOn(console, "error").mockImplementation(() => {}); + + try { + const payload = enc.encode("sensitive-payload-that-must-not-be-parsed-or-logged"); + const result = parseKiroEvent("someUnknownFutureEvent", payload); + + expect(result).toBeNull(); + expect(error).not.toHaveBeenCalled(); + expect(getDebugLogEntries()).toHaveLength(0); + } finally { + error.mockRestore(); + } + }); +}); diff --git a/tests/providers/kiro/kiro-metering-usage.test.ts b/tests/providers/kiro/kiro-metering-usage.test.ts new file mode 100644 index 00000000000..0cbcf940046 --- /dev/null +++ b/tests/providers/kiro/kiro-metering-usage.test.ts @@ -0,0 +1,64 @@ +import { expect, test } from "bun:test"; +import { parseKiroStream } from "../../../src/adapters/kiro"; +import { encodeMessage } from "../../../src/lib/eventstream-decoder"; +import { normalizeUsageEntryForTest } from "../../../src/usage/log"; +import { createTestTranslatorBudget } from "../../helpers/translator-budget"; + +const credit = 0.04582331509121062; +const conversationId = "11111111-1111-4111-8111-111111111111"; +const tokens = { uncachedInputTokens: 10, outputTokens: 2, totalTokens: 12 }; + +async function terminal(frames: Array<[string, unknown]>) { + const response = new Response(new ReadableStream({ + start(controller) { + for (const [type, payload] of frames) { + controller.enqueue(encodeMessage({ ":message-type": "event", ":event-type": type }, + new TextEncoder().encode(JSON.stringify(payload)))); + } + controller.close(); + }, + })); + const events = await Array.fromAsync(parseKiroStream(response, createTestTranslatorBudget())); + const result = events.at(-1); + if (!result || !("usage" in result)) throw new Error("missing terminal usage"); + return result; +} + +test.each([true, false])("credit snapshot survives token metadata (metering first=%s)", async first => { + const metering: [string, unknown] = ["meteringEvent", { unit: "credit", unitPlural: "credits", usage: credit }]; + const metadata: [string, unknown] = ["metadataEvent", { tokenUsage: tokens }]; + const result = await terminal([ + ["initial-response", { conversationId }], + ["assistantResponseEvent", { content: "ok" }], + ...(first ? [metering, metadata] : [metadata, metering]), + ]); + expect(result).toMatchObject({ type: "done", usage: { providerCredits: credit, inputTokens: 10, outputTokens: 2 }, + providerState: { kiro: { conversationId } } }); + const persisted = normalizeUsageEntryForTest({ requestId: "metering", timestamp: 1, provider: "kiro", + model: "test", status: 200, durationMs: 1, usageStatus: "reported", usage: result.usage }); + expect(persisted.usage?.providerCredits).toBe(credit); +}); + +test("repeated per-request readings replace the prior snapshot, including zero", async () => { + const result = await terminal([ + ["assistantResponseEvent", { content: "ok" }], + ["meteringEvent", { unit: "credit", usage: credit }], + ["meteringEvent", { unit: "credits", amount: 0 }], + ]); + expect(result.usage).toMatchObject({ providerCredits: 0, estimated: true }); +}); + +test("unreported credits and other units do not become measured zero", async () => { + for (const extra of [[], [["meteringEvent", { unit: "token", usage: 10 }]] as Array<[string, unknown]>]) { + const result = await terminal([["assistantResponseEvent", { content: "ok" }], ...extra]); + expect(result.usage).not.toHaveProperty("providerCredits"); + } +}); + +test("a stream error preserves credits already reported", async () => { + const result = await terminal([ + ["meteringEvent", { unit: "credit", usage: credit }], + ["error", { message: "upstream failed" }], + ]); + expect(result).toMatchObject({ type: "error", usage: { providerCredits: credit } }); +}); diff --git a/tests/responses/empty-completion-guard.test.ts b/tests/responses/empty-completion-guard.test.ts index b1985f31b0e..227b5c4936b 100644 --- a/tests/responses/empty-completion-guard.test.ts +++ b/tests/responses/empty-completion-guard.test.ts @@ -125,13 +125,13 @@ describe("empty-completion guard retry", () => { const events = await collect(guardEmptyCompletionEventStream({ firstEvents: eventsOf( { type: "thinking_delta", thinking: "..." }, - { type: "done", usage: { inputTokens: 100, outputTokens: 0, cachedInputTokens: 40 } }, + { type: "done", usage: { inputTokens: 100, outputTokens: 0, cachedInputTokens: 40, providerCredits: 0.04 } }, ), continuation: () => eventsOf( { type: "tool_call_start", id: "c1", name: "run" }, { type: "tool_call_delta", arguments: "{}" }, { type: "tool_call_end" }, - { type: "done", usage: { inputTokens: 200, outputTokens: 30, reasoningOutputTokens: 12 } }, + { type: "done", usage: { inputTokens: 200, outputTokens: 30, reasoningOutputTokens: 12, providerCredits: 0.01 } }, ), })); @@ -142,6 +142,7 @@ describe("empty-completion guard retry", () => { totalTokens: 330, cachedInputTokens: 40, reasoningOutputTokens: 12, + providerCredits: 0.05, }); }); diff --git a/tests/server/server-kiro-completion-e2e.test.ts b/tests/server/server-kiro-completion-e2e.test.ts index 01327b482c8..4d35b2100d3 100644 --- a/tests/server/server-kiro-completion-e2e.test.ts +++ b/tests/server/server-kiro-completion-e2e.test.ts @@ -7,6 +7,7 @@ import { saveConfig } from "../../src/config"; import { encodeMessage } from "../../src/lib/eventstream-decoder"; import { startServer } from "../../src/server"; import { clearRequestLogsForTests, getRequestLogEntries } from "../../src/server/request-log"; +import { readUsageEntries, resetUsageReadCacheForTests } from "../../src/usage/log"; import type { OcxConfig } from "../../src/types"; import { installIsolatedCodexHome, type IsolatedCodexHome } from "../helpers/isolated-codex-home"; import { removeTreeWithRetry } from "../helpers/remove-tree"; @@ -132,8 +133,8 @@ function anthropicEvents(sse: string): Array<{ name: string; data: Record { test("/v1/responses keeps progress nonterminal and lets only the bounded fallback complete", async () => { const upstream = scriptedKiroUpstream([ - [textFrame("Checking the workspace.")], - completionFrames("The workspace is ready."), + [textFrame("Checking the workspace."), eventFrame("meteringEvent", { unit: "credit", usage: 0.04582331509121062 })], + [...completionFrames("The workspace is ready."), eventFrame("meteringEvent", { unit: "credit", amount: 0.01 })], ]); saveConfig(kiroConfig(upstream.server.url.toString())); const proxy = startServer(0); @@ -164,6 +165,13 @@ describe("Kiro completion through public server endpoints", () => { expect(messages.map((item: { phase?: string }) => item.phase)).toEqual(["commentary", "final_answer"]); expect(wire).not.toContain(KIRO_COMPLETION_TOOL_NAME); + const expectedCredits = 0.04582331509121062 + 0.01; + const log = getRequestLogEntries().find(entry => entry.provider === "kiro-test"); + expect(log?.usage).toMatchObject({ providerCredits: expectedCredits, estimated: true }); + resetUsageReadCacheForTests(); + const persisted = readUsageEntries().find(entry => entry.requestId === log?.requestId); + expect(persisted?.usage).toMatchObject({ providerCredits: expectedCredits, estimated: true }); + expect(upstream.requests).toHaveLength(2); expect(kiroToolNames(upstream.requests[0])).toEqual(["bash", KIRO_COMPLETION_TOOL_NAME]); expect(kiroToolNames(upstream.requests[1])).toEqual(["bash", KIRO_COMPLETION_TOOL_NAME]); diff --git a/tests/server/terminal-guard.test.ts b/tests/server/terminal-guard.test.ts index 61234f80a29..2370a4e6cec 100644 --- a/tests/server/terminal-guard.test.ts +++ b/tests/server/terminal-guard.test.ts @@ -169,7 +169,7 @@ describe("terminal guard", () => { parsed: parsed("请检查这个问题并修复代码"), firstEvents: (async function* () { yield { type: "text_delta", text: "我接下来会修改相关文件。" } as AdapterEvent; - yield { type: "done", usage: { inputTokens: 10, outputTokens: 2 } } as AdapterEvent; + yield { type: "done", usage: { inputTokens: 10, outputTokens: 2, providerCredits: 0.04 } } as AdapterEvent; })(), continuation: next => { continuations += 1; @@ -177,7 +177,7 @@ describe("terminal guard", () => { return (async function* () { yield { type: "tool_call_start", id: "call_1", name: "exec_command" } as AdapterEvent; yield { type: "tool_call_end" } as AdapterEvent; - yield { type: "done", usage: { inputTokens: 20, outputTokens: 3 } } as AdapterEvent; + yield { type: "done", usage: { inputTokens: 20, outputTokens: 3, providerCredits: 0.01 } } as AdapterEvent; })(); }, adapterName: "anthropic", @@ -186,7 +186,7 @@ describe("terminal guard", () => { expect(continuations).toBe(1); expect(actual.filter(event => event.type === "done")).toHaveLength(1); expect(actual.some(event => event.type === "assistant_boundary")).toBe(true); - expect(actual.at(-1)).toMatchObject({ usage: { inputTokens: 30, outputTokens: 5, totalTokens: 35 } }); + expect(actual.at(-1)).toMatchObject({ usage: { inputTokens: 30, outputTokens: 5, totalTokens: 35, providerCredits: 0.05 } }); }); diff --git a/tests/usage/key-attribution.test.ts b/tests/usage/key-attribution.test.ts index 9b2c096b674..70dd335b7a0 100644 --- a/tests/usage/key-attribution.test.ts +++ b/tests/usage/key-attribution.test.ts @@ -17,14 +17,17 @@ describe("key attempt accounting", () => { const key = (reference: string) => ({ adapter: "openai-chat" as const, authMode: "key" as const, baseUrl: "https://example.test", _apiKeyAttempt: { reference } }); noteProviderAttemptSend(child, "test", key("synthetic-a"), undefined); - recordKeyAttemptUsage(child, { inputTokens: 100, outputTokens: 10 }); + recordKeyAttemptUsage(child, { inputTokens: 100, outputTokens: 10, providerCredits: 0.04 }); const parent = { ...child }; noteProviderAttemptSend(child, "test", key("synthetic-b"), undefined, "key-429"); - recordKeyAttemptUsage(child, { inputTokens: 200, outputTokens: 20 }); + recordKeyAttemptUsage(child, { inputTokens: 200, outputTokens: 20, providerCredits: 0.01 }); const rows: RequestLogEntry[] = []; addFinalRequestLog("stream-key-switch", Date.now(), parent, 200, undefined, row => rows.push(row)); expect(rows[0].attempts?.map(attempt => attempt.usage?.inputTokens)).toEqual([100, 200]); - expect(rows[0].usage).toMatchObject({ inputTokens: 300, outputTokens: 30 }); + expect(rows[0].usage).toMatchObject({ inputTokens: 300, outputTokens: 30, providerCredits: 0.05 }); + const persisted = normalizeUsageEntryForTest({ ...rows[0], timestamp: 1, durationMs: 1 }); + expect(persisted.usage?.providerCredits).toBe(0.05); + expect(persisted.attempts?.map(attempt => attempt.usage?.providerCredits)).toEqual([0.04, 0.01]); }); test("a reader takes the attempts or the request total, never both", () => { From 3d0b581e3a96b3a7f565ef06fbbe52889d0c9b03 Mon Sep 17 00:00:00 2001 From: RHODIZSECURITY Date: Sun, 27 Sep 2026 02:04:39 +0900 Subject: [PATCH 07/11] fix(status): trust attested live startup health (#5977) Carried from #5977 as one squashed commit. Co-authored-by: RHODIZSECURITY --- src/cli/status.ts | 45 +++++++++-- tests/cli/cli-status-startup-health.test.ts | 86 +++++++++++++++++++++ 2 files changed, 125 insertions(+), 6 deletions(-) create mode 100644 tests/cli/cli-status-startup-health.test.ts diff --git a/src/cli/status.ts b/src/cli/status.ts index 9079a2b5f90..21cc55f302d 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -29,6 +29,8 @@ import { tokenCollidesWithAdmin } from "../lib/admin-secrets"; export { proxyHealthFailureReason, isConnectionRefused, isUncleanExitEvidence, probeUncleanExitState } from "./status-probes"; export type { ListenTarget } from "./status-probes"; import { checkProxyHealth, probeUncleanExitState, type ListenTarget } from "./status-probes"; +import { LOCAL_MANAGEMENT_READ_PATHS } from "../lib/local-management-capability"; +import { fetchBoundLocalManagementRead } from "../server/local-management-read-client"; /** * The state of the data-plane admission secret the SERVICE will use. State only -- never the value. @@ -233,6 +235,34 @@ function statusDashboardUrl(config: StatusListenConfig, hostname: string | undef return `http://${dashboardHostname}:${port}/`; } +const STARTUP_HEALTH_BOOLEAN_FIELDS = [ + "routingInjected", "localRoutingDependency", "autostartEnabled", "rebootSafe", + "serviceInstalled", "serviceViable", "serviceEnabled", "serviceRunning", + "serviceStale", "serviceConflict", "shimInstalled", "shimHealthy", + "serviceSupported", "diagnosticStale", +] as const; + +export async function fetchLiveStartupHealth( + live: NonNullable>>, + deps: Parameters[2] = {}, +): Promise { + const result = await fetchBoundLocalManagementRead( + live, LOCAL_MANAGEMENT_READ_PATHS.startupHealth, { timeoutMs: 1_500, ...deps }, + ); + if (result.kind !== "response" || !result.response.ok) return null; + let payload: unknown; + try { payload = await result.response.json(); } catch { return null; } + if (!payload || typeof payload !== "object" || Array.isArray(payload)) return null; + const row = payload as Record; + if (row.status !== "native" && row.status !== "protected" && row.status !== "at-risk") return null; + if (row.protection !== "service" && row.protection !== "shim" && row.protection !== "none") return null; + if (row.routingKind !== "native" && row.routingKind !== "opencodex-local" + && row.routingKind !== "custom-local" && row.routingKind !== "custom-remote" && row.routingKind !== "unknown") return null; + if (row.shimCoverage !== "full" && row.shimCoverage !== "cli-only" && row.shimCoverage !== "none") return null; + for (const key of STARTUP_HEALTH_BOOLEAN_FIELDS) if (typeof row[key] !== "boolean") return null; + return payload as StartupHealth; +} + /** * The hub block, or null when this machine is not a hub. * @@ -634,16 +664,19 @@ export async function collectStatus(): Promise { hostname: config.hostname, }); const bunRuntime = durableBunRuntime(); + const liveStartup = live ? await fetchLiveStartupHealth(live) : null; const service = diagnoseService(); // A service can be registered and still not serve: the manager reports the job - // either way. `live` was already identity-probed a few lines above, so cross-check - // rather than print registration as if it were service. - const serviceSummary = service.installed && !live - ? `${service.summary} — registered but NOT serving; see ${serviceLogPath()} and re-run 'ocx service repair'` - : service.summary; + // either way. When the identity-probed live proxy provides an attested startup verdict, + // prefer it over a shell-local service-manager probe that lacks the service environment. + const serviceSummary = liveStartup?.protection === "service" && liveStartup.serviceViable + ? `running under the live managed service (logs: ${serviceLogPath()})` + : service.installed && !live + ? `${service.summary} — registered but NOT serving; see ${serviceLogPath()} and re-run 'ocx service repair'` + : service.summary; const codexShim = diagnoseCodexShim(); const codexShimSummary = codexShim.summary; - const startup = collectStartupHealth(config, { + const startup = liveStartup ?? collectStartupHealth(config, { service, shim: codexShim, routingKind: getCodexRoutingKind(), diff --git a/tests/cli/cli-status-startup-health.test.ts b/tests/cli/cli-status-startup-health.test.ts new file mode 100644 index 00000000000..751a85faba3 --- /dev/null +++ b/tests/cli/cli-status-startup-health.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, test } from "bun:test"; +import { fetchLiveStartupHealth } from "../../src/cli/status"; + +const LIVE = { + pid: 4242, + port: 10101, + hostname: "127.0.0.1", + source: "runtime" as const, +}; + +const SECRET = "A".repeat(43); +const NONCE = "B".repeat(43); + +function startupPayload() { + return { + status: "protected", + routingKind: "opencodex-local", + routingInjected: true, + localRoutingDependency: true, + autostartEnabled: true, + rebootSafe: true, + protection: "service", + serviceInstalled: true, + serviceViable: true, + serviceEnabled: true, + serviceRunning: true, + serviceStale: false, + serviceConflict: false, + shimInstalled: true, + shimHealthy: true, + shimCoverage: "cli-only", + serviceSupported: true, + platform: "linux", + diagnosticStale: false, + recommendedCommand: null, + commands: { + installService: "ocx service install", + repairService: "ocx service repair", + installShim: "ocx codex-shim install", + restoreNative: "ocx restore", + }, + }; +} + +function deps(body: unknown) { + return { + readRuntime: () => ({ + pid: LIVE.pid, + port: LIVE.port, + hostname: LIVE.hostname, + attestationSecret: SECRET, + }), + createNonce: () => NONCE, + now: () => 1_000, + fetchImpl: async () => new Response(JSON.stringify(body), { + status: 200, + headers: { "content-type": "application/json" }, + }), + }; +} + +describe("ocx status live startup health", () => { + test("uses an attested live startup verdict when the shell-local service probe would disagree", async () => { + const observed = await fetchLiveStartupHealth(LIVE, deps(startupPayload())); + expect(observed?.status).toBe("protected"); + expect(observed?.rebootSafe).toBe(true); + expect(observed?.serviceViable).toBe(true); + expect(observed?.protection).toBe("service"); + }); + + test("rejects malformed live startup payloads", async () => { + const observed = await fetchLiveStartupHealth(LIVE, deps({ + ...startupPayload(), + serviceRunning: "yes", + })); + expect(observed).toBeNull(); + }); + + test("fails closed when the runtime attestation cannot bind the live PID", async () => { + const observed = await fetchLiveStartupHealth(LIVE, { + ...deps(startupPayload()), + readRuntime: () => null, + }); + expect(observed).toBeNull(); + }); +}); \ No newline at end of file From 0008586b1cd4385ae456ee13cf5f93ca193e11a2 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 02:16:08 +0900 Subject: [PATCH 08/11] fix(status): validate live startup health before trusting it Reject incomplete live verdicts and malformed adoption evidence before the CLI uses them. Register the status regression explicitly and document the read/fallback behavior. Red: four malformed-verdict cases and the layout owner check failed. Green: 25 focused status/layout tests passed; typecheck, structure check and docs build passed. --- .../content/docs/reference/cli/lifecycle.md | 5 +++++ scripts/test-layout/layout.json | 1 + src/cli/status.ts | 21 +++++++++++++++++++ structure/ops/docs-and-release.md | 2 ++ tests/cli/cli-status-startup-health.test.ts | 11 +++++++++- tests/fixtures/test-layout-expected.json | 1 + tests/test-layout-tooling.test.ts | 1 + 7 files changed, 41 insertions(+), 1 deletion(-) diff --git a/docs-site/src/content/docs/reference/cli/lifecycle.md b/docs-site/src/content/docs/reference/cli/lifecycle.md index 21dbee3153a..b5a0a0beb06 100644 --- a/docs-site/src/content/docs/reference/cli/lifecycle.md +++ b/docs-site/src/content/docs/reference/cli/lifecycle.md @@ -164,6 +164,11 @@ default provider, Codex autostart setting, service state, shim state, and the re home. Only the explicit, high-confidence Windows Orca runtime-home signature adds an actionable App-home mismatch warning; it never changes `CODEX_HOME` automatically. +When the running proxy is identity-verified, status reads its startup-safety verdict through a +short-lived local capability. This keeps the service protection result accurate when a shell lacks +the service manager's environment. If that read is unavailable or malformed, status uses the local +diagnostic instead. + Human output also includes an **OAuth health** block after the OAuth logins summary: `OAuth health: ok` when every known account is healthy, or `OAuth health: warning` with one redacted line per non-healthy account (provider, masked account id, status such as reauthentication required, rate or diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 97af5d79fe9..8f42701e226 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -544,6 +544,7 @@ "cli-status-hub-state.test.ts": "cli", "cli-status-json.test.ts": "cli", "cli-status-oauth-health.test.ts": "cli", + "cli-status-startup-health.test.ts": "cli", "cli-stop-json.test.ts": "cli", "cli-storage-inspect.test.ts": "cli", "cli-transport-honesty.test.ts": "cli", diff --git a/src/cli/status.ts b/src/cli/status.ts index 21cc55f302d..00245b6fd60 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -260,6 +260,27 @@ export async function fetchLiveStartupHealth( && row.routingKind !== "custom-local" && row.routingKind !== "custom-remote" && row.routingKind !== "unknown") return null; if (row.shimCoverage !== "full" && row.shimCoverage !== "cli-only" && row.shimCoverage !== "none") return null; for (const key of STARTUP_HEALTH_BOOLEAN_FIELDS) if (typeof row[key] !== "boolean") return null; + if (typeof row.platform !== "string") return null; + if (row.recommendedCommand !== null && typeof row.recommendedCommand !== "string") return null; + if (!row.commands || typeof row.commands !== "object" || Array.isArray(row.commands)) return null; + const commands = row.commands as Record; + for (const key of ["installService", "repairService", "installShim", "restoreNative"] as const) { + if (typeof commands[key] !== "string") return null; + } + if (row.routingAdoption !== undefined) { + if (!row.routingAdoption || typeof row.routingAdoption !== "object" || Array.isArray(row.routingAdoption)) return null; + const adoption = row.routingAdoption as Record; + if (adoption.adoption !== "not-applicable" && adoption.adoption !== "adopted" + && adoption.adoption !== "pending-client-restart" && adoption.adoption !== "unknown") return null; + if (adoption.injectedAtMs !== null && (typeof adoption.injectedAtMs !== "number" || !Number.isFinite(adoption.injectedAtMs))) return null; + if (!Number.isSafeInteger(adoption.observedClients) || (adoption.observedClients as number) < 0) return null; + if (!Array.isArray(adoption.staleClients) || !adoption.staleClients.every((client: unknown) => { + if (!client || typeof client !== "object" || Array.isArray(client)) return false; + const row = client as Record; + return Number.isSafeInteger(row.pid) && (row.pid as number) > 0 + && typeof row.startedAtMs === "number" && Number.isFinite(row.startedAtMs); + })) return null; + } return payload as StartupHealth; } diff --git a/structure/ops/docs-and-release.md b/structure/ops/docs-and-release.md index b2f84c4a57c..d4ee30bbd0d 100644 --- a/structure/ops/docs-and-release.md +++ b/structure/ops/docs-and-release.md @@ -116,6 +116,8 @@ sharing a model-name fragment remain distinct from current Codex-native support. The Remote Hub guide distinguishes selected-runtime readiness from general runtime diagnostics; `tests/cli/cli-connect-readiness.test.ts` exercises that boundary and general status's single discovery pass with isolated executable fixtures. +The CLI status reference follows `src/cli/status.ts`: after identity verification, status prefers the live proxy's `/api/startup-health` verdict through a PID/port-bound local read capability. It validates the returned shape and falls back to local service, shim, and routing diagnostics for unavailable or malformed responses. This avoids a shell-only service-manager environment changing the reported protection; no reusable management credential is copied into the CLI. + The provider guide's OrcaRouter login section in English and all seven translated sources follows the [bounded ingestion contract](../transports/inventory.md#bounded-response-ingestion-and-orcarouter-login): 64 KiB of valid UTF-8 JSON and one 30-second deadline covering headers and body. These are login diff --git a/tests/cli/cli-status-startup-health.test.ts b/tests/cli/cli-status-startup-health.test.ts index 751a85faba3..83f3590a41a 100644 --- a/tests/cli/cli-status-startup-health.test.ts +++ b/tests/cli/cli-status-startup-health.test.ts @@ -76,6 +76,15 @@ describe("ocx status live startup health", () => { expect(observed).toBeNull(); }); + test.each([ + ["missing commands", { commands: undefined }], + ["non-string recovery command", { recommendedCommand: 42 }], + ["missing service log platform", { platform: undefined }], + ["malformed client adoption", { routingAdoption: { adoption: "pending-client-restart" } }], + ])("rejects %s rather than accepting an unusable live verdict", async (_label, changed) => { + expect(await fetchLiveStartupHealth(LIVE, deps({ ...startupPayload(), ...changed }))).toBeNull(); + }); + test("fails closed when the runtime attestation cannot bind the live PID", async () => { const observed = await fetchLiveStartupHealth(LIVE, { ...deps(startupPayload()), @@ -83,4 +92,4 @@ describe("ocx status live startup health", () => { }); expect(observed).toBeNull(); }); -}); \ No newline at end of file +}); diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index c16c5386ed4..bf431d60d22 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -370,6 +370,7 @@ "cli-status-hub-state.test.ts": "cli", "cli-status-json.test.ts": "cli", "cli-status-oauth-health.test.ts": "cli", + "cli-status-startup-health.test.ts": "cli", "cli-stop-json.test.ts": "cli", "cli-storage-inspect.test.ts": "cli", "cli-transport-honesty.test.ts": "cli", diff --git a/tests/test-layout-tooling.test.ts b/tests/test-layout-tooling.test.ts index 792859a3683..22588459bd6 100644 --- a/tests/test-layout-tooling.test.ts +++ b/tests/test-layout-tooling.test.ts @@ -270,6 +270,7 @@ describe("membership oracle", () => { // it" and "the table says which domain owns it". test("the merged regression tests are classified explicitly, not by seed", () => { const owners = { + "cli-status-startup-health.test.ts": "cli", "ci-structure-gate.test.ts": "ci-workflows", "responses-code-mode-patch-compile.test.ts": "responses", "gui-codex-usage-score-parity.test.ts": "gui", From d73277e352f6f89ebb498a4ae7b2d72444fee44e Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 02:17:57 +0900 Subject: [PATCH 09/11] fix(kiro): keep unknown event headers out of debug logs Unknown Smithy event-type headers are upstream controlled. Record only their length in opt-in diagnostics, preserving the unknown-event signal without writing raw header text to logs. Red: the diagnostic regression exposed the raw event type. Green: 7 focused tests, typecheck and structure check passed. --- src/adapters/kiro-events.ts | 3 ++- structure/providers/kiro.md | 2 +- tests/providers/kiro/kiro-metering-events.test.ts | 5 +++-- 3 files changed, 6 insertions(+), 4 deletions(-) diff --git a/src/adapters/kiro-events.ts b/src/adapters/kiro-events.ts index 1ef837d9f62..f49e73e2b5d 100644 --- a/src/adapters/kiro-events.ts +++ b/src/adapters/kiro-events.ts @@ -116,7 +116,8 @@ function parseTokenUsage(eventType: string, value: unknown): OcxUsage | undefine export function parseKiroEvent(eventType: string, payload: Uint8Array): ParsedKiroEvent | null { // Unknown event types are intentionally ignored without parsing or logging their payload. if (!KNOWN_EVENT_TYPES.has(eventType)) { - debugProviderDiagnostic("kiro", "unknown_event", { eventType }); + // The Smithy header is upstream-controlled too; a raw value can contain private data. + debugProviderDiagnostic("kiro", "unknown_event", { eventTypeLength: eventType.length }); return null; } const parsed = parseObject(eventType, payload); diff --git a/structure/providers/kiro.md b/structure/providers/kiro.md index 1745968844e..08e288d7ade 100644 --- a/structure/providers/kiro.md +++ b/structure/providers/kiro.md @@ -138,7 +138,7 @@ Spend arrives in `meteringEvent` as **credits, not tokens**. No captured respons is a snapshot; separate completion-fallback responses add their credits. Missing metering stays absent and measured zero stays zero. `initial-response` carries `conversationId` through the same validated provider-state path as `messageMetadataEvent`. Unknown event types produce opt-in -`debugProviderDiagnostic` entries containing only the event type, never the payload. +`debugProviderDiagnostic` entries containing only the event-type length, never the raw header or payload. Coverage: `tests/providers/kiro/kiro-metering-events.test.ts`, `tests/providers/kiro/kiro-metering-usage.test.ts`, and `tests/server/server-kiro-completion-e2e.test.ts`. diff --git a/tests/providers/kiro/kiro-metering-events.test.ts b/tests/providers/kiro/kiro-metering-events.test.ts index 4a1dfdccfed..1bec4e9d772 100644 --- a/tests/providers/kiro/kiro-metering-events.test.ts +++ b/tests/providers/kiro/kiro-metering-events.test.ts @@ -131,7 +131,7 @@ describe("parseKiroEvent - unknown event diagnostics", () => { resetDebugLogBufferForTests(); }); - test("unknown event type calls debugProviderDiagnostic with { eventType } and does not log/parse payload when debug enabled", () => { + test("unknown event diagnostics omit upstream-controlled header and payload bytes", () => { setDebugSettings({ debug: true }); const error = spyOn(console, "error").mockImplementation(() => {}); @@ -144,7 +144,8 @@ describe("parseKiroEvent - unknown event diagnostics", () => { const line = String(error.mock.calls[0]?.[0] ?? ""); expect(line).toContain("[ocx:kiro:unknown_event]"); - expect(line).toContain('"eventType":"someUnknownFutureEvent"'); + expect(line).toContain('"eventTypeLength":22'); + expect(line).not.toContain("someUnknownFutureEvent"); expect(line).not.toContain("sensitive-payload"); const logEntries = getDebugLogEntries(); From b473efe023d9d3c077a54e1ad9641ecad9449d76 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 03:09:37 +0900 Subject: [PATCH 10/11] Revert "fix(status): validate live startup health before trusting it" This reverts commit 0008586b1cd4385ae456ee13cf5f93ca193e11a2. --- .../content/docs/reference/cli/lifecycle.md | 5 ----- scripts/test-layout/layout.json | 1 - src/cli/status.ts | 21 ------------------- structure/ops/docs-and-release.md | 2 -- tests/cli/cli-status-startup-health.test.ts | 11 +--------- tests/fixtures/test-layout-expected.json | 1 - tests/test-layout-tooling.test.ts | 1 - 7 files changed, 1 insertion(+), 41 deletions(-) diff --git a/docs-site/src/content/docs/reference/cli/lifecycle.md b/docs-site/src/content/docs/reference/cli/lifecycle.md index b5a0a0beb06..21dbee3153a 100644 --- a/docs-site/src/content/docs/reference/cli/lifecycle.md +++ b/docs-site/src/content/docs/reference/cli/lifecycle.md @@ -164,11 +164,6 @@ default provider, Codex autostart setting, service state, shim state, and the re home. Only the explicit, high-confidence Windows Orca runtime-home signature adds an actionable App-home mismatch warning; it never changes `CODEX_HOME` automatically. -When the running proxy is identity-verified, status reads its startup-safety verdict through a -short-lived local capability. This keeps the service protection result accurate when a shell lacks -the service manager's environment. If that read is unavailable or malformed, status uses the local -diagnostic instead. - Human output also includes an **OAuth health** block after the OAuth logins summary: `OAuth health: ok` when every known account is healthy, or `OAuth health: warning` with one redacted line per non-healthy account (provider, masked account id, status such as reauthentication required, rate or diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 8f42701e226..97af5d79fe9 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -544,7 +544,6 @@ "cli-status-hub-state.test.ts": "cli", "cli-status-json.test.ts": "cli", "cli-status-oauth-health.test.ts": "cli", - "cli-status-startup-health.test.ts": "cli", "cli-stop-json.test.ts": "cli", "cli-storage-inspect.test.ts": "cli", "cli-transport-honesty.test.ts": "cli", diff --git a/src/cli/status.ts b/src/cli/status.ts index 00245b6fd60..21cc55f302d 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -260,27 +260,6 @@ export async function fetchLiveStartupHealth( && row.routingKind !== "custom-local" && row.routingKind !== "custom-remote" && row.routingKind !== "unknown") return null; if (row.shimCoverage !== "full" && row.shimCoverage !== "cli-only" && row.shimCoverage !== "none") return null; for (const key of STARTUP_HEALTH_BOOLEAN_FIELDS) if (typeof row[key] !== "boolean") return null; - if (typeof row.platform !== "string") return null; - if (row.recommendedCommand !== null && typeof row.recommendedCommand !== "string") return null; - if (!row.commands || typeof row.commands !== "object" || Array.isArray(row.commands)) return null; - const commands = row.commands as Record; - for (const key of ["installService", "repairService", "installShim", "restoreNative"] as const) { - if (typeof commands[key] !== "string") return null; - } - if (row.routingAdoption !== undefined) { - if (!row.routingAdoption || typeof row.routingAdoption !== "object" || Array.isArray(row.routingAdoption)) return null; - const adoption = row.routingAdoption as Record; - if (adoption.adoption !== "not-applicable" && adoption.adoption !== "adopted" - && adoption.adoption !== "pending-client-restart" && adoption.adoption !== "unknown") return null; - if (adoption.injectedAtMs !== null && (typeof adoption.injectedAtMs !== "number" || !Number.isFinite(adoption.injectedAtMs))) return null; - if (!Number.isSafeInteger(adoption.observedClients) || (adoption.observedClients as number) < 0) return null; - if (!Array.isArray(adoption.staleClients) || !adoption.staleClients.every((client: unknown) => { - if (!client || typeof client !== "object" || Array.isArray(client)) return false; - const row = client as Record; - return Number.isSafeInteger(row.pid) && (row.pid as number) > 0 - && typeof row.startedAtMs === "number" && Number.isFinite(row.startedAtMs); - })) return null; - } return payload as StartupHealth; } diff --git a/structure/ops/docs-and-release.md b/structure/ops/docs-and-release.md index d4ee30bbd0d..b2f84c4a57c 100644 --- a/structure/ops/docs-and-release.md +++ b/structure/ops/docs-and-release.md @@ -116,8 +116,6 @@ sharing a model-name fragment remain distinct from current Codex-native support. The Remote Hub guide distinguishes selected-runtime readiness from general runtime diagnostics; `tests/cli/cli-connect-readiness.test.ts` exercises that boundary and general status's single discovery pass with isolated executable fixtures. -The CLI status reference follows `src/cli/status.ts`: after identity verification, status prefers the live proxy's `/api/startup-health` verdict through a PID/port-bound local read capability. It validates the returned shape and falls back to local service, shim, and routing diagnostics for unavailable or malformed responses. This avoids a shell-only service-manager environment changing the reported protection; no reusable management credential is copied into the CLI. - The provider guide's OrcaRouter login section in English and all seven translated sources follows the [bounded ingestion contract](../transports/inventory.md#bounded-response-ingestion-and-orcarouter-login): 64 KiB of valid UTF-8 JSON and one 30-second deadline covering headers and body. These are login diff --git a/tests/cli/cli-status-startup-health.test.ts b/tests/cli/cli-status-startup-health.test.ts index 83f3590a41a..751a85faba3 100644 --- a/tests/cli/cli-status-startup-health.test.ts +++ b/tests/cli/cli-status-startup-health.test.ts @@ -76,15 +76,6 @@ describe("ocx status live startup health", () => { expect(observed).toBeNull(); }); - test.each([ - ["missing commands", { commands: undefined }], - ["non-string recovery command", { recommendedCommand: 42 }], - ["missing service log platform", { platform: undefined }], - ["malformed client adoption", { routingAdoption: { adoption: "pending-client-restart" } }], - ])("rejects %s rather than accepting an unusable live verdict", async (_label, changed) => { - expect(await fetchLiveStartupHealth(LIVE, deps({ ...startupPayload(), ...changed }))).toBeNull(); - }); - test("fails closed when the runtime attestation cannot bind the live PID", async () => { const observed = await fetchLiveStartupHealth(LIVE, { ...deps(startupPayload()), @@ -92,4 +83,4 @@ describe("ocx status live startup health", () => { }); expect(observed).toBeNull(); }); -}); +}); \ No newline at end of file diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index bf431d60d22..c16c5386ed4 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -370,7 +370,6 @@ "cli-status-hub-state.test.ts": "cli", "cli-status-json.test.ts": "cli", "cli-status-oauth-health.test.ts": "cli", - "cli-status-startup-health.test.ts": "cli", "cli-stop-json.test.ts": "cli", "cli-storage-inspect.test.ts": "cli", "cli-transport-honesty.test.ts": "cli", diff --git a/tests/test-layout-tooling.test.ts b/tests/test-layout-tooling.test.ts index 22588459bd6..792859a3683 100644 --- a/tests/test-layout-tooling.test.ts +++ b/tests/test-layout-tooling.test.ts @@ -270,7 +270,6 @@ describe("membership oracle", () => { // it" and "the table says which domain owns it". test("the merged regression tests are classified explicitly, not by seed", () => { const owners = { - "cli-status-startup-health.test.ts": "cli", "ci-structure-gate.test.ts": "ci-workflows", "responses-code-mode-patch-compile.test.ts": "responses", "gui-codex-usage-score-parity.test.ts": "gui", From 1637ad3e538602215d4fade86e75e2bbbefc4de9 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 03:09:42 +0900 Subject: [PATCH 11/11] Revert "fix(status): trust attested live startup health (#5977)" This reverts commit 3d0b581e3a96b3a7f565ef06fbbe52889d0c9b03. --- src/cli/status.ts | 45 ++--------- tests/cli/cli-status-startup-health.test.ts | 86 --------------------- 2 files changed, 6 insertions(+), 125 deletions(-) delete mode 100644 tests/cli/cli-status-startup-health.test.ts diff --git a/src/cli/status.ts b/src/cli/status.ts index 21cc55f302d..9079a2b5f90 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -29,8 +29,6 @@ import { tokenCollidesWithAdmin } from "../lib/admin-secrets"; export { proxyHealthFailureReason, isConnectionRefused, isUncleanExitEvidence, probeUncleanExitState } from "./status-probes"; export type { ListenTarget } from "./status-probes"; import { checkProxyHealth, probeUncleanExitState, type ListenTarget } from "./status-probes"; -import { LOCAL_MANAGEMENT_READ_PATHS } from "../lib/local-management-capability"; -import { fetchBoundLocalManagementRead } from "../server/local-management-read-client"; /** * The state of the data-plane admission secret the SERVICE will use. State only -- never the value. @@ -235,34 +233,6 @@ function statusDashboardUrl(config: StatusListenConfig, hostname: string | undef return `http://${dashboardHostname}:${port}/`; } -const STARTUP_HEALTH_BOOLEAN_FIELDS = [ - "routingInjected", "localRoutingDependency", "autostartEnabled", "rebootSafe", - "serviceInstalled", "serviceViable", "serviceEnabled", "serviceRunning", - "serviceStale", "serviceConflict", "shimInstalled", "shimHealthy", - "serviceSupported", "diagnosticStale", -] as const; - -export async function fetchLiveStartupHealth( - live: NonNullable>>, - deps: Parameters[2] = {}, -): Promise { - const result = await fetchBoundLocalManagementRead( - live, LOCAL_MANAGEMENT_READ_PATHS.startupHealth, { timeoutMs: 1_500, ...deps }, - ); - if (result.kind !== "response" || !result.response.ok) return null; - let payload: unknown; - try { payload = await result.response.json(); } catch { return null; } - if (!payload || typeof payload !== "object" || Array.isArray(payload)) return null; - const row = payload as Record; - if (row.status !== "native" && row.status !== "protected" && row.status !== "at-risk") return null; - if (row.protection !== "service" && row.protection !== "shim" && row.protection !== "none") return null; - if (row.routingKind !== "native" && row.routingKind !== "opencodex-local" - && row.routingKind !== "custom-local" && row.routingKind !== "custom-remote" && row.routingKind !== "unknown") return null; - if (row.shimCoverage !== "full" && row.shimCoverage !== "cli-only" && row.shimCoverage !== "none") return null; - for (const key of STARTUP_HEALTH_BOOLEAN_FIELDS) if (typeof row[key] !== "boolean") return null; - return payload as StartupHealth; -} - /** * The hub block, or null when this machine is not a hub. * @@ -664,19 +634,16 @@ export async function collectStatus(): Promise { hostname: config.hostname, }); const bunRuntime = durableBunRuntime(); - const liveStartup = live ? await fetchLiveStartupHealth(live) : null; const service = diagnoseService(); // A service can be registered and still not serve: the manager reports the job - // either way. When the identity-probed live proxy provides an attested startup verdict, - // prefer it over a shell-local service-manager probe that lacks the service environment. - const serviceSummary = liveStartup?.protection === "service" && liveStartup.serviceViable - ? `running under the live managed service (logs: ${serviceLogPath()})` - : service.installed && !live - ? `${service.summary} — registered but NOT serving; see ${serviceLogPath()} and re-run 'ocx service repair'` - : service.summary; + // either way. `live` was already identity-probed a few lines above, so cross-check + // rather than print registration as if it were service. + const serviceSummary = service.installed && !live + ? `${service.summary} — registered but NOT serving; see ${serviceLogPath()} and re-run 'ocx service repair'` + : service.summary; const codexShim = diagnoseCodexShim(); const codexShimSummary = codexShim.summary; - const startup = liveStartup ?? collectStartupHealth(config, { + const startup = collectStartupHealth(config, { service, shim: codexShim, routingKind: getCodexRoutingKind(), diff --git a/tests/cli/cli-status-startup-health.test.ts b/tests/cli/cli-status-startup-health.test.ts deleted file mode 100644 index 751a85faba3..00000000000 --- a/tests/cli/cli-status-startup-health.test.ts +++ /dev/null @@ -1,86 +0,0 @@ -import { describe, expect, test } from "bun:test"; -import { fetchLiveStartupHealth } from "../../src/cli/status"; - -const LIVE = { - pid: 4242, - port: 10101, - hostname: "127.0.0.1", - source: "runtime" as const, -}; - -const SECRET = "A".repeat(43); -const NONCE = "B".repeat(43); - -function startupPayload() { - return { - status: "protected", - routingKind: "opencodex-local", - routingInjected: true, - localRoutingDependency: true, - autostartEnabled: true, - rebootSafe: true, - protection: "service", - serviceInstalled: true, - serviceViable: true, - serviceEnabled: true, - serviceRunning: true, - serviceStale: false, - serviceConflict: false, - shimInstalled: true, - shimHealthy: true, - shimCoverage: "cli-only", - serviceSupported: true, - platform: "linux", - diagnosticStale: false, - recommendedCommand: null, - commands: { - installService: "ocx service install", - repairService: "ocx service repair", - installShim: "ocx codex-shim install", - restoreNative: "ocx restore", - }, - }; -} - -function deps(body: unknown) { - return { - readRuntime: () => ({ - pid: LIVE.pid, - port: LIVE.port, - hostname: LIVE.hostname, - attestationSecret: SECRET, - }), - createNonce: () => NONCE, - now: () => 1_000, - fetchImpl: async () => new Response(JSON.stringify(body), { - status: 200, - headers: { "content-type": "application/json" }, - }), - }; -} - -describe("ocx status live startup health", () => { - test("uses an attested live startup verdict when the shell-local service probe would disagree", async () => { - const observed = await fetchLiveStartupHealth(LIVE, deps(startupPayload())); - expect(observed?.status).toBe("protected"); - expect(observed?.rebootSafe).toBe(true); - expect(observed?.serviceViable).toBe(true); - expect(observed?.protection).toBe("service"); - }); - - test("rejects malformed live startup payloads", async () => { - const observed = await fetchLiveStartupHealth(LIVE, deps({ - ...startupPayload(), - serviceRunning: "yes", - })); - expect(observed).toBeNull(); - }); - - test("fails closed when the runtime attestation cannot bind the live PID", async () => { - const observed = await fetchLiveStartupHealth(LIVE, { - ...deps(startupPayload()), - readRuntime: () => null, - }); - expect(observed).toBeNull(); - }); -}); \ No newline at end of file