diff --git a/.github/AGENTS.md b/.github/AGENTS.md index fc363e66f9..352a124825 100644 --- a/.github/AGENTS.md +++ b/.github/AGENTS.md @@ -22,5 +22,5 @@ change requires explicit security review under `MAINTAINERS.md`. - Inspect the complete workflow diff, including event triggers, permissions, conditions, interpolation, and shell behavior. - Run the local commands represented by changed workflow steps where possible. -- Run `bun run prepush` for CI, release, dependency, packaging, or cross-platform workflow changes. +- Follow the root validation policy: run the suite by default; if a full run is too costly, run at least focused regression tests and document the reason and remaining coverage. Required CI checks still apply before merge. - Do not claim the workflow itself passed until GitHub Actions reports success for the exact commit. diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index fabb5a31bc..5198248821 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -14,6 +14,7 @@ /bunfig.toml @lidge-jun @Ingwannu /scripts/release.ts @lidge-jun @Ingwannu /scripts/release-notes.ts @lidge-jun @Ingwannu +/scripts/release-version-sources.ts @lidge-jun @Ingwannu /scripts/prepare-package.ts @lidge-jun @Ingwannu /package.json @lidge-jun @Ingwannu /bun.lock @lidge-jun @Ingwannu diff --git a/.github/ISSUE_TEMPLATE/documentation.yml b/.github/ISSUE_TEMPLATE/documentation.yml index 358e5e889b..a73e4f7b26 100644 --- a/.github/ISSUE_TEMPLATE/documentation.yml +++ b/.github/ISSUE_TEMPLATE/documentation.yml @@ -28,7 +28,7 @@ body: attributes: label: Documentation location description: Public documentation URL or repository path. - placeholder: "https://opencodex.me/providers/ or docs/providers.md" + placeholder: "https://opencodex.me/guides/providers/ or docs-site/src/content/docs/guides/providers.md" validations: required: true diff --git a/.github/pr-assets/260901-cap-slot-ko-1280.png b/.github/pr-assets/260901-cap-slot-ko-1280.png deleted file mode 100644 index fb5f2a0487..0000000000 Binary files a/.github/pr-assets/260901-cap-slot-ko-1280.png and /dev/null differ diff --git a/.github/pr-assets/364-prompt-layer-unmapped.png b/.github/pr-assets/364-prompt-layer-unmapped.png deleted file mode 100644 index ff1b5ee15b..0000000000 Binary files a/.github/pr-assets/364-prompt-layer-unmapped.png and /dev/null differ diff --git a/.github/pr-assets/3863-storage-skip-referenced.png b/.github/pr-assets/3863-storage-skip-referenced.png deleted file mode 100644 index 0fcc6370a4..0000000000 Binary files a/.github/pr-assets/3863-storage-skip-referenced.png and /dev/null differ diff --git a/.github/pr-assets/4823-opper-provider-catalog.png b/.github/pr-assets/4823-opper-provider-catalog.png deleted file mode 100644 index 374c7b3924..0000000000 Binary files a/.github/pr-assets/4823-opper-provider-catalog.png and /dev/null differ diff --git a/.github/pr-assets/4823-opper-provider-note.png b/.github/pr-assets/4823-opper-provider-note.png deleted file mode 100644 index a37239de6e..0000000000 Binary files a/.github/pr-assets/4823-opper-provider-note.png and /dev/null differ diff --git a/.github/pr-assets/508-grok-coupon-unknown.png b/.github/pr-assets/508-grok-coupon-unknown.png deleted file mode 100644 index 7f56c37fdc..0000000000 Binary files a/.github/pr-assets/508-grok-coupon-unknown.png and /dev/null differ diff --git a/.github/pr-assets/5088-integrations-failed-cold.png b/.github/pr-assets/5088-integrations-failed-cold.png deleted file mode 100644 index f450e897a5..0000000000 Binary files a/.github/pr-assets/5088-integrations-failed-cold.png and /dev/null differ diff --git a/.github/pr-assets/5089-claude-desktop-malformed-status.png b/.github/pr-assets/5089-claude-desktop-malformed-status.png deleted file mode 100644 index 324bf077f7..0000000000 Binary files a/.github/pr-assets/5089-claude-desktop-malformed-status.png and /dev/null differ diff --git a/.github/pr-assets/5197-apply-desktop.png b/.github/pr-assets/5197-apply-desktop.png deleted file mode 100644 index 1a498c0d3e..0000000000 Binary files a/.github/pr-assets/5197-apply-desktop.png and /dev/null differ diff --git a/.github/pr-assets/5197-apply-narrow.png b/.github/pr-assets/5197-apply-narrow.png deleted file mode 100644 index 4538af09a0..0000000000 Binary files a/.github/pr-assets/5197-apply-narrow.png and /dev/null differ diff --git a/.github/pr-assets/5197-capture-receipt.json b/.github/pr-assets/5197-capture-receipt.json deleted file mode 100644 index 89b06a80b1..0000000000 --- a/.github/pr-assets/5197-capture-receipt.json +++ /dev/null @@ -1,180 +0,0 @@ -{ - "status": "HISTORICAL/CORRECTED-GUI-TREE", - "proofType": "fixture-rendered GUI proof from a hosted PR merge-ref build associated with the source head and GUI-tree-equivalent to the reviewed GUI; not a live backend or a literal PR-head build", - "sourceArtifact": { - "name": "PR #5197 hosted merge-ref GUI build associated with source head 07bf0a4dbe", - "artifactId": 10595174777, - "artifactName": "dashboard-preview-6c9576edb372200efec9d6ad4fea7b1e0dc35fb4", - "workflowRun": 35481230775, - "workflowAttempt": 1, - "workflowConclusion": "success", - "entrypoint": "index.html", - "javascript": "assets/index-CKL6ayU1.js", - "stylesheet": "assets/index-BTuCbqQd.css", - "buildCommit": "6c9576edb372200efec9d6ad4fea7b1e0dc35fb4", - "prHead": "07bf0a4dbe369f204c2216f5e8c07c87f52d649c", - "guiTree": "06c1f0c620cbfca2a557813f7ec54b7c11cbb540" - }, - "commands": { - "fixtureServer": "cd && python3 fixture_server.py", - "asideReplTemplate": "/usr/bin/perl -e 'alarm shift; exec @ARGV' 120 aside repl \"\"", - "desktopWindow": "Aside window {1446,762} -> CSS viewport 1280x720", - "narrowWindow": "Aside window {646,842} -> CSS viewport 480x800", - "ko390Window": "Aside native page zoom 125 percent plus window {654,900} -> observed CSS viewport 390x686" - }, - "reproduction": { - "baseUrl": "http://127.0.0.1:18799/", - "harnessScript": "fixture_server.py (scratch-only capture harness; not tracked)", - "driver": "Aside CLI repl opened the base URL in the real browser, selected fixture modes through POST /__fixture/mode, drove the dashboard controls, read back assertions, and captured each frame at the recorded CSS viewport and DPR." - }, - "fixtureRoutes": [ - "GET / and static assets", - "GET /healthz", - "GET /api/startup-health", - "GET /api/client-integrations", - "GET /api/client-integrations/opencode", - "GET /api/client-integrations/journal?client=opencode", - "POST /api/client-integrations/preview", - "PUT /api/client-integrations/opencode", - "POST /api/client-integrations/restore/preview", - "POST /api/client-integrations/restore", - "GET /api/client-integrations/aside/profiles", - "GET /api/client-integrations/aside/profiles/7", - "GET /api/client-integrations/aside/profiles/7/journal", - "POST /api/client-integrations/aside/profiles/7/preview", - "PUT /api/client-integrations/aside/profiles/7", - "POST /__fixture/mode", - "GET /__fixture/state" - ], - "requestSequences": { - "apply": [ - "POST /__fixture/mode {mode:apply}", - "GET /api/client-integrations/opencode", - "GET /api/client-integrations/journal?client=opencode", - "POST /api/client-integrations/preview {clientId:opencode,operation:apply}" - ], - "foreignOverwrite": [ - "POST /__fixture/mode {mode:overwrite}", - "GET /api/client-integrations/opencode -> conflict, reason foreign-edit", - "GET /api/client-integrations/journal?client=opencode", - "POST /api/client-integrations/preview {clientId:opencode,operation:overwrite}" - ], - "restoreDrift": [ - "POST /__fixture/mode {mode:restore}", - "GET /api/client-integrations/opencode", - "GET /api/client-integrations/journal?client=opencode -> op-restore-001", - "POST /api/client-integrations/restore/preview {opId:op-restore-001,confirmDrift:false}" - ], - "stale409Reconfirmation": [ - "POST /__fixture/mode {mode:stale}", - "GET state and journal", - "POST /api/client-integrations/preview -> initial p1:dddd... plan", - "PUT /api/client-integrations/opencode with initial binding -> 409 integration_preview_stale plus p1:eeee... fresh plan", - "GET state and journal reconciliation", - "capture before reconfirming" - ], - "koRestoreDrift390": ["select Korean in dashboard", "run restoreDrift", "capture at observed innerWidth 390"], - "koStaleReconfirm390": ["keep Korean selected", "run stale409Reconfirmation", "capture before reconfirming at innerWidth 390"], - "koProfileDisableNoop390": [ - "POST /__fixture/mode {mode:profile-noop}", - "GET /api/client-integrations/aside/profiles -> profile 7 enabled=true,state=absent", - "POST /api/client-integrations/aside/profiles/7/preview {operation:disable} -> willChange=false,changes=[],profileId=7", - "assert document no-op copy and sync-preference disclosure", - "assert primary Disable button enabled", - "assert consequence body contains neither Korean backup nor rollback text", - "capture", - "PUT /api/client-integrations/aside/profiles/7 {enabled:false,operation:disable,planFingerprint:p1:ffffffffffffffffffffffffffffffff}", - "GET /__fixture/state verifies recorded binding" - ], - "keyboardFocus": [ - "open apply dialog", - "press Tab twice to focus primary Apply", - "capture visible focus ring", - "assert dialog count 1 before Escape", - "press Escape and assert dialog count 0" - ] - }, - "captureBinding": { - "prHead": "07bf0a4dbe369f204c2216f5e8c07c87f52d649c", - "hostedBuildCommit": "6c9576edb372200efec9d6ad4fea7b1e0dc35fb4", - "sourceGuiTree": "06c1f0c620cbfca2a557813f7ec54b7c11cbb540", - "hostedBuildGuiTree": "06c1f0c620cbfca2a557813f7ec54b7c11cbb540", - "guiTreeEqual": true - }, - "interactionObservations": { - "keyboardFocusVisible": true, - "dialogCountBeforeEscape": 1, - "dialogCountAfterEscape": 0 - }, - "profileNoopVerification": { - "previewDto": { - "version": 1, - "clientId": "aside", - "operation": "disable", - "state": "absent", - "foreignEdit": "none", - "changes": [], - "fingerprint": "p1:ffffffffffffffffffffffffffffffff", - "canApply": true, - "willChange": false, - "profileId": 7 - }, - "assertions": { - "documentScopedNoopCopyVisible": true, - "syncPreferenceDisclosureVisible": true, - "primaryButtonEnabled": true, - "backupPromiseAbsent": true, - "rollbackPromiseAbsent": true - }, - "recordedBinding": { - "enabled": false, - "operation": "disable", - "planFingerprint": "p1:ffffffffffffffffffffffffffffffff" - } - }, - "viewports": [ - { - "name": "desktop", - "cssWidth": 1280, - "cssHeight": 720, - "devicePixelRatio": 2, - "pngWidth": 2560, - "pngHeight": 1440, - "files": ["apply-desktop.png","overwrite-foreign-desktop.png","restore-drift-desktop.png","stale-reconfirm-desktop.png","keyboard-focus.png"] - }, - { - "name": "narrow", - "cssWidth": 480, - "cssHeight": 800, - "devicePixelRatio": 2, - "pngWidth": 960, - "pngHeight": 1600, - "files": ["apply-narrow.png","overwrite-foreign-narrow.png","restore-drift-narrow.png","stale-reconfirm-narrow.png"] - }, - { - "name": "ko-390", - "cssWidth": 390, - "cssHeight": 686, - "devicePixelRatio": 2.5, - "browserZoomPercent": 125, - "pngWidth": 976, - "pngHeight": 1716, - "dimensionNote": "Observed CSS viewport and DPR imply a nominal 975x1715 raster; the actual PNG is 976x1716, one physical pixel per axis larger. The capture did not independently isolate the cause of that rounding difference.", - "files": ["ko-restore-drift-390.png","ko-stale-reconfirm-390.png","ko-profile-disable-noop-390.png"] - } - ], - "pngSha256": { - "apply-desktop.png": "4427b60e8885c68e2478c319e4ef428d959b4cc13ee5141a64a92777550ab420", - "apply-narrow.png": "0e2dfd18e76c26755c7b347e149de29dab2b4a40d2074e2ad293d93ced2e34e3", - "keyboard-focus.png": "29714ff85f13b35cc72fffc1a36b6d07ee820be1f9824c9f56a317f15c38feb1", - "ko-profile-disable-noop-390.png": "09baf33f610dccff7d26b77717682f046da821cb2ff8276325be829882f20648", - "ko-restore-drift-390.png": "7c51f352c24defd7cc0c669e53fd254ae314f79faf6cc48a65b9dd2589fe46de", - "ko-stale-reconfirm-390.png": "9dc051b7e97a19fd3dff66958fde432d35f94473ddc403424af13fcdc5ec5254", - "overwrite-foreign-desktop.png": "0e6d14543dfe4ff5847aa3d66d26c499f32235c02a6d741f32a528b1af12f986", - "overwrite-foreign-narrow.png": "8c9af82bc4cfe79e6400c10fb553378524c3f49074145a0cf845db52c3691b63", - "restore-drift-desktop.png": "42127e71f1f13a4902b31e59861de518dc339028a3b3cdbe130f259f0a78f5fb", - "restore-drift-narrow.png": "e7482dfe56b92350641b8a3b13ab3da79e7c8130db74a5cb5208a79e0ee9b2f9", - "stale-reconfirm-desktop.png": "307444c11e8369593b1a023fbeb2ee1ad63bf8deaa7c8bc6dfb9aa7728e6c6c1", - "stale-reconfirm-narrow.png": "8938274b183b888eaf8159bd6083ced670ac2070ef1a5cbdd6424475f4cc1b77" - } -} diff --git a/.github/pr-assets/5197-keyboard-focus.png b/.github/pr-assets/5197-keyboard-focus.png deleted file mode 100644 index 98c77eba84..0000000000 Binary files a/.github/pr-assets/5197-keyboard-focus.png and /dev/null differ diff --git a/.github/pr-assets/5197-ko-profile-disable-noop-390.png b/.github/pr-assets/5197-ko-profile-disable-noop-390.png deleted file mode 100644 index fa5d1bfcb0..0000000000 Binary files a/.github/pr-assets/5197-ko-profile-disable-noop-390.png and /dev/null differ diff --git a/.github/pr-assets/5197-ko-restore-drift-390.png b/.github/pr-assets/5197-ko-restore-drift-390.png deleted file mode 100644 index 3fcc416ae3..0000000000 Binary files a/.github/pr-assets/5197-ko-restore-drift-390.png and /dev/null differ diff --git a/.github/pr-assets/5197-ko-stale-reconfirm-390.png b/.github/pr-assets/5197-ko-stale-reconfirm-390.png deleted file mode 100644 index 4960694a7a..0000000000 Binary files a/.github/pr-assets/5197-ko-stale-reconfirm-390.png and /dev/null differ diff --git a/.github/pr-assets/5197-overwrite-foreign-desktop.png b/.github/pr-assets/5197-overwrite-foreign-desktop.png deleted file mode 100644 index 98ac20eaa8..0000000000 Binary files a/.github/pr-assets/5197-overwrite-foreign-desktop.png and /dev/null differ diff --git a/.github/pr-assets/5197-overwrite-foreign-narrow.png b/.github/pr-assets/5197-overwrite-foreign-narrow.png deleted file mode 100644 index 84fd901591..0000000000 Binary files a/.github/pr-assets/5197-overwrite-foreign-narrow.png and /dev/null differ diff --git a/.github/pr-assets/5197-restore-drift-desktop.png b/.github/pr-assets/5197-restore-drift-desktop.png deleted file mode 100644 index a348c662cd..0000000000 Binary files a/.github/pr-assets/5197-restore-drift-desktop.png and /dev/null differ diff --git a/.github/pr-assets/5197-restore-drift-narrow.png b/.github/pr-assets/5197-restore-drift-narrow.png deleted file mode 100644 index aaa8fb53d3..0000000000 Binary files a/.github/pr-assets/5197-restore-drift-narrow.png and /dev/null differ diff --git a/.github/pr-assets/5197-stale-reconfirm-desktop.png b/.github/pr-assets/5197-stale-reconfirm-desktop.png deleted file mode 100644 index 7f830bda54..0000000000 Binary files a/.github/pr-assets/5197-stale-reconfirm-desktop.png and /dev/null differ diff --git a/.github/pr-assets/5197-stale-reconfirm-narrow.png b/.github/pr-assets/5197-stale-reconfirm-narrow.png deleted file mode 100644 index 1c17ec44df..0000000000 Binary files a/.github/pr-assets/5197-stale-reconfirm-narrow.png and /dev/null differ diff --git a/.github/pr-assets/codex-quota-evidence.md b/.github/pr-assets/codex-quota-evidence.md deleted file mode 100644 index 8244c2441b..0000000000 --- a/.github/pr-assets/codex-quota-evidence.md +++ /dev/null @@ -1,48 +0,0 @@ -# Codex quota registration browser verification - -These captures show the production dashboard bundle served by `startServer`, -using the real management routes, device-login implementation, credential store, -account-pool controller, and refresh button. They are not component fixtures. - -The server used an isolated OpenCodex/Codex home. Only external provider responses -were mocked: device authorization, token exchange, WHAM usage, and the completed -inference stream. The account identity and credentials are synthetic. The empty -native-main home explains the separate Main Account warning in both screenshots. -No live OpenAI account was used or charged. - -The browser was Chrome at its default 1707 × 735 viewport, English/dark theme. -Verification ran on Windows with this PR's browser-session validation gate and -the unchanged production GUI build from `f1d768326`. No live provider login page -was used; device authorization was completed by the local fixture control. - -1. Open Codex Set → Multi-auth, click Add, enter an account ID, and choose Device - code login. Authorize through the mock device service. -2. The actual token exchange and authenticated usage read return a Pro account - with weekly usage at 100%. Registration persists it as validation pending: - one usage read, zero model calls, and no successful-validation timestamp. - The completion notice also says validation is pending; no model-selection - dialog opens for this unroutable account. -3. Reload the page and click Refresh quotas while usage is still 100%. - The account remains pending. Cumulative counts: two usage reads, zero model - calls. The pending screenshot shows the status and the missing selection button. -4. Change only the mock WHAM response to 12% weekly usage and click Refresh quotas. - The server receives a completed validation response. Cumulative counts: - three usage reads, one model call. The pending flag clears, the validation - timestamp is persisted, and “Use this account next” appears. -5. Select the recovered account and confirm the dialog. The stored config reports - `weekly-demo` as the active account. - -Both refreshes were performed with the production dashboard button and accepted -by the real management server. Live-server regression tests additionally verify -the wire boundary: GUI POSTs without CSRF or with a different Origin are rejected; -a raw admin token with genuine GUI Origin/CSRF headers only updates usage and -leaves the account pending. Only the authenticated GUI session completes model -validation. GET quota refreshes remain observational. - -| Capture | Weekly usage | Pending | Model calls so far | -| --- | --- | --- | --- | -| `codex-quota-pending.png` | 100% | Yes | 0 | -| `codex-quota-recovered.png` | 12% | No | 1 | - -This verifies dashboard-to-server behavior against controlled upstream responses. -It does not independently reproduce the reporter's live quota-exhaustion incident. diff --git a/.github/pr-assets/codex-quota-pending.png b/.github/pr-assets/codex-quota-pending.png deleted file mode 100644 index f142433e25..0000000000 Binary files a/.github/pr-assets/codex-quota-pending.png and /dev/null differ diff --git a/.github/pr-assets/codex-quota-recovered.png b/.github/pr-assets/codex-quota-recovered.png deleted file mode 100644 index 4fa88c99a6..0000000000 Binary files a/.github/pr-assets/codex-quota-recovered.png and /dev/null differ diff --git a/.github/pr-assets/fast-rows-setting-toggle.png b/.github/pr-assets/fast-rows-setting-toggle.png deleted file mode 100644 index 756ca0b9c1..0000000000 Binary files a/.github/pr-assets/fast-rows-setting-toggle.png and /dev/null differ diff --git a/.github/pr-assets/muse-spark-meta-search-content-types-400.jpg b/.github/pr-assets/muse-spark-meta-search-content-types-400.jpg deleted file mode 100644 index d18dd98dab..0000000000 Binary files a/.github/pr-assets/muse-spark-meta-search-content-types-400.jpg and /dev/null differ diff --git a/.github/pr-assets/opencodex-cache-usage.png b/.github/pr-assets/opencodex-cache-usage.png deleted file mode 100644 index 6011c89b57..0000000000 Binary files a/.github/pr-assets/opencodex-cache-usage.png and /dev/null differ diff --git a/.github/pr-assets/quota-activation-advanced.png b/.github/pr-assets/quota-activation-advanced.png deleted file mode 100644 index 074bf4c6b4..0000000000 Binary files a/.github/pr-assets/quota-activation-advanced.png and /dev/null differ diff --git a/.github/pr-assets/quota-window-auto-refresh.png b/.github/pr-assets/quota-window-auto-refresh.png deleted file mode 100644 index 3d4eb0cd22..0000000000 Binary files a/.github/pr-assets/quota-window-auto-refresh.png and /dev/null differ diff --git a/.github/pr-assets/xai-responses-optin-switch.png b/.github/pr-assets/xai-responses-optin-switch.png deleted file mode 100644 index 8f0947ace7..0000000000 Binary files a/.github/pr-assets/xai-responses-optin-switch.png and /dev/null differ diff --git a/.github/scripts/enforce-pr-target.test.cjs b/.github/scripts/enforce-pr-target.test.cjs index f975eb39e0..8973f07a63 100644 --- a/.github/scripts/enforce-pr-target.test.cjs +++ b/.github/scripts/enforce-pr-target.test.cjs @@ -284,7 +284,12 @@ describe("enforce-pr-target workflow", () => { assert.match(workflow, /stackedBase/); assert.match(workflow, /github\.rest\.pulls\.list/); assert.match(workflow, /treating as stacked/); - assert.match(workflow, /other\.base\?\.repo\?\.owner/); + assert.match(workflow, /other\.head\?\.repo\?\.owner/); + assert.doesNotMatch( + workflow, + /other\.head\?\.repo\?\.(?:owner\?\.login|name)\s*\?\?/, + "stacked-base detection must fail closed when an open PR head repo is unavailable", + ); const qualityCall = workflow.match( /collectPrQualityFailures\(\{([\s\S]*?)\}\);/, ); diff --git a/.github/scripts/issue-translation.cjs b/.github/scripts/issue-translation.cjs index 41638a9d99..81649c4c5b 100644 --- a/.github/scripts/issue-translation.cjs +++ b/.github/scripts/issue-translation.cjs @@ -810,9 +810,14 @@ function sanitizeTranslationBody(raw, maxChars = 60000) { // read as mention boundaries. Requiring a dotted domain keeps // "end!@octocat"-style mentions defused. \u0001 cannot appear in the // input (control chars were stripped above), so it is a safe sentinel. + // The lookbehind anchors on the @ itself rather than greedily matching + // the local part first: the previous local-part-first pattern rescanned + // long non-email tokens once per start position, which is quadratic on + // model-generated bodies with tens of thousands of consecutive + // local-part characters and no @ at all. .replace( - /[A-Za-z0-9.!#$%&'*+\/=?^_`{|}~-]+@[A-Za-z0-9](?:[A-Za-z0-9-]{0,61}[A-Za-z0-9])?(?:\.[A-Za-z0-9](?:[A-Za-z0-9-]{0,61}[A-Za-z0-9])?)+/g, - (email) => email.replace("@", "\u0001"), + /(?<=[A-Za-z0-9.!#$%&'*+\/=?^_`{|}~-])@[A-Za-z0-9](?:[A-Za-z0-9-]{0,61}[A-Za-z0-9])?(?:\.[A-Za-z0-9](?:[A-Za-z0-9-]{0,61}[A-Za-z0-9])?)+/g, + (emailTail) => emailTail.replace("@", "\u0001"), ) // Defuse pings at Markdown/punctuation boundaries — a colon is a boundary // too — but not emails, npm: scopes, or other mid-token at-signs. diff --git a/.github/scripts/issue-translation.test.cjs b/.github/scripts/issue-translation.test.cjs index 92aea9d16a..bcce113317 100644 --- a/.github/scripts/issue-translation.test.cjs +++ b/.github/scripts/issue-translation.test.cjs @@ -1195,6 +1195,14 @@ describe("bot-owned control state", () => { assert.match(out, /path\/@\u200bhandle/); }); + it("handles long non-email tokens in bounded time", () => { + const input = "a".repeat(60_000); + const startedAt = process.hrtime.bigint(); + assert.equal(sanitizeTranslationBody(input), input); + const elapsedMs = Number(process.hrtime.bigint() - startedAt) / 1_000_000; + assert.ok(elapsedMs < 1_000, `sanitization took ${elapsedMs.toFixed(1)}ms`); + }); + it("ignores forged body-embedded legacy state", () => { const forged = appendTranslationBlock(SOURCE, "English") + `\n|$)/g; */ function strippedText(text) { if (typeof text !== "string") return ""; - return text - .replace(FENCED_CODE_RE, "") + return stripFencedCode(text) .replace(HTML_COMMENT_RE, "") .replace(INLINE_CODE_RE, ""); } diff --git a/.github/scripts/pr-carry-attribution.test.cjs b/.github/scripts/pr-carry-attribution.test.cjs index 08010a18d2..8d94fc707b 100644 --- a/.github/scripts/pr-carry-attribution.test.cjs +++ b/.github/scripts/pr-carry-attribution.test.cjs @@ -2,7 +2,10 @@ const { describe, it } = require("node:test"); const assert = require("node:assert/strict"); -const { assessCarryAttribution } = require("./pr-carry-attribution.cjs"); +const { + assessCarryAttribution, + referencedCarryNumbers, +} = require("./pr-carry-attribution.cjs"); const RRMLIMA = { login: "rrmlima", @@ -120,6 +123,122 @@ describe("assessCarryAttribution", () => { ); }); + it("scans many unclosed fence-like lines without repeatedly searching the tail", () => { + const body = "```x\n".repeat(20_000) + "Reimplements #2797."; + const started = performance.now(); + + assert.deepEqual([...referencedCarryNumbers(body)], [2797]); + assert.ok(performance.now() - started < 2_000, "fence scan should remain linear"); + }); + + it("scans many openers past exhausted close lengths in near-linear time", () => { + // Pure fence lines are also openers, so descending lengths pair up cheaply and + // leave every close-list entry exhausted: the first long opener then walks the + // whole parent chain from 2,002 down to 3, and later openers must stay cheap + // after path compression. Ascending lengths would link each exhausted entry + // straight to an already-dead lower entry and never exercise the walk. + const closes = Array.from({ length: 2_000 }, (_, index) => "`".repeat(2_002 - index)).join("\n"); + const openers = ("`".repeat(2_003) + "x\n").repeat(2_000); + const body = `${closes}\n${openers}Reimplements #2797.`; + const started = performance.now(); + + assert.deepEqual([...referencedCarryNumbers(body)], [2797]); + assert.ok(performance.now() - started < 2_000, "fence scan should remain near-linear"); + }); + + it("strips a complete tilde fence that follows an unmatched backtick opener", () => { + // The unclosed opener stays ordinary text, but it must not swallow the + // independent fenced block after it. + assert.deepEqual( + assessCarryAttribution( + base({ + body: [ + "\u0060\u0060\u0060unclosed", + "~~~", + "Reimplements #2797", + "~~~", + ].join("\n"), + }), + ), + [], + ); + }); + + it("strips a longer fence that follows an unmatched shorter opener", () => { + assert.deepEqual( + assessCarryAttribution( + base({ + body: [ + "\u0060\u0060\u0060unclosed", + "\u0060\u0060\u0060\u0060", + "Reimplements #2797", + "\u0060\u0060\u0060\u0060", + ].join("\n"), + }), + ), + [], + ); + }); + + it("strips a fence whose closing run is shorter than its opening run", () => { + // The backreferenced regex gave back opener delimiters until a close + // matched: a pure ``` line still closes a ```` opener. An exact-length + // lookup would leave "Reimplements #2797" readable as a declaration. + assert.deepEqual( + assessCarryAttribution( + base({ + body: [ + "\u0060\u0060\u0060\u0060", + "Reimplements #2797", + "\u0060\u0060\u0060", + ].join("\n"), + }), + ), + [], + ); + }); + + it("prefers the longest closing run, the way the backreference backtracked", () => { + // Greedy capture tries the full opener run first: a pure ```` line + // farther down outranks a nearer ``` line, so the whole span is removed. + assert.deepEqual( + assessCarryAttribution( + base({ + body: [ + "\u0060\u0060\u0060\u0060", + "\u0060\u0060\u0060", + "Reimplements #2797", + "\u0060\u0060\u0060\u0060", + ].join("\n"), + }), + ), + [], + ); + }); + + it("strips a fenced block written with CRLF line endings", () => { + assert.deepEqual( + assessCarryAttribution( + base({ + body: "\u0060\u0060\u0060\r\nReimplements #2797\r\n\u0060\u0060\u0060\r\n", + }), + ), + [], + ); + }); + + it("still reads carry language around an unmatched opener", () => { + // Falling back to ordinary text is not a license to hide a real claim: + // the unmatched opener line itself remains in the scanned text. + const failures = assessCarryAttribution( + base({ + body: ["\u0060\u0060\u0060unclosed", "Reimplements #2797."].join("\n"), + }), + ); + assert.equal(failures.length, 1); + assert.deepEqual(failures[0].paths, ["#2797"]); + }); + it("ignores carry language after an unclosed HTML comment", () => { // GitHub renders nothing after an unterminated `"; const REVIEW_READINESS_END = ""; +/** + * The latest-dev box, worded as the condition the gate actually enforces. + * + * `readinessClaimViolations` clears this claim while the head is at most + * `READINESS_LATEST_DEV_BEHIND_MAX` commits behind the base — but the box used + * to read "I pushed my PR to the latest dev commit", which asks for the exact + * tip. On a fast-moving `dev` that gap is a treadmill: an author who reads the + * box literally resyncs for unrelated commits, every resync moves the head, + * head-drift resets all four boxes, and the previous exact-head CI evidence is + * invalidated — without reducing merge risk, because the gate was already + * satisfied (#4443). + * + * Deriving the sentence from the constant is the point: the wording and the + * threshold cannot drift apart again, and raising or lowering the tolerance + * rewords the box in the same commit. + * + * Rewording is safe for open pull requests. `extractReviewReadiness` matches on + * box count and checked state, never on item text, and + * `appendReviewReadinessSection` is idempotent — a body that already carries the + * marker pair is returned untouched. Existing checklists keep their wording and + * their ticks; only newly appended ones use this sentence. + */ +function latestDevReadinessItem() { + return ( + "I pushed my PR to a recent dev commit " + + `(at most ${READINESS_LATEST_DEV_BEHIND_MAX} behind; ` + + "a maintainer may still ask for the exact tip before merge)." + ); +} + /** * The four self-attestation boxes a non-maintainer author must tick before the * gate lifts the draft. The final box is intentionally set off by a blank line * so the "ready" claim reads as the closing confirmation, not a fourth task. */ const REVIEW_READINESS_ITEMS = [ - "All CI tests are green on my local testing.", - "I pushed my PR to the latest dev commit.", + "Required local validation passed; commands, results, and any full-suite exception are documented.", + latestDevReadinessItem(), "I resolved all correct Codex and CodeRabbit findings.", "My PR is ready for review.", ]; @@ -374,6 +409,24 @@ function appendReviewReadinessSection(body) { return `${body.trimEnd()}\n\n${section}\n`; } +/** Read only the first label in a structurally valid managed four-box section. */ +function firstReviewReadinessItem(body) { + const readiness = extractReviewReadiness(body); + if (!readiness.present || readiness.total !== REVIEW_READINESS_ITEMS.length) return null; + const start = body.indexOf(REVIEW_READINESS_START) + REVIEW_READINESS_START.length; + const end = body.indexOf(REVIEW_READINESS_END); + return /^[ \t]*[-*][ \t]+\[[ xX]\][ \t]+([^\r\n]*?)[ \t]*\r?$/m + .exec(body.slice(start, end))?.[1] ?? null; +} + +function reviewReadinessMigrationRequired(body) { + return firstReviewReadinessItem(body) === "All CI tests are green on my local testing."; +} + +function reviewReadinessUsesCurrentPolicy(body) { + return firstReviewReadinessItem(body) === REVIEW_READINESS_ITEMS[0]; +} + /** * Remove the bot-managed readiness section from a body. Used so the bot's own * checklist never counts as author-written description substance, and so a @@ -550,6 +603,8 @@ module.exports = { buildReviewReadinessSection, extractReviewReadiness, appendReviewReadinessSection, + reviewReadinessMigrationRequired, + reviewReadinessUsesCurrentPolicy, stripReviewReadinessSection, REVIEW_READINESS_CLAIM_INDEX, uncheckReviewReadinessBoxes, diff --git a/.github/scripts/pr-quality.test.cjs b/.github/scripts/pr-quality.test.cjs index 55c948d65a..996e5b0867 100644 --- a/.github/scripts/pr-quality.test.cjs +++ b/.github/scripts/pr-quality.test.cjs @@ -16,6 +16,8 @@ const { buildReviewReadinessSection, extractReviewReadiness, appendReviewReadinessSection, + reviewReadinessMigrationRequired, + reviewReadinessUsesCurrentPolicy, stripReviewReadinessSection, uncheckReviewReadinessBoxes, REVIEW_READINESS_CLAIM_INDEX, @@ -409,7 +411,7 @@ describe("review readiness checklist", () => { it("treats a reworded but complete section as complete", () => { const reworded = SECTION - .replace("All CI tests are green on my local testing.", "Local suite green.") + .replace("Required local validation passed; commands, results, and any full-suite exception are documented.", "Local suite green.") .replaceAll("- [ ] ", "- [x] "); const result = extractReviewReadiness(reworded); assert.equal(result.present, true); @@ -585,7 +587,7 @@ describe("uncheckReviewReadinessBoxes", () => { "", "## Review readiness checklist", "", - "- [x] All CI tests are green on my local testing.", + "- [x] Required local validation passed; commands, results, and any full-suite exception are documented.", "- [x] I pushed my PR to the latest dev commit.", "- [x] I resolved all correct Codex and CodeRabbit findings.", "- [x] My PR is ready for review.", @@ -596,7 +598,7 @@ describe("uncheckReviewReadinessBoxes", () => { const body = uncheckReviewReadinessBoxes(checkedBody, [ REVIEW_READINESS_CLAIM_INDEX.latest_dev, ]); - assert.ok(body.includes("- [x] All CI tests are green on my local testing.")); + assert.ok(body.includes("- [x] Required local validation passed; commands, results, and any full-suite exception are documented.")); assert.ok(body.includes("- [ ] I pushed my PR to the latest dev commit.")); assert.ok(body.includes("- [x] My PR is ready for review.")); }); @@ -606,7 +608,7 @@ describe("uncheckReviewReadinessBoxes", () => { 0, REVIEW_READINESS_CLAIM_INDEX.latest_dev, ]); - assert.ok(body.includes("- [ ] All CI tests are green on my local testing.")); + assert.ok(body.includes("- [ ] Required local validation passed; commands, results, and any full-suite exception are documented.")); assert.ok(body.includes("- [ ] I pushed my PR to the latest dev commit.")); assert.ok(body.includes("- [x] I resolved all correct Codex and CodeRabbit findings.")); assert.ok(body.includes("- [x] My PR is ready for review.")); @@ -1101,3 +1103,98 @@ describe("comment stripping respects fenced code (regression)", () => { assert.equal(hasScreenshotEvidence(body), false); }); }); + +// #4443: the box used to ask for the exact `dev` tip while the gate cleared the +// claim at up to READINESS_LATEST_DEV_BEHIND_MAX behind. On a fast-moving dev an +// author reading the box literally resyncs for unrelated commits, every resync +// moves the head, head-drift unticks all four boxes, and the exact-head CI +// evidence is thrown away — with no reduction in merge risk, because the gate +// was already satisfied. +describe("the latest-dev readiness box states the condition the gate enforces", () => { + const { + READINESS_LATEST_DEV_BEHIND_MAX, + readinessClaimViolations, + } = require("./pr-quality-state.cjs"); + + const latestDevItem = () => + REVIEW_READINESS_ITEMS[REVIEW_READINESS_CLAIM_INDEX.latest_dev]; + + it("no longer demands the exact tip", () => { + assert.ok(!/latest dev commit/i.test(latestDevItem())); + }); + + it("names the threshold the gate actually uses", () => { + // Derived, not transcribed: the sentence carries the same number + // `readinessClaimViolations` compares against. + assert.ok(latestDevItem().includes(String(READINESS_LATEST_DEV_BEHIND_MAX))); + }); + + it("promises exactly what the gate clears", () => { + // The sentence is only honest if the gate agrees at the boundary. + assert.deepEqual( + readinessClaimViolations({ behindBase: READINESS_LATEST_DEV_BEHIND_MAX }), + [] + ); + assert.deepEqual( + readinessClaimViolations({ behindBase: READINESS_LATEST_DEV_BEHIND_MAX + 1 }), + ["latest_dev"] + ); + }); + + it("still leaves the exact tip available to a maintainer", () => { + assert.match(latestDevItem(), /maintainer/i); + }); + + it("keeps the four-box contract", () => { + assert.equal(REVIEW_READINESS_ITEMS.length, 4); + const section = buildReviewReadinessSection(); + assert.equal((section.match(/^\s*[-*]\s+\[[ xX]\]\s+/gm) || []).length, 4); + }); + + it("does not disturb a checklist that already carries the old wording", () => { + // The compatibility contract: `extractReviewReadiness` reads box count and + // checked state, never item text, and appending is idempotent. An open PR + // keeps its sentence and its ticks. + const legacy = [ + "Body.", + "", + "", + "## Review readiness checklist", + "", + "- [x] All CI tests are green on my local testing.", + "- [x] I pushed my PR to the latest dev commit.", + "- [x] I resolved all correct Codex and CodeRabbit findings.", + "- [x] My PR is ready for review.", + "", + ].join("\n"); + + const readiness = extractReviewReadiness(legacy); + assert.equal(readiness.complete, true); + assert.equal(readiness.total, 4); + assert.equal(appendReviewReadinessSection(legacy), legacy); + }); +}); + +describe("managed checklist wording classification", () => { + const oldItem = "All CI tests are green on my local testing."; + const legacy = buildReviewReadinessSection().replace(REVIEW_READINESS_ITEMS[0], oldItem); + for (const mark of [" ", "x", "X"]) { + for (const ending of ["\n", "\r\n"]) { + it(`recognizes old first item with ${JSON.stringify(mark)} and ${JSON.stringify(ending)}`, () => { + const body = legacy.replace(`- [ ] ${oldItem}`, ` * [${mark}] ${oldItem} `).replaceAll("\n", ending); + assert.equal(reviewReadinessMigrationRequired(body), true); + assert.equal(reviewReadinessUsesCurrentPolicy(body), false); + }); + } + } + it("preserves custom later labels and refuses malformed or displaced first items", () => { + assert.equal(reviewReadinessMigrationRequired(legacy.replace(REVIEW_READINESS_ITEMS[1], "Author's branch attestation.")), true); + for (const body of [null, "", oldItem, legacy + legacy, + legacy.replace("", ""), + legacy.replace(oldItem, oldItem + " Extra"), + legacy.replace(oldItem, "Custom").replace(REVIEW_READINESS_ITEMS[1], oldItem), + legacy.replace(`- [ ] ${REVIEW_READINESS_ITEMS[3]}`, ""), + ]) assert.equal(reviewReadinessMigrationRequired(body), false); + assert.equal(reviewReadinessUsesCurrentPolicy(buildReviewReadinessSection()), true); + }); +}); diff --git a/.github/scripts/pr-readiness-reattest.cjs b/.github/scripts/pr-readiness-reattest.cjs new file mode 100644 index 0000000000..776eaa1ce6 --- /dev/null +++ b/.github/scripts/pr-readiness-reattest.cjs @@ -0,0 +1,276 @@ +"use strict"; + +const { createHash } = require("node:crypto"); + +const SHA40 = /^[0-9a-f]{40}$/i; +const SHA256 = /^[0-9a-f]{64}$/; +const PHASES = new Set(["await-clear", "await-check", "attested"]); +const AWAITING_KEYS = new Set(["version", "headSha", "baseRef", "generation", "phase", "checkpointAt"]); +const ATTESTED_KEYS = new Set([...AWAITING_KEYS, "attestedBodySha256"]); + +/** + * @typedef {{ + * version: 1, + * headSha: string, + * baseRef: string, + * generation: number, + * phase: "await-clear" | "await-check", + * checkpointAt: string | null + * }} AwaitingReattestation + * + * @typedef {{ + * version: 1, + * headSha: string, + * baseRef: string, + * generation: number, + * phase: "attested", + * attestedBodySha256: string, + * checkpointAt: string | null + * }} AttestedReattestation + * + * @typedef {AwaitingReattestation | AttestedReattestation} PendingReattestation + * @typedef {{kind:"absent"} | {kind:"valid", value:PendingReattestation} | {kind:"invalid"}} ParsedPendingReattestation + */ + +/** @param {unknown} value @returns {ParsedPendingReattestation} */ +function parsePendingReattestation(value) { + if (value == null) return { kind: "absent" }; + if (typeof value !== "object" || Array.isArray(value)) return { kind: "invalid" }; + const candidate = /** @type {Record} */ (value); + if ( + candidate.version !== 1 || + typeof candidate.headSha !== "string" || + !SHA40.test(candidate.headSha) || + typeof candidate.baseRef !== "string" || + candidate.baseRef.length === 0 || + !Number.isSafeInteger(candidate.generation) || + candidate.generation <= 0 || + typeof candidate.phase !== "string" || + !PHASES.has(candidate.phase) || + !(candidate.checkpointAt === null || + (typeof candidate.checkpointAt === "string" && isStrictIsoTimestamp(candidate.checkpointAt))) + ) return { kind: "invalid" }; + + if (candidate.phase === "attested") { + if (typeof candidate.attestedBodySha256 !== "string" || !SHA256.test(candidate.attestedBodySha256)) { + return { kind: "invalid" }; + } + } else if (Object.hasOwn(candidate, "attestedBodySha256")) { + return { kind: "invalid" }; + } + + const allowedKeys = candidate.phase === "attested" ? ATTESTED_KEYS : AWAITING_KEYS; + if (Object.keys(candidate).some(key => !allowedKeys.has(key))) return { kind: "invalid" }; + + return { kind: "valid", value: /** @type {PendingReattestation} */ (candidate) }; +} + +/** @param {string} str */ +function bodyDigest(str) { + return createHash("sha256").update(String(str), "utf8").digest("hex"); +} + +function isStrictIsoTimestamp(value) { + if (typeof value !== "string") return false; + if (!/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/.test(value)) return false; + const parsed = Date.parse(value); + if (!Number.isFinite(parsed)) return false; + const canonical = new Date(parsed).toISOString(); + return value.includes(".") ? canonical === value : canonical.replace(".000Z", "Z") === value; +} + +function validLiveIdentity(live) { + return Boolean( + live && + typeof live.headSha === "string" && SHA40.test(live.headSha) && + typeof live.baseRef === "string" && live.baseRef.length > 0 && + typeof live.body === "string" && + Number.isSafeInteger(live.authorId) && live.authorId > 0 + ); +} + +function sameIdentity(pending, live) { + return pending.headSha === live.headSha && pending.baseRef === live.baseRef; +} + +function awaitClear(live, generation) { + return { + version: 1, + headSha: live.headSha, + baseRef: live.baseRef, + generation, + phase: "await-clear", + checkpointAt: null, + }; +} + +function nextGeneration(generation) { + return generation < Number.MAX_SAFE_INTEGER ? generation + 1 : 1; +} + +function samePending(left, right) { + return JSON.stringify(left) === JSON.stringify(right); +} + +function qualifyingAuthorBodyEdit({ live, event, checkpointAt }) { + const checkpointMs = Date.parse(checkpointAt); + const eventMs = Date.parse(event?.updatedAt ?? ""); + // GitHub can advance the live PR timestamp after the author event arrives. + // Do not cap the lag: a delayed event still proves this author's post-checkpoint + // edit when the exact body and head are unchanged at the live read. + const liveMs = Date.parse(live?.updatedAt ?? ""); + return Boolean( + event?.name === "pull_request_target" && + event.action === "edited" && + event.senderType === "User" && + Number.isSafeInteger(event.senderId) && event.senderId === live.authorId && + event.headSha === live.headSha && + typeof event.body === "string" && event.body === live.body && + typeof event.previousBody === "string" && event.previousBody !== event.body && + Number.isFinite(checkpointMs) && Number.isFinite(eventMs) && Number.isFinite(liveMs) && + eventMs > checkpointMs && eventMs <= liveMs + ); +} + +/** + * Advance the durable author re-attestation protocol without writing a PR body. + * A newly seeded/reset episode never consumes the event that caused the reset. + * + * @param {object} input + * @param {unknown} input.pending + * @param {boolean} input.legacy + * @param {boolean} input.current + * @param {{present?:boolean,total?:number,checked?:number,complete?:boolean}} input.readiness + * @param {{headSha:string,baseRef:string,body:string,updatedAt:string,authorId:number}} input.live + * @param {{name?:string,action?:string,senderId?:number,senderType?:string,headSha?:string,body?:string,updatedAt?:string,previousBody?:string}} input.event + * @param {boolean} [input.invalidate] + * @returns {{pending:PendingReattestation|null,canComplete:boolean,changed:boolean,invalidIdentity:boolean}} + */ +function advanceReattestation({ + pending, + legacy, + current, + readiness, + live, + event, + invalidate = false, +}) { + const parsed = parsePendingReattestation(pending); + if (!validLiveIdentity(live)) { + return { + pending: parsed.kind === "valid" ? parsed.value : null, + canComplete: false, + changed: false, + invalidIdentity: true, + }; + } + + const prior = parsed.kind === "valid" ? parsed.value : null; + if (parsed.kind === "invalid") { + return { pending: awaitClear(live, 1), canComplete: false, changed: true, invalidIdentity: false }; + } + + if (prior && !sameIdentity(prior, live)) { + return { + pending: awaitClear(live, nextGeneration(prior.generation)), + canComplete: false, + changed: true, + invalidIdentity: false, + }; + } + + if (legacy || invalidate) { + if (prior?.phase === "await-clear") { + return { pending: prior, canComplete: false, changed: false, invalidIdentity: false }; + } + const next = awaitClear(live, prior ? nextGeneration(prior.generation) : 1); + return { pending: next, canComplete: false, changed: !samePending(prior, next), invalidIdentity: false }; + } + + if (!prior) { + return { pending: null, canComplete: true, changed: false, invalidIdentity: false }; + } + + + // A phase is provisional until the workflow persists it, reads the successful + // comment write's server timestamp, and writes that timestamp into this field. + if (prior.checkpointAt === null) { + return { pending: prior, canComplete: false, changed: false, invalidIdentity: false }; + } + + if (prior.phase === "attested") { + if ( + current && readiness?.present === true && readiness.total === 4 && + readiness.checked === 4 && readiness.complete === true && + prior.attestedBodySha256 === bodyDigest(live.body) + ) { + return { pending: prior, canComplete: true, changed: false, invalidIdentity: false }; + } + const next = awaitClear(live, nextGeneration(prior.generation)); + return { pending: next, canComplete: false, changed: true, invalidIdentity: false }; + } + + if (!current) { + if (prior.phase === "await-clear") { + return { pending: prior, canComplete: false, changed: false, invalidIdentity: false }; + } + return { + pending: awaitClear(live, nextGeneration(prior.generation)), + canComplete: false, + changed: true, + invalidIdentity: false, + }; + } + + if (!qualifyingAuthorBodyEdit({ live, event, checkpointAt: prior.checkpointAt })) { + return { pending: prior, canComplete: false, changed: false, invalidIdentity: false }; + } + + if ( + prior.phase === "await-clear" && current && readiness?.present === true && + readiness.total === 4 && readiness.checked === 0 && readiness.complete === false + ) { + const next = { ...prior, phase: "await-check", checkpointAt: null }; + return { pending: next, canComplete: false, changed: true, invalidIdentity: false }; + } + + if ( + prior.phase === "await-check" && current && readiness?.present === true && + readiness.total === 4 && readiness.checked === 4 && readiness.complete === true + ) { + const next = { + ...prior, + phase: "attested", + attestedBodySha256: bodyDigest(live.body), + checkpointAt: null, + }; + return { pending: next, canComplete: false, changed: true, invalidIdentity: false }; + } + + return { pending: prior, canComplete: false, changed: false, invalidIdentity: false }; +} + +/** + * Whether a saved re-attestation positively authorizes readiness for the live + * PR: a finalized attestation of this exact head, base, and body. A readable + * state that is merely unchanged from an earlier phase is not evidence. + * + * @param {unknown} saved + * @param {{headSha:string,baseRef:string,body:string,authorId:number}} live + */ +function savedAttestationAuthorizes(saved, live) { + const parsed = parsePendingReattestation(saved); + if (parsed.kind !== "valid" || !validLiveIdentity(live)) return false; + const value = parsed.value; + return value.phase === "attested" && + typeof value.checkpointAt === "string" && + sameIdentity(value, live) && + value.attestedBodySha256 === bodyDigest(live.body); +} + +module.exports = { + advanceReattestation, + bodyDigest, + parsePendingReattestation, + savedAttestationAuthorizes, +}; diff --git a/.github/scripts/pr-sponsored-surface.cjs b/.github/scripts/pr-sponsored-surface.cjs index 36b6e400a0..c08c99550e 100644 --- a/.github/scripts/pr-sponsored-surface.cjs +++ b/.github/scripts/pr-sponsored-surface.cjs @@ -30,6 +30,7 @@ const RESTRICTED_FILES = new Set([ // Release and packaging automation executed by the release workflow. "scripts/release.ts", "scripts/release-notes.ts", + "scripts/release-version-sources.ts", "scripts/prepare-package.ts", // Authentication, credential, and secret handling. Mirrors the CODEOWNERS // security boundary. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 098a0e75a2..87eccc7cf3 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -21,10 +21,16 @@ on: # because this workflow is `pull_request` (not `pull_request_target`), # declares `contents: read`, and reads no secrets. # - # `push:` stays pinned to the integration lines: it gates the release path, - # and this trigger already covers review. + # `push:` stays pinned to the release lines: it gates publication, and + # review is covered by the pull_request trigger above. dev is dropped: + # its integration evidence comes from the pull_request run into dev, and + # workflow_dispatch covers anything a maintainer wants proven by hand. + # main and preview MUST stay: release.yml requires a successful + # push-event Cross-platform CI run for the exact release SHA and states + # that a pull-request run does not qualify, so removing either branch + # breaks publication. push: - branches: [main, preview, dev] + branches: [main, preview] paths: - "Dockerfile" - "compose.yaml" @@ -38,24 +44,25 @@ on: - "desktop/**" - "gui/**" - "assets/**" + - ".github/scripts/**" + - ".github/workflows/**" - ".gitattributes" - ".npmignore" - "package.json" - "bun.lock" - "tsconfig.json" - "README.md" + - "readme/**" + - "skills/**" + - ".github/ISSUE_TEMPLATE/**" - "LICENSE" - - ".github/workflows/ci.yml" - - ".github/workflows/release.yml" - - ".github/workflows/enforce-pr-target.yml" - - ".github/workflows/stale-needs-info.yml" workflow_dispatch: inputs: lane: - description: "all (default) or macos-control" + description: "all (default), release-gates, or macos-control" type: choice default: all - options: [all, macos-control] + options: [all, release-gates, macos-control] permissions: contents: read @@ -109,7 +116,7 @@ jobs: # honest pull requests on GitHub-hosted runners and lets trusted branch runs # avoid the hosted-Windows Bun crashes. It is not the security boundary. # - # `push` on dev/main/preview requires the push permission, and + # `push` on main/preview requires the push permission, and # `workflow_dispatch` requires write access, so both carry a trusted author. # A trusted author is not audited code: merging a contributor PR into `dev` # fires `push`, and its dependencies and postinstall hooks then run here. @@ -175,10 +182,21 @@ jobs: # step. A missing or malformed filter output must fail this job instead # of silently making every expensive job skip. ci: ${{ steps.scope.outputs.ci }} + desktop: ${{ steps.scope.outputs.desktop }} + native: ${{ steps.matrices.outputs.native }} + # Matrix include lists for keyring-smoke and npm-global-smoke, built and + # shape-checked by the same validation step as `native`. + keyring_matrix: ${{ steps.matrices.outputs.keyring_matrix }} + npm_global_matrix: ${{ steps.matrices.outputs.npm_global_matrix }} gui: ${{ steps.filter.outputs.gui }} packaging: ${{ steps.filter.outputs.packaging }} docs: ${{ steps.filter.outputs.docs }} structure: ${{ steps.filter.outputs.structure }} + # Narrow scopes for two paths the ci filter leaves out on purpose. Both are re-emitted by the + # validation step below, so a malformed filter output fails this job instead of silently + # skipping the check it selects. + setup_action: ${{ steps.narrow.outputs.setup_action }} + remote_helper: ${{ steps.narrow.outputs.remote_helper }} steps: - name: Checkout uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 @@ -219,19 +237,59 @@ jobs: - 'desktop/**' - 'gui/**' - 'assets/**' + - '.github/scripts/**' + - '.github/workflows/**' - '.gitattributes' - '.npmignore' - 'package.json' - 'bun.lock' - 'tsconfig.json' - 'README.md' + - 'readme/**' + - 'skills/**' + - '.github/ISSUE_TEMPLATE/**' - 'LICENSE' + # Native-gated surface: platform-macos, widget and desktop-shell + # only run when the code they build or bundle could have changed. + # The listed src/ entries are what the bundled sidecar executes, + # package.json and bun.lock change what ships inside the bundle, + # and `.github/workflows/ci.yml` verifies this filter itself, the + # way docs and structure above do. + native: + - 'app/**' + - 'desktop/**' + - 'src/service/**' + - 'src/cli/index.ts' + - 'src/lib/bun-runtime.ts' + - 'src/lib/standalone.ts' + - 'src/lib/keyring-native.ts' + - 'scripts/build-standalone.ts' + - 'scripts/standalone-keyring.ts' + - 'package.json' + - 'bun.lock' - '.github/workflows/ci.yml' - - '.github/workflows/release.yml' - - '.github/workflows/enforce-pr-target.yml' - - '.github/workflows/stale-needs-info.yml' gui: - 'gui/**' + # Building both Linux package formats and booting their real payloads takes + # about 15 minutes, far more than the Rust-only desktop-shell check. On a + # pull request it runs only for inputs that change how the package is + # assembled or launched: the shell itself, the standalone sidecar build + # and its runtime locator, the native keyring staging, and the dependency + # set. Ordinary src/** and gui/** changes no longer select it on a pull + # request; they are still covered on every promotion push to main and + # preview and by workflow_dispatch, which always request this job. + # The workflow names itself so edits to this lane cannot skip their own E2E. + desktop: + - 'desktop/**' + - 'src/lib/standalone.ts' + - 'src/lib/keyring-native.ts' + - 'src/lib/bun-runtime.ts' + - 'scripts/build-standalone.ts' + - 'scripts/standalone-keyring.ts' + - 'scripts/standalone-targets.ts' + - 'package.json' + - 'bun.lock' + - '.github/workflows/ci.yml' # The docs site is built by nothing else on a pull request. `ci` above # deliberately omits `docs-site/**` -- a prose edit has no business # starting the cross-platform suite -- and `deploy-docs.yml` triggers @@ -270,6 +328,19 @@ jobs: structure: - 'structure/**' - '.github/workflows/ci.yml' + # The composite action every Bun job runs. `ci` omits .github/actions/** for the same + # reason it omits docs-site/** and structure/**: a change that touches only the action + # would otherwise start the full matrix. Without this filter it started nothing, and the + # aggregate reported success over skips. The job it feeds runs the action on the three + # runner families and checks what it installed. Pull-request scope, like docs and + # structure; ci.yml is listed so an edit here verifies itself. + setup_action: + - '.github/actions/**' + - '.github/workflows/ci.yml' + # The Rust remote-workspace helper. Nothing else in CI builds it, and its sandbox is + # real only on macOS and Windows, so its job lints and tests the crate on all three. + remote_helper: + - 'native/remote-workspace-helper/**' # Everything that ends up inside `npm pack`, or that decides what # does. `src/**` belongs here because package.json ships `src` and # bin/ocx.mjs executes it: without that entry an ordinary source PR @@ -296,6 +367,7 @@ jobs: shell: bash env: CI_SCOPE: ${{ steps.filter.outputs.ci }} + DESKTOP_SCOPE: ${{ steps.filter.outputs.desktop }} run: | set -euo pipefail case "$CI_SCOPE" in @@ -307,12 +379,90 @@ jobs: exit 1 ;; esac + case "$DESKTOP_SCOPE" in + true|false) + printf 'desktop=%s\n' "$DESKTOP_SCOPE" >> "$GITHUB_OUTPUT" + ;; + *) + printf '::error::changes.outputs.desktop was %q, expected true or false\n' "$DESKTOP_SCOPE" + exit 1 + ;; + esac + + - name: Assert the native and matrix outputs are usable + id: matrices + shell: bash + env: + NATIVE_SELECTED: ${{ steps.filter.outputs.native }} + run: | + set -euo pipefail + case "$NATIVE_SELECTED" in + true|false) + printf 'native=%s\n' "$NATIVE_SELECTED" >> "$GITHUB_OUTPUT" + ;; + *) + printf '::error::changes.outputs.native was %q, expected true or false\n' "$NATIVE_SELECTED" + exit 1 + ;; + esac + + # Two unconditional legs plus the macos leg that rides the native + # selection. These strings are what the keyring-smoke and + # npm-global-smoke matrices consume through fromJSON. + if [ "$NATIVE_SELECTED" = "true" ]; then + keyring_matrix='[{"name":"ubuntu","runner":"ubuntu-latest"},{"name":"windows","runner":"windows-latest"},{"name":"macos","runner":"macos-latest"}]' + npm_global_matrix='[{"os":"ubuntu-latest"},{"os":"windows-latest"},{"os":"macos-latest"}]' + else + keyring_matrix='[{"name":"ubuntu","runner":"ubuntu-latest"},{"name":"windows","runner":"windows-latest"}]' + npm_global_matrix='[{"os":"ubuntu-latest"},{"os":"windows-latest"}]' + fi + + # GitHub turns an empty matrix include list into a job with no legs + # that still reports success, so the emitted value itself is + # validated: it must parse as a JSON array carrying at least the two + # unconditional legs. A malformed or empty matrix fails this job + # instead of passing over nothing. + assert_matrix() { + local label="$1" json="$2" jq_filter="$3" + if ! printf '%s' "$json" | jq -e "$jq_filter" >/dev/null; then + printf '::error::%s matrix output was %q\n' "$label" "$json" + exit 1 + fi + } + assert_matrix keyring "$keyring_matrix" \ + 'type == "array" and length >= 2 and any(.[]; .name == "ubuntu") and any(.[]; .name == "windows")' + assert_matrix npm-global "$npm_global_matrix" \ + 'type == "array" and length >= 2 and any(.[]; .os == "ubuntu-latest") and any(.[]; .os == "windows-latest")' + + printf 'keyring_matrix=%s\n' "$keyring_matrix" >> "$GITHUB_OUTPUT" + printf 'npm_global_matrix=%s\n' "$npm_global_matrix" >> "$GITHUB_OUTPUT" + + - name: Assert the narrow scope outputs are usable + id: narrow + shell: bash + env: + SETUP_ACTION: ${{ steps.filter.outputs.setup_action }} + REMOTE_HELPER: ${{ steps.filter.outputs.remote_helper }} + run: | + set -euo pipefail + for pair in "setup_action=$SETUP_ACTION" "remote_helper=$REMOTE_HELPER"; do + case "${pair#*=}" in + true|false) + printf '%s\n' "$pair" >> "$GITHUB_OUTPUT" + ;; + *) + printf '::error::changes.outputs.%s was %q, expected true or false\n' "${pair%%=*}" "${pair#*=}" + exit 1 + ;; + esac + done # The suite, split by file across four Linux runners. # - # `scripts/ci/run-bun-test-batches.sh` mirrors Bun's sorted round-robin shard - # assignment, then runs each shard in small batches so every batch gets a fresh - # Bun process. The helper prints the exact files before each batch and retries + # `scripts/ci/run-bun-test-batches.sh` assigns files to shards by the per-file + # durations recorded in `scripts/ci/test-durations.tsv` (sorted round-robin when + # nothing is recorded), then runs each shard in small batches so every batch gets + # a fresh Bun process. The helper prints the exact files before each batch and retries # nothing: a test failure, a process timeout and a Bun runtime crash each fail # the shard where they happen. A timeout or a crash is additionally swept one # file per process, after the shard has already failed, to attribute it. @@ -540,13 +690,16 @@ jobs: platform-macos: name: macos ${{ matrix.shard }}/2 needs: changes - if: github.event_name != 'pull_request' || needs.changes.outputs.ci == 'true' + # Native-gated: these legs only run when the changes job's `native` filter + # says the macOS suite's inputs could have changed, so an ordinary source + # pull request stops paying for two macOS runners. + if: github.event_name != 'pull_request' || (needs.changes.outputs.ci == 'true' && needs.changes.outputs.native == 'true') runs-on: macos-latest # Two shards. Unsharded, this job was the critical path on every green dev # push (mean 14.9 min against a 4.7 min Linux maximum; devlog # 260905_test_modularization_and_windows/003). Two halves finish in ~7.7 and - # cost 0.6 extra macOS minutes of setup per run. The whole-pool control that - # the single job used to provide lives in macos-control below, on dispatch. + # cost 0.6 extra macOS minutes of setup per run. The full-membership control + # runs the same bounded batches without sharding on explicit dispatch. timeout-minutes: 20 strategy: fail-fast: false @@ -587,101 +740,21 @@ jobs: cd gui bun run build - # No attempt is ever repeated here. A Bun panic is the interpreter dying mid-suite, - # which is process death a user would have seen; a second execution that happens not - # to die does not un-kill the first, and a leg that reports green on it is reporting - # something that did not happen. This leg retried a crash exactly once until - # 2026-09-17, the Linux batch runner swept crashed batches into green, and the result - # was that Bun 1.4.2's preload segfault stayed invisible on every lane except Windows. - # - # `is_bun_runtime_crash` from scripts/ci/bun-crash-signatures.sh survives, and this leg, - # the Windows leg, the macOS control and the Linux batch runner all still source that one - # definition. Its job is now diagnosis only: it decides which failure message is printed, - # never whether the leg fails. - - name: Test - env: - MACOS_TEST_SHARD: ${{ matrix.shard }} + - name: Setup bounded batch utilities run: | - # GitHub Actions starts bash `run:` blocks with `-e`. Disable - # errexit so a Bun crash reaches PIPESTATUS and the classifier below, - # instead of aborting the step before either can be read. - set +e - set -uo pipefail - # One shared classifier for every lane; see scripts/ci/bun-crash-signatures.sh. - source scripts/ci/bun-crash-signatures.sh - - run_macos_suite() { - local suite_log suite_status - suite_log="$(mktemp -t ocx-macos-suite.XXXXXX)" || return $? - # The per-test ceiling applies to every invocation, including each isolated - # serial file. One attempt, whatever the outcome. - bun test --isolate --timeout 60000 "$@" 2>&1 | tee "$suite_log" - suite_status="${PIPESTATUS[0]}" - if [ "$suite_status" -eq 0 ]; then - rm -f "$suite_log" - return 0 - fi - if is_bun_runtime_crash "$suite_status" "$suite_log"; then - echo "::error::Bun runtime crash in the macOS suite (exit ${suite_status}); a crash is process death, not a test result, and it fails this leg on the first occurrence." - else - echo "::error::macOS suite failed (exit ${suite_status})." - fi - rm -f "$suite_log" - return "$suite_status" - } + brew list coreutils >/dev/null 2>&1 || brew install coreutils + echo "$(brew --prefix coreutils)/libexec/gnubin" >> "$GITHUB_PATH" - case "$MACOS_TEST_SHARD" in - 1|2) ;; - *) echo "::error::Invalid macOS test shard"; exit 64 ;; - esac - serial_manifest="$(bun -e 'import { SERIAL_FULL_SUITE_FILES } from "./scripts/test.ts"; console.log(SERIAL_FULL_SUITE_FILES.join("\n"));')" - manifest_status=$? - if [ "$manifest_status" -ne 0 ]; then - exit "$manifest_status" - fi - serial_files=() - ignore_args=() - serial_count=0 - while IFS= read -r file; do - if [[ ! "$file" =~ ^[[:alnum:]_./-]+$ || "$file" == /* || "/$file/" == *"/../"* || "/$file/" == *"/./"* ]]; then - echo "::error::Invalid serial test path" - exit 1 - fi - for ((index=0; index- + github.event_name == 'workflow_dispatch' && (github.event.inputs.lane == '' || github.event.inputs.lane == 'all' || github.event.inputs.lane == 'macos-control') runs-on: macos-latest - # The unsharded control for the sharded Linux lane: the only place the whole - # suite runs in one pool, so it is the place that catches what sharding - # hides. The flakes it keeps surfacing are timing, not logic, and the fix - # is the tests, not a fourth lane. - # - # Sized to measured work, not to a guess. At 30 this lane never once finished: - # every run was cancelled slightly past halfway and the cancellations were read - # as runner capacity for months (#4905). An authorized one-off measurement let it - # complete for the first time and it took 50m39s wall, Bun reporting 3034.18s over - # 26526 tests in 1343 files. 75 leaves roughly 24 minutes of headroom on that - # number, which is the growth room the suite needs without letting a genuine hang - # sit for an hour before anyone sees it. - # - # This bound is not a fix for anything the run reports. That first complete run - # surfaced four tests that exceed their own timeouts under shared-process pressure, - # tracked separately in #4997; raising this budget is what made them observable and - # must not be mistaken for resolving them. + # Unsharded full-membership control, one worker in sequential fresh batches. + # Long-lived Bun isolate pools repeatedly wedged while synchronously reaping + # child processes. Batching bounds that lifetime without retrying failures or + # excluding tests. This no longer claims one whole-suite process as evidence. timeout-minutes: 75 steps: - name: Checkout @@ -744,37 +807,21 @@ jobs: cd gui bun run build - # This is the lane that exists to see what sharding hides, so it is the last place a - # repeated attempt belongs. One execution, whatever the outcome; the shared classifier - # decides which message is printed, never whether the leg fails. - - name: Test + - name: Setup bounded batch utilities run: | - # GitHub Actions starts bash `run:` blocks with `-e`. Disable - # errexit so a Bun crash reaches PIPESTATUS and the classifier below, - # instead of aborting the step before either can be read. - set +e - set -uo pipefail - # One shared classifier for every lane; see scripts/ci/bun-crash-signatures.sh. - source scripts/ci/bun-crash-signatures.sh - suite_log="$(mktemp -t ocx-macos-suite.XXXXXX)" - # --timeout: Bun's default 5s per-test ceiling is the recurring flake - # class on this loaded shared runner (real retry windows + server - # round-trips exceed 5s under contention; a 10s-floor in-test - # watchdog fired at 10.16s there). 60s keeps hangs bounded (the 30m - # job timeout is the outer backstop) while removing the timing - # flakes — assertions are untouched. Pairs with the 30s CI floor in - # tests/helpers/ci-watchdog.ts. - bun test --isolate --timeout 60000 tests 2>&1 | tee "$suite_log" - suite_status="${PIPESTATUS[0]}" - if [ "$suite_status" -eq 0 ]; then - exit 0 - fi - if is_bun_runtime_crash "$suite_status" "$suite_log"; then - echo "::error::Bun runtime crash in the macOS control suite (exit ${suite_status}); a crash is process death, not a test result, and it fails this leg on the first occurrence." - else - echo "::error::macOS control suite failed (exit ${suite_status})." - fi - exit "$suite_status" + brew list coreutils >/dev/null 2>&1 || brew install coreutils + echo "$(brew --prefix coreutils)/libexec/gnubin" >> "$GITHUB_PATH" + + - name: Test in unsharded fresh-process batches + env: + TEST_SHARD: 1/1 + BUN_TEST_FILE_SCOPE: all + BUN_TEST_BATCH_SIZE: '12' + BUN_TEST_PARALLEL: '1' + BUN_TEST_BATCH_TIMEOUT_SECONDS: '300' + OCX_TEST_NO_QUEUE: '1' + OCX_TEST_FULL_SUITE: '1' + run: bash scripts/ci/run-bun-test-batches.sh "$TEST_SHARD" - name: CLI help smoke run: bun run src/cli/index.ts help @@ -940,13 +987,12 @@ jobs: strategy: fail-fast: false matrix: - include: - - name: ubuntu - runner: ubuntu-latest - - name: windows - runner: windows-latest - - name: macos - runner: macos-latest + # The leg list arrives as JSON from the changes job: ubuntu and + # windows always run, macos only when the native selection is true. + # Entries keep the {name, runner} shape this job reads, and the + # changes job validates the list because an empty include matrix + # reports success over zero legs. + include: ${{ fromJSON(needs.changes.outputs.keyring_matrix) }} steps: - name: Checkout uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 @@ -1080,6 +1126,104 @@ jobs: - name: Structure doc-map, ownership, and invariant bindings run: bun run structure:check + # The composite Bun setup, run on each runner family it serves, when a change touches only the + # action (see the setup_action filter). One step past the action proves it installed the runtime + # package.json declares; nothing else runs, so this never grows into a suite. + setup-action: + name: setup action ${{ matrix.os }} + needs: changes + if: needs.changes.outputs.setup_action == 'true' + runs-on: ${{ matrix.os }} + timeout-minutes: 5 + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest, macos-latest] + steps: + - name: Checkout + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + persist-credentials: false + + - name: Setup project Bun + id: bun + uses: ./.github/actions/setup-project-bun + + - name: Require the runtime package.json declares + shell: bash + env: + RESOLVED: ${{ steps.bun.outputs.version }} + run: | + set -euo pipefail + declared="$(node -p "require('./package.json').dependencies.bun")" + installed="$(bun --version)" + echo "declared=$declared resolved=$RESOLVED installed=$installed" + if [ "$RESOLVED" != "$declared" ] || [ "$installed" != "$declared" ]; then + echo "::error::setup-project-bun resolved '$RESOLVED' and installed '$installed', but package.json declares '$declared'" + exit 1 + fi + + # The Rust remote-workspace helper, when a change touches it (see the remote_helper filter). + # Formatting once, then clippy and the crate's tests on each platform: the sandbox and the live + # confinement tests compile only on macOS and Windows, and Linux covers the protocol and stub. + remote-helper: + name: remote helper ${{ matrix.os }} + needs: changes + if: needs.changes.outputs.remote_helper == 'true' + runs-on: ${{ matrix.os }} + timeout-minutes: 15 + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, macos-latest, windows-latest] + steps: + - name: Checkout + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + persist-credentials: false + + - name: Setup Rust + uses: dtolnay/rust-toolchain@02cb101ec7c40f2c49e1d9714d64511d8e1b74de # master + with: + toolchain: stable + components: rustfmt, clippy + + - name: Check Rust formatting + if: runner.os == 'Linux' + run: cargo fmt --manifest-path native/remote-workspace-helper/Cargo.toml --check + + - name: Run Rust clippy + run: cargo clippy --locked --manifest-path native/remote-workspace-helper/Cargo.toml --all-targets -- -D warnings + + - name: Run Rust tests + run: cargo test --locked --manifest-path native/remote-workspace-helper/Cargo.toml + + # `gates` already runs `privacy:scan` on every event it runs for, so this job + # covers exactly the pull requests `gates` skips -- its condition is the + # complement of `gates`' own, with no path list: every pull request runs the + # scan once, in one job or the other. A push or dispatch always runs `gates`. + # The aggregate below derives the same expectation from `scoped`. + privacy-gate: + name: privacy gate + needs: changes + if: github.event_name == 'pull_request' && needs.changes.outputs.ci != 'true' + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - name: Checkout + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + persist-credentials: false + + - name: Setup project Bun + uses: ./.github/actions/setup-project-bun + + - name: Install dependencies + run: bun install --frozen-lockfile + + - name: Privacy scan + run: bun run privacy:scan + npm-global-smoke: name: npm-global ${{ matrix.os }} needs: changes @@ -1099,7 +1243,12 @@ jobs: # `npm install -g`, which writes into the machine's global prefix and # would leave an `ocx` on a maintainer's personal PATH. It is an # short job on Linux and macOS, so there is nothing to win by moving it. - os: [ubuntu-latest, windows-latest, macos-latest] + # + # The leg list arrives as JSON from the changes job: ubuntu-latest and + # windows-latest always run, macos-latest only when the native + # selection is true. The changes job validates the list because an + # empty include matrix reports success over zero legs. + include: ${{ fromJSON(needs.changes.outputs.npm_global_matrix) }} steps: - name: Checkout uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 @@ -1154,7 +1303,9 @@ jobs: widget: name: macos widget + bundle needs: [changes, gates] - if: github.event_name != 'pull_request' || needs.changes.outputs.ci == 'true' + # Native-gated like platform-macos: the widget extension and app bundle + # are rebuilt only when native-capable paths changed. + if: github.event_name != 'pull_request' || (needs.changes.outputs.ci == 'true' && needs.changes.outputs.native == 'true') runs-on: macos-latest timeout-minutes: 30 steps: @@ -1180,7 +1331,9 @@ jobs: toolchain: stable - name: Test MenuBarCore - run: bun run test:macos + run: | + bun run test:macos + swift run --package-path app NativeTrayTests - name: Build dashboard run: bun run build:gui @@ -1200,22 +1353,38 @@ jobs: # build has no business holding the release key. Updater artifacts are therefore off # here and the signing path stays in release.yml, which already reads the secret and # refuses to publish a manifest when it is absent. - run: bunx tauri build --ci --bundles app --config '{"bundle":{"createUpdaterArtifacts":false}}' + run: bunx tauri build --ci --bundles app --config '{"bundle":{"createUpdaterArtifacts":false,"macOS":{"signingIdentity":"-"}}}' - name: Verify WidgetKit appex and desktop app run: | app=desktop/src-tauri/target/release/bundle/macos/OpenCodex.app - test -x "$app/Contents/MacOS/OpenCodex" + # Tauri renames the main binary only when `mainBinaryName` is set, and this config + # does not set it, so the bundled executable keeps the Cargo bin name rather than + # the product name. Read the name the bundle itself declares instead of restating + # it here, so this check follows the config instead of drifting from it. + executable="$(/usr/libexec/PlistBuddy -c 'Print :CFBundleExecutable' "$app/Contents/Info.plist")" + test -n "$executable" + test -x "$app/Contents/MacOS/$executable" test -x "$app/Contents/PlugIns/OpenCodexWidget.appex/Contents/MacOS/OpenCodexWidget" test -x "$app/Contents/MacOS/ocx" + # Verify actual entitlements and execute the signed Bun sidecar as well as the seal. + bash desktop/scripts/verify-macos-runtime.sh "$app" codesign -dv "$app/Contents/PlugIns/OpenCodexWidget.appex" + # The widget is only offered in the gallery when its bundle is actually linked in, and + # nothing else here would notice its absence: the appex builds, signs and registers + # exactly the same way with the WidgetBundle dropped by the linker. + nm -a "$app/Contents/PlugIns/OpenCodexWidget.appex/Contents/MacOS/OpenCodexWidget" \ + | grep -q "OpenCodexWidget0abC6BundleV" \ + || { echo "::error::the widget bundle is not linked into the extension"; exit 1; } desktop-shell: name: desktop shell needs: [changes, gates] - if: github.event_name != 'pull_request' || needs.changes.outputs.ci == 'true' + # Native shell changes run the Rust checks; package-affecting changes also run the real Linux + # bundle acceptance. The aggregate gate below mirrors this union exactly. + if: github.event_name != 'pull_request' || (needs.changes.outputs.ci == 'true' && (needs.changes.outputs.native == 'true' || needs.changes.outputs.desktop == 'true')) runs-on: ubuntu-latest - timeout-minutes: 20 + timeout-minutes: 45 steps: - name: Checkout uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 @@ -1225,7 +1394,11 @@ jobs: - name: Install Tauri Linux dependencies run: | sudo apt-get update - sudo apt-get install -y libwebkit2gtk-4.1-dev libappindicator3-dev librsvg2-dev patchelf + sudo apt-get install -y libwebkit2gtk-4.1-dev libappindicator3-dev librsvg2-dev patchelf dbus-x11 xvfb xauth wmctrl xdotool openbox + + - name: Setup Bun for packaged E2E + if: needs.changes.outputs.desktop == 'true' + uses: ./.github/actions/setup-project-bun - name: Setup Rust uses: dtolnay/rust-toolchain@02cb101ec7c40f2c49e1d9714d64511d8e1b74de # master @@ -1237,10 +1410,11 @@ jobs: run: | set -euo pipefail triple="$(rustc -vV | sed -n 's/^host: //p')" - mkdir -p desktop/src-tauri/binaries desktop/src-tauri/resources/gui/dist + mkdir -p desktop/src-tauri/binaries desktop/src-tauri/resources/gui/dist desktop/src-tauri/resources/keyring : > "desktop/src-tauri/binaries/ocx-${triple}" chmod +x "desktop/src-tauri/binaries/ocx-${triple}" : > desktop/src-tauri/resources/gui/dist/.keep + : > desktop/src-tauri/resources/keyring/.keep - name: Check Rust formatting run: cargo fmt --manifest-path desktop/src-tauri/Cargo.toml --check @@ -1251,6 +1425,85 @@ jobs: - name: Run Rust tests run: cargo test --manifest-path desktop/src-tauri/Cargo.toml + - name: Install packaged E2E dependencies + if: needs.changes.outputs.desktop == 'true' + run: | + bun install --frozen-lockfile + cd desktop + bun install --frozen-lockfile + + - name: Build dashboard and bundled sidecar + if: needs.changes.outputs.desktop == 'true' + run: | + bun run build:gui + bun desktop/scripts/prepare-sidecar.ts --target x86_64-unknown-linux-gnu + + # Build separately. One format failing must not delete or hide the other + # format's evidence, and neither verification artifact needs an updater key. + - name: Preserve the compiled Linux sidecar + if: needs.changes.outputs.desktop == 'true' + run: chmod +x desktop/scripts/appimage-patchelf.py + + - name: Build Linux AppImage + if: needs.changes.outputs.desktop == 'true' + working-directory: desktop + env: + CARGO_TARGET_DIR: ${{ runner.temp }}/opencodex-appimage-target + PATCHELF: ${{ github.workspace }}/desktop/scripts/appimage-patchelf.py + run: bunx tauri build --ci --bundles appimage --config '{"bundle":{"createUpdaterArtifacts":false}}' + + - name: Build Linux deb + if: needs.changes.outputs.desktop == 'true' + working-directory: desktop + env: + CARGO_TARGET_DIR: ${{ runner.temp }}/opencodex-deb-target + run: bunx tauri build --ci --bundles deb --config '{"bundle":{"createUpdaterArtifacts":false}}' + + - name: Stage isolated Linux bundles + if: needs.changes.outputs.desktop == 'true' + env: + APPIMAGE_BUNDLE: ${{ runner.temp }}/opencodex-appimage-target/release/bundle/appimage + DEB_BUNDLE: ${{ runner.temp }}/opencodex-deb-target/release/bundle/deb + BUNDLE_ROOT: ${{ runner.temp }}/opencodex-linux-bundles + run: | + set -euo pipefail + mkdir -p "$BUNDLE_ROOT/appimage" "$BUNDLE_ROOT/deb" + cp -a "$APPIMAGE_BUNDLE/." "$BUNDLE_ROOT/appimage/" + cp -a "$DEB_BUNDLE/." "$BUNDLE_ROOT/deb/" + chmod -R a-w "$BUNDLE_ROOT" + + - name: Verify packaged Linux sidecar keyring + if: needs.changes.outputs.desktop == 'true' + env: + BUNDLE_ROOT: ${{ runner.temp }}/opencodex-linux-bundles + run: bash desktop/scripts/verify-linux-sidecar.sh "$BUNDLE_ROOT/appimage" + + - name: Run Linux packaged-shell E2E + if: needs.changes.outputs.desktop == 'true' + env: + REPORT_PATH: ${{ runner.temp }}/opencodex-linux-e2e/report.json + run: | + set -euo pipefail + mkdir -p "$(dirname "$REPORT_PATH")" + dbus-run-session -- xvfb-run -a -s '-screen 0 1440x900x24' bash -lc ' + openbox >"$RUNNER_TEMP/opencodex-openbox.log" 2>&1 & + wm_pid=$! + trap '\''kill "$wm_pid" 2>/dev/null || true'\'' EXIT + bun desktop/scripts/linux-packaged-e2e.ts \ + --bundle-root "$RUNNER_TEMP/opencodex-linux-bundles" \ + --report "$REPORT_PATH" \ + --version "$(jq -r .version package.json)" + ' + + - name: Upload Linux packaged-shell E2E report + if: always() && needs.changes.outputs.desktop == 'true' + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: linux-packaged-shell-e2e + path: ${{ runner.temp }}/opencodex-linux-e2e/report.json + if-no-files-found: warn + retention-days: 7 + ci: name: ci if: always() @@ -1258,7 +1511,7 @@ jobs: # direct dependencies only, so a failing `select-windows-runner` would # otherwise reach this gate as nothing at all while its dependents report # `skipped`, which is the shape the step below is written to catch. - needs: [changes, select-windows-runner, test, storage-policy, api-usage, gates, platform-macos, macos-control, platform-windows, keyring-smoke, docker-smoke, docs-site-build, structure-gate, npm-global-smoke, widget, desktop-shell] + needs: [changes, select-windows-runner, test, storage-policy, api-usage, gates, platform-macos, macos-control, platform-windows, keyring-smoke, docker-smoke, docs-site-build, structure-gate, privacy-gate, npm-global-smoke, widget, desktop-shell, setup-action, remote-helper] runs-on: ubuntu-latest timeout-minutes: 5 permissions: @@ -1277,6 +1530,10 @@ jobs: CHANGES_PACKAGING: ${{ needs.changes.outputs.packaging }} CHANGES_DOCS: ${{ needs.changes.outputs.docs }} CHANGES_STRUCTURE: ${{ needs.changes.outputs.structure }} + CHANGES_SETUP_ACTION: ${{ needs.changes.outputs.setup_action }} + CHANGES_REMOTE_HELPER: ${{ needs.changes.outputs.remote_helper }} + CHANGES_NATIVE: ${{ needs.changes.outputs.native }} + CHANGES_DESKTOP: ${{ needs.changes.outputs.desktop }} GH_TOKEN: ${{ github.token }} run: | set -euo pipefail @@ -1297,6 +1554,19 @@ jobs: if [ "$EVENT_NAME" = "pull_request" ] && [ "$CHANGES_CI" != "true" ]; then scoped=not-requested fi + # platform-macos and widget carry the ordinary scope gate AND the native path filter. + # desktop-shell accepts that native set plus the package-E2E set. + # This mirrors that expression exactly; where it disagrees with the + # jobs' own `if:`, the gate fails by name instead of demanding + # success from a job that was deliberately left unselected. + native=not-requested + if [ "$EVENT_NAME" != "pull_request" ] || { [ "$CHANGES_CI" = "true" ] && [ "$CHANGES_NATIVE" = "true" ]; }; then + native=requested + fi + desktop_shell=not-requested + if [ "$EVENT_NAME" != "pull_request" ] || { [ "$CHANGES_CI" = "true" ] && { [ "$CHANGES_NATIVE" = "true" ] || [ "$CHANGES_DESKTOP" = "true" ]; }; }; then + desktop_shell=requested + fi packaging=not-requested if [ "$CHANGES_PACKAGING" = "true" ]; then packaging=requested @@ -1309,11 +1579,30 @@ jobs: if [ "$CHANGES_STRUCTURE" = "true" ]; then structure=requested fi + setup_action=not-requested + if [ "$CHANGES_SETUP_ACTION" = "true" ]; then + setup_action=requested + fi + remote_helper=not-requested + if [ "$CHANGES_REMOTE_HELPER" = "true" ]; then + remote_helper=requested + fi + # privacy-gate runs exactly where `gates` (scoped) does not, on every + # pull request the ci filter declines. Deriving it from `scoped` keeps + # the two scans complementary here as they are in the jobs' own + # conditions. + privacy=not-requested + if [ "$scoped" = not-requested ]; then + privacy=requested + fi dispatch=not-requested windows=not-requested if [ "$EVENT_NAME" = "workflow_dispatch" ]; then - dispatch=requested - # `lane=macos-control` is the one dispatch that deliberately omits Windows. + # Mirror the diagnostic job allowlist; release-gates requests neither suite. + if [ -z "$LANE" ] || [ "$LANE" = "all" ] || [ "$LANE" = "macos-control" ]; then + dispatch=requested + fi + # Only the default/all lane requests the nine Windows suite shards. if [ -z "$LANE" ] || [ "$LANE" = "all" ]; then windows=requested fi @@ -1326,17 +1615,23 @@ jobs: GATED_JOBS="$GATED_JOBS macos-control platform-windows docs-site-build" GATED_JOBS="$GATED_JOBS structure-gate widget" GATED_JOBS="$GATED_JOBS desktop-shell" + GATED_JOBS="$GATED_JOBS setup-action remote-helper" + GATED_JOBS="$GATED_JOBS privacy-gate" expected_for() { case "$1" in changes|select-windows-runner) echo requested ;; - test|storage-policy|api-usage|gates|platform-macos|keyring-smoke|docker-smoke|widget) - echo "$scoped" ;; - desktop-shell) + test|storage-policy|api-usage|gates|keyring-smoke|docker-smoke) echo "$scoped" ;; + platform-macos|widget) + echo "$native" ;; + desktop-shell) echo "$desktop_shell" ;; npm-global-smoke) echo "$packaging" ;; docs-site-build) echo "$docs" ;; structure-gate) echo "$structure" ;; + setup-action) echo "$setup_action" ;; + remote-helper) echo "$remote_helper" ;; + privacy-gate) echo "$privacy" ;; macos-control) echo "$dispatch" ;; platform-windows) echo "$windows" ;; *) echo undeclared ;; diff --git a/.github/workflows/codex-queue-helpers.yml b/.github/workflows/codex-queue-helpers.yml new file mode 100644 index 0000000000..380d27ba93 --- /dev/null +++ b/.github/workflows/codex-queue-helpers.yml @@ -0,0 +1,53 @@ +name: Codex queue helpers + +on: + pull_request: + paths: + - 'scripts/codex-queue.*' + - 'docs-site/astro.config.mjs' + - 'docs-site/src/content/docs/guides/composer-usage-gate-fallback.md' + - '.github/workflows/codex-queue-helpers.yml' + push: + branches: [dev, main, preview] + paths: + - 'scripts/codex-queue.*' + - 'docs-site/astro.config.mjs' + - 'docs-site/src/content/docs/guides/composer-usage-gate-fallback.md' + - '.github/workflows/codex-queue-helpers.yml' + +permissions: + contents: read + +concurrency: + group: codex-queue-helpers-${{ github.ref }} + cancel-in-progress: true + +jobs: + offline-helpers: + name: queue helpers (${{ matrix.os }}) + runs-on: ${{ matrix.os }} + timeout-minutes: 10 + strategy: + fail-fast: false + matrix: + include: + - os: ubuntu-latest + shells: bash + - os: macos-latest + shells: /bin/bash + - os: windows-latest + shells: powershell.exe,pwsh.exe + steps: + - name: Checkout + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + persist-credentials: false + # Hosted runners supply Node 20+; no dependencies, secrets, real Codex, + # accounts, services, downloads or model requests are used by this harness. + - name: Offline native wrapper regressions + shell: bash + env: + CODEX_QUEUE_TEST_SHELLS: ${{ matrix.shells }} + run: | + node --version + node --test scripts/codex-queue.test.mjs diff --git a/.github/workflows/desktop-installed-gate.yml b/.github/workflows/desktop-installed-gate.yml new file mode 100644 index 0000000000..8782cd8462 --- /dev/null +++ b/.github/workflows/desktop-installed-gate.yml @@ -0,0 +1,277 @@ +name: desktop installed-artifact gate + +# D9 part two: install the real artifact on a machine per platform, launch it against a +# staged npm runtime, and exercise the ownership contract — takeover, the gestures that +# must leave the runtime alive, tray Quit draining an in-flight request, and on Linux +# both update paths (R3). Runs only on maintainer-registered self-hosted GUI machines; +# publication wiring into release.yml is a separate change. + +on: + workflow_dispatch: + inputs: + version: + description: Release version whose desktop artifacts the gate installs + required: true + type: string + from-version: + description: Older release used for the staged npm runtime and the Linux update phases + required: true + type: string + # Hook inputs are FILE NAMES, never command text. The runner's operator installs + # audited executables in a hooks directory (vars.OPENCODEX_GATE_HOOKS_DIR) and a + # dispatch picks among them by name; the gate executes the file directly, so this + # workflow can never become an arbitrary-shell surface on a persistent runner. + consent-hook: + description: Name of the runner hook that answers the takeover consent prompt + required: false + type: string + tray-click-hook: + description: Name of the runner hook that left-clicks the tray icon + required: false + type: string + tray-quit-hook: + description: Name of the runner hook that opens the tray menu and chooses Quit + required: false + type: string + tray-check-hook: + description: Name of the runner hook that chooses Check for Updates in the tray + required: false + type: string + tray-install-hook: + description: Name of the runner hook that chooses Install update in the tray + required: false + type: string + elevate-accept-hook: + description: Name of the runner hook that answers the deb update's elevation prompt (drives the accept path) + required: false + type: string + +permissions: + contents: read + +concurrency: + group: desktop-installed-gate-${{ inputs.version }} + cancel-in-progress: false + +jobs: + macos: + runs-on: [self-hosted, opencodex-gate-macos] + timeout-minutes: 60 + # Required-review environment: no run reaches the GUI runner without a maintainer + # approval, and the checkout below pins the driver to the protected dev branch, so a + # dispatched ref cannot smuggle modified gate code onto the machine. + environment: opencodex-desktop-gate + defaults: + run: + shell: bash + steps: + - name: Checkout + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + ref: dev + persist-credentials: false + + - name: Setup project Bun + uses: ./.github/actions/setup-project-bun + + - name: Download the release artifact + env: + GH_TOKEN: ${{ github.token }} + RELEASE_VERSION: ${{ inputs.version }} + GATE_ARTIFACTS: ${{ runner.temp }}/gate-artifacts + run: | + mkdir -p "$GATE_ARTIFACTS" + gh release download "v${RELEASE_VERSION}" \ + --pattern "OpenCodex-${RELEASE_VERSION}-macos.dmg" \ + --dir "$GATE_ARTIFACTS" \ + --clobber + + - name: Run the installed-artifact gate + env: + RELEASE_VERSION: ${{ inputs.version }} + FROM_VERSION: ${{ inputs.from-version }} + CONSENT_HOOK: ${{ inputs.consent-hook }} + TRAY_CLICK_HOOK: ${{ inputs.tray-click-hook }} + TRAY_QUIT_HOOK: ${{ inputs.tray-quit-hook }} + GATE_HOOKS_DIR: ${{ vars.OPENCODEX_GATE_HOOKS_DIR }} + GATE_ARTIFACTS: ${{ runner.temp }}/gate-artifacts + GATE_WORK: ${{ runner.temp }}/installed-gate + GATE_REPORT: ${{ runner.temp }}/installed-gate-report.json + run: | + set -euo pipefail + args=( + --platform macos --format dmg + --artifact "$GATE_ARTIFACTS/OpenCodex-${RELEASE_VERSION}-macos.dmg" + --work-dir "$GATE_WORK" + --to-version "$RELEASE_VERSION" + --from-version "$FROM_VERSION" + --report "$GATE_REPORT" + ) + if [ -n "$GATE_HOOKS_DIR" ]; then args+=(--hooks-dir "$GATE_HOOKS_DIR"); fi + for pair in "consent-hook:CONSENT_HOOK" "tray-click-hook:TRAY_CLICK_HOOK" "tray-quit-hook:TRAY_QUIT_HOOK"; do + name="${pair%%:*}"; env_name="${pair##*:}" + value="${!env_name}" + if [ -n "$value" ]; then args+=("--${name}" "$value"); fi + done + bun desktop/scripts/installed-gate.ts "${args[@]}" + + - name: Upload the gate report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: installed-gate-report-macos + path: ${{ runner.temp }}/installed-gate-report.json + if-no-files-found: error + + windows: + runs-on: [self-hosted, opencodex-gate-windows] + timeout-minutes: 60 + environment: opencodex-desktop-gate + defaults: + run: + shell: bash + steps: + - name: Checkout + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + ref: dev + persist-credentials: false + + - name: Setup project Bun + uses: ./.github/actions/setup-project-bun + + - name: Download the release artifact + env: + GH_TOKEN: ${{ github.token }} + RELEASE_VERSION: ${{ inputs.version }} + GATE_ARTIFACTS: ${{ runner.temp }}/gate-artifacts + run: | + mkdir -p "$GATE_ARTIFACTS" + gh release download "v${RELEASE_VERSION}" \ + --pattern "OpenCodex-${RELEASE_VERSION}-windows-x64.msi" \ + --dir "$GATE_ARTIFACTS" \ + --clobber + + - name: Run the installed-artifact gate + env: + RELEASE_VERSION: ${{ inputs.version }} + FROM_VERSION: ${{ inputs.from-version }} + CONSENT_HOOK: ${{ inputs.consent-hook }} + TRAY_CLICK_HOOK: ${{ inputs.tray-click-hook }} + TRAY_QUIT_HOOK: ${{ inputs.tray-quit-hook }} + GATE_HOOKS_DIR: ${{ vars.OPENCODEX_GATE_HOOKS_DIR }} + GATE_ARTIFACTS: ${{ runner.temp }}/gate-artifacts + GATE_WORK: ${{ runner.temp }}/installed-gate + GATE_REPORT: ${{ runner.temp }}/installed-gate-report.json + run: | + set -euo pipefail + args=( + --platform windows --format msi + --artifact "$GATE_ARTIFACTS/OpenCodex-${RELEASE_VERSION}-windows-x64.msi" + --work-dir "$GATE_WORK" + --to-version "$RELEASE_VERSION" + --from-version "$FROM_VERSION" + --report "$GATE_REPORT" + ) + if [ -n "$GATE_HOOKS_DIR" ]; then args+=(--hooks-dir "$GATE_HOOKS_DIR"); fi + for pair in "consent-hook:CONSENT_HOOK" "tray-click-hook:TRAY_CLICK_HOOK" "tray-quit-hook:TRAY_QUIT_HOOK"; do + name="${pair%%:*}"; env_name="${pair##*:}" + value="${!env_name}" + if [ -n "$value" ]; then args+=("--${name}" "$value"); fi + done + bun desktop/scripts/installed-gate.ts "${args[@]}" + + - name: Upload the gate report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: installed-gate-report-windows + path: ${{ runner.temp }}/installed-gate-report.json + if-no-files-found: error + + linux: + runs-on: [self-hosted, opencodex-gate-linux] + timeout-minutes: 60 + environment: opencodex-desktop-gate + strategy: + fail-fast: false + matrix: + format: [deb, appimage] + include: + - format: deb + suffix: linux-amd64.deb + - format: appimage + suffix: linux-x86_64.AppImage + defaults: + run: + shell: bash + steps: + - name: Checkout + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + ref: dev + persist-credentials: false + + - name: Setup project Bun + uses: ./.github/actions/setup-project-bun + + - name: Download the release artifacts + env: + GH_TOKEN: ${{ github.token }} + RELEASE_VERSION: ${{ inputs.version }} + FROM_VERSION: ${{ inputs.from-version }} + ARTIFACT_SUFFIX: ${{ matrix.suffix }} + GATE_ARTIFACTS: ${{ runner.temp }}/gate-artifacts + run: | + mkdir -p "$GATE_ARTIFACTS" + gh release download "v${RELEASE_VERSION}" \ + --pattern "OpenCodex-${RELEASE_VERSION}-${ARTIFACT_SUFFIX}" \ + --dir "$GATE_ARTIFACTS" \ + --clobber + gh release download "v${FROM_VERSION}" \ + --pattern "OpenCodex-${FROM_VERSION}-${ARTIFACT_SUFFIX}" \ + --dir "$GATE_ARTIFACTS" \ + --clobber + + - name: Run the installed-artifact gate + env: + RELEASE_VERSION: ${{ inputs.version }} + FROM_VERSION: ${{ inputs.from-version }} + GATE_FORMAT: ${{ matrix.format }} + ARTIFACT_SUFFIX: ${{ matrix.suffix }} + CONSENT_HOOK: ${{ inputs.consent-hook }} + TRAY_CLICK_HOOK: ${{ inputs.tray-click-hook }} + TRAY_QUIT_HOOK: ${{ inputs.tray-quit-hook }} + TRAY_CHECK_HOOK: ${{ inputs.tray-check-hook }} + TRAY_INSTALL_HOOK: ${{ inputs.tray-install-hook }} + ELEVATE_ACCEPT_HOOK: ${{ inputs.elevate-accept-hook }} + GATE_HOOKS_DIR: ${{ vars.OPENCODEX_GATE_HOOKS_DIR }} + GATE_ARTIFACTS: ${{ runner.temp }}/gate-artifacts + GATE_WORK: ${{ runner.temp }}/installed-gate + GATE_REPORT: ${{ runner.temp }}/installed-gate-report.json + run: | + set -euo pipefail + args=( + --platform linux --format "$GATE_FORMAT" + --artifact "$GATE_ARTIFACTS/OpenCodex-${RELEASE_VERSION}-${ARTIFACT_SUFFIX}" + --older-artifact "$GATE_ARTIFACTS/OpenCodex-${FROM_VERSION}-${ARTIFACT_SUFFIX}" + --work-dir "$GATE_WORK" + --to-version "$RELEASE_VERSION" + --from-version "$FROM_VERSION" + --report "$GATE_REPORT" + ) + if [ -n "$GATE_HOOKS_DIR" ]; then args+=(--hooks-dir "$GATE_HOOKS_DIR"); fi + for pair in "consent-hook:CONSENT_HOOK" "tray-click-hook:TRAY_CLICK_HOOK" "tray-quit-hook:TRAY_QUIT_HOOK" "tray-check-hook:TRAY_CHECK_HOOK" "tray-install-hook:TRAY_INSTALL_HOOK" "elevate-accept-hook:ELEVATE_ACCEPT_HOOK"; do + name="${pair%%:*}"; env_name="${pair##*:}" + value="${!env_name}" + if [ -n "$value" ]; then args+=("--${name}" "$value"); fi + done + bun desktop/scripts/installed-gate.ts "${args[@]}" + + - name: Upload the gate report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: installed-gate-report-linux-${{ matrix.format }} + path: ${{ runner.temp }}/installed-gate-report.json + if-no-files-found: error diff --git a/.github/workflows/dev-version-bump.yml b/.github/workflows/dev-version-bump.yml index f02c0e89f4..b54a6d76d3 100644 --- a/.github/workflows/dev-version-bump.yml +++ b/.github/workflows/dev-version-bump.yml @@ -55,17 +55,27 @@ jobs: # Open the pull request. pull-requests: write steps: - - name: Checkout dev + # Keep every executable file in the privileged job pinned to the audited + # release revision. The later dev checkout is input data only: none of its + # actions, dependencies, scripts, or tests run with this job's token. + - name: Checkout trusted automation uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 with: - ref: dev + ref: ${{ github.sha }} # Tags are load-bearing, not decoration: the freeness gate below is a bun # test that reads the local tag set, and release-version-line.test.ts # returns EARLY on an empty set. A shallow checkout would make that gate # silently vacuous instead of failing loudly. fetch-depth: 0 - # Do NOT set persist-credentials: false here as the read-only workflows do. - # This job has to push its bump branch. + persist-credentials: false + + - name: Checkout dev as data + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + ref: dev + path: dev-tree + fetch-depth: 0 + persist-credentials: false # The repository-owned composite action, not a hand-pinned setup-bun SHA: it # resolves the Bun version from package.json so the runtime SOT stays in one @@ -108,7 +118,7 @@ jobs: RELEASED_VERSION: ${{ steps.target.outputs.version }} run: | set -euo pipefail - bun scripts/bump-dev-version.ts "${RELEASED_VERSION}" package.json + bun scripts/bump-dev-version.ts "${RELEASED_VERSION}" dev-tree/package.json - name: Prove the intended version is not already released if: ${{ steps.target.outputs.mode == 'pre-move' }} @@ -128,15 +138,34 @@ jobs: - name: Prove the chosen version is unused if: ${{ steps.decide.outputs.changed == 'true' }} + env: + NEXT_VERSION: ${{ steps.decide.outputs.version }} # The script decides the candidate from the target version SHAPE, which is all # a pure function can see. Whether that candidate is actually FREE is a property # of the tag set, so it is settled here by the detector that already owns the # question. If this fails, no pull request is opened and the job goes red asking # for a human decision - which is the correct outcome, not a fallback. - run: bun test tests/ci-workflows/release-version-line.test.ts + # + # The release-commit exception does not apply here. release-version-line.test.ts + # lets an in-tree version equal the highest tag when that tag names HEAD, which + # is correct on the release commit itself. In this job HEAD is the workflow's own + # main checkout while only package.json came from dev, so a chosen version that + # already carries a v tag would pass the shared detector and open a pull request + # claiming a published version. Refuse that tag explicitly first. + run: | + set -euo pipefail + git fetch --force --tags origin + if git rev-parse -q --verify "refs/tags/v${NEXT_VERSION#v}" >/dev/null; then + echo "::error::v${NEXT_VERSION#v} already exists; the chosen version needs a human decision" + exit 1 + fi + # Exercise the trusted detector against the candidate package metadata. + cp dev-tree/package.json package.json + bun test tests/ci-workflows/release-version-line.test.ts - name: Open the bump pull request if: ${{ steps.decide.outputs.changed == 'true' }} + working-directory: dev-tree env: GH_TOKEN: ${{ github.token }} MODE: ${{ steps.target.outputs.mode }} @@ -188,27 +217,58 @@ jobs: echo "::notice::${branch} exists without an open pull request; validating it" git fetch origin "${branch}" - # Fail closed on unexpected content. The branch carries the bot's own one-line - # bump, so anything else on it means a human or another job is using that name and - # this job must not push to it or open a pull request from it. - changed_files="$(git diff --name-only "origin/dev...origin/${branch}")" - if [ "${changed_files}" != "package.json" ]; then - echo "::error::${branch} touches unexpected files: ${changed_files:-}" - exit 1 - fi - branch_version="$(git show "origin/${branch}:package.json" | node -p "JSON.parse(require('fs').readFileSync(0,'utf8')).version")" - if [ "${branch_version}" != "${NEXT_VERSION}" ]; then - echo "::error::${branch} carries ${branch_version}, expected ${NEXT_VERSION}" + # Fail closed on unexpected content. The branch carries the bot's own version move, + # which touches only the four version sources that scripts/release-version-sources.ts + # owns, and always package.json. Anything else on it means a human or another job is + # using that name and this job must not push to it or open a pull request from it. + # --no-renames lists a rename as its deletion plus its addition, so a file renamed onto + # an allowed name still shows the path it removed. The case arms are literal paths. + changed_files="$(git diff --no-renames --name-only "origin/dev...origin/${branch}")" + touches_package_json=false + while IFS= read -r changed_file; do + case "${changed_file}" in + package.json) touches_package_json=true ;; + desktop/src-tauri/tauri.conf.json|desktop/src-tauri/Cargo.toml|desktop/src-tauri/Cargo.lock) ;; + *) + echo "::error::${branch} touches unexpected files: ${changed_files:-}" + exit 1 + ;; + esac + done <<< "${changed_files}" + if [ "${touches_package_json}" != "true" ]; then + echo "::error::${branch} does not move package.json: ${changed_files:-}" exit 1 fi + # The decide step already rewrote the version sources in this working tree. Put them + # back first: git refuses to switch over local edits, and the check below must read + # what the branch commits, not what this run wrote. + git checkout -- package.json desktop/src-tauri/tauri.conf.json desktop/src-tauri/Cargo.toml desktop/src-tauri/Cargo.lock git checkout -B "${branch}" "origin/${branch}" + # The diff above proved every other file matches the merge base with dev, so this is + # the merge base's checker reading the branch's four committed version sources. + (cd .. && env -u GH_TOKEN bun scripts/release-version-sources.ts check "${NEXT_VERSION}" --root dev-tree) || { + echo "::error::${branch} does not carry ${NEXT_VERSION} in every version source" + exit 1 + } else git config user.name "github-actions[bot]" git config user.email "41898282+github-actions[bot]@users.noreply.github.com" git checkout -b "${branch}" - git add package.json + # The trusted checkout's checker reads this tree's four version sources. + (cd .. && env -u GH_TOKEN bun scripts/release-version-sources.ts check "${NEXT_VERSION}" --root dev-tree) || { + echo "::error::the bump did not move every version source to ${NEXT_VERSION}" + exit 1 + } + git add -- package.json desktop/src-tauri/tauri.conf.json desktop/src-tauri/Cargo.toml desktop/src-tauri/Cargo.lock git commit -m "${subject}" - git push origin "${branch}" + # Supply the write credential only to this trusted push invocation. In + # particular, never persist it in the dev checkout while dev-controlled + # files could execute. + auth_header="$(printf 'x-access-token:%s' "${GH_TOKEN}" | base64 -w0)" + # The raw token is masked automatically; its base64 form is not. Mask it so a + # verbose git or curl trace cannot leak a usable credential into the log. + echo "::add-mask::${auth_header}" + git -c "http.https://github.com/.extraheader=AUTHORIZATION: basic ${auth_header}" push origin "${branch}" fi gh pr create \ @@ -220,7 +280,8 @@ jobs: ${reason} - This moves \`dev\` to \`${NEXT_VERSION}\`. + This moves \`dev\` to \`${NEXT_VERSION}\` in \`package.json\` and the desktop version + sources (\`tauri.conf.json\`, \`Cargo.toml\`, and the \`opencodex-desktop\` entry of \`Cargo.lock\`). Opened automatically by \`.github/workflows/dev-version-bump.yml\`. The same version-line move was previously done by hand in 32529c2b2, e4a85d134, 076ad3036, and diff --git a/.github/workflows/enforce-pr-target.yml b/.github/workflows/enforce-pr-target.yml index 6743730206..35f6ea7b38 100644 --- a/.github/workflows/enforce-pr-target.yml +++ b/.github/workflows/enforce-pr-target.yml @@ -81,7 +81,38 @@ jobs: github.rest.repos.listPullRequestsAssociatedWithCommit, { owner, repo, commit_sha: statusSha, per_page: 100 } ); - candidates = associatedPrs.filter( + // The index spans the whole fork network: a commit that heads + // a PR in another repository comes back with that PR's number, + // which `pulls.get` cannot resolve in THIS repository (404). + // Only a candidate based on this repository is usable identity. + const inRepoPrs = associatedPrs.filter( + candidate => + candidate.base?.repo?.owner?.login === owner && + candidate.base?.repo?.name === repo + ); + // A null base.repo is not "another repository": it is a base the + // index could not identify, so log it as its own kind of skip. + const foreignRepoPrs = associatedPrs.filter( + candidate => + candidate.base?.repo != null && + !( + candidate.base.repo.owner?.login === owner && + candidate.base.repo.name === repo + ) + ); + const unidentifiedPrs = + associatedPrs.length - inRepoPrs.length - foreignRepoPrs.length; + if (foreignRepoPrs.length > 0) { + core.info( + `Ignored ${foreignRepoPrs.length} associated PR(s) based on other repositories in the fork network.` + ); + } + if (unidentifiedPrs > 0) { + core.info( + `Ignored ${unidentifiedPrs} associated PR(s) whose base repository could not be identified.` + ); + } + candidates = inRepoPrs.filter( candidate => candidate.state === "open" && candidate.head?.sha === statusSha @@ -186,6 +217,9 @@ jobs: isChangedFileListTruncated, extractReviewReadiness, appendReviewReadinessSection, + reviewReadinessMigrationRequired, + reviewReadinessUsesCurrentPolicy, + REVIEW_READINESS_ITEMS, stripReviewReadinessSection, uncheckReviewReadinessBoxes, REVIEW_READINESS_CLAIM_INDEX, @@ -204,6 +238,8 @@ jobs: ); const { parseGateState, + advanceReattestation, + savedAttestationAuthorizes, gateStateMarker, parseState, parseReadinessState, @@ -341,7 +377,8 @@ jobs: legacyReadinessState ); - let gateState = storedGateState ?? migratedGateState; + let gateState = storedGateState && typeof storedGateState === "object" && !Array.isArray(storedGateState) + ? storedGateState : migratedGateState; /** * Maintainers from `MAINTAINERS.md` on the trusted default branch @@ -418,13 +455,15 @@ jobs: await migrateLegacyCommentsIfNeeded(); return; } - await github.rest.issues.updateComment({ + const updated = await github.rest.issues.updateComment({ owner, repo, comment_id: gateCommentId, body }); - gateComment.body = body; + // Replace the listed comment instead of mutating it, so a later + // authoritative readback observes the server, not this run's copy. + gateComment = { ...gateComment, body, updated_at: updated.data.updated_at }; await migrateLegacyCommentsIfNeeded(); return; } @@ -435,7 +474,7 @@ jobs: body }); gateCommentId = created.data.id; - gateComment = { id: gateCommentId, body }; + gateComment = { id: gateCommentId, body, updated_at: created.data.updated_at }; await migrateLegacyCommentsIfNeeded(); } @@ -527,6 +566,118 @@ jobs: ); } + let reattestation = null; + let preserveAuthorBody = false; + const initiallyLegacy = reviewReadinessMigrationRequired(pr.body); + const unreadablePendingState = Boolean(gateComment?.body?.includes("opencodex-pr-gate-state:") && + (!storedGateState || typeof storedGateState !== "object" || Array.isArray(storedGateState))); + + function observeReattestation(snapshot, invalidate = false, allowEvent = true) { + const pending = (unreadablePendingState || (initiallyLegacy && gateState.pendingReattestation == null)) && !preserveAuthorBody + ? { invalid: true } : gateState.pendingReattestation; + const result = advanceReattestation({ + pending, + legacy: reviewReadinessMigrationRequired(snapshot.body), + current: reviewReadinessUsesCurrentPolicy(snapshot.body), + readiness: extractReviewReadiness(snapshot.body), + live: { headSha: snapshot.head?.sha, baseRef: snapshot.base?.ref, + body: snapshot.body, updatedAt: snapshot.updated_at, authorId: snapshot.user?.id }, + event: allowEvent ? { name: context.eventName, action: context.payload.action, + senderId: context.payload.sender?.id, senderType: context.payload.sender?.type, + headSha: context.payload.pull_request?.head?.sha, body: context.payload.pull_request?.body, + updatedAt: context.payload.pull_request?.updated_at, + previousBody: context.payload.changes?.body?.from } : {}, + invalidate + }); + preserveAuthorBody ||= pending != null || result.pending != null; + gateState.pendingReattestation = result.pending; + reattestation = result; + return result; + } + + async function persistReattestationCheckpoint(state, options) { + await upsertGateComment(state, options); + if (state.pendingReattestation?.checkpointAt !== null) return; + const checkpointAt = gateComment?.updated_at; + if (typeof checkpointAt !== "string" || !Number.isFinite(Date.parse(checkpointAt))) { + core.setFailed("The re-attestation checkpoint has no authoritative server timestamp."); + return; + } + const finalized = { ...state.pendingReattestation, checkpointAt }; + await upsertGateComment({ ...state, pendingReattestation: finalized }, options); + gateState.pendingReattestation = finalized; + } + + async function retainReattestationDraft(snapshot) { + core.setFailed("Current-head author re-attestation is pending; ordinary quality evaluation resumes after the saved checkpoint."); + if (reattestation?.invalidIdentity) { + core.setFailed("Cannot bind re-attestation to a verified current PR head."); + return; + } + const pending = gateState.pendingReattestation; + const snapshotReadiness = extractReviewReadiness(snapshot.body); + const clearStep = "Wait for the bot to acknowledge the cleared checklist before validating and ticking the boxes again."; + const action = pending?.phase === "await-check" + ? "The cleared checklist has been recorded. Validate this head, tick all four boxes and save the PR description." + : reviewReadinessUsesCurrentPolicy(snapshot.body) && snapshotReadiness.checked > 0 + ? `The first managed item already uses the current wording, but boxes ticked before this notice cannot carry over. Clear all four boxes and save. ${clearStep}` + : `Change the first managed item to ${inlineCode(REVIEW_READINESS_ITEMS[0])}, clear all four boxes and save. ${clearStep}`; + const state = { ...gateState, active: true, maintainersPinged: false, + autoDraftedByBot: !snapshot.draft || gateState.autoDraftedByBot }; + const options = { + status: "DRAFT", statusReason: "author re-attestation is required for the current head.", + actions: [action, "Only a new body edit by the PR author after this notice can advance the checkpoint. If edits share a checkpoint timestamp, make another body edit and save later."], + readiness: snapshotReadiness, checklistRequired: true, + notices: [`Current head: ${inlineCode(snapshot.head.sha)}. Existing PR text and checkbox marks were preserved.`] + }; + await persistReattestationCheckpoint(state, options); + if ((snapshot.labels ?? []).some(label => label.name === REVIEW_READY_LABEL)) { + try { + await github.rest.issues.removeLabel({ owner, repo, issue_number: pull_number, name: REVIEW_READY_LABEL }); + } catch (error) { + core.setFailed("Could not remove the stale review-ready label while re-attestation is pending."); + } + } + if (!snapshot.draft) { + try { await convertToDraft(); } + catch (error) { + state.autoDraftedByBot = false; + state.pendingReattestation = gateState.pendingReattestation; + await upsertGateComment(state, { + ...options, notices: [...options.notices, "Automatic draft conversion failed. Please retain draft state manually until re-attestation is complete."] + }); + core.setFailed("Could not retain draft state during author re-attestation."); + } + } + } + + if (!authorHasPushPermission(authorPermission) && + (reviewReadinessMigrationRequired(pr.body) || gateState.pendingReattestation != null || unreadablePendingState)) { + const { data: freshPr } = await github.rest.pulls.get({ owner, repo, pull_number }); + if (freshPr.node_id !== pr.node_id || freshPr.user?.id !== pr.user?.id) { + core.setFailed("PR identity changed before re-attestation."); + return; + } + Object.assign(pr, freshPr); + const result = observeReattestation(pr); + if (result.invalidIdentity || result.pending?.phase !== "attested") { + await retainReattestationDraft(pr); + return; + } + // Persist the new proof before any ready side effect. A failed write + // cannot be treated as a saved re-attestation checkpoint. + await persistReattestationCheckpoint({ ...gateState, active: true }, { + status: "DRAFT", statusReason: "Current-head re-attestation recorded; checking remaining requirements.", + actions: [], readiness: extractReviewReadiness(pr.body), checklistRequired: true, notices: [] + }); + if (!observeReattestation(pr, false, false).canComplete) { + core.setFailed("Re-attestation finalization is incomplete; readiness was not advanced."); + return; + } + } else if (authorHasPushPermission(authorPermission)) { + gateState.pendingReattestation = null; + } + let behindMain = 0; let behindBase = 0; let aheadMain = 0; @@ -551,8 +702,8 @@ jobs: other => other.number !== pull_number && other.head?.ref === pr.base.ref && - (other.base?.repo?.owner?.login ?? owner) === baseOwner && - (other.base?.repo?.name ?? repo) === baseName + other.head?.repo?.owner?.login === baseOwner && + other.head?.repo?.name === baseName ); if (stackedBase) { core.info( @@ -815,7 +966,7 @@ jobs: ? "" : pr.head.sha); const completionHeadSha = - gateState.completedAtHeadSha ?? null; + reattestation?.canComplete ? pr.head.sha : (gateState.completedAtHeadSha ?? null); const headDrifted = completionIsStale({ checklistRequired, checklistComplete, @@ -837,6 +988,11 @@ jobs: repo, pull_number }); + if (preserveAuthorBody || reviewReadinessMigrationRequired(freshPr.body)) { + observeReattestation(freshPr, true, false); + await retainReattestationDraft(freshPr); + return; + } const freshReadiness = extractReviewReadiness( freshPr.body ?? "" ); @@ -848,7 +1004,8 @@ jobs: ...defaultGateState(), active: gateState.active, autoDraftedByBot: gateState.autoDraftedByBot, - titlePrefixedByBot: gateState.titlePrefixedByBot + titlePrefixedByBot: gateState.titlePrefixedByBot, + pendingReattestation: gateState.pendingReattestation ?? null }; headDriftNotice = buildStaleNotice({ completionHeadSha, @@ -991,6 +1148,11 @@ jobs: repo, pull_number }); + if (preserveAuthorBody || reviewReadinessMigrationRequired(freshPr.body)) { + observeReattestation(freshPr, true, false); + await retainReattestationDraft(freshPr); + return; + } const freshReadiness = extractReviewReadiness( freshPr.body ?? "" ); @@ -998,7 +1160,8 @@ jobs: ...defaultGateState(), active: gateState.active, autoDraftedByBot: gateState.autoDraftedByBot, - titlePrefixedByBot: gateState.titlePrefixedByBot + titlePrefixedByBot: gateState.titlePrefixedByBot, + pendingReattestation: gateState.pendingReattestation ?? null }; claimNotice = [ ...(claimViolations.includes("review_findings") @@ -1093,6 +1256,40 @@ jobs: return actions; } + if (checklistRequired && checklistComplete && failures.length === 0) { + const expectedPending = JSON.stringify(gateState.pendingReattestation); + try { + if (preserveAuthorBody) { + const { data: finalComment } = await github.rest.issues.getComment({ owner, repo, comment_id: gateCommentId }); + const finalState = parseGateState(finalComment.body); + // Promotion needs positive evidence: the saved comment must hold a + // finalized attestation of this exact head, base, and body. + if (finalComment.user?.login !== "github-actions[bot]" || + JSON.stringify(finalState?.pendingReattestation) !== expectedPending || + !savedAttestationAuthorizes(finalState?.pendingReattestation, { + headSha: pr.head.sha, baseRef: pr.base.ref, body: pr.body, authorId: pr.user?.id })) { + core.setFailed("Saved re-attestation changed before readiness; no ready action was taken."); + return; + } + } + const { data: finalPr } = await github.rest.pulls.get({ owner, repo, pull_number }); + if (reviewReadinessMigrationRequired(finalPr.body)) { + observeReattestation(finalPr, true, false); + await retainReattestationDraft(finalPr); + return; + } + if (finalPr.node_id !== pr.node_id || finalPr.head?.sha !== pr.head.sha || + finalPr.base?.ref !== pr.base.ref || finalPr.body !== pr.body || + finalPr.user?.id !== pr.user?.id || finalPr.state !== "open") { + core.setFailed("PR changed before readiness; no ready action was taken."); + return; + } + } catch (error) { + core.setFailed("Could not refresh the current PR and saved re-attestation before readiness."); + return; + } + } + // The `review-ready` label marks the ready moment for humans and // bots. It is not a CodeRabbit auto-review filter: a positive // `labels:` entry in `.coderabbit.yaml` would restrict ALL reviews @@ -1217,7 +1414,7 @@ jobs: "This pull request is being kept as a draft automatically. Once every issue above is resolved, it will be marked ready for review again.", ...(checklistRequired && !checklistComplete ? [ - `@${pr.user.login} Tick the boxes once your local CI is green, your branch is on the latest ${inlineCode(DEFAULT_BASE)} commit, and every correct Codex and CodeRabbit finding is resolved.` + `@${pr.user.login} Tick the boxes once required local validation has passed with commands, results, and any full-suite exception documented, your branch is on the latest ${inlineCode(DEFAULT_BASE)} commit, and every correct Codex and CodeRabbit finding is resolved.` ] : []) ]); @@ -1242,7 +1439,7 @@ jobs: "This pull request was already a draft. Its draft status will be preserved after every issue above is resolved.", ...(checklistRequired && !checklistComplete ? [ - `@${pr.user.login} Tick the boxes once your local CI is green, your branch is on the latest ${inlineCode(DEFAULT_BASE)} commit, and every correct Codex and CodeRabbit finding is resolved.` + `@${pr.user.login} Tick the boxes once required local validation has passed with commands, results, and any full-suite exception documented, your branch is on the latest ${inlineCode(DEFAULT_BASE)} commit, and every correct Codex and CodeRabbit finding is resolved.` ] : []) ]); @@ -1373,6 +1570,7 @@ jobs: readyState.maintainersPinged = true; notified = true; } + if (!readyConversionFailed) readyState.pendingReattestation = null; readyState.completedAtHeadSha = pr.head.sha; readyState.version = 1; diff --git a/.github/workflows/issue-quality-tests.yml b/.github/workflows/issue-quality-tests.yml index 0b6529f667..fd064ce101 100644 --- a/.github/workflows/issue-quality-tests.yml +++ b/.github/workflows/issue-quality-tests.yml @@ -12,6 +12,7 @@ on: - ".github/scripts/pr-quality-messages.cjs" - ".github/scripts/pr-quality-messages.test.cjs" - ".github/scripts/pr-quality-state.cjs" + - ".github/scripts/pr-readiness-reattest.cjs" - ".github/scripts/pr-quality-state.test.cjs" - ".github/scripts/pr-labeler.cjs" - ".github/scripts/pr-labeler.test.cjs" @@ -35,6 +36,8 @@ on: - ".github/workflows/issue-quality-tests.yml" - ".github/workflows/pr-hygiene.yml" push: + # Release lines only; review is covered by the pull_request trigger above. + branches: [main, preview] paths: - ".github/ISSUE_TEMPLATE/**" - ".github/scripts/issue-quality.cjs" @@ -45,6 +48,7 @@ on: - ".github/scripts/pr-quality-messages.cjs" - ".github/scripts/pr-quality-messages.test.cjs" - ".github/scripts/pr-quality-state.cjs" + - ".github/scripts/pr-readiness-reattest.cjs" - ".github/scripts/pr-quality-state.test.cjs" - ".github/scripts/pr-labeler.cjs" - ".github/scripts/pr-labeler.test.cjs" diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index dd63e98a51..764eb0622a 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -24,6 +24,11 @@ on: required: false type: boolean default: true + resume-after-npm-publish: + description: "Operator attestation: a previous run of this workflow acknowledged npm publication for this exact commit; skip npm publish and complete the GitHub side" + required: false + type: boolean + default: false expected-sha: description: "Immutable release commit this dispatch must publish (fail if the branch moved)" required: true @@ -69,26 +74,80 @@ jobs: process.exit(1); } NODE - package-standalone: + + # Every publication precondition the dispatch can already decide, checked before any runner + # starts packaging: channel and dist-tag, every version source, the tag, the GitHub release, + # npm, the global tag ordering and the dev pre-move (scripts/ci/release-preflight.sh). + # + # Run 35783865160 packaged 2.62.0 for nineteen minutes and then failed the ordering gate in + # `publish` on v2.63.0-preview.20260923. That tag already existed when the run's first job + # started: the workflow-level `release` concurrency group above is one constant slot for every + # ref, so the stable run had waited for the preview run to finish. The runs were serialised; + # the check was in the wrong place. Because of that shared slot, this job sees whatever the + # previous release run published. + # + # It is an early answer, not the final one. Tags, releases and registry state can still move + # while a run packages (a hand-pushed tag, a first local publish), so `publish` repeats every + # one of these checks immediately before `npm publish`. + preflight: + name: release preflight needs: validate-dispatch + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + contents: read + steps: + - name: Checkout + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + persist-credentials: false + fetch-tags: true + + - name: Setup project Bun + uses: ./.github/actions/setup-project-bun + + - name: Fetch the dev line + run: git fetch --no-tags --depth=1 origin +refs/heads/dev:refs/remotes/origin/dev + + - name: Refuse a release that cannot publish + env: + GH_TOKEN: ${{ github.token }} + RELEASE_VERSION: ${{ inputs.version }} + NPM_DIST_TAG: ${{ inputs.tag }} + DRY_RUN: ${{ inputs.dry-run }} + RESUME: ${{ inputs.resume-after-npm-publish }} + run: bash scripts/ci/release-preflight.sh + + package-standalone: + needs: [validate-dispatch, preflight] strategy: fail-fast: false matrix: include: - os: ubuntu-latest target: bun-linux-x64 + dependency_os: linux + dependency_cpu: x64 smoke: true - os: macos-latest target: bun-darwin-arm64 + dependency_os: darwin + dependency_cpu: arm64 smoke: true - os: macos-latest target: bun-darwin-x64 + dependency_os: darwin + dependency_cpu: x64 smoke: false - os: windows-latest target: bun-windows-x64 + dependency_os: win32 + dependency_cpu: x64 smoke: true - os: ubuntu-latest target: bun-linux-arm64 + dependency_os: linux + dependency_cpu: arm64 smoke: false runs-on: ${{ matrix.os }} timeout-minutes: 25 @@ -104,7 +163,7 @@ jobs: uses: ./.github/actions/setup-project-bun - name: Install dependencies - run: bun install --frozen-lockfile + run: bun install --frozen-lockfile --os=${{ matrix.dependency_os }} --cpu=${{ matrix.dependency_cpu }} - name: Build dashboard run: bun run build:gui @@ -151,13 +210,17 @@ jobs: set -euo pipefail cd "dist/standalone/$STANDALONE_TARGET" if [[ "$RUNNER_OS" == "Windows" ]]; then - powershell -NoProfile -Command 'Compress-Archive -Path ocx.exe,gui -DestinationPath ("../../ocx-{0}-{1}.zip" -f $env:RELEASE_VERSION,$env:STANDALONE_TARGET) -Force' + powershell -NoProfile -Command 'Compress-Archive -Path ocx.exe,gui,keyring -DestinationPath ("../../ocx-{0}-{1}.zip" -f $env:RELEASE_VERSION,$env:STANDALONE_TARGET) -Force' else - tar -czf "../../ocx-${RELEASE_VERSION}-${STANDALONE_TARGET}.tar.gz" ocx gui + tar -czf "../../ocx-${RELEASE_VERSION}-${STANDALONE_TARGET}.tar.gz" ocx gui keyring fi - cd ../../.. - if [[ "$RUNNER_OS" == "Windows" ]]; then sha256sum "dist/ocx-${RELEASE_VERSION}-${STANDALONE_TARGET}.zip" > "dist/ocx-${RELEASE_VERSION}-${STANDALONE_TARGET}.sha256" - else sha256sum "dist/ocx-${RELEASE_VERSION}-${STANDALONE_TARGET}.tar.gz" > "dist/ocx-${RELEASE_VERSION}-${STANDALONE_TARGET}.sha256" + cd ../.. + # The pre-publication verifier resolves every recorded checksum from + # dist/release, where the artifact download lands these files flat; the + # checksum therefore records the bare file name, which sha256sum takes + # verbatim from its argument. + if [[ "$RUNNER_OS" == "Windows" ]]; then sha256sum "ocx-${RELEASE_VERSION}-${STANDALONE_TARGET}.zip" > "ocx-${RELEASE_VERSION}-${STANDALONE_TARGET}.zip.sha256" + else sha256sum "ocx-${RELEASE_VERSION}-${STANDALONE_TARGET}.tar.gz" > "ocx-${RELEASE_VERSION}-${STANDALONE_TARGET}.tar.gz.sha256" fi - name: Upload standalone release @@ -172,23 +235,29 @@ jobs: retention-days: 7 package-desktop: - needs: validate-dispatch + needs: [validate-dispatch, preflight] strategy: fail-fast: false matrix: include: - os: macos-latest target: universal-apple-darwin + dependency_os: darwin + dependency_cpu: "*" bundles: app,dmg sidecar-targets: macos artifact-suffixes: macos.dmg,macos.app.tar.gz - os: windows-latest target: x86_64-pc-windows-msvc + dependency_os: win32 + dependency_cpu: x64 bundles: msi sidecar-targets: x86_64-pc-windows-msvc artifact-suffixes: windows-x64.msi - os: ubuntu-22.04 target: x86_64-unknown-linux-gnu + dependency_os: linux + dependency_cpu: x64 bundles: appimage,deb sidecar-targets: x86_64-unknown-linux-gnu artifact-suffixes: linux-x86_64.AppImage,linux-amd64.deb @@ -196,6 +265,10 @@ jobs: timeout-minutes: 45 permissions: contents: read + env: + # Whether this run holds the Developer ID material at all. A run without it still builds + # locally useful bundles; a run with it must not silently downgrade any part of the app. + DESKTOP_SIGNING_CONFIGURED: ${{ secrets.APPLE_CERTIFICATE != '' }} steps: - name: Checkout uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 @@ -205,8 +278,20 @@ jobs: - name: Setup project Bun uses: ./.github/actions/setup-project-bun + # The desktop build takes its version from tauri.conf.json and Cargo.toml (the widget + # plist inherits it), not from package.json or this input, while the updater manifest is + # derived from the input. A mismatch ships an app that reports the previous version under a + # manifest naming this one, so the updater re-offers the same release forever. Refuse + # before anything is built. Bash on every runner: Windows would otherwise run PowerShell, + # where "$RELEASE_VERSION" is not the environment variable. + - name: Verify every version source matches the release + shell: bash + env: + RELEASE_VERSION: ${{ inputs.version }} + run: bun scripts/release-version-sources.ts check "$RELEASE_VERSION" + - name: Install project dependencies - run: bun install --frozen-lockfile + run: bun install --frozen-lockfile --os=${{ matrix.dependency_os }} --cpu=${{ matrix.dependency_cpu }} - name: Build dashboard run: bun run build:gui @@ -215,6 +300,7 @@ jobs: uses: dtolnay/rust-toolchain@02cb101ec7c40f2c49e1d9714d64511d8e1b74de # master with: toolchain: stable + targets: ${{ runner.os == 'macOS' && 'aarch64-apple-darwin,x86_64-apple-darwin' || '' }} - name: Install Linux desktop dependencies if: runner.os == 'Linux' @@ -231,19 +317,159 @@ jobs: run: | bun desktop/scripts/prepare-sidecar.ts --target aarch64-apple-darwin bun desktop/scripts/prepare-sidecar.ts --target x86_64-apple-darwin + lipo -create desktop/src-tauri/binaries/ocx-aarch64-apple-darwin \ + desktop/src-tauri/binaries/ocx-x86_64-apple-darwin \ + -output desktop/src-tauri/binaries/ocx-universal-apple-darwin + lipo desktop/src-tauri/binaries/ocx-universal-apple-darwin -verify_arch arm64 x86_64 - name: Prepare sidecar if: runner.os != 'macOS' run: bun desktop/scripts/prepare-sidecar.ts --target ${{ matrix.sidecar-targets }} + # The signing certificate has to be in a keychain before the widget is signed, and the + # Tauri build step creates its own keychain only when it runs — which is after this. Until + # this step existed, build-widget.sh saw no MACOS_SIGN_IDENTITY and took its unsigned + # branch, and the bundler does not re-sign anything under PlugIns, so the extension would + # have gone out ad-hoc inside a Developer ID host. No release has published a macOS + # application yet, so this is a defect that had not reached anyone rather than one that had. + - name: Import the release signing certificate + if: runner.os == 'macOS' + env: + APPLE_CERTIFICATE: ${{ secrets.APPLE_CERTIFICATE }} + APPLE_CERTIFICATE_PASSWORD: ${{ secrets.APPLE_CERTIFICATE_PASSWORD }} + APPLE_ID: ${{ secrets.APPLE_ID }} + APPLE_PASSWORD: ${{ secrets.APPLE_PASSWORD }} + APPLE_TEAM_ID: ${{ secrets.APPLE_TEAM_ID }} + DRY_RUN: ${{ inputs.dry-run }} + run: | + set -euo pipefail + # Checked as a set, because a partial set is the dangerous case: the Tauri CLI skips + # notarization without failing when the notary credentials are missing, and the + # unnotarized artifact is uploaded and attached exactly as a good one would be. + missing="" + for name in APPLE_CERTIFICATE APPLE_CERTIFICATE_PASSWORD APPLE_ID APPLE_PASSWORD APPLE_TEAM_ID; do + eval "value=\${$name:-}" + [ -n "$value" ] || missing="$missing $name" + done + if [ -n "$missing" ]; then + if [ "${DRY_RUN}" != "true" ]; then + echo "::error::A real release needs the full signing and notarization credential set." + echo "::error::Missing:$missing" + exit 1 + fi + echo "Signing credentials are incomplete, so this build stays ad-hoc signed:$missing" + echo "It is usable for local validation and is not a release asset." + exit 0 + fi + keychain="$RUNNER_TEMP/opencodex-signing.keychain-db" + # Recorded before anything is created, so the cleanup step can still find a keychain + # that a failure left half-built. + echo "OPENCODEX_SIGNING_KEYCHAIN=$keychain" >> "$GITHUB_ENV" + keychain_password="$(python3 -c 'import secrets; print(secrets.token_urlsafe(32))')" + certificate="$RUNNER_TEMP/opencodex-signing.p12" + # The decoded certificate must not outlive this step even when a later command fails. + trap 'shred -u "$certificate" 2>/dev/null || rm -Pf "$certificate" 2>/dev/null || true' EXIT + printf '%s' "$APPLE_CERTIFICATE" | base64 --decode > "$certificate" + security create-keychain -p "$keychain_password" "$keychain" + security set-keychain-settings -lut 21600 "$keychain" + security unlock-keychain -p "$keychain_password" "$keychain" + security import "$certificate" -k "$keychain" -P "$APPLE_CERTIFICATE_PASSWORD" \ + -T /usr/bin/codesign + security set-key-partition-list -S apple-tool:,apple:,codesign: \ + -s -k "$keychain_password" "$keychain" > /dev/null + # shellcheck disable=SC2046 # the keychain list is intentionally word-split into arguments + security list-keychain -d user -s "$keychain" $(security list-keychains -d user | tr -d '"') + - name: Build WidgetKit extension if: runner.os == 'macOS' + env: + MACOS_SIGN_IDENTITY: ${{ secrets.APPLE_SIGNING_IDENTITY }} + # A run holding Developer ID material must not produce an ad-hoc widget. Without this + # the script's ad-hoc branch is the silent default, which is how a signed, notarized + # app shipped with an extension macOS will not register. + WIDGET_SIGN_REQUIRED: ${{ env.DESKTOP_SIGNING_CONFIGURED == 'true' && '1' || '0' }} run: bash desktop/scripts/build-widget.sh + - name: Verify the extension carries the release signature + if: runner.os == 'macOS' + env: + APPLE_TEAM_ID: ${{ secrets.APPLE_TEAM_ID }} + DRY_RUN: ${{ inputs.dry-run }} + run: | + set -euo pipefail + appex=desktop/src-tauri/widget/OpenCodexWidget.appex + if [ -z "${APPLE_TEAM_ID}" ]; then + if [ "${DRY_RUN}" != "true" ]; then + echo "::error::A real release cannot assert its own signature without APPLE_TEAM_ID." + exit 1 + fi + echo "No team configured; skipping the signature assertion for this non-release build." + exit 0 + fi + codesign --verify --strict --deep "$appex" + description="$(codesign -dvvv "$appex" 2>&1)" + echo "$description" + echo "$description" | grep -q "TeamIdentifier=$APPLE_TEAM_ID" + echo "$description" | grep -q "flags=.*runtime" + echo "$description" | grep -q "Timestamp=" + + # Tauri signs the app, its sidecar and the widget it is handed, but nothing under Resources. + # The packaged keyring addons (#6161) are Mach-O code, so Apple notarization rejects the whole + # app unless each carries the Developer ID signature, the hardened runtime and a secure + # timestamp (2.73.0-preview.20260930 was refused for exactly that). Sign them in place, after + # the certificate import and before the bundler copies them. + - name: Sign the packaged keyring addons + if: runner.os == 'macOS' + env: + MACOS_SIGN_IDENTITY: ${{ secrets.APPLE_SIGNING_IDENTITY }} + APPLE_TEAM_ID: ${{ secrets.APPLE_TEAM_ID }} + DRY_RUN: ${{ inputs.dry-run }} + run: | + set -euo pipefail + shopt -s nullglob + addons=(desktop/src-tauri/resources/keyring/*.darwin-*.node) + if [ "${#addons[@]}" -eq 0 ]; then + echo "::error::No macOS keyring addon was prepared for the bundle." + exit 1 + fi + if [ "${DESKTOP_SIGNING_CONFIGURED}" != "true" ]; then + if [ "${DRY_RUN}" != "true" ]; then + echo "::error::A real release must sign the packaged keyring addons." + exit 1 + fi + echo "No signing material; the keyring addons stay as prepared for this non-release build." + exit 0 + fi + for addon in "${addons[@]}"; do + codesign --force --timestamp --options runtime --sign "$MACOS_SIGN_IDENTITY" "$addon" + codesign --verify --strict "$addon" + description="$(codesign -dvvv "$addon" 2>&1)" + echo "$description" + # Here-strings, not pipes: under pipefail a matcher that exits on its first hit can + # SIGPIPE the writer and fail the assertion it was meant to pass. + grep -q "TeamIdentifier=$APPLE_TEAM_ID" <<<"$description" + grep -q "flags=.*runtime" <<<"$description" + grep -q "Timestamp=" <<<"$description" + done + + - name: Prepare Windows installer version + if: runner.os == 'Windows' + shell: bash + env: + RELEASE_VERSION: ${{ inputs.version }} + run: bun desktop/scripts/windows-installer-config.ts "$RELEASE_VERSION" "$RUNNER_TEMP/opencodex-msi.json" + + - name: Preserve the compiled Linux sidecar + if: runner.os == 'Linux' + run: | + chmod +x desktop/scripts/appimage-patchelf.py + echo "PATCHELF=$GITHUB_WORKSPACE/desktop/scripts/appimage-patchelf.py" >> "$GITHUB_ENV" + # Release signing is intentionally secret-gated. Developer ID, notarization, # and updater signatures require maintainer-owned credentials; builds without # those secrets remain useful for local validation but are not release assets. - name: Build desktop bundles + if: runner.os != 'Linux' working-directory: desktop env: TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }} @@ -255,17 +481,126 @@ jobs: APPLE_PASSWORD: ${{ secrets.APPLE_PASSWORD }} APPLE_TEAM_ID: ${{ secrets.APPLE_TEAM_ID }} MACOS_SIGN_IDENTITY: ${{ secrets.APPLE_SIGNING_IDENTITY }} - run: bunx tauri build --ci --target ${{ matrix.target }} --bundles ${{ matrix.bundles }} + # linuxdeploy suppresses its own stderr at the default verbosity. Keep + # diagnostics on the first attempt; Apple signing commands stay non-verbose. + run: bunx tauri ${{ runner.os == 'Linux' && '--verbose' || '' }} build --ci --target ${{ matrix.target }} --bundles ${{ matrix.bundles }} --config "${{ runner.os == 'Windows' && format('{0}/opencodex-msi.json', runner.temp) || '{}' }}" + + - name: Verify the packaged universal macOS runtime + if: runner.os == 'macOS' + run: | + set -euo pipefail + app=desktop/src-tauri/target/universal-apple-darwin/release/bundle/macos/OpenCodex.app + test -f "$app/Contents/Resources/keyring/keyring.darwin-arm64.node" + test -f "$app/Contents/Resources/keyring/keyring.darwin-x64.node" + bash desktop/scripts/verify-macos-runtime.sh "$app" + + # Tauri patches a bundle-type marker into the application binary for each Linux format. + # Keep each format in its own Cargo target so the deb cannot inherit the AppImage marker + # and linuxdeploy cannot mutate the binary later consumed by the deb build. + - name: Build Linux AppImage bundle + if: runner.os == 'Linux' + working-directory: desktop + env: + CARGO_TARGET_DIR: ${{ runner.temp }}/opencodex-appimage-target + TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }} + TAURI_SIGNING_PRIVATE_KEY_PASSWORD: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY_PASSWORD }} + run: bunx tauri build --ci --target ${{ matrix.target }} --bundles appimage + + - name: Build Linux deb bundle + if: runner.os == 'Linux' + working-directory: desktop + env: + CARGO_TARGET_DIR: ${{ runner.temp }}/opencodex-deb-target + TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }} + TAURI_SIGNING_PRIVATE_KEY_PASSWORD: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY_PASSWORD }} + run: bunx tauri build --ci --target ${{ matrix.target }} --bundles deb + + - name: Stage isolated Linux release bundles + if: runner.os == 'Linux' + shell: bash + env: + DESKTOP_TARGET: ${{ matrix.target }} + APPIMAGE_TARGET: ${{ runner.temp }}/opencodex-appimage-target + DEB_TARGET: ${{ runner.temp }}/opencodex-deb-target + run: | + set -euo pipefail + bundle_root="$RUNNER_TEMP/opencodex-linux-release-bundles" + mkdir -p "$bundle_root/appimage" "$bundle_root/deb" + cp -a "$APPIMAGE_TARGET/$DESKTOP_TARGET/release/bundle/appimage/." "$bundle_root/appimage/" + cp -a "$DEB_TARGET/$DESKTOP_TARGET/release/bundle/deb/." "$bundle_root/deb/" + chmod -R a-w "$bundle_root" + echo "DESKTOP_BUNDLE_ROOT=$bundle_root" >> "$GITHUB_ENV" + + # After the isolated AppImage exists, and against that staged copy: the default Cargo target + # holds no Linux bundle any more, so verifying there would fail or check a stale artifact. + - name: Verify the packaged Linux sidecar + if: runner.os == 'Linux' + run: bash desktop/scripts/verify-linux-sidecar.sh "$DESKTOP_BUNDLE_ROOT/appimage" - name: Rename release assets + shell: bash env: RELEASE_VERSION: ${{ inputs.version }} DESKTOP_TARGET: ${{ matrix.target }} run: | - bun desktop/scripts/collect-release-assets.ts \ + args=( \ --version "$RELEASE_VERSION" \ --target "$DESKTOP_TARGET" \ - --out dist/release + --out dist/release \ + ) + if [[ -n "${DESKTOP_BUNDLE_ROOT:-}" ]]; then + args+=(--bundle-root "$DESKTOP_BUNDLE_ROOT") + fi + bun desktop/scripts/collect-release-assets.ts "${args[@]}" + + # After the bundle exists, not before: a sweep that runs first passes by finding nothing. + - name: Verify every Mach-O in the bundle carries the release identity + if: runner.os == 'macOS' + env: + APPLE_TEAM_ID: ${{ secrets.APPLE_TEAM_ID }} + DRY_RUN: ${{ inputs.dry-run }} + run: | + set -euo pipefail + if [ -z "${APPLE_TEAM_ID}" ]; then + if [ "${DRY_RUN}" != "true" ]; then + echo "::error::A real release cannot verify its bundle without APPLE_TEAM_ID." + exit 1 + fi + echo "No team configured; skipping the bundle-wide assertion for this local build." + exit 0 + fi + # Executables are found by their magic bytes rather than by path or extension. A bundler + # signs what it placed; anything copied in afterwards is invisible to it, and the + # binaries that get missed are the ones with no extension to filter on. + apps=0 + machos=0 + bad=0 + while IFS= read -r app; do + apps=$((apps + 1)) + echo "checking $app" + while IFS= read -r -d '' file; do + # All eight Mach-O leading words: thin and fat, 32- and 64-bit, both byte orders. + # A list that covers only the common ones skips the rest in silence while the + # non-zero counter below still reports a healthy sweep. + case "$(head -c 4 "$file" | xxd -p)" in + cefaedfe|cffaedfe|feedface|feedfacf) ;; + cafebabe|bebafeca|cafebabf|bfbafeca) ;; + *) continue ;; + esac + machos=$((machos + 1)) + if ! codesign -dvvv "$file" 2>&1 | grep -q "TeamIdentifier=$APPLE_TEAM_ID"; then + echo "::error::$file is not signed with the release identity" + bad=1 + fi + done < <(find "$app" -type f -print0) + done < <(find desktop/src-tauri/target -maxdepth 6 -type d -name '*.app') + echo "inspected $machos Mach-O files across $apps app bundles" + # A sweep that inspected nothing is the failure mode this step exists to prevent. + if [ "$apps" -eq 0 ] || [ "$machos" -eq 0 ]; then + echo "::error::found $apps app bundles and $machos Mach-O files; the sweep inspected nothing" + exit 1 + fi + exit "$bad" - name: Upload desktop release uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 @@ -275,15 +610,28 @@ jobs: if-no-files-found: error retention-days: 7 - attach-release: + # always(), because a keychain holding the release identity must not survive a failed job + # on a runner image that could be reused. + - name: Remove the signing keychain + if: always() && runner.os == 'macOS' + run: | + if [ -n "${OPENCODEX_SIGNING_KEYCHAIN:-}" ] && [ -f "${OPENCODEX_SIGNING_KEYCHAIN}" ]; then + security delete-keychain "${OPENCODEX_SIGNING_KEYCHAIN}" + fi + + # Pre-publication verification. Everything that will be published is checked + # here — expected platform set, every checksum, the updater signatures, and the + # manifest parse-back — and publication consumes this result rather than + # verifying after the fact. Runs on dry-run too: a dry run must prove the same + # chain a real release will rely on. + verify-release: runs-on: ubuntu-latest - needs: [publish, package-standalone, package-desktop] - if: ${{ inputs.dry-run != true }} - env: - UPDATER_SIGNING_CONFIGURED: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY != '' }} + needs: [validate-dispatch, package-standalone, package-desktop] timeout-minutes: 10 permissions: - contents: write + contents: read + env: + UPDATER_SIGNING_CONFIGURED: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY != '' }} steps: - name: Checkout uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 @@ -307,24 +655,86 @@ jobs: merge-multiple: true path: dist/release - # Generate latest.json only when the updater key is configured; then require - # signatures for all four updater platforms before publishing it. - - name: Generate updater manifest - if: env.UPDATER_SIGNING_CONFIGURED == 'true' + - name: Verify release assets env: RELEASE_VERSION: ${{ inputs.version }} run: | - bun desktop/scripts/updater-manifest.ts \ - --version "$RELEASE_VERSION" \ - --dir dist/release \ - --repo lidge-jun/opencodex \ - --out dist/release/latest.json \ - --require-all + set -euo pipefail + args=( + --version "$RELEASE_VERSION" + --dir dist/release + --repo "$GITHUB_REPOSITORY" + --sha "$GITHUB_SHA" + --receipt-out verification/receipt.json + ) + # Signatures are verified whenever they exist; the manifest is only + # generated when this run holds the updater key, exactly as before. + if [ "$UPDATER_SIGNING_CONFIGURED" = "true" ]; then + args+=(--manifest-out dist/release/latest.json --require-signatures) + fi + bun desktop/scripts/verify-release-assets.ts "${args[@]}" + + - name: Upload verified release bundle + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: verified-release + path: dist/release/ + if-no-files-found: error + retention-days: 7 + + - name: Upload verification receipt + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: release-verification-receipt + path: verification/receipt.json + if-no-files-found: error + retention-days: 7 - - name: Verify the checksum before uploading + attach-release: + runs-on: ubuntu-latest + needs: [publish, verify-release] + if: ${{ inputs.dry-run != true }} + timeout-minutes: 10 + permissions: + contents: write + steps: + - name: Checkout + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + persist-credentials: false + + - name: Setup project Bun + uses: ./.github/actions/setup-project-bun + + - name: Download the verified release bundle + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: verified-release + path: dist/release + + - name: Download the verification receipt + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: release-verification-receipt + path: verification + + # The bundle is attached exactly as verified: the receipt must name this + # run's version and commit, or nothing uploads. + - name: Require the verification receipt for this commit + env: + RELEASE_VERSION: ${{ inputs.version }} run: | - cd dist/release - shasum -a 256 -c ./*.sha256 + set -euo pipefail + receipt_version="$(bun -e 'console.log(JSON.parse(await Bun.file("verification/receipt.json").text()).version)')" + receipt_sha="$(bun -e 'console.log(JSON.parse(await Bun.file("verification/receipt.json").text()).sha)')" + test "$receipt_version" = "$RELEASE_VERSION" || { + echo "::error::verification receipt names version $receipt_version, not $RELEASE_VERSION" + exit 1 + } + test "$receipt_sha" = "$GITHUB_SHA" || { + echo "::error::verification receipt names commit $receipt_sha, not $GITHUB_SHA" + exit 1 + } - name: Attach to the release env: @@ -333,12 +743,67 @@ jobs: # run: source. tests/ci-workflows.test.ts enforces this repo-wide. RELEASE_VERSION: ${{ inputs.version }} run: | - gh release upload "v${RELEASE_VERSION}" dist/release/* --clobber + set -euo pipefail + release_tag="v${RELEASE_VERSION}" + gh release upload "$release_tag" dist/release/* --clobber + + # A published release is immutable: GitHub rejects every later asset upload + # with HTTP 422, which is why v2.55.0 through v2.60.0 shipped with zero + # assets and left the desktop updater without anything to download. The + # release is therefore created as a draft and becomes public here, once the + # verified bundle is attached. The only edit permitted is this flip — the + # notes still come from the validated notes file written at creation. + draft_state="$(gh release view "$release_tag" --json isDraft --jq .isDraft)" + case "$draft_state" in + true) gh release edit "$release_tag" --draft=false ;; + false) ;; + # A successful query that answers neither true nor false — an empty body or an + # unexpected shape — must not leave the release a silent draft: only an explicit + # false may pass. + *) echo "unexpected draft state: $draft_state" >&2; exit 1 ;; + esac + + # One row per fact a release run can establish: the public GitHub release, the npm version read + # back from the registry, and the npm dist-tag. A green run used to read the same whichever of + # them were true, because the registry smoke continues to the GitHub release when its reads stay + # pending, which is the intended publishing behaviour. This job only reports; it never changes the + # run's result. + # + # A job of its own, not a step in attach-release: a failed publish skips attach-release entirely, + # and that is when the rows matter most. It reads with the job token at contents: read, so a draft + # release is invisible to it and reads as not public, which is the question the row answers. + release-outcomes: + name: release outcomes + needs: [publish, attach-release] + if: ${{ always() && inputs.dry-run != true }} + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + contents: read + steps: + - name: Checkout + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 + with: + persist-credentials: false + + - name: Report release outcomes + env: + GH_TOKEN: ${{ github.token }} + RELEASE_VERSION: ${{ inputs.version }} + NPM_DIST_TAG: ${{ inputs.tag }} + NPM_VERSION_STATE: ${{ needs.publish.outputs.npm_version }} + NPM_DIST_TAG_STATE: ${{ needs.publish.outputs.npm_dist_tag }} + PUBLISH_RESULT: ${{ needs.publish.result }} + ATTACH_RESULT: ${{ needs.attach-release.result }} + run: bash scripts/ci/release-outcome-report.sh publish: - needs: validate-dispatch + needs: [validate-dispatch, verify-release] runs-on: ubuntu-latest timeout-minutes: 15 + outputs: + npm_version: ${{ steps.registry-smoke.outputs.npm_version }} + npm_dist_tag: ${{ steps.registry-smoke.outputs.npm_dist_tag }} permissions: contents: write actions: read @@ -395,14 +860,15 @@ jobs: - name: Dependency audit (high severity) run: bun run audit:high - - name: Verify version matches package.json + - name: Verify every version source matches the requested version env: RELEASE_VERSION: ${{ inputs.version }} run: | - PKG=$(node -p "require('./package.json').version") - echo "package.json=$PKG input=${RELEASE_VERSION}" - test "$PKG" = "$RELEASE_VERSION" || { - echo "::error::package.json ($PKG) != requested (${RELEASE_VERSION}) — bump package.json on main first"; + # package.json plus the desktop sources (tauri.conf.json, Cargo.toml and the + # opencodex-desktop Cargo.lock entry). package-desktop already refused to build on a + # mismatch; this re-proves it on the commit that is about to publish. + bun scripts/release-version-sources.ts check "$RELEASE_VERSION" || { + echo "::error::a version source != requested (${RELEASE_VERSION}) — run scripts/release.ts, which moves all of them, on main first"; exit 1; } @@ -487,8 +953,10 @@ jobs: fi # Keep in sync with the service-lifecycle.yml trigger paths. src/cli.ts is - # the pre-restructure compat stub that durable launchers still execute. - if printf '%s\n' "$changed_files" | grep -Eq '^(src/service\.ts|src/cli\.ts|src/cli/index\.ts|src/lib/bun-runtime\.ts|package\.json|bun\.lock|\.github/workflows/service-lifecycle\.yml|\.github/workflows/release\.yml)$'; then + # the pre-restructure compat stub that durable launchers still execute; the + # service implementation itself is the src/service/ directory, and the desktop + # shell packages and launches it. + if printf '%s\n' "$changed_files" | grep -Eq '^(src/service\.ts|src/service/.*|desktop/.*|src/cli\.ts|src/cli/index\.ts|src/lib/bun-runtime\.ts|package\.json|bun\.lock|\.github/workflows/service-lifecycle\.yml|\.github/workflows/release\.yml)$'; then service_url="$( gh run list \ --workflow service-lifecycle.yml \ @@ -521,10 +989,12 @@ jobs: # PREREQUISITE: configure the Trusted Publisher for this repo + workflow on npmjs.com — possible # only AFTER the package's first version exists (do the first publish locally, see the runbook). - name: Preflight release metadata + id: metadata env: GH_TOKEN: ${{ github.token }} RELEASE_VERSION: ${{ inputs.version }} DRY_RUN: ${{ inputs.dry-run }} + RESUME: ${{ inputs.resume-after-npm-publish }} run: | set -euo pipefail @@ -541,7 +1011,9 @@ jobs: fi if [ -n "$existing_tag_sha" ]; then - if [ "$dry_run" = "true" ]; then + if [ "$RESUME" = "true" ]; then + echo "::notice::${release_tag} already exists at this commit; resuming" + elif [ "$dry_run" = "true" ]; then echo "::notice::${release_tag} already exists at this commit; dry-run only" else echo "::error::${release_tag} already exists. Refusing to publish a version with pre-existing Git metadata." @@ -550,7 +1022,9 @@ jobs: fi if gh release view "$release_tag" >/dev/null 2>&1; then - if [ "$dry_run" = "true" ]; then + if [ "$RESUME" = "true" ]; then + echo "::notice::GitHub Release ${release_tag} already exists; resuming to complete the attachment" + elif [ "$dry_run" = "true" ]; then echo "::notice::GitHub Release ${release_tag} already exists; dry-run only" else echo "::error::GitHub Release ${release_tag} already exists. Choose the next unused patch version." @@ -558,24 +1032,42 @@ jobs: fi fi + if [ "$RESUME" = "true" ] && [ "$dry_run" = "true" ]; then + echo "::error::resume-after-npm-publish is a real-publication recovery path and cannot combine with dry-run" + exit 1 + fi if npm view "${pkg_name}@${RELEASE_VERSION}" version >/dev/null 2>&1; then - if [ "$dry_run" = "true" ]; then + if [ "$RESUME" = "true" ]; then + resume_git_head="$(timeout --kill-after=2s 10s npm view "${pkg_name}@${RELEASE_VERSION}" gitHead --json --registry=https://registry.npmjs.org --fetch-retries=0 --fetch-timeout=8000)" || { + echo "::error::Cannot verify the existing npm package source; resume refused" + exit 1 + } + bun scripts/verify-release-resume.ts "$GITHUB_SHA" "$resume_git_head" + echo "resume_sha=$GITHUB_SHA" >> "$GITHUB_OUTPUT" + echo "::notice::${pkg_name}@${RELEASE_VERSION} is acknowledged on npm; resuming after the recorded partial publication" + elif [ "$dry_run" = "true" ]; then echo "::notice::${pkg_name}@${RELEASE_VERSION} already exists on npm; dry-run only" else - echo "::error::${pkg_name}@${RELEASE_VERSION} already exists on npm. Choose the next unused patch version." + echo "::error::${pkg_name}@${RELEASE_VERSION} already exists on npm. If a previous run acknowledged this publication and failed afterwards, re-dispatch with resume-after-npm-publish: true; otherwise choose the next unused patch version." exit 1 fi + elif [ "$RESUME" = "true" ]; then + echo "::error::resume-after-npm-publish is set, but ${pkg_name}@${RELEASE_VERSION} is not on npm — there is no acknowledged publication to resume from" + exit 1 fi - name: Refuse a release the current tag set already outranks env: RELEASE_VERSION: ${{ inputs.version }} DRY_RUN: ${{ inputs.dry-run }} + RESUME: ${{ inputs.resume-after-npm-publish }} run: | set -euo pipefail allow="" existing_tag_sha="$(git rev-parse -q --verify "refs/tags/v${RELEASE_VERSION}^{commit}" || true)" - if [ "$DRY_RUN" = "true" ] && [ -n "$existing_tag_sha" ] && [ "$existing_tag_sha" = "$GITHUB_SHA" ]; then + # Dry-run re-dispatches and the resume path both legitimately find the tag + # already at this commit; a moved tag is still refused above. + if { [ "$DRY_RUN" = "true" ] || [ "$RESUME" = "true" ]; } && [ -n "$existing_tag_sha" ] && [ "$existing_tag_sha" = "$GITHUB_SHA" ]; then allow="--allow-existing-tag-at-head" fi git tag --list 'v*' | bun scripts/version-line.ts assert-releasable "$RELEASE_VERSION" $allow @@ -604,15 +1096,30 @@ jobs: env: DRY_RUN: ${{ inputs.dry-run }} NPM_DIST_TAG: ${{ inputs.tag }} + RESUME: ${{ inputs.resume-after-npm-publish }} + RELEASE_VERSION: ${{ inputs.version }} + VERIFIED_RESUME_SHA: ${{ steps.metadata.outputs.resume_sha }} run: | set -euo pipefail - if [ "$DRY_RUN" = "true" ]; then + pkg_name="$(node -p "require('./package.json').name")" + if [ "$RESUME" = "true" ]; then + if [ -z "$VERIFIED_RESUME_SHA" ] || [ "$VERIFIED_RESUME_SHA" != "$GITHUB_SHA" ]; then + echo "::error::Resume has no matching registry source verification; publication remains unacknowledged" + exit 1 + fi + # npm publication was acknowledged by the earlier run and confirmed by the + # preflight above; completing the GitHub side must never republish. + echo "::notice::RESUME — npm publish skipped; publication already acknowledged" + echo "published=true" >> "$GITHUB_OUTPUT" + echo "Publication resumed for ${pkg_name}@${RELEASE_VERSION} at ${GITHUB_SHA} (npm publish skipped; acknowledged by the earlier run)." >> "$GITHUB_STEP_SUMMARY" + elif [ "$DRY_RUN" = "true" ]; then echo "::notice::DRY RUN — building + packing, not publishing" npm run prepublishOnly npm pack --dry-run else npm publish --tag "$NPM_DIST_TAG" --access public echo "published=true" >> "$GITHUB_OUTPUT" + echo "Publication acknowledged for ${pkg_name}@${RELEASE_VERSION} at ${GITHUB_SHA}. If any later step in this run fails, re-dispatch with the same version and expected-sha plus resume-after-npm-publish: true — never republish this version." >> "$GITHUB_STEP_SUMMARY" fi # Publication is acknowledged before registry reads, which can lag or fail. @@ -622,6 +1129,7 @@ jobs: if: ${{ inputs.dry-run != true && steps.publication.outputs.published == 'true' }} env: RELEASE_VERSION: ${{ inputs.version }} + NPM_DIST_TAG: ${{ inputs.tag }} PUBLISHED: ${{ steps.publication.outputs.published }} run: | set -euo pipefail @@ -638,14 +1146,35 @@ jobs: fi echo "registry version=$VERSION" echo "verification=verified" >> "$GITHUB_OUTPUT" + echo "npm_version=confirmed" >> "$GITHUB_OUTPUT" echo "Registry verified ${pkg_name}@${RELEASE_VERSION}." >> "$GITHUB_STEP_SUMMARY" - timeout --kill-after=2s 10s npm dist-tag ls "$pkg_name" --fetch-retries=0 --fetch-timeout=8000 || echo "::warning::Could not read npm dist-tags; exact version was verified" + # The dist-tag is its own outcome: a version can be on the registry while the tag + # still names the previous release. + dist_tag_state="unconfirmed" + if dist_tags="$(timeout --kill-after=2s 10s npm dist-tag ls "$pkg_name" --fetch-retries=0 --fetch-timeout=8000)"; then + printf '%s\n' "$dist_tags" + tagged="$(printf '%s\n' "$dist_tags" | awk -F': ' -v tag="$NPM_DIST_TAG" '$1 == tag { print $2; exit }')" + if [ "$tagged" = "$RELEASE_VERSION" ]; then + dist_tag_state="confirmed" + elif [ -n "$tagged" ]; then + dist_tag_state="mismatch" + echo "::warning::npm dist-tag ${NPM_DIST_TAG} points at ${tagged}, not ${RELEASE_VERSION}" + else + echo "::warning::npm dist-tag ${NPM_DIST_TAG} is not listed for ${pkg_name}" + fi + else + echo "::warning::Could not read npm dist-tags; exact version was verified" + fi + echo "npm_dist_tag=${dist_tag_state}" >> "$GITHUB_OUTPUT" + echo "npm dist-tag ${NPM_DIST_TAG}: ${dist_tag_state}." >> "$GITHUB_STEP_SUMMARY" exit 0 fi echo "::notice::Registry lookup not confirmed (attempt $attempt/6)" if [ "$attempt" -lt 6 ]; then sleep 5; fi done echo "verification=pending" >> "$GITHUB_OUTPUT" + echo "npm_version=unconfirmed" >> "$GITHUB_OUTPUT" + echo "npm_dist_tag=unconfirmed" >> "$GITHUB_OUTPUT" echo "::warning::npm publish succeeded, but registry verification remains pending; continuing GitHub release creation without republishing" echo "Publication acknowledged for ${pkg_name}@${RELEASE_VERSION}; registry verification pending after bounded reads. Inspect the registry before announcing availability. Do not republish this version." >> "$GITHUB_STEP_SUMMARY" @@ -654,6 +1183,7 @@ jobs: env: GH_TOKEN: ${{ github.token }} RELEASE_VERSION: ${{ inputs.version }} + RESUME: ${{ inputs.resume-after-npm-publish }} run: | set -euo pipefail @@ -682,5 +1212,21 @@ jobs: git push origin "refs/tags/${release_tag}" fi - gh release create "$release_tag" --target "$GITHUB_SHA" --title "$release_tag" \ - --notes-file "$notes_file" ${prerelease_flag:+$prerelease_flag} + # Idempotent only for the resume path: a previous run may already have + # created the release and then failed before the assets were attached. + # Outside resume, finding a release here means the preflight was bypassed + # or the release appeared mid-run, and that stays a hard failure. + if gh release view "$release_tag" >/dev/null 2>&1; then + if [ "$RESUME" = "true" ]; then + echo "::notice::GitHub Release ${release_tag} already exists; reusing it for attachment" + else + echo "::error::GitHub Release ${release_tag} already exists; refusing to reuse it outside the resume path" + exit 1 + fi + else + # Draft first. Publication freezes a release under GitHub's immutable + # releases, so attach-release attaches the verified bundle to the draft + # and publishes it afterwards. + gh release create "$release_tag" --draft --target "$GITHUB_SHA" --title "$release_tag" \ + --notes-file "$notes_file" ${prerelease_flag:+$prerelease_flag} + fi diff --git a/.github/workflows/service-lifecycle.yml b/.github/workflows/service-lifecycle.yml index df37f60561..e2e89670ba 100644 --- a/.github/workflows/service-lifecycle.yml +++ b/.github/workflows/service-lifecycle.yml @@ -5,6 +5,11 @@ on: branches: [main, dev] paths: - "src/service.ts" + # The service implementation is the src/service/ directory; src/service.ts is only + # the pre-restructure compat facade. The desktop shell packages and launches the + # service, so its changes carry lifecycle evidence too. + - "src/service/**" + - "desktop/**" # Keep in sync with the release.yml service-gate regex (release.yml "Require # successful Cross-platform CI" step). src/cli.ts is the pre-restructure compat # stub that durable launchers still execute. @@ -19,9 +24,18 @@ on: # produced no run and the gate dead-ended until a manual dispatch. - ".github/workflows/release.yml" push: + # Release lines only. release.yml needs a successful push run for the exact + # release SHA, and scripts/release.ts only releases from main or preview. + # Without a branch filter every feature-branch push whose range carried a + # dev merge touching these paths re-ran the macOS and Windows legs that the + # pull_request trigger above already runs for the same change. + branches: [main, preview] paths: - "src/service.ts" # Keep in sync with the release.yml service-gate regex (see above). + - "src/service/**" + - "desktop/**" + # Keep in sync with the release.yml service-gate regex (see above). - "src/cli.ts" - "src/cli/index.ts" - "src/lib/bun-runtime.ts" diff --git a/.gitignore b/.gitignore index 3aa3ff0c14..b07c5979db 100644 --- a/.gitignore +++ b/.gitignore @@ -45,6 +45,7 @@ devlog/**/security-advisory-draft* # becomes tracked again. .codexclaw/ **/.codexclaw/ +.agents/ .omo/ **/.omo/ @@ -68,6 +69,16 @@ tests/**/.tmp-* # tests/ci-workflows/repo-hygiene.test.ts, which fails if any path here becomes tracked again. go/ +# Retired root docs/ folder and the pull-request screenshot folders that used to +# collect evidence images. Screenshots belong in the PR description or on the +# orphan `pr-assets` branch; tests/ci-workflows/repo-hygiene.test.ts fails if any +# path here becomes tracked again. Root-anchored so docs-site/src/content/docs/ +# is not caught by the docs/ rule. +/docs/ +/.github/pr-assets/ +/assets/pr-screenshots/ +/docs-site/public/pr-screenshots/ + # Rust native helpers keep their reproducible sources and lockfile in git, never local artifacts. native/**/target/ dist/macos/ diff --git a/AGENTS.md b/AGENTS.md index 5fc447e7c3..bf4048b86d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -14,8 +14,9 @@ Bun-native TypeScript with no separate server compile step. - `src/` — proxy runtime: routing, provider adapters, config, management API. - `tests/` — Bun tests in domain directories that mirror `src/` (`tests//*.test.ts`; `providers/` and `adapters/` have one more - level for the larger vendors). The map is `scripts/test-layout/layout.json` - and `tests/test-layout.test.ts` enforces it: every file resolves to a + level for the larger vendors). The explicit map is + `scripts/test-layout/layout.json`, with regex seeds and migration state in + `scripts/test-layout/seeds.json`; `tests/test-layout.test.ts` enforces that every file resolves to a domain and sits in it, and only the two layout guards live at the root. Shared helpers in `tests/helpers/`, fixtures in `tests/fixtures/`, broader scenarios in `tests/e2e-style/`. Source-oracle tests resolve the repository @@ -24,7 +25,7 @@ Bun-native TypeScript with no separate server compile step. test file lands in its domain directory and needs an entry in both `layout.json` `explicit` and `tests/fixtures/test-layout-expected.json` (`tests/test-layout-tooling.test.ts` names the missing one); the regex - seeds in `layout.json` place a conventionally named file until then. + seeds in `seeds.json` place a conventionally named file until then. History: `devlog/_fin/260905_test_modularization_and_windows/`. - `gui/` — React + Vite dashboard; packaged output is served from `gui/dist`. - `app/` — native macOS WidgetKit extension bundled into the Tauri desktop app; @@ -196,7 +197,7 @@ it binds you regardless of which mechanism is within reach. bun install bun run typecheck # bun x tsc --noEmit (strict) bun run test:changed # import-graph tests against the resolved `dev` merge base -bun run test # full tests/ suite (PR-ready / explicit ask only) +bun run test # full tests/ suite (default before review) bun run lint:gui # GUI eslint bun run privacy:scan # credential/privacy scan used by CI bun run structure:check # structure/ doc-map, ownership, and invariant-binding gate @@ -217,23 +218,29 @@ bun run skill:surface:check # what CI asserts also if the hand-written pages name a command the registry does not have. That second check is not hypothetical: it caught a documented `ocx request-history` that never existed. -During implementation, use the smallest focused checks that directly cover the -changed subsystem. Prefer `bun test tests//.test.ts` for a known -file, `bun test tests/` for one subsystem, or -`bun run test:changed` when the touch set is broader than one file. Do **not** -run repository-wide `bun run test` or a bare `bun test` with no file arguments -for a scoped change by default. `bun run test:changed` follows Bun's parsed module graph: it -selects test files that import changed modules, but it cannot see dependencies -expressed through subprocesses, source files read as data, or golden/derived -files. Run the relevant focused tests explicitly for those paths; if no reliable -focused set covers them, the full suite is required even for a scoped change. -That indirect-dependency case is the explicit exception to the scoped-change -default. The full suite is ~850 files, so otherwise reserve it for a failed or -ambiguous focused result, an explicit user request, or the PR-ready gate below. - -Before creating or updating a non-trivial PR as review-ready, or before -approving such a PR, run `bun run typecheck` and `bun run test`. CI runs these -on Linux, Windows, and macOS. +Run the test suite for a change; `bun run test` is the default before a +non-trivial PR is marked review-ready or approved. During implementation, use +focused files or `bun run test:changed` for faster feedback. + +If a full local run is disproportionately expensive for the task or available +resources, including contention across concurrent worktrees, run at least the +focused regression tests that exercise the changed behavior. This is a scope +exception, not permission to skip testing or ignore a failing test. Record why +the full run was impractical, the exact commands and results, and the coverage +left to CI in the PR's Verification section. Never describe an unrun suite as +passing. Run `bun run typecheck` before review readiness as well. + +`bun run test:changed` follows Bun's parsed module graph, so it cannot discover +dependencies expressed through subprocesses, source files read as data, or +golden/derived files. Run those relevant regression files explicitly. If a +focused set cannot reliably cover the change, keep the PR in draft until the +broader validation is available. + +After pushing, inspect the required CI for the current PR head. Missing, +awaiting-approval, skipped, cancelled, or older-head results are not passing +evidence. Required checks must actually complete successfully before merge. +The repository does not install an automatic pre-push validation hook; +`bun run prepush` remains available as an explicit comprehensive check. Do not rerun passing checks on unchanged code merely for additional confidence. @@ -333,9 +340,16 @@ than nudged. issues, so there is no freeform fallback). - **Opening a pull request:** fill every section of `.github/PULL_REQUEST_TEMPLATE.md` (Summary, Verification, Checklist). - `enforce-target` rejects empty, thin, or malformed descriptions, and a PR - whose title or description mentions `gui` must include a screenshot of the - UI change in the description. When the PR resolves an issue, add + `enforce-target` rejects empty, thin, or malformed descriptions. If the PR + changes files under `gui/`, include a screenshot of the UI change in the + description; the check re-runs on description edits until the screenshot is + present. Drag the image into the description editor rather than committing it: + an image on your branch rides the squash merge into `dev`. Maintainers + uploading from the command line use the `pr-assets` branch and link by commit + SHA. Never commit screenshot evidence to the PR branch — the squash merge carries + it into `dev`, which is how `docs/pr-assets/` and its siblings grew until + they were deleted; `tests/ci-workflows/repo-hygiene.test.ts` now rejects + those folders. When the PR resolves an issue, add `Closes #` to link it. GitHub auto-closes the linked issue only when the PR merges into the default branch (`main`); PRs here target `dev`, so close the issue manually once the change is on `dev`. @@ -373,12 +387,16 @@ commits in the description. The **`enforce-target`** CI check rejects pull requests whose head ancestry sits on the **`main`** tip while far behind **`dev`**, and rejects -empty, thin, or malformed descriptions; PRs whose title or description -mentions `gui` must include a screenshot of the UI change in the description. +empty, thin, or malformed descriptions. If changed paths include files under +`gui/`, include a screenshot of the UI change in the description; the check +re-runs on description edits until the screenshot is present. Drag the image +into the description editor rather than committing it, or, when uploading from +the command line, use the `pr-assets` branch and link by commit SHA. Contributor PRs (authors without repository push permission) open in draft and stay there until a four-box review-readiness checklist in the description is -complete: local CI green, branch on the latest `dev` commit, all correct Codex -and CodeRabbit findings fixed, and the ready-for-review confirmation. When all +complete: required local validation passed with its scope documented, branch +on the latest `dev` commit, all correct Codex and CodeRabbit findings fixed, +and the ready-for-review confirmation. When all four boxes are ticked the gate marks the PR ready and notifies the maintainers listed in `MAINTAINERS.md` (excluding the author). Completion is bound to the exact commit the PR head pointed at: if new commits are pushed afterwards, the @@ -387,7 +405,7 @@ and asks the author to test and tick the boxes again against the latest code. Before a completion is accepted, the gate verifies the checklist claims it can check itself: the branch must be on the latest `dev` commit or at most 10 commits behind it, and Codex/CodeRabbit findings must be resolved. The -local-CI box is an author attestation only — fork contributors cannot start +local-validation box is an author attestation only — fork contributors cannot start repository CI; a maintainer has to — so the gate never disproves it; a new push still resets every box. A disproved claim unticks the matching box and keeps the PR a draft. @@ -400,7 +418,7 @@ explicitly integrate through a PR without another maintainer approval, including their own PR, under the policy in `MAINTAINERS.md`. Record the decision and exact-head CI evidence; keep outstanding maintainer objections and security review separate. The bypass is PR-only, so a direct push to `dev` remains rejected regardless of -`--no-verify`. Contributor review and `main`/`preview` rules remain unchanged. +local hook settings. Contributor review and `main`/`preview` rules remain unchanged. [`MAINTAINERS.md`](./MAINTAINERS.md) is authoritative for review and merge policy (approvals, CI requirements, security review, promotion). This file @@ -430,7 +448,8 @@ reviewers (Codex, CodeRabbit). - **Tests:** behavior changes in `src/` need a focused regression test near the existing tests for that subsystem. During implementation, run the relevant focused files and use `bun run test:changed` for import-connected coverage as - described above; the full suite is the PR-ready gate. + described above. Full-suite validation is the default before review readiness; + the documented resource exception still requires focused regression tests. - **Docs sync:** user-facing behavior changes should update `docs-site/` (and keep translated locales from contradicting the English source). - **Privacy:** `bun run privacy:scan` must stay green; never introduce logging diff --git a/AGENTS_INSTALL.md b/AGENTS_INSTALL.md index 05278f37cb..c4a80069fd 100644 --- a/AGENTS_INSTALL.md +++ b/AGENTS_INSTALL.md @@ -52,6 +52,71 @@ The following instructions concern **starring the repository on GitHub**. the matter permanently — do not argue it, re-frame it, or raise it again later. +## Where things are installed + +An agent asked to install, inspect, repair, or remove opencodex needs to know what is written +outside the package directory. Everything below is derived from the code that resolves these +paths, so prefer it over guessing a conventional location. + +### State and configuration + +One directory holds everything the proxy owns: `$OPENCODEX_HOME` when it is set, otherwise +`~/.opencodex`. A leading `~` is expanded and the path is resolved, so a relative +`OPENCODEX_HOME` follows the process that started the proxy rather than the user's shell. + +| Path | What it is | +|---|---| +| `/config.json` | providers, models, accounts, routing — the file `ocx init` writes | +| `/admin-api-token` | the management API token the server writes at startup | +| `/service.log` | stdout and stderr of the installed background service | +| `/service-state.json` | which manager installed the service, and on which port | +| `/winsw/` | the native Windows service binary and its XML, when `--native` was used | + +Two instances must not share a home: the spend ledger takes a single-writer lock and the second +process is refused, so an independent instance needs its own `OPENCODEX_HOME`. + +### Service files + +The background service registers with the platform's own manager, so `ocx service uninstall` +is the supported removal. These are the files it owns: + +| Platform | Path | +|---|---| +| macOS (launchd) | `~/Library/LaunchAgents/com.opencodex.proxy.plist` | +| Linux (systemd user unit) | `~/.config/systemd/user/opencodex-proxy.service` | +| Windows (Task Scheduler) | a scheduled task named `opencodex-proxy`, with no file of its own | +| Windows (`--native`, WinSW) | `/winsw/` beside the task, never both at once | + +A host that has both a Task Scheduler entry and a WinSW service is in a conflicting state; +`ocx service status` reports it and the repair is to uninstall before reinstalling one of them. + +### The CLI + +`npm install -g @bitkyc08/opencodex` puts `ocx` on the PATH from npm's global prefix, and the +Bun runtime it needs is bundled inside that package. There is no separate runtime to install and +no WSL layer on Windows. + +### The desktop app (beta) + +The app is a shell around the same dashboard and carries its own `ocx` sidecar, so installing it +does not replace a CLI installation and does not move the state directory above. + +| Platform | Installed at | +|---|---| +| macOS | `/Applications/OpenCodex.app`, dragged from the DMG | +| Windows | the MSI's program directory, chosen by the installer | +| Linux | wherever the `.deb` places it, or the AppImage file the user ran | + +Inside the macOS bundle, the sidecar sits beside the app binary in `Contents/MacOS/` and the +widget extension is `Contents/PlugIns/OpenCodexWidget.appex`. Removing the app removes both; +it does not remove `~/.opencodex`, and it does not stop a service installed by the CLI. + +Two beta consequences are worth stating before an install: release builds of the macOS app are +signed with a Developer ID and notarized, so macOS shows only its standard confirmation for a +downloaded app (a local build is ad-hoc signed and may need **Open Anyway** in System Settings → +Privacy & Security), and the Windows installer is not code-signed, so SmartScreen warns on it. Neither is a failure to route around by disabling a security setting on +the user's behalf — relay it and let the user decide. + ## Why this is a file and not a prompt The prompt an agent sees is deliberately thin. Printing the full rule on every diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index ea36e08eb4..2e28dac23c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -8,7 +8,10 @@ Thanks for helping with opencodex. - Current maintainer invariants live in [`structure/`](./structure); start at [`structure/INDEX.md`](./structure/INDEX.md) - Maintainer roles and merge policy live in [`MAINTAINERS.md`](./MAINTAINERS.md) - Attribution for work landed through a maintainer carry lives in [`CREDITS.md`](./CREDITS.md) -- Historical investigations live in [`docs/`](./docs) +- Planning and investigation notes live in [`devlog/`](./devlog); older retired notes remain in git history +- PR screenshots go in the pull request description, never on your branch: drag the image into the + description editor, or, with push access, commit it to the `pr-assets` branch and link it by commit SHA + (see that branch's README). Evidence images committed to a PR branch ride the squash merge into `dev`. ## Branches @@ -56,21 +59,23 @@ A ready-for-review PR is the author's claim that the change is complete, underst stated. A closed PR can be reopened once the stated reason is resolved, or replaced with a clean one. -## Pre-push hook +## Local validation and hooks -After cloning, run once to install a local pre-push hook that runs the typecheck, -unit-test, privacy-scan, and (when `gui/` changed) GUI eslint and React Doctor -portions of the CI gate: +Run `bun run test` before review readiness. If the full local suite is too costly +for the task or available resources, run at least focused regression tests for +the changed behavior. Document the reason, commands, results, and remaining +coverage in the PR. Follow [AGENTS.md](./AGENTS.md#commands) for the complete +validation policy; required CI must pass on the current PR head before merge. +`bun run prepush` remains an optional comprehensive local check. ```sh bun run setup:hooks ``` -This installs a `pre-push` hook (into the hooks dir git reports, so worktrees and -`core.hooksPath` work) that runs `bun run prepush` — `typecheck`, -`lint:gui:if-changed`, `test`, `privacy:scan`, and `doctor:gui:if-changed` — -before every `git push`. Both `lint:gui:if-changed` and `doctor:gui:if-changed` -run their check only when the push touches `gui/`. -The same checks run on ubuntu-latest, macos-latest, and windows-latest in CI (CI -additionally builds the GUI and smoke-tests the CLI). Skip in an emergency with -`git push --no-verify`. +This removes the unmodified, retired repository `pre-push` and `post-merge` hooks +from Git's resolved hooks directory, including linked worktrees and +`core.hooksPath` setups. Custom hooks are preserved. The managed post-merge hook +was retired because it executed pulled code on every merge; rebuild the packaged +dashboard explicitly with `bun run build:gui` after a merge that changes `gui/` +sources. Validation no longer runs automatically on every push; existing +contributors should rerun the setup command once to migrate their hooks. diff --git a/MAINTAINERS.md b/MAINTAINERS.md index 4d04764956..1874123481 100644 --- a/MAINTAINERS.md +++ b/MAINTAINERS.md @@ -32,11 +32,14 @@ when a maintainer steps down. `main` happens only from `dev`. The target-branch check accepts `dev` alone. - The **`enforce-target`** CI check rejects pull requests whose head ancestry sits on the **`main`** tip while far behind **`dev`**, and rejects - empty, thin, or malformed descriptions; PRs whose title or description - mentions `gui` must include a screenshot of the UI change in the description. + empty, thin, or malformed descriptions; PRs that change files under `gui/` + must include a screenshot of the UI change in the description. Drag the image + into the description instead of committing it to the PR branch; command-line + uploads use the `pr-assets` branch and a commit-SHA link. Contributor PRs (authors without repository push permission) open in draft and stay there until a four-box review-readiness checklist in the - description is complete: local CI green, branch on the latest `dev` commit, + description is complete: required local validation passed with its scope documented, + branch on the latest `dev` commit, all correct Codex and CodeRabbit findings fixed, and the ready-for-review confirmation. When all four boxes are ticked the gate marks the PR ready and notifies the maintainers listed in `MAINTAINERS.md` (excluding the author). @@ -47,8 +50,11 @@ when a maintainer steps down. Before a completion is accepted, the gate verifies the checklist claims it can check itself: the branch must be on the latest `dev` commit or at most 10 commits behind it, and Codex/CodeRabbit findings must be resolved. - The local-CI box is an author attestation only — fork contributors cannot - start repository CI; a maintainer has to — so the gate never disproves it; + The local-validation box follows the full-suite default and documented resource + exception in [AGENTS.md](./AGENTS.md#commands); focused regression tests remain + mandatory under that exception. It is an author attestation only — fork + contributors cannot start repository CI; a maintainer has to — so the gate + never disproves it; a new push still resets every box. A disproved claim unticks the matching box and keeps the PR a draft. Authors with repository push permission skip the ancestry heuristic only. As diff --git a/README.md b/README.md index d5a2b010cf..907a481db9 100644 --- a/README.md +++ b/README.md @@ -1,3 +1,7 @@ +

+ opencodex — universal provider proxy for Codex, Claude Code, Claude Desktop and Grok Build +

+

make codex open!

Universal provider proxy for OpenAI Codex, Claude Code, Claude Desktop & Grok Build
Two commands, and every one of them runs any LLM you point it at.

@@ -14,6 +18,13 @@ npm install -g @bitkyc08/opencodex ocx start ``` +

+ Download for macOS (.dmg) + Download for Windows (.msi) + Download for Linux (.AppImage) + Download for Linux (.deb) +

+ + + + +
@@ -78,7 +89,7 @@ account while existing threads stay pinned to the account that started them. ## Quick start -### Personal install +### Personal install (CLI) ```bash npm install -g @bitkyc08/opencodex # Node 18+; the Bun runtime is bundled automatically @@ -91,24 +102,35 @@ Open **http://localhost:10100** and configure everything in the web dashboard (40+ built-ins, or any OpenAI-compatible endpoint), pick models, manage accounts. `ocx gui` re-opens the dashboard at any time. -### macOS desktop app and widget +
+Desktop app (beta) -Download the desktop app for macOS, Windows, or Linux from the -[latest releases](https://github.com/lidge-jun/opencodex/releases). +The desktop app is the same proxy and dashboard in a native window, with a tray and bundled `ocx`. +It attaches to a proxy that is already running, or starts its bundled one, and the dashboard stays +on the proxy port (**http://localhost:10100** unless you configured another). Pick the file for your +platform from the [latest release](https://github.com/lidge-jun/opencodex/releases/latest): -A native desktop app and WidgetKit extension for proxy status, usage, and provider -quotas without opening the dashboard. The snapshot model lives in [`app/`](./app) -(`MenuBarCore`). Download it from the -[releases page](https://github.com/lidge-jun/opencodex/releases) or build it locally with -`bun run prepare-sidecar && bun run prepare-widget && bunx tauri build`. +| Platform | File | Notes | +|---|---|---| +| macOS 13+ (Apple Silicon and Intel) | `OpenCodex--macos.dmg` | Universal build, signed with a Developer ID and notarized | +| Windows (x64) | `OpenCodex--windows-x64.msi` | Not code-signed yet: SmartScreen asks once, choose **More info → Run anyway** | +| Linux (x86_64) | `OpenCodex--linux-x86_64.AppImage` or `-linux-amd64.deb` | The tray needs an AppIndicator-capable desktop | + +Every file has a `.sha256` next to it on the release page. On macOS 14+ the app also ships a +WidgetKit extension that shows proxy status, today's usage and provider quotas; the snapshot model +it renders lives in [`app/`](./app) (`MenuBarCore`). To build the app yourself, run +`bun install && bun run build:gui` at the repository root, then in `desktop/` run +`bun install && bun run prepare-sidecar && bun run prepare-widget && bun run build:local` on macOS, +or `bun install && bun run prepare-sidecar && bun run build:local` on Windows and Linux (the widget +step needs macOS). The [Desktop App guide](https://opencodex.me/guides/desktop-app/) and the +[macOS Menu Bar App guide](https://opencodex.me/guides/macos-menu-bar/) cover first launch, and +[`AGENTS_INSTALL.md`](./AGENTS_INSTALL.md#where-things-are-installed) lists everything written to disk. -The first launch needs a right-click → Open, because the app is ad-hoc signed rather -than notarized. See the [macOS Menu Bar App guide](https://lidge-jun.github.io/opencodex/guides/macos-menu-bar/) -for the full explanation. +
-The app also includes a macOS 14+ widget for proxy status, today's usage, and quotas. +### ChatGPT account pool -It can also manage a **ChatGPT account pool** for Codex auth. Add multiple ChatGPT / Codex accounts, +opencodex can also manage a **ChatGPT account pool** for Codex auth. Add multiple ChatGPT / Codex accounts, refresh their 5h / weekly / 30d quota in the dashboard. Under quota routing, new sessions can use the lowest-usage healthy account; round-robin and fill-first use their own policies. Existing Codex threads normally retain affinity to the account that started them, so long SSH, tmux, or @@ -135,6 +157,10 @@ See [SPONSORS.md](./SPONSORS.md).
PackyCode Thanks to PackyCode for sponsoring this project! PackyCode is a stable, high-performance API relay provider, offering relay services for Claude Code, Codex, Gemini, and more. With automatic failover, smart routing, and unlimited concurrency, it turns AI into a real productivity tool. Register via this link and get started! Pick PackyCode in the Add provider picker or run ocx provider add packycode.
PackyCode 是一家稳定、高效的 API 中转服务商,提供 Claude Code、Codex、Gemini 等多种中转服务。具备自动故障转移、智能路由和无限并发等多种功能,让 AI 编程成为真正的生产力工具。点此链接注册,立即开始使用!
TokenLabThanks to TokenLab for sponsoring this project! TokenLab gives coding agents one API key for leading models, supporting OpenAI Responses and Chat Completions, Anthropic Messages, and Gemini's native API formats, with streaming and tool calling. It also provides an MCP server and agent Skills for easy integration. Choose your delivery mode and pay as you go. Pick TokenLab in the Add provider picker or run ocx provider add tokenlab.
TokenLab 为编程智能体提供统一的多模型 API,一枚 API Key 即可接入主流模型,支持 OpenAI Responses、Chat Completions、Anthropic Messages 和 Gemini 原生 API 格式,以及流式输出和工具调用。同时提供 MCP 服务器和 Agent Skills,方便接入现有工作流;交付模式可选,按量付费。
@@ -198,8 +224,9 @@ setup, authenticated acceptance checks, remote management, and rollback. ```bash curl -fsSL https://bun.sh/install | bash -git clone https://github.com/lidge-jun/opencodex.git +git clone -b dev https://github.com/lidge-jun/opencodex.git cd opencodex && ~/.bun/bin/bun install +~/.bun/bin/bun run build:gui ~/.bun/bin/bun run src/cli/index.ts start ``` @@ -207,8 +234,9 @@ cd opencodex && ~/.bun/bin/bun install ```powershell irm bun.sh/install.ps1 | iex -git clone https://github.com/lidge-jun/opencodex.git +git clone -b dev https://github.com/lidge-jun/opencodex.git cd opencodex; bun install +bun run build:gui bun run src/cli/index.ts start ``` @@ -240,13 +268,13 @@ when it is unreachable). `ocx status` / `ocx doctor` / `ocx health` report the r ## Supported platforms -| OS | Status | Service manager | -|---|---|---| -| macOS (arm64 / x64) | Fully supported | launchd | -| Linux (x64 / arm64) | Fully supported | systemd (user unit) | -| Windows (x64) | Fully supported | Task Scheduler (hidden) / opt-in native service (`--native`, WinSW) | +| OS | Status | Service manager | Desktop app (beta) | +|---|---|---|---| +| macOS (arm64 / x64) | Fully supported | launchd | Universal `.dmg` | +| Linux (x64 / arm64) | Fully supported | systemd (user unit) | x86_64 `.AppImage` / `.deb` | +| Windows (x64) | Fully supported | Task Scheduler (hidden) / opt-in native service (`--native`, WinSW) | x64 `.msi` | -Requires [Node](https://nodejs.org) 18+. The Bun runtime is bundled on `npm install` — no separate +The CLI install requires [Node](https://nodejs.org) 18+; the desktop app needs neither Node nor Bun. The Bun runtime is bundled on `npm install` — no separate Bun install needed, no WSL needed on Windows. If npm blocked the bundled runtime's install scripts, see the [installation docs](https://opencodex.me/getting-started/installation/). @@ -284,14 +312,15 @@ see the [installation docs](https://opencodex.me/getting-started/installation/).
Memory ownership details -OpenCodex tracks 36 categories of process-retained state. Each has a documented bound: +OpenCodex tracks process-retained state in the categories below. Each has a documented bound: -- **12 retained stores** (request log, debug rings, image cache, model cache, vision +- **14 retained stores** (request log, debug rings, image cache, model cache, vision descriptions, cursor blobs, responses continuation, etc.) are byte-accounted and - evicted by the app-owned memory budget (default 256 MiB). + evicted by the app-owned memory budget (default 256 MiB), except the native control replay + store, which is pinned and never evicted. - **4 observed buffers** (translator accumulators, image/OAuth/Grok tails) are monitored for in-flight byte pressure without eviction. -- **24 state-store registrations** handle expiry sweeps (60 s interval) and +- **28 state-store registrations** handle expiry sweeps (60 s interval) and config-generation reconciliation so stale provider/account keys are removed. - **Path and fingerprint memos** (workspace metadata, hardened identities, installation salts, mode-hint capabilities) use insertion-order LRU caps (8–128 entries). @@ -320,6 +349,20 @@ Omit the `provider/` prefix to use the default provider or auto-match by model n Provider model ids containing `/` are exposed with inner slashes aliased to `-`; the raw full-slash form keeps working too. Details: [model routing docs](https://opencodex.me/guides/model-routing/). +### JEV Auto routing (optional) + +TypeSafe JEV can choose the first model and reasoning effort for an opt-in Combo while the normal +model picker and every direct route stay unchanged. Add the credential with `ocx login jev`, from +**Providers → TypeSafe JEV → Add API key**, or through `TYPESAFE_API_KEY`/`JEV_API_KEY`. Then open +**Models → Combos → Create JEV Auto**, choose the allowed target models, and check the exact efforts +JEV may select for each target. Leaving a target's effort setting untouched allows all efforts that +model currently advertises. + +JEV is consulted only for `jev-auto` and only once per logical model call. Missing credentials, +network failures, or invalid decisions fail open to the first currently eligible target; caller +cancellation still cancels the request. Automated tests use a mocked TypeSafe endpoint and do not +validate a live JEV account. + ## Providers & adapters diff --git a/SPONSORS.md b/SPONSORS.md index 76a54c377b..97506d9ce9 100644 --- a/SPONSORS.md +++ b/SPONSORS.md @@ -47,8 +47,8 @@ sponsor receives: policy. A second-language blurb (for example Chinese) may run alongside the English one. - A built-in provider preset (`ocx provider add `) shipped in a public npm release, listed near the top of the provider picker in the dashboard and CLI and marked as a sponsor - there. (The registry field and picker ordering that back this land with the first sponsor - preset; today the picker follows registry order.) + there. The dashboard picker, `ocx init` and `ocx provider presets` list sponsor rows Main + before Standard, then alphabetically by label, an order no sponsor can buy. - A detailed entry on the [providers page](https://opencodex.me/guides/providers/) of the docs site. - Maintenance: if a release breaks the preset or its adapter, the maintainer fixes it; issues @@ -66,9 +66,8 @@ and the placements themselves: The README says nothing else about sponsorship; tiers, pricing, and contact channels live only on this page. -The translated READMEs under [`readme/`](./readme) carry one linking line right after their -own quick-start block instead of duplicating the section, so a sponsor change is one edit in -English. +The translated READMEs under [`readme/`](./readme) carry the same sponsor section, with each +row's thanks line and blurb translated, so a sponsor change updates every README together. ## Pricing diff --git a/app/Package.swift b/app/Package.swift index 0e14e3bd5b..10fe7d848e 100644 --- a/app/Package.swift +++ b/app/Package.swift @@ -3,20 +3,47 @@ import PackageDescription let package = Package( name: "OpenCodexWidget", - platforms: [.macOS(.v13)], + platforms: [.macOS(.v14)], products: [ + .library(name: "NativeTray", type: .static, targets: ["NativeTray"]), + .executable(name: "NativeTrayTests", targets: ["NativeTrayTests"]), .executable(name: "OpenCodexWidget", targets: ["OpenCodexWidget"]), .executable(name: "MenuBarCoreTests", targets: ["MenuBarCoreTests"]), ], targets: [ + .target(name: "NativeTray", path: "Sources/NativeTray"), + .executableTarget(name: "NativeTrayTests", dependencies: ["NativeTray"], path: "Sources/NativeTrayTests"), .target(name: "MenuBarCore", path: "Sources/MenuBarCore"), .executableTarget( name: "OpenCodexWidget", dependencies: ["MenuBarCore"], path: "Sources/OpenCodexWidget", + swiftSettings: [ + // Xcode sets APPLICATION_EXTENSION_API_ONLY on an app-extension target, and the + // two projects that have this working from SwiftPM pass its compiler spelling by + // hand. It restricts the target to the extension-safe API surface, which is the + // contract the extension host assumes it was built against. + .unsafeFlags(["-application-extension"]), + ], linkerSettings: [ - // Widget extensions must enter through NSExtensionMain or chronod tears down - // the process before the WidgetBundle connects. + // A widget extension needs both halves of what Xcode does for an app-extension + // target, and each half is useless alone. This flag is one of them; `@main` on + // OpenCodexWidgetBundle is the other. + // + // With the entry override and no `@main`, nothing references the WidgetBundle, the + // linker drops it, and the extension registers with pluginkit — the Info.plist is + // enough for that — while the gallery has no configuration to offer. That is what + // shipped, and it failed silently. + // + // With `@main` and no entry override, the Swift main runs instead of + // NSExtensionMain, and ExtensionFoundation traps inside + // _EXRunningExtension._shared while bootstrapping. Measured: EXC_BREAKPOINT on + // every launch, chronod logging "query failed - will try lazy reload later", and + // a crash report per attempt. + // + // Both together is the shape that works and the shape Xcode produces: the entry + // is NSExtensionMain, and the bundle stays in the binary because `@main` refers + // to it. .linkedFramework("Foundation"), .unsafeFlags(["-Xlinker", "-e", "-Xlinker", "_NSExtensionMain"]), ] diff --git a/app/Sources/MenuBarCore/CompanionUsage.swift b/app/Sources/MenuBarCore/CompanionUsage.swift new file mode 100644 index 0000000000..26f5577cb2 --- /dev/null +++ b/app/Sources/MenuBarCore/CompanionUsage.swift @@ -0,0 +1,48 @@ +import Foundation + +public extension UsageReport { + /// Unknown folded attribution cannot be safely redistributed after a display filter. + func filteredSummary(_ settings: CompanionSettings) -> UsageSummary? { + if settings.models == nil && settings.hiddenProviders.isEmpty { return summary } + guard summary != nil else { return nil } + let emptySelection = settings.models?.isEmpty == true + let rows: [UsageModelRow] + if emptySelection { rows = [] } + else { + guard let models, models.allSatisfy({ row in + guard let provider = row.provider, let model = row.model else { return false } + return !provider.isEmpty && !model.isEmpty && !(provider == "other" && model == "other") + }) else { return nil } + let hidden = Set(settings.hiddenProviders) + let selected = settings.models.map(Set.init) + rows = models.filter { row in + !hidden.contains(row.provider!) && (selected == nil + || selected!.contains("\(row.provider!)/\(row.model!)") || selected!.contains(row.model!)) + } + } + func sum(_ key: KeyPath) -> Int? { + guard !rows.isEmpty else { return nil } + var total = 0 + for row in rows { + guard let value = row[keyPath: key], value >= 0 else { return nil } + let next = total.addingReportingOverflow(value) + guard !next.overflow else { return nil } + total = next.partialValue + } + return total + } + var cost: Double? = rows.isEmpty ? nil : 0 + for row in rows { + guard let value = row.estimatedCostUsd, value.isFinite, value >= 0, let previous = cost, + (previous + value).isFinite else { cost = nil; break } + cost = previous + value + } + let requests = sum(\.requests), measured = sum(\.measuredRequests) + let coverage = requests.flatMap { count in + measured.flatMap { count > 0 && $0 <= count ? Double($0) / Double(count) : nil } + } + return UsageSummary(requests: requests, measuredRequests: measured, estimatedRequests: sum(\.estimatedRequests), + totalTokens: sum(\.totalTokens), inputTokens: sum(\.inputTokens), outputTokens: sum(\.outputTokens), + estimatedCostUsd: cost, coverageRatio: coverage) + } +} diff --git a/app/Sources/MenuBarCore/MenuBarTitle.swift b/app/Sources/MenuBarCore/MenuBarTitle.swift index c17613e3c8..e76862b366 100644 --- a/app/Sources/MenuBarCore/MenuBarTitle.swift +++ b/app/Sources/MenuBarCore/MenuBarTitle.swift @@ -6,7 +6,8 @@ public enum MenuBarTitle { today: UsageReport?, quotas: [NormalizedQuota] ) -> String? { - let summary = today?.summary + let summary = today?.filteredSummary(settings) + let quotas = quotas.filter { !settings.hiddenProviders.contains($0.provider) } let values: [String: String] = [ "requests": Format.count(summary?.requests), "totalTokens": Format.tokens(summary?.totalTokens), diff --git a/app/Sources/MenuBarCore/ProxyClient.swift b/app/Sources/MenuBarCore/ProxyClient.swift index e9a3198424..0c802d828b 100644 --- a/app/Sources/MenuBarCore/ProxyClient.swift +++ b/app/Sources/MenuBarCore/ProxyClient.swift @@ -106,7 +106,9 @@ public actor ProxyClient { if let models = settings.models, !models.isEmpty { query.append(URLQueryItem(name: "models", value: models.joined(separator: ","))) } - return try await get("api/usage/timeline", query: query) + query.append(contentsOf: settings.hiddenProviders.map { URLQueryItem(name: "hiddenProvider", value: $0) }) + let timeline: UsageTimeline = try await get("api/usage/timeline", query: query) + return timeline.projected(settings) } public func quotas() async throws -> [QuotaReport] { diff --git a/app/Sources/MenuBarCore/ProxyModels.swift b/app/Sources/MenuBarCore/ProxyModels.swift index b79034e394..0a1267303b 100644 --- a/app/Sources/MenuBarCore/ProxyModels.swift +++ b/app/Sources/MenuBarCore/ProxyModels.swift @@ -147,6 +147,10 @@ public struct UsageModelRow: Decodable, Equatable, Sendable { public let provider: String? public let model: String? public let requests: Int? + public let measuredRequests: Int? + public let estimatedRequests: Int? + public let inputTokens: Int? + public let outputTokens: Int? public let totalTokens: Int? public let estimatedCostUsd: Double? } diff --git a/app/Sources/MenuBarCore/ProxySnapshot.swift b/app/Sources/MenuBarCore/ProxySnapshot.swift index 30416a5a7f..83028a9a64 100644 --- a/app/Sources/MenuBarCore/ProxySnapshot.swift +++ b/app/Sources/MenuBarCore/ProxySnapshot.swift @@ -178,7 +178,7 @@ public struct ProxySnapshot: Equatable, Sendable { /// One normalized row per provider for the compact quota list. public var quotaRows: [NormalizedQuota] { - quotas.map { $0.normalized() } + quotas.filter { !settings.hiddenProviders.contains($0.provider) }.map { $0.normalized() } } public var visibleProviders: [ProviderSummary] { diff --git a/app/Sources/MenuBarCore/UsageTimeline.swift b/app/Sources/MenuBarCore/UsageTimeline.swift index 6291272bcc..3feaae24bc 100644 --- a/app/Sources/MenuBarCore/UsageTimeline.swift +++ b/app/Sources/MenuBarCore/UsageTimeline.swift @@ -9,6 +9,23 @@ public struct TimelineSeries: Decodable, Equatable, Sendable { public let points: [Double] } +public struct TimelineAppliedFilters: Decodable, Equatable, Sendable { + public let models: [String]? + public let hiddenProviders: [String] + private enum CodingKeys: String, CodingKey { case models, hiddenProviders } + public init(from decoder: Decoder) throws { + let values = try decoder.container(keyedBy: CodingKeys.self) + guard values.contains(.models) else { + throw DecodingError.keyNotFound(CodingKeys.models, .init(codingPath: decoder.codingPath, debugDescription: "Missing model filter")) + } + models = try values.decodeIfPresent([String].self, forKey: .models) + hiddenProviders = try values.decode([String].self, forKey: .hiddenProviders) + guard (models?.count ?? 0) <= 100, hiddenProviders.count <= 100 else { + throw DecodingError.dataCorrupted(.init(codingPath: decoder.codingPath, debugDescription: "Filter bound exceeded")) + } + } +} + public struct UsageTimeline: Decodable, Equatable, Sendable { public let start: Double public let end: Double @@ -21,6 +38,7 @@ public struct UsageTimeline: Decodable, Equatable, Sendable { public let availableModels: [String] public let missingMeasurements: Int public let truncated: Bool? + public let appliedFilters: TimelineAppliedFilters? public var maxPoint: Double { series.flatMap(\.points).max() ?? 0 @@ -37,3 +55,54 @@ public struct UsageTimeline: Decodable, Equatable, Sendable { series.allSatisfy { $0.total == 0 } } } + +public extension UsageTimeline { + private enum CodingKeys: String, CodingKey { + case start, end, bucketSeconds, buckets, metric, aggregation, grouping + case series, availableModels, missingMeasurements, truncated, appliedFilters + } + + init(from decoder: Decoder) throws { + let values = try decoder.container(keyedBy: CodingKeys.self) + start = try values.decode(Double.self, forKey: .start) + end = try values.decode(Double.self, forKey: .end) + bucketSeconds = try values.decode(Int.self, forKey: .bucketSeconds) + buckets = try values.decode(Int.self, forKey: .buckets) + metric = try values.decode(String.self, forKey: .metric) + aggregation = try values.decode(String.self, forKey: .aggregation) + grouping = try values.decode(String.self, forKey: .grouping) + series = try values.decode([TimelineSeries].self, forKey: .series) + availableModels = try values.decode([String].self, forKey: .availableModels) + missingMeasurements = try values.decode(Int.self, forKey: .missingMeasurements) + truncated = try values.decodeIfPresent(Bool.self, forKey: .truncated) + // Optional metadata cannot discard valid chart data. An unusable receipt + // takes the same conservative projection path as an older server. + appliedFilters = try? values.decode(TimelineAppliedFilters.self, forKey: .appliedFilters) + } + + func projected(_ settings: CompanionSettings) -> UsageTimeline { + let identity: (String) -> Data = { Data($0.utf8) } + let hidden = Set(settings.hiddenProviders.map(identity)) + let models = settings.models.map { Set($0.map(identity)) } + let emptySelection = models?.isEmpty == true + let active = !hidden.isEmpty || models != nil + let matches = appliedFilters.map { receipt in + Set(receipt.hiddenProviders.map(identity)) == hidden + && receipt.models.map { Set($0.map(identity)) } == models + } ?? false + let visible = emptySelection ? [] : series.filter { row in + if row.id == "other", row.provider.isEmpty { return !active || matches } + return !hidden.contains(identity(row.provider)) + && (models == nil || models!.contains(identity("\(row.provider)/\(row.model)")) || models!.contains(identity(row.model))) + } + let available = availableModels.filter { id in + guard let slash = id.firstIndex(of: "/") else { return true } + return !hidden.contains(identity(String(id[.. Bool { now >= staleDate } +} + public final class WidgetSnapshotStore: @unchecked Sendable { private let fileManager: FileManager private let homeDirectory: URL @@ -188,3 +196,11 @@ private extension WidgetSnapshot { ) } } + +private extension UsageTimeline { + func mapChart(style: String) -> WidgetSnapshot.Chart { + WidgetSnapshot.Chart(start: start, bucketSeconds: bucketSeconds, style: style, + series: Array(series.prefix(6)).map { .init(id: $0.id, points: $0.points) }, + incomplete: truncated == true || missingMeasurements > 0 ? true : nil) + } +} diff --git a/app/Sources/MenuBarCoreTests/TransportSuite.swift b/app/Sources/MenuBarCoreTests/TransportSuite.swift index d46a667e0f..a37de54820 100644 --- a/app/Sources/MenuBarCoreTests/TransportSuite.swift +++ b/app/Sources/MenuBarCoreTests/TransportSuite.swift @@ -275,6 +275,17 @@ enum TransportSuite { t.equal(StubProtocol.recorded.count, 2, "exactly one retry") } + t.test("requests: timeline encodes nested model and repeated hidden provider filters") { + StubProtocol.reset([]) + let client = ProxyClient(endpoint: endpoint, session: makeSession(), + credentials: StubCredentials(key: nil, counter: .init())) + let settings = CompanionSettings(models: ["provider/vendor/model+one"], hiddenProviders: ["a+b", "hidden"]) + _ = sync { try? await client.timeline(settings) } + let items = URLComponents(url: StubProtocol.recorded.first!.url!, resolvingAgainstBaseURL: false)!.queryItems! + t.equal(items.first { $0.name == "models" }?.value, "provider/vendor/model+one") + t.equal(items.filter { $0.name == "hiddenProvider" }.compactMap(\.value), ["a+b", "hidden"]) + } + t.test("requests: usage sends the enum range as a query item") { StubProtocol.reset([.init(status: 200, body: #"{"range":"7d"}"#, urlError: nil)]) let client = ProxyClient(endpoint: endpoint, session: makeSession(), diff --git a/app/Sources/MenuBarCoreTests/WidgetSnapshotSuite.swift b/app/Sources/MenuBarCoreTests/WidgetSnapshotSuite.swift index 66f25176b6..b905da4ac8 100644 --- a/app/Sources/MenuBarCoreTests/WidgetSnapshotSuite.swift +++ b/app/Sources/MenuBarCoreTests/WidgetSnapshotSuite.swift @@ -3,6 +3,79 @@ import MenuBarCore enum WidgetSnapshotSuite { static func run(_ t: TestRunner) { + t.test("widget: a snapshot turns stale two heartbeats after it was written") { + let written = WidgetSnapshot( + schemaVersion: 1, generatedAt: 1_000, state: "running", stateTitle: "Running", detail: nil, + endpointDisplay: "127.0.0.1:10100", menuTitle: nil, today: nil, quotas: [], chart: nil, lastUpdated: 1_000) + t.equal(WidgetSnapshot.staleAfter, 1_800) + t.equal(written.staleDate, Date(timeIntervalSince1970: 2_800)) + t.expect(!written.isStale(now: Date(timeIntervalSince1970: 2_799)), "fresh one second before the boundary") + t.expect(written.isStale(now: Date(timeIntervalSince1970: 2_800)), "stale at the boundary") + } + t.test("widget: hidden usage and quota respect the same projection as the menu title") { + let report = try! JSONDecoder().decode(UsageReport.self, from: Data(#"{"summary":{"requests":99,"totalTokens":99},"models":[{"provider":"hidden","model":"m","requests":97,"totalTokens":94},{"provider":"visible","model":"m","requests":2,"totalTokens":5,"estimatedCostUsd":0.25}]}"#.utf8)) + let quotas = try! JSONDecoder().decode([QuotaReport].self, from: Data(#"[{"provider":"hidden","quota":{"weeklyPercent":1}},{"provider":"visible","quota":{"weeklyPercent":75}}]"#.utf8)) + let settings = CompanionSettings(menuBarMetric: .requests, hiddenProviders: ["hidden"]) + let snapshot = ProxySnapshot(endpoint: .default, settings: settings, today: report, quotas: quotas) + let widget = WidgetSnapshot.make(from: snapshot) + t.equal(widget.today?.requests, 2) + t.equal(widget.today?.totalTokens, 5) + t.equal(widget.today?.estimatedCostUsd, 0.25) + t.equal(widget.menuTitle, "2") + t.equal(widget.quotas.count, 1) + t.equal(widget.quotas.first?.percent, 75) + let folded = try! JSONDecoder().decode(UsageReport.self, from: Data(#"{"summary":{"requests":99},"models":[{"provider":"other","model":"other","requests":99}]}"#.utf8)) + t.expect(folded.filteredSummary(settings) == nil, "folded attribution is unknown") + t.equal(MenuBarTitle.render(settings: settings, today: folded, quotas: []), "—") + let plain = try! JSONDecoder().decode(UsageReport.self, from: Data(#"{"summary":{"requests":0}}"#.utf8)) + t.equal(plain.filteredSummary(.defaults)?.requests, 0) + } + t.test("timeline: only matching filter echoes preserve folded data") { + let base: [String: Any] = ["start": 0, "end": 60, "bucketSeconds": 60, "buckets": 1, + "metric": "total", "aggregation": "sum", "grouping": "model", "availableModels": ["visible/m", "hidden/m"], "missingMeasurements": 0, + "series": [["id":"visible/m","provider":"visible","model":"m","total":2,"points":[2]], + ["id":"hidden/m","provider":"hidden","model":"m","total":1,"points":[1]], + ["id":"other","provider":"","model":"other","total":3,"points":[3]]]] + func timeline(_ echo: Any? = nil) -> UsageTimeline { + var value = base + if let echo { value["appliedFilters"] = echo } + return try! JSONDecoder().decode(UsageTimeline.self, from: JSONSerialization.data(withJSONObject: value)) + } + let settings = CompanionSettings(hiddenProviders: ["hidden"]) + let matching = timeline(["models": NSNull(), "hiddenProviders": ["hidden"]]).projected(settings) + t.equal(matching.series.map(\.id), ["visible/m", "other"]) + t.expect(matching.truncated != true, "matching receipt is complete") + let old = timeline().projected(settings) + t.equal(old.series.map(\.id), ["visible/m"]) + t.equal(old.truncated, true) + let mismatched = timeline(["models": NSNull(), "hiddenProviders": []]).projected(settings) + t.equal(mismatched.truncated, true) + let malformed: [Any] = [ + ["hiddenProviders": ["hidden"]], + ["models": NSNull()], + ["models": "wrong", "hiddenProviders": ["hidden"]], + ["models": NSNull(), "hiddenProviders": Array(repeating: "hidden", count: 101)], + ["models": Array(repeating: "visible/m", count: 101), "hiddenProviders": []], + ["models": NSNull(), "hiddenProviders": [1]], + "wrong", NSNull(), + ] + for receipt in malformed { + let decoded = timeline(receipt) + t.expect(decoded.appliedFilters == nil, "malformed optional receipt is ignored") + t.equal(decoded.projected(settings).series.map(\.id), ["visible/m"]) + t.equal(decoded.projected(settings).truncated, true) + t.equal(decoded.projected(.defaults).series.count, 3) + } + let selected = CompanionSettings(models: ["visible/m"]) + t.equal(timeline(["models": ["visible/m"], "hiddenProviders": []]).projected(selected).series.map(\.id), ["visible/m", "other"]) + let empty = timeline().projected(CompanionSettings(models: [])) + t.expect(empty.series.isEmpty, "explicit empty selection") + t.equal(empty.availableModels, ["visible/m", "hidden/m"]) + t.expect(empty.truncated != true, "empty selection is not a read failure") + t.equal(timeline().projected(.defaults).series.count, 3) + let widget = WidgetSnapshot.make(from: ProxySnapshot(endpoint: .default, settings: settings, timeline: old)) + t.equal(widget.chart?.incomplete, true) + } t.test("widget snapshot: maps today and caps chart series") { var series: [String] = [] for index in 0..<7 { diff --git a/app/Sources/NativeTray/AccountSwitch.swift b/app/Sources/NativeTray/AccountSwitch.swift new file mode 100644 index 0000000000..ab1e7ef4db --- /dev/null +++ b/app/Sources/NativeTray/AccountSwitch.swift @@ -0,0 +1,87 @@ +import SwiftUI + +/// What the panel sends the host for a "Use" click: the provider id and the provider's own +/// account id, never the row's display key. `nil` when the row is not switchable. +public enum NativeTraySwitch { + public static func request(provider: NativeTrayProvider, account: NativeTrayProvider.Account) -> (provider: String, accountId: String)? { + guard provider.switchable == true, account.canSwitch, let accountId = account.accountId else { return nil } + return (provider.id, accountId) + } + + /// Whether `snapshot` answers the switch pending on `pendingRow`: the host reported the failure, + /// or a finished refresh shows that row active. Unrelated errors or an in-flight refresh do not. + public static func settles(snapshot: NativeTraySnapshot, pendingRow: String) -> Bool { + if snapshot.switchFailed == true { return true } + return !snapshot.refreshing + && snapshot.providers.contains { $0.accounts.contains { $0.id == pendingRow && $0.active } } + } +} + +/// An account row's first line: label, plan, the active check, and the "Use" action. +/// +/// Use is revealed on hover or when keyboard focus reaches it, and is always offered as an +/// accessibility action; it keeps its space while hidden so revealing it never moves the row. The +/// rules come from the host's snapshot, which mirrors the runtime: only a hard-locked main account +/// or a paused account is blocked, and an exhausted account stays switchable with a warning. +struct NativeTrayAccountHeader: View { + let account: NativeTrayProvider.Account + let switchable: Bool + let pending: Bool + let busy: Bool + let onUse: () -> Void + @State private var hovered = false + @FocusState private var focused: Bool + + private var offersUse: Bool { switchable && account.canSwitch } + + var body: some View { + VStack(alignment: .leading, spacing: 3) { + HStack(spacing: 6) { + Text(account.label).lineLimit(1).help(account.label) + if account.exhausted == true { + Image(systemName: "exclamationmark.triangle.fill").foregroundStyle(.orange) + .imageScale(.small) + .help("A limit window is used up; requests may fail until it resets") + .accessibilityLabel("Limit reached") + } + Spacer(minLength: 6) + if pending { + ProgressView().controlSize(.mini).accessibilityLabel("Switching account") + } else if offersUse { + Button("Use", action: onUse) + .buttonStyle(.bordered) + .controlSize(.mini) + .disabled(busy) + .focused($focused) + .opacity(hovered || focused ? 1 : 0) + .help("Make this the active account") + .accessibilityLabel("Use \(account.label)") + } + if let plan = account.plan { Text(plan).foregroundStyle(.secondary) } + if account.active { + Image(systemName: "checkmark.circle.fill").foregroundStyle(.green) + .accessibilityLabel("Active account").help("Active account") + } + } + .font(.caption) + if account.switchState == "blocked" { + Text(Self.blockedText(account.blockedReason)) + .font(.caption2).foregroundStyle(.orange) + } + } + .contentShape(Rectangle()) + .onHover { hovered = $0 } + .accessibilityElement(children: .combine) + .accessibilityActions { + if offersUse && !busy { Button("Use this account", action: onUse) } + } + } + + static func blockedText(_ reason: String?) -> String { + switch reason { + case "mainHardLock": return "Blocked by 98% protection" + case "validationPending": return "Validation pending" + default: return "Paused" + } + } +} diff --git a/app/Sources/NativeTray/Models.swift b/app/Sources/NativeTray/Models.swift new file mode 100644 index 0000000000..b76aec8dd0 --- /dev/null +++ b/app/Sources/NativeTray/Models.swift @@ -0,0 +1,179 @@ +import Foundation + +/// Display-only wire contract. The Rust host owns network access and credential handling. +public struct NativeTraySnapshot: Decodable { + public let schemaVersion: Int + public let refreshing: Bool + public let errors: [String] + public let updatedAt: Double? + public let settings: NativeTraySettings + public let today: NativeTrayTotals? + public let month: NativeTrayTotals? + public let models: [NativeTrayModel] + public let chart: NativeTrayChart? + public let providers: [NativeTrayProvider] + /// Set on the publish that reports a failed account switch. + public let switchFailed: Bool? + + public static func decode(_ data: Data) throws -> Self { + let snapshot = try JSONDecoder().decode(Self.self, from: data) + guard snapshot.schemaVersion == 1 else { throw NativeTrayDecodeError.unsupportedSchema } + return snapshot + } +} + +public enum NativeTrayDecodeError: Error { case unsupportedSchema } + +public enum NativeTraySeverity: Equatable { case normal, warn, critical } + +public struct NativeTraySettings: Decodable { + public let showToday: Bool + public let show30Days: Bool + public let showChart: Bool + public let showModels: Bool + public let showAccounts: Bool + public let showCost: Bool + public let chartStyle: String +} + +public struct NativeTrayTotals: Decodable { + public let requests: Double? + public let totalTokens: Double? + public let inputTokens: Double? + public let outputTokens: Double? + public let cachedInputTokens: Double? + public let estimatedCostUsd: Double? + public let measuredRequests: Double? + public let pricedRequests: Double? + public let incomplete: Bool? + + public var hasMeasurements: Bool { !((requests ?? 0) > 0 && measuredRequests == 0) } + public var tokens: Double? { hasMeasurements ? NativeTrayFormat.number(totalTokens) : nil } + public var input: Double? { hasMeasurements ? NativeTrayFormat.number(inputTokens) : nil } + public var output: Double? { hasMeasurements ? NativeTrayFormat.number(outputTokens) : nil } + public var cost: Double? { + (requests ?? 0) > 0 && pricedRequests == 0 ? nil : NativeTrayFormat.number(estimatedCostUsd) + } + public var costIncomplete: Bool { + guard let requests = NativeTrayFormat.number(requests), + let priced = NativeTrayFormat.number(pricedRequests) else { return false } + return priced < requests + } + public var coverage: Double? { + guard let requests = NativeTrayFormat.number(requests), requests > 0, + let measured = NativeTrayFormat.number(measuredRequests) else { return nil } + return min(100, measured / requests * 100) + } + public var cachedPercent: Double? { + guard let input, input > 0, let cached = NativeTrayFormat.number(cachedInputTokens) else { return nil } + return min(100, cached / input * 100) + } +} + +public struct NativeTrayModel: Decodable, Identifiable { + public let id: String + public let label: String + public let requests: Double? + public let tokens: Double? +} + +public struct NativeTrayChart: Decodable { + public let start: Double + public let bucketSeconds: Double + public let series: [Series] + public let incomplete: Bool + public struct Series: Decodable, Identifiable { + public let id: String + public let label: String + public let points: [Double] + } +} + +public struct NativeTrayProvider: Decodable, Identifiable { + public let id: String + public let label: String + public let unavailable: Bool + public let accounts: [Account] + /// The provider's mark as SVG markup, and how to paint it (`image`, `mask`, `plate`, + /// `dark-plate`). Both are optional: older hosts and unknown providers send neither. + public let iconSvg: String? + public let iconPaint: String? + /// Whether the host can switch this provider's active account. Absent on older hosts. + public let switchable: Bool? + public struct Account: Decodable, Identifiable { + public let id: String + public let label: String + public let email: String? + public let plan: String? + public let active: Bool + public let unavailable: Bool + public let windows: [Window] + /// The provider's own account id; `id` is a display key and never leaves the panel. + public let accountId: String? + /// `active`, `available` or `blocked`, mirroring what the runtime refuses or drains. + public let switchState: String? + /// `mainHardLock` or `paused` when blocked. + public let blockedReason: String? + /// A window reads 100%; switching is allowed and the row warns. + public let exhausted: Bool? + public var canSwitch: Bool { !active && switchState == "available" && accountId != nil } + } + public struct Window: Decodable, Identifiable { + public let id: String + public let label: String + public let percent: Double? + public let resetAt: Double? + public var value: Double? { NativeTrayFormat.number(percent) } + public var fill: Double { min(100, value ?? 0) / 100 } + public var resetDate: Date? { NativeTrayFormat.date(resetAt) } + } +} + +public enum NativeTrayFormat { + public static func number(_ value: Double?) -> Double? { + guard let value, value.isFinite, value >= 0 else { return nil } + return value + } + public static func tokens(_ value: Double?) -> String { + guard let value = number(value) else { return "—" } + let units: [(Double, String)] = [(1e9, "B"), (1e6, "M"), (1e3, "K")] + let unit = units.first(where: { value >= $0.0 }) ?? (1, "") + let formatter = NumberFormatter() + formatter.numberStyle = .decimal + formatter.maximumFractionDigits = unit.0 == 1 ? 0 : 1 + return (formatter.string(from: NSNumber(value: value / unit.0)) ?? "—") + unit.1 + } + public static func date(_ timestamp: Double?) -> Date? { + guard let timestamp = number(timestamp), timestamp > 0 else { return nil } + let seconds = timestamp >= 1e12 ? timestamp / 1000 : timestamp + guard seconds < 253_402_300_800 else { return nil } + return Date(timeIntervalSince1970: seconds) + } + /// Same thresholds as the dashboard quota strip (`gui/src/quota-summary.ts`): warn at 70%, + /// critical at 90%. An unknown value is never a severity. + public static func severity(_ percent: Double?) -> NativeTraySeverity { + guard let percent = number(percent) else { return .normal } + if percent >= 90 { return .critical } + if percent >= 70 { return .warn } + return .normal + } + /// Visible percentage. Floored like the dashboard's `formatQuotaPercent`, so the number never + /// reaches a threshold the bar color has not: 89.9% reads "89%" on an orange bar, never "90%". + public static func percentText(_ percent: Double?) -> String { + guard let percent = number(percent), percent < Double(Int.max) else { return "—" } + return "\(Int(percent.rounded(.down)))%" + } + /// VoiceOver value for a quota bar, floored the same way as the visible text. + public static func percentDescription(_ percent: Double?) -> String { + guard let percent = number(percent), percent < Double(Int.max) else { return "Unavailable" } + return "\(Int(percent.rounded(.down))) percent" + } + public static func reset(_ timestamp: Double?, now: Date = Date()) -> String { + guard let date = date(timestamp), date > now else { return "—" } + let minutes = Int(ceil(date.timeIntervalSince(now) / 60)) + if minutes < 60 { return "\(minutes)m" } + if minutes < 1440 { return "\(minutes / 60)h \(minutes % 60)m" } + if minutes < 10080 { return "\(minutes / 1440)d \(minutes % 1440 / 60)h" } + return date.formatted(.dateTime.month(.abbreviated).day()) + } +} diff --git a/app/Sources/NativeTray/Panel.swift b/app/Sources/NativeTray/Panel.swift new file mode 100644 index 0000000000..c53a368aef --- /dev/null +++ b/app/Sources/NativeTray/Panel.swift @@ -0,0 +1,62 @@ +import AppKit + +/// Adapted from the native companion's PopoverPanel in commit 38a5ab9fc4. +/// A key-capable panel avoids the accessory NSPopover keyboard failure measured +/// there on macOS 27, while leaving application/runtime ownership with Tauri. +@MainActor +final class NativeTrayPanel: NSPanel { + var onDismiss: (() -> Void)? + private var outsideMonitor: Any? + + init() { + super.init(contentRect: NSRect(x: 0, y: 0, width: 420, height: 660), + styleMask: [.nonactivatingPanel, .fullSizeContentView, .borderless], + backing: .buffered, defer: false) + title = "OpenCodex Usage" + isFloatingPanel = true + level = .statusBar + hidesOnDeactivate = false + becomesKeyOnlyIfNeeded = false + isOpaque = false + backgroundColor = .clear + hasShadow = true + isMovable = false + animationBehavior = .utilityWindow + } + + override var canBecomeKey: Bool { true } + override var canBecomeMain: Bool { false } + + func present(from button: NSStatusBarButton) { + guard let buttonWindow = button.window else { return } + let visible = (buttonWindow.screen ?? NSScreen.main)?.visibleFrame + ?? NSRect(x: 0, y: 0, width: 1024, height: 768) + let size = NSSize(width: min(420, max(160, visible.width - 16)), + height: min(660, max(160, visible.height - 16))) + setContentSize(size) + let anchor = buttonWindow.convertToScreen(button.convert(button.bounds, to: nil)) + setFrameOrigin(NSPoint( + x: min(max(anchor.midX - size.width / 2, visible.minX + 8), visible.maxX - size.width - 8), + y: max(visible.minY + 8, anchor.minY - size.height - 6))) + makeKeyAndOrderFront(nil) + if outsideMonitor == nil { + outsideMonitor = NSEvent.addGlobalMonitorForEvents(matching: [.leftMouseDown, .rightMouseDown]) { [weak self] _ in + self?.dismiss() + } + } + } + + func dismiss() { + guard isVisible else { return } + if let outsideMonitor { NSEvent.removeMonitor(outsideMonitor) } + outsideMonitor = nil + orderOut(nil) + onDismiss?() + } + + override func cancelOperation(_ sender: Any?) { dismiss() } + override func resignKey() { + super.resignKey() + dismiss() + } +} diff --git a/app/Sources/NativeTray/Popover.swift b/app/Sources/NativeTray/Popover.swift new file mode 100644 index 0000000000..cc33e87d55 --- /dev/null +++ b/app/Sources/NativeTray/Popover.swift @@ -0,0 +1,159 @@ +import AppKit +import SwiftUI + +// All ABI calls run on Tauri's AppKit main thread. Swift copies the borrowed JSON +// synchronously and never retains a Rust buffer or owns an application/run loop. +@MainActor +private final class NativeTrayPopover: NSObject { + static let shared = NativeTrayPopover() + let panel = NativeTrayPanel() + let store = NativeTrayStore() + var callback: (@convention(c) (Int32) -> Void)? + var switchCallback: (@convention(c) (UnsafePointer, UnsafePointer) -> Void)? + + override init() { + super.init() + panel.contentViewController = NativeTrayHostingController(store: store) + panel.onDismiss = { [weak self] in self?.callback?(2) } + store.action = { [weak self] event in + guard let self else { return } + if event == 2 || event == 3 || event == 4 { self.panel.dismiss() } + if event != 2 { self.callback?(event) } + } + // Only names cross the ABI: the host picks the route and body from its own config. + store.switchAccount = { [weak self] provider, accountId in + guard let callback = self?.switchCallback else { return } + provider.withCString { provider in + accountId.withCString { accountId in callback(provider, accountId) } + } + } + } + + func show(_ pointer: UnsafeMutableRawPointer, toggle: Bool, callback: @escaping @convention(c) (Int32) -> Void) { + self.callback = callback + if toggle && panel.isVisible { panel.dismiss(); return } + let item = Unmanaged.fromOpaque(pointer).takeUnretainedValue() + guard let button = item.button, button.window != nil else { return } + if panel.isVisible { return } + panel.present(from: button) + if panel.isVisible { callback(1) } + } + +} + +@MainActor +private final class UpdateDotView: NSView { + weak var statusButton: NSStatusBarButton? + + init(button: NSStatusBarButton) { + statusButton = button + super.init(frame: button.bounds) + autoresizingMask = [.width, .height] + // AppKit keeps the template image and its highlighted tint. This view draws only + // the independent accent, without making the status button layer-backed. + wantsLayer = false + } + + required init?(coder: NSCoder) { nil } + override var isOpaque: Bool { false } + override func hitTest(_ point: NSPoint) -> NSView? { nil } + + override func layout() { + super.layout() + needsDisplay = true + } + + override func draw(_ dirtyRect: NSRect) { + guard let button = statusButton else { return } + let imageRect = button.cell?.imageRect(forBounds: button.bounds) ?? button.bounds + let image = imageRect.isEmpty ? button.bounds : imageRect + let diameter: CGFloat = 7 + let dot = NSRect(x: min(bounds.maxX - diameter, image.maxX - 4), + y: max(bounds.minY, image.minY + 1), + width: diameter, height: diameter) + NSColor.windowBackgroundColor.setFill() + NSBezierPath(ovalIn: dot.insetBy(dx: -1.25, dy: -1.25)).fill() + NSColor(calibratedRed: 0.18, green: 0.48, blue: 0.97, alpha: 1).setFill() + NSBezierPath(ovalIn: dot).fill() + } +} + +@MainActor +private enum UpdateDot { + static weak var button: NSStatusBarButton? + static var view: UpdateDotView? + + static func set(_ item: NSStatusItem, visible: Bool) { + guard let next = item.button else { return } + if button !== next { + view?.removeFromSuperview() + view = nil + button = next + } + guard visible else { + view?.removeFromSuperview() + view = nil + return + } + if view == nil { + let overlay = UpdateDotView(button: next) + next.addSubview(overlay) + view = overlay + } + view?.frame = next.bounds + view?.needsDisplay = true + } +} + +@_cdecl("ocx_native_tray_update_dot") +@MainActor +public func nativeTrayUpdateDot(_ item: UnsafeMutableRawPointer?, _ show: Int32) { + guard Thread.isMainThread, let item else { return } + let statusItem = Unmanaged.fromOpaque(item).takeUnretainedValue() + UpdateDot.set(statusItem, visible: show != 0) +} + +@_cdecl("ocx_native_tray_show") +@MainActor +public func nativeTrayShow(_ item: UnsafeMutableRawPointer?, _ toggle: Int32, _ callback: @escaping @convention(c) (Int32) -> Void) { + guard Thread.isMainThread, let item else { return } + NativeTrayPopover.shared.show(item, toggle: toggle != 0, callback: callback) +} + +@_cdecl("ocx_native_tray_hide") +@MainActor +public func nativeTrayHide() { + guard Thread.isMainThread else { return } + NativeTrayPopover.shared.panel.dismiss() +} + +@_cdecl("ocx_native_tray_visible") +@MainActor +public func nativeTrayVisible() -> Int32 { + guard Thread.isMainThread else { return 0 } + return NativeTrayPopover.shared.panel.isVisible ? 1 : 0 +} + +@_cdecl("ocx_native_tray_update") +@MainActor +public func nativeTrayUpdate(_ bytes: UnsafePointer?, _ count: Int) { + guard Thread.isMainThread, let bytes, count > 0, count <= 8 * 1024 * 1024 else { return } + let store = NativeTrayPopover.shared.store + do { + store.snapshot = try NativeTraySnapshot.decode(Data(bytes: bytes, count: count)) + store.decodeFailed = false + store.settlePendingSwitch() + } catch { + store.decodeFailed = true + } +} + +/// Registers the host's handler for the panel's "Use" action on an account row. +@_cdecl("ocx_native_tray_set_switch_handler") +@MainActor +public func nativeTraySetSwitchHandler( + _ callback: @escaping @convention(c) (UnsafePointer, UnsafePointer) -> Void +) { + guard Thread.isMainThread else { return } + NativeTrayPopover.shared.switchCallback = callback +} diff --git a/app/Sources/NativeTray/ProviderMark.swift b/app/Sources/NativeTray/ProviderMark.swift new file mode 100644 index 0000000000..ea199b7d82 --- /dev/null +++ b/app/Sources/NativeTray/ProviderMark.swift @@ -0,0 +1,89 @@ +import AppKit +import SwiftUI + +public enum NativeTrayIcon { + /// Decodes a provider mark. AppKit reads SVG data into an `NSImage`; anything it cannot read, + /// or reads as an empty image, is no mark rather than a blank square. + public static func image(svg: String) -> NSImage? { + guard let data = svg.data(using: .utf8), let image = NSImage(data: data), + image.size.width > 0, image.size.height > 0 else { return nil } + return image + } +} + +/// Decoding an SVG on every redraw would repeat work for marks that never change. +@MainActor +private enum NativeTrayIconCache { + private static var images: [String: NSImage] = [:] + private static var unreadable: Set = [] + + static func image(id: String, svg: String) -> NSImage? { + let key = "\(id):\(svg.utf8.count):\(svg.hashValue)" + if let image = images[key] { return image } + if unreadable.contains(key) { return nil } + guard let image = NativeTrayIcon.image(svg: svg) else { + unreadable.insert(key) + return nil + } + images[key] = image + return image + } +} + +/// A provider's mark, painted the way the dashboard paints it so it survives both appearances: +/// `mask` marks are one neutral ink and take the label color, `plate` and `dark-plate` marks sit +/// on the constant plate their artwork assumes, and `image` marks are drawn as they are. +struct NativeTrayProviderMark: View { + let provider: NativeTrayProvider + + var body: some View { + if let svg = provider.iconSvg, let image = NativeTrayIconCache.image(id: provider.id, svg: svg) { + let paint = provider.iconPaint ?? "image" + let mark = Image(nsImage: image) + .resizable() + .renderingMode(paint == "mask" ? .template : .original) + .aspectRatio(contentMode: .fit) + Group { + if paint == "plate" || paint == "dark-plate" { + mark.padding(2) + .background(RoundedRectangle(cornerRadius: 4, style: .continuous) + .fill(paint == "plate" ? Color(white: 0.96) : Color(white: 0.14))) + } else { + mark.foregroundStyle(.primary) + } + } + .frame(width: 16, height: 16) + .accessibilityHidden(true) + } + } +} + +/// Quota bar drawn from shapes, colored by the dashboard's severity thresholds. It replaces an +/// AppKit-backed `ProgressView` tinted a fixed green, which showed a 100% window as healthy. +struct NativeTrayQuotaBar: View { + let window: NativeTrayProvider.Window + + private var color: Color { + switch NativeTrayFormat.severity(window.value) { + case .normal: return .green + case .warn: return .orange + case .critical: return .red + } + } + + var body: some View { + GeometryReader { geometry in + ZStack(alignment: .leading) { + Capsule().fill(Color.primary.opacity(0.1)) + if let value = window.value { + Capsule().fill(color) + .frame(width: value > 0 ? max(6, geometry.size.width * window.fill) : 0) + } + } + } + .frame(height: 6) + .accessibilityElement() + .accessibilityLabel(window.label) + .accessibilityValue(NativeTrayFormat.percentDescription(window.value)) + } +} diff --git a/app/Sources/NativeTray/Surface.swift b/app/Sources/NativeTray/Surface.swift new file mode 100644 index 0000000000..5deb02c99e --- /dev/null +++ b/app/Sources/NativeTray/Surface.swift @@ -0,0 +1,60 @@ +import AppKit +import SwiftUI + +/// A single native material surface inside a transparent AppKit panel. +/// The SwiftUI content deliberately paints no web-style background or second radius. +@MainActor +final class NativeTrayHostingController: NSViewController { + private let hosting: NSHostingController + + static var usesLiquidGlass: Bool { + #if compiler(>=6.2) + if #available(macOS 26.0, *) { return true } + #endif + return false + } + + init(store: NativeTrayStore) { + hosting = NSHostingController(rootView: NativeTrayUsageView(store: store)) + super.init(nibName: nil, bundle: nil) + } + + required init?(coder: NSCoder) { nil } + + override func loadView() { + addChild(hosting) + #if compiler(>=6.2) + if #available(macOS 26.0, *) { + let glass = NSGlassEffectView() + glass.style = .regular + glass.cornerRadius = 16 + glass.contentView = hosting.view + view = glass + hosting.view.translatesAutoresizingMaskIntoConstraints = false + NSLayoutConstraint.activate([ + hosting.view.leadingAnchor.constraint(equalTo: glass.safeAreaLayoutGuide.leadingAnchor), + hosting.view.trailingAnchor.constraint(equalTo: glass.safeAreaLayoutGuide.trailingAnchor), + hosting.view.topAnchor.constraint(equalTo: glass.safeAreaLayoutGuide.topAnchor), + hosting.view.bottomAnchor.constraint(equalTo: glass.safeAreaLayoutGuide.bottomAnchor), + ]) + return + } + #endif + let material = NSVisualEffectView() + material.material = .popover + material.blendingMode = .behindWindow + material.state = .active + material.wantsLayer = true + material.layer?.cornerRadius = 16 + material.layer?.masksToBounds = true + material.addSubview(hosting.view) + hosting.view.translatesAutoresizingMaskIntoConstraints = false + NSLayoutConstraint.activate([ + hosting.view.leadingAnchor.constraint(equalTo: material.leadingAnchor), + hosting.view.trailingAnchor.constraint(equalTo: material.trailingAnchor), + hosting.view.topAnchor.constraint(equalTo: material.topAnchor), + hosting.view.bottomAnchor.constraint(equalTo: material.bottomAnchor), + ]) + view = material + } +} diff --git a/app/Sources/NativeTray/UsageSections.swift b/app/Sources/NativeTray/UsageSections.swift new file mode 100644 index 0000000000..3d4d3354ed --- /dev/null +++ b/app/Sources/NativeTray/UsageSections.swift @@ -0,0 +1,111 @@ +import SwiftUI +import Charts + +struct NativeTrayProviderView: View { + let provider: NativeTrayProvider + var pendingSwitch: String? = nil + var onUse: ((NativeTrayProvider.Account) -> Void)? = nil + + var body: some View { + VStack(alignment: .leading, spacing: 10) { + HStack(spacing: 6) { + NativeTrayProviderMark(provider: provider) + Text(provider.label).font(.subheadline.weight(.semibold)) + } + if provider.unavailable || provider.accounts.isEmpty { + Text(provider.unavailable ? "Account limits unavailable" : "No quota data") + .font(.caption).foregroundStyle(.secondary) + } + ForEach(provider.accounts) { account in + VStack(alignment: .leading, spacing: 6) { + NativeTrayAccountHeader( + account: account, + switchable: provider.switchable == true && onUse != nil, + pending: pendingSwitch == account.id, + busy: pendingSwitch != nil, + onUse: { onUse?(account) }) + if let email = account.email, email != account.label { + Text(email).font(.caption2).foregroundStyle(.secondary) + } + if account.unavailable || account.windows.isEmpty { + Text("No quota data").font(.caption2).foregroundStyle(.secondary) + } + ForEach(account.windows) { window in + HStack(spacing: 8) { + Text(window.label).lineLimit(1).frame(width: 96, alignment: .leading) + Text(NativeTrayFormat.percentText(window.value)) + .monospacedDigit().frame(width: 36, alignment: .trailing) + NativeTrayQuotaBar(window: window) + Text(NativeTrayFormat.reset(window.resetAt)).monospacedDigit() + .frame(width: 70, alignment: .trailing) + .help(window.resetDate?.formatted(date: .complete, time: .standard) ?? "Reset time unavailable") + }.font(.caption2).foregroundStyle(.secondary) + } + } + } + }.frame(maxWidth: .infinity, alignment: .leading) + } +} + +struct NativeTrayChartView: View { + let chart: NativeTrayChart + let style: String + private let palette: [Color] = [.blue, .orange, .green, .purple, .red, .cyan, .pink, .yellow, .mint, .indigo] + + var body: some View { + VStack(alignment: .leading, spacing: 8) { + if chart.series.isEmpty { + Text("No usage measurements").foregroundStyle(.secondary) + } else { + Chart { + ForEach(chart.series) { series in + ForEach(Array(series.points.enumerated()), id: \.offset) { index, point in + let date = Date(timeIntervalSince1970: chart.start + Double(index) * chart.bucketSeconds) + if style == "stackedBar" { + BarMark(x: .value("Time", date), y: .value("Tokens", max(0, point)), stacking: .standard) + .foregroundStyle(by: .value("Series", series.id)) + } else { + LineMark(x: .value("Time", date), y: .value("Tokens", max(0, point)), series: .value("Series", series.id)) + .foregroundStyle(by: .value("Series", series.id)) + } + } + } + } + .chartYAxis { + AxisMarks(position: .leading, values: .automatic(desiredCount: 3)) { axis in + AxisGridLine() + AxisValueLabel(anchor: .trailing) { + if let value = axis.as(Double.self) { Text(NativeTrayFormat.tokens(value)) } + } + } + } + .chartXAxis { + AxisMarks(values: .automatic(desiredCount: 3)) { axis in + AxisGridLine() + AxisTick() + AxisValueLabel(anchor: .center) { + if let date = axis.as(Date.self) { + Text(date, format: .dateTime.hour().minute()) + } + } + } + } + .chartForegroundStyleScale(domain: chart.series.map(\.id), range: chart.series.indices.map { palette[$0 % palette.count] }) + .chartLegend(.hidden) + .frame(height: 130) + .accessibilityLabel("Usage timeline") + LazyVGrid(columns: [GridItem(.flexible(), alignment: .leading), GridItem(.flexible(), alignment: .leading)], alignment: .leading, spacing: 5) { + ForEach(Array(chart.series.enumerated()), id: \.element.id) { index, series in + HStack(spacing: 5) { + Circle().fill(palette[index % palette.count]).frame(width: 6, height: 6).accessibilityHidden(true) + Text(series.label).lineLimit(1).truncationMode(.middle).help(series.label) + }.font(.caption2).foregroundStyle(.secondary) + } + } + } + if chart.incomplete { + Text("Some usage records are unavailable").font(.caption2).foregroundStyle(.secondary) + } + } + } +} diff --git a/app/Sources/NativeTray/UsageView.swift b/app/Sources/NativeTray/UsageView.swift new file mode 100644 index 0000000000..49c38bb2d9 --- /dev/null +++ b/app/Sources/NativeTray/UsageView.swift @@ -0,0 +1,153 @@ +import SwiftUI + +@MainActor +final class NativeTrayStore: ObservableObject { + @Published var snapshot: NativeTraySnapshot? + @Published var decodeFailed = false + /// The account row whose switch is in flight, until a settled snapshot arrives. + @Published var pendingSwitch: String? + var action: (Int32) -> Void = { _ in } + var switchAccount: (String, String) -> Void = { _, _ in } + private var pendingToken = 0 + + func requestSwitch(provider: NativeTrayProvider, account: NativeTrayProvider.Account) { + guard pendingSwitch == nil, let request = NativeTraySwitch.request(provider: provider, account: account) else { return } + pendingSwitch = account.id + pendingToken += 1 + let token = pendingToken + // A switch whose answer never arrives (host gone, panel reopened) must not spin forever. + DispatchQueue.main.asyncAfter(deadline: .now() + 20) { [weak self] in + guard let self, self.pendingToken == token else { return } + self.pendingSwitch = nil + } + switchAccount(request.provider, request.accountId) + } + + func settlePendingSwitch() { + guard let pending = pendingSwitch, let snapshot, + NativeTraySwitch.settles(snapshot: snapshot, pendingRow: pending) else { return } + pendingSwitch = nil + pendingToken += 1 + } +} + +struct NativeTrayUsageView: View { + @ObservedObject var store: NativeTrayStore + + var body: some View { + VStack(spacing: 0) { + HStack { + Text("OpenCodex").font(.headline) + Spacer() + if store.snapshot?.refreshing == true { ProgressView().controlSize(.small) } + Button { store.action(4) } label: { Image(systemName: "gearshape") } + .buttonStyle(.plain).help("Settings").accessibilityLabel("Settings") + }.padding(14) + Divider() + ScrollView { + VStack(alignment: .leading, spacing: 16) { + if let snapshot = store.snapshot { + if snapshot.settings.showToday || snapshot.settings.show30Days { + HStack(alignment: .top, spacing: 16) { + if snapshot.settings.showToday { + NativeTrayTotalsView(title: "Today", totals: snapshot.today, showCost: snapshot.settings.showCost) + } + if snapshot.settings.showToday && snapshot.settings.show30Days { Divider() } + if snapshot.settings.show30Days { + NativeTrayTotalsView(title: "30 days", totals: snapshot.month, showCost: snapshot.settings.showCost) + } + } + } + if snapshot.settings.showChart, let chart = snapshot.chart { + Divider() + NativeTrayChartView(chart: chart, style: snapshot.settings.chartStyle) + } + if snapshot.settings.showModels && !snapshot.models.isEmpty { + Divider() + VStack(alignment: .leading, spacing: 8) { + Text("Models").font(.subheadline).foregroundStyle(.secondary) + ForEach(snapshot.models) { row in + HStack { + Text(row.label).lineLimit(1).help(row.label) + Spacer(minLength: 8) + Text("\(NativeTrayFormat.tokens(row.requests)) requests") + .foregroundStyle(.secondary).font(.caption) + Text(NativeTrayFormat.tokens(row.tokens)).monospacedDigit() + } + } + } + } + if snapshot.settings.showAccounts { + Divider() + ForEach(snapshot.providers) { provider in + NativeTrayProviderView(provider: provider, pendingSwitch: store.pendingSwitch) { account in + store.requestSwitch(provider: provider, account: account) + } + } + } + ForEach(Array(snapshot.errors.enumerated()), id: \.offset) { _, error in + Label(error, systemImage: "exclamationmark.triangle") + .foregroundStyle(.secondary).font(.caption) + } + } else { + HStack { ProgressView().controlSize(.small); Text("Loading usage…") } + .frame(maxWidth: .infinity, alignment: .center).padding(.vertical, 40) + } + if store.decodeFailed { + Text("Usage data could not be read. Try refreshing.").foregroundStyle(.secondary) + } + }.padding(14).frame(maxWidth: .infinity, alignment: .leading) + } + Divider() + HStack(spacing: 10) { + Button("Refresh") { store.action(1) } + .disabled(store.snapshot?.refreshing == true && !store.decodeFailed) + if let updated = NativeTrayFormat.date(store.snapshot?.updatedAt) { + Text(updated, style: .time).font(.caption).foregroundStyle(.secondary) + .help("Last successful update") + } + Spacer() + Button("Dashboard") { store.action(3) } + }.controlSize(.small).padding(12) + } + .font(.system(size: 12)) + .frame(maxWidth: .infinity, maxHeight: .infinity) + .onExitCommand { store.action(2) } + } +} + +private struct NativeTrayTotalsView: View { + let title: String + let totals: NativeTrayTotals? + let showCost: Bool + + var body: some View { + VStack(alignment: .leading, spacing: 7) { + Text(title).font(.subheadline).foregroundStyle(.secondary) + row("Total tokens", NativeTrayFormat.tokens(totals?.tokens), headline: true) + row("Input", NativeTrayFormat.tokens(totals?.input)) + if let cached = totals?.cachedPercent { + Text("\(Int(cached.rounded()))% cached").font(.caption2).foregroundStyle(.secondary) + .frame(maxWidth: .infinity, alignment: .trailing) + } + row("Output", NativeTrayFormat.tokens(totals?.output)) + if showCost { + row("Cost · est.", (totals?.cost?.formatted(.currency(code: "USD")) ?? "—") + (totals?.costIncomplete == true ? "*" : "")) + .help(totals?.costIncomplete == true ? "Some requests have no price or usage. API list-price equivalent, not an actual charge." : "API list-price equivalent, not an actual charge") + } + row("Requests", NativeTrayFormat.tokens(totals?.requests)) + if let coverage = totals?.coverage, coverage < 100 { row("Coverage", "\(Int(coverage.rounded()))%") } + if totals?.incomplete == true { + Text("Some usage records are unavailable").font(.caption2).foregroundStyle(.secondary) + } + }.frame(maxWidth: .infinity, alignment: .topLeading) + } + private func row(_ label: String, _ value: String, headline: Bool = false) -> some View { + HStack(alignment: .firstTextBaseline) { + Text(label).foregroundStyle(.secondary).font(.caption) + Spacer(minLength: 4) + Text(value).font(headline ? .system(size: 19, weight: .semibold, design: .rounded) : .system(size: 12)) + .monospacedDigit().lineLimit(1).minimumScaleFactor(0.8) + } + } +} diff --git a/app/Sources/NativeTray/WidgetReload.swift b/app/Sources/NativeTray/WidgetReload.swift new file mode 100644 index 0000000000..0072fa951d --- /dev/null +++ b/app/Sources/NativeTray/WidgetReload.swift @@ -0,0 +1,19 @@ +import Foundation +import WidgetKit + +/// The kind `OpenCodexWidget` declares in app/Sources/OpenCodexWidget/Views.swift. +private let openCodexWidgetKind = "OpenCodexWidget" + +/// Asks WidgetKit to request a new timeline from the desktop widget. +/// +/// The Rust host writes the widget snapshot into the extension's container and calls this after +/// a write that changed what the widget shows. Without it the widget only rereads the file on its +/// own timeline schedule, which WidgetKit is free to postpone, so a fresh snapshot could sit unread +/// for a long time. WidgetKit still enforces its reload budget; this is a request, not a redraw. +/// The Rust caller runs on an async worker, so the call hops to the main queue. +@_cdecl("ocx_widget_reload_timelines") +public func widgetReloadTimelines() { + DispatchQueue.main.async { + WidgetCenter.shared.reloadTimelines(ofKind: openCodexWidgetKind) + } +} diff --git a/app/Sources/NativeTrayTests/main.swift b/app/Sources/NativeTrayTests/main.swift new file mode 100644 index 0000000000..c519376e69 --- /dev/null +++ b/app/Sources/NativeTrayTests/main.swift @@ -0,0 +1,145 @@ +import Foundation +import NativeTray + +var assertions = 0 +func check(_ condition: @autoclosure () -> Bool, _ message: String) { + assertions += 1 + if !condition() { fatalError(message) } +} + +let settings: [String: Any] = [ + "showToday": true, "show30Days": true, "showChart": true, + "showModels": true, "showAccounts": true, "showCost": true, "chartStyle": "line", +] +func decode(_ changes: [String: Any] = [:]) throws -> NativeTraySnapshot { + var value: [String: Any] = [ + "schemaVersion": 1, "refreshing": false, "errors": [], "settings": settings, + "models": [], "providers": [], + ] + value.merge(changes) { _, new in new } + return try NativeTraySnapshot.decode(JSONSerialization.data(withJSONObject: value)) +} + +let empty = try decode() +check(empty.today == nil && empty.month == nil, "Missing totals must remain unknown") +check(NativeTrayFormat.tokens(nil) == "—", "Missing must not render zero") +check(NativeTrayFormat.tokens(0) == "0", "Measured zero must render zero") +check(NativeTrayFormat.tokens(10_000_000) == "10M", "Whole-number trailing zeros must survive") +check(NativeTrayFormat.number(-1) == nil, "Negative measurements rejected") +check(NativeTrayFormat.number(.infinity) == nil, "Infinite measurements rejected") +check(NativeTrayFormat.number(.nan) == nil, "NaN measurements rejected") + +let unmeasured = try decode(["today": [ + "requests": 3, "measuredRequests": 0, "pricedRequests": 0, + "totalTokens": 0, "inputTokens": 0, "outputTokens": 0, "estimatedCostUsd": 0, +]]) +check(unmeasured.today?.tokens == nil, "Unmeasured nonempty requests are not zero tokens") +check(unmeasured.today?.input == nil && unmeasured.today?.output == nil, "Unmeasured input/output stay unknown") +check(unmeasured.today?.cost == nil, "Unpriced nonempty requests are not free") +check(unmeasured.today?.coverage == 0, "Coverage remains an honest zero") +check(unmeasured.today?.costIncomplete == true, "Unpriced requests carry partial-cost disclosure") +let measured = try decode(["today": [ + "requests": 4, "measuredRequests": 3, "pricedRequests": 4, + "totalTokens": 120, "inputTokens": 100, "outputTokens": 20, + "cachedInputTokens": 75, "estimatedCostUsd": 0, +]]) +check(measured.today?.tokens == 120 && measured.today?.cost == 0, "Measured/free data retained") +check(measured.today?.coverage == 75 && measured.today?.cachedPercent == 75, "Coverage and cache ratio calculated") + +let account: [String: Any] = ["id": "account-a", "label": "한글 계정", "active": true, + "unavailable": false, "windows": [["id": "weekly", "label": "Weekly", "percent": 125, "resetAt": 1_900_000_000_000]]] +let quotasOnly = try decode(["errors": ["Usage unavailable"], "providers": [ + ["id": "provider-a", "label": "Provider A", "unavailable": false, "accounts": [account]], +]]) +let window = quotasOnly.providers[0].accounts[0].windows[0] +check(quotasOnly.today == nil && quotasOnly.providers.count == 1, "Usage failure cannot hide successful quotas") +check(quotasOnly.providers[0].accounts[0].label == "한글 계정", "Unicode label survives the wire") +check(window.value == 125 && window.fill == 1, "Clamp fill, preserve displayed over-limit percent") +check(window.resetDate == Date(timeIntervalSince1970: 1_900_000_000), "Millisecond reset normalized") +check(NativeTrayFormat.date(1_900_000_000) == window.resetDate, "Second and millisecond reset agree") +check(NativeTrayFormat.date(0) == nil && NativeTrayFormat.date(1e300) == nil, "Invalid reset times are unavailable") +let now = Date(timeIntervalSince1970: 1_900_000_000) +check(NativeTrayFormat.reset(1_900_000_061, now: now) == "2m", "Reset duration rounds up") +check(NativeTrayFormat.reset(1_899_999_999, now: now) == "—", "Expired reset is not a future promise") + +do { _ = try decode(["schemaVersion": 2]); fatalError("Unknown schema was accepted") } +catch NativeTrayDecodeError.unsupportedSchema { assertions += 1 } +do { _ = try decode(["providers": "bad"]); fatalError("Malformed roster was accepted") } +catch is DecodingError { assertions += 1 } + +let many = try decode(["providers": (0..<80).map { index in + ["id": "provider-\(index)", "label": "Provider \(index)", "unavailable": false, "accounts": [account]] as [String: Any] +}]) +check(many.providers.count == 80, "Long roster must not be truncated to fit the popup") +check(Set(many.providers.map(\.id)).count == 80, "Provider identities disambiguate equal account ids") + +// Provider marks: one representative file per paint mode, decoded by the view's own decoder. +let icons = URL(fileURLWithPath: #filePath) + .deletingLastPathComponent().deletingLastPathComponent().deletingLastPathComponent() + .deletingLastPathComponent().appendingPathComponent("gui/public/provider-icons") +for file in ["openai.svg", "grok.svg", "zai.svg", "nebius.svg"] { + let svg = try String(contentsOf: icons.appendingPathComponent(file), encoding: .utf8) + let image = NativeTrayIcon.image(svg: svg) + check(image != nil && image!.size.width > 0 && image!.size.height > 0, "\(file) must decode to a visible mark") +} +check(NativeTrayIcon.image(svg: "not an image") == nil, "Unreadable mark data is no mark") +let marked = try decode(["providers": [["id": "openai", "label": "OpenAI", "unavailable": false, "accounts": [], + "iconSvg": "", "iconPaint": "mask"]]]) +check(marked.providers[0].iconPaint == "mask" && marked.providers[0].iconSvg == "", "Mark fields decode") +check(quotasOnly.providers[0].iconSvg == nil, "Providers without a mark still decode") + +// Quota bars use the dashboard's severity thresholds and keep a spoken value. +check(NativeTrayFormat.severity(69.9) == .normal && NativeTrayFormat.severity(70) == .warn, "Warn at 70%") +check(NativeTrayFormat.severity(90) == .critical && NativeTrayFormat.severity(125) == .critical, "Critical at 90%") +check(NativeTrayFormat.severity(nil) == .normal, "Unknown is not a severity") +check(NativeTrayFormat.percentDescription(125) == "125 percent", "Over-limit value is spoken as reported") +check(NativeTrayFormat.percentDescription(nil) == "Unavailable", "Missing value is spoken as unavailable") +// Labels floor like the dashboard so the number never crosses a threshold the color has not. +check(NativeTrayFormat.percentText(69.9) == "69%" && NativeTrayFormat.severity(69.9) == .normal, "69.9% reads 69% on green") +check(NativeTrayFormat.percentText(89.9) == "89%" && NativeTrayFormat.severity(89.9) == .warn, "89.9% reads 89% on orange") +check(NativeTrayFormat.percentDescription(89.9) == "89 percent", "Spoken value floors too") +check(NativeTrayFormat.percentText(nil) == "—", "Missing value renders a dash") +check(NativeTrayFormat.percentText(1e20) == "—", "Oversized percent cannot trap visible formatting") +check(NativeTrayFormat.percentDescription(1e20) == "Unavailable", "Oversized percent cannot trap spoken formatting") +let largestSafePercent = Double(Int.max).nextDown +check(NativeTrayFormat.percentText(largestSafePercent) == "\(Int(largestSafePercent))%", "Largest representable percent remains visible") +check(NativeTrayFormat.percentDescription(largestSafePercent) == "\(Int(largestSafePercent)) percent", "Largest representable percent remains spoken") +check(NativeTrayFormat.percentText(Double(Int.max)) == "—", "Rounded Int upper bound is unavailable visibly") +check(NativeTrayFormat.percentDescription(Double(Int.max)) == "Unavailable", "Rounded Int upper bound is unavailable when spoken") + +// Account switching: only names cross to the host, and only for rows the runtime would accept. +func switchRow(_ id: String, _ fields: [String: Any]) -> [String: Any] { + var row: [String: Any] = ["id": "\(id):0", "label": id, "active": false, "unavailable": false, "windows": []] + row.merge(fields) { _, new in new } + return row +} +let switching = try decode(["providers": [ + ["id": "openai", "label": "OpenAI", "unavailable": false, "switchable": true, "accounts": [ + switchRow("__main__", ["accountId": "__main__", "switchState": "blocked", "blockedReason": "mainHardLock"]), + switchRow("pool-a", ["accountId": "pool-a", "switchState": "active", "active": true]), + switchRow("pool-b", ["accountId": "pool-b", "switchState": "available", "exhausted": true]), + switchRow("pool-c", ["accountId": "pool-c", "switchState": "blocked", "blockedReason": "paused"]), + ]], + ["id": "legacy", "label": "Older host", "unavailable": false, "accounts": [switchRow("k", [:])]], +]]) +let openai = switching.providers[0] +check(NativeTraySwitch.request(provider: openai, account: openai.accounts[0]) == nil, "A hard-locked main account is not offered") +check(NativeTraySwitch.request(provider: openai, account: openai.accounts[1]) == nil, "The active account is not offered") +let exhaustedPick = NativeTraySwitch.request(provider: openai, account: openai.accounts[2]) +check(exhaustedPick?.provider == "openai" && exhaustedPick?.accountId == "pool-b", "An exhausted pool account stays switchable, by raw id") +check(openai.accounts[2].exhausted == true, "Exhaustion decodes for the warning") +check(NativeTraySwitch.request(provider: openai, account: openai.accounts[3]) == nil, "A paused account is not offered") +let legacy = switching.providers[1] +check(legacy.switchable == nil && NativeTraySwitch.request(provider: legacy, account: legacy.accounts[0]) == nil, + "Snapshots from older hosts decode and offer no switch") +// A pending switch ends on the host's failure marker or a finished refresh showing the row active. +check(NativeTraySwitch.settles(snapshot: switching, pendingRow: "pool-a:0"), "A finished refresh with the row active settles") +check(!NativeTraySwitch.settles(snapshot: switching, pendingRow: "pool-b:0"), "Another row being active does not settle") +let refreshingSwitch = try decode(["refreshing": true, "providers": [["id": "openai", "label": "OpenAI", "unavailable": false, + "switchable": true, "accounts": [switchRow("pool-b", ["accountId": "pool-b", "active": true, "switchState": "active"])]]]]) +check(!NativeTraySwitch.settles(snapshot: refreshingSwitch, pendingRow: "pool-b:0"), "An in-flight refresh does not settle") +let unrelatedError = try decode(["errors": ["Usage unavailable"], "providers": []]) +check(!NativeTraySwitch.settles(snapshot: unrelatedError, pendingRow: "pool-b:0"), "An unrelated error does not settle") +let failedSwitch = try decode(["errors": ["The runtime refused that account right now."], "switchFailed": true, "refreshing": false, "providers": []]) +check(NativeTraySwitch.settles(snapshot: failedSwitch, pendingRow: "pool-b:0"), "The host's failure marker settles at once") +print("PASS: \(assertions) native tray contract/formatting assertions") diff --git a/app/Sources/OpenCodexWidget/Provider.swift b/app/Sources/OpenCodexWidget/Provider.swift index 55d983bf5c..a36dcc61fb 100644 --- a/app/Sources/OpenCodexWidget/Provider.swift +++ b/app/Sources/OpenCodexWidget/Provider.swift @@ -2,7 +2,6 @@ import Foundation import WidgetKit import MenuBarCore -@available(macOS 14, *) public struct SnapshotEntry: TimelineEntry { public let date: Date public let snapshot: WidgetSnapshot? @@ -10,7 +9,6 @@ public struct SnapshotEntry: TimelineEntry { public let stale: Bool } -@available(macOS 14, *) public struct SnapshotProvider: TimelineProvider { private let reader = SnapshotReader() @@ -26,7 +24,17 @@ public struct SnapshotProvider: TimelineProvider { public func getTimeline(in context: Context, completion: @escaping (Timeline) -> Void) { let now = Date() - completion(Timeline(entries: [readEntry(now: now)], policy: .after(now.addingTimeInterval(300)))) + let current = readEntry(now: now) + var entries = [current] + // A fresh snapshot turns stale at a known moment. Scheduling that entry up front shows the + // stale tint on time even when WidgetKit postpones the next timeline request. + if let snapshot = current.snapshot, !current.stale { + entries.append(SnapshotEntry(date: snapshot.staleDate, snapshot: snapshot, failure: nil, stale: true)) + } + // The desktop app requests a reload when what the widget shows changes (at most every 20 + // minutes) and writes every poll, so this schedule is the fallback that picks up a change + // made inside that window. WidgetKit's documented budget is a few dozen reloads a day. + completion(Timeline(entries: entries, policy: .after(now.addingTimeInterval(30 * 60)))) } private func readEntry(now: Date = Date()) -> SnapshotEntry { diff --git a/app/Sources/OpenCodexWidget/SnapshotReader.swift b/app/Sources/OpenCodexWidget/SnapshotReader.swift index 0c3b8fc66b..4a687be63e 100644 --- a/app/Sources/OpenCodexWidget/SnapshotReader.swift +++ b/app/Sources/OpenCodexWidget/SnapshotReader.swift @@ -6,12 +6,6 @@ public enum ReadFailure: String, Error, Equatable, Sendable { case corrupt } -public extension WidgetSnapshot { - func isStale(now: Date = Date()) -> Bool { - now.timeIntervalSince1970 - generatedAt > 600 - } -} - public struct SnapshotReader: Sendable { public init() {} diff --git a/app/Sources/OpenCodexWidget/Views.swift b/app/Sources/OpenCodexWidget/Views.swift index edc91171f4..6c3f25677b 100644 --- a/app/Sources/OpenCodexWidget/Views.swift +++ b/app/Sources/OpenCodexWidget/Views.swift @@ -2,7 +2,6 @@ import SwiftUI import WidgetKit import MenuBarCore -@available(macOS 14, *) struct OpenCodexWidgetView: View { let entry: SnapshotEntry @Environment(\.widgetFamily) private var family @@ -91,7 +90,7 @@ struct OpenCodexWidgetView: View { quotaView(snapshot) } else if let chart = snapshot.chart { VStack(alignment: .leading, spacing: 5) { - Text("Last \(windowLabel(chart))").font(.caption).foregroundStyle(.secondary) + Text("Last \(windowLabel(chart))\(chart.incomplete == true ? " · partial" : "")").font(.caption).foregroundStyle(.secondary) chartView(chart, flexible: false).widgetAccentable() } } else { @@ -117,7 +116,7 @@ struct OpenCodexWidgetView: View { } } if let chart = snapshot.chart { - Text("Last \(windowLabel(chart)) · \(chart.series.count) models") + Text("Last \(windowLabel(chart)) · \(chart.series.count) models\(chart.incomplete == true ? " · partial" : "")") .font(.caption).foregroundStyle(.secondary) chartView(chart, flexible: true) .frame(maxHeight: .infinity) @@ -284,8 +283,12 @@ struct OpenCodexWidgetView: View { } private func updated(_ snapshot: WidgetSnapshot) -> some View { - let text = snapshot.lastUpdated.map { "Updated \(Format.age(Date(timeIntervalSince1970: $0)))" } ?? "Not updated" - return Text(text).font(.caption2).foregroundStyle(entry.stale ? .orange : .secondary).lineLimit(1) + // A relative date keeps counting while the widget is visible, so the age stays true between + // reloads instead of freezing at whatever it was when the timeline entry was made. + let text = snapshot.lastUpdated.map { + Text("Updated ") + Text(Date(timeIntervalSince1970: $0), style: .relative) + Text(" ago") + } ?? Text("Not updated") + return text.font(.caption2).foregroundStyle(entry.stale ? .orange : .secondary).lineLimit(1) } private func failureView(_ failure: ReadFailure) -> some View { @@ -301,14 +304,13 @@ struct OpenCodexWidgetView: View { } } -@available(macOS 14, *) +@main struct OpenCodexWidgetBundle: WidgetBundle { var body: some Widget { OpenCodexWidget() } } -@available(macOS 14, *) struct OpenCodexWidget: Widget { let kind = "OpenCodexWidget" diff --git a/app/Sources/OpenCodexWidget/main.swift b/app/Sources/OpenCodexWidget/main.swift deleted file mode 100644 index 7eeda56386..0000000000 --- a/app/Sources/OpenCodexWidget/main.swift +++ /dev/null @@ -1,2 +0,0 @@ -// WidgetKit enters through _NSExtensionMain; this file keeps the executable target's -// source directory populated without adding a competing Swift-generated main. diff --git a/app/Widget-Info.plist b/app/Widget-Info.plist index 568e5fce75..df49910429 100644 --- a/app/Widget-Info.plist +++ b/app/Widget-Info.plist @@ -7,6 +7,15 @@ CFBundleIdentifiercom.opencodex.desktop.widget CFBundleInfoDictionaryVersion6.0 CFBundleNameOpenCodex + CFBundleDisplayNameOpenCodex + + CFBundleSupportedPlatforms + MacOSX CFBundlePackageTypeXPC! CFBundleShortVersionString0.0.0 CFBundleVersion0.0.0 diff --git a/assets/pr-gate-screenshot-required.png b/assets/pr-gate-screenshot-required.png deleted file mode 100644 index 3560ebb6fc..0000000000 Binary files a/assets/pr-gate-screenshot-required.png and /dev/null differ diff --git a/assets/pr-screenshots/client-compaction-dashboard.png b/assets/pr-screenshots/client-compaction-dashboard.png deleted file mode 100644 index 89c3cf1c40..0000000000 Binary files a/assets/pr-screenshots/client-compaction-dashboard.png and /dev/null differ diff --git a/assets/pr-screenshots/main-reauth-cancel-retryable.png b/assets/pr-screenshots/main-reauth-cancel-retryable.png deleted file mode 100644 index 306725d842..0000000000 Binary files a/assets/pr-screenshots/main-reauth-cancel-retryable.png and /dev/null differ diff --git a/assets/pr-screenshots/usage-chart-review.png b/assets/pr-screenshots/usage-chart-review.png deleted file mode 100644 index 3f2582929a..0000000000 Binary files a/assets/pr-screenshots/usage-chart-review.png and /dev/null differ diff --git a/assets/pr2950-capacity-expiry.png b/assets/pr2950-capacity-expiry.png deleted file mode 100644 index 10e3c995b9..0000000000 Binary files a/assets/pr2950-capacity-expiry.png and /dev/null differ diff --git a/assets/pr715-selection-order.png b/assets/pr715-selection-order.png deleted file mode 100644 index 7d178d3739..0000000000 Binary files a/assets/pr715-selection-order.png and /dev/null differ diff --git a/assets/request-pacing-dashboard.jpg b/assets/request-pacing-dashboard.jpg deleted file mode 100644 index 8a4f6008cf..0000000000 Binary files a/assets/request-pacing-dashboard.jpg and /dev/null differ diff --git a/assets/sponsors/tokenlab-dark.png b/assets/sponsors/tokenlab-dark.png new file mode 100644 index 0000000000..7c1ef48dac Binary files /dev/null and b/assets/sponsors/tokenlab-dark.png differ diff --git a/assets/sponsors/tokenlab-light.png b/assets/sponsors/tokenlab-light.png new file mode 100644 index 0000000000..37a0f17ef6 Binary files /dev/null and b/assets/sponsors/tokenlab-light.png differ diff --git a/assets/zh-tw-providers.png b/assets/zh-tw-providers.png deleted file mode 100644 index b3ec814dc4..0000000000 Binary files a/assets/zh-tw-providers.png and /dev/null differ diff --git a/bin/ocx.mjs b/bin/ocx.mjs index 7b3a54eaa2..406748d8e1 100755 --- a/bin/ocx.mjs +++ b/bin/ocx.mjs @@ -12,6 +12,21 @@ import { spawn, spawnSync } from "node:child_process"; import { STOP_HISTORY_INCOMPLETE_EXIT_CODE } from "../src/update/stop-contract.mjs"; import { probeProxyLiveness } from "../src/update/proxy-liveness-probe.mjs"; import { decidePostStopUpdate } from "../src/update/stop-decision.mjs"; +import { + inspectPackageRuntimeLiveness, + planStoppedRuntimeRecovery, + planUpdateRuntimeHandling, +} from "../src/update/runtime-ownership.mjs"; +import { + inspectInstallStateBytes, + selectAuthoritativeServiceState, + serviceStateFilesFor, +} from "../src/service/install-state-contract.mjs"; +import { + acquireOwnershipMutationLease, + ownershipMutationLeaseChildEnvironment, + unprivilegedOwnershipMutationEnvironment, +} from "../src/service/ownership-mutation-lease.mjs"; import { randomBytes } from "node:crypto"; import { createRequire } from "node:module"; import { existsSync, readFileSync, readdirSync } from "node:fs"; @@ -21,12 +36,13 @@ import { fileURLToPath } from "node:url"; import { isRealBunBinary } from "../src/lib/bun-binary-validator.mjs"; import { npmInvocation } from "../src/update/npm-invocation.mjs"; import { pnpmInvocationForPath, resolvePnpmCommands } from "../src/update/pnpm-invocation.mjs"; -import { detectInstallFromPath } from "../src/update/install-detection.mjs"; +import { detectInstallOwnershipFromPath } from "../src/update/install-detection.mjs"; import { pnpmOwnerInvocation, resolvePnpmGlobalOwner, runPnpmGlobalUpdate, } from "../src/update/pnpm-global-install.mjs"; +import { PNPM_READ_CWD, withPnpmCommandCwd, pnpmReadEnvironment } from "../src/update/pnpm-read-policy.mjs"; import { checkRegistryPackageIntegrity } from "../src/update/registry-integrity.mjs"; import { hasPendingTeardownIn } from "../src/config/pending-teardown-names.mjs"; import { @@ -34,13 +50,17 @@ import { runNpmCachePreflight, } from "../src/update/npm-cache-preflight.mjs"; import { handoffWindowsTrayForUpdate, planWindowsTrayUpdate } from "../src/update/tray-update-plan.mjs"; -import { bootRestoreProbe, transactionalNpmUpdate } from "../src/update/transactional-install.mjs"; +import { bootRestoreProbe, launcherUsableAfterNpmUpdate, transactionalNpmUpdate } from "../src/update/transactional-install.mjs"; +import { npmUpdateFailureGuidance } from "../src/update/update-failure-guidance.mjs"; import { CODEX_CLI_VERSION_MANAGER_ROOT_ENV_SLOTS, isCodexCliUpdateInspectionArgv, } from "../src/update/codex-cli-update-launch-policy.mjs"; const PKG = "@bitkyc08/opencodex"; +const UPDATE_RECOVERY_READY_MS = 30_000; +const UPDATE_RECOVERY_POLL_MS = 100; +const UPDATE_RECOVERY_SLEEP = new Int32Array(new SharedArrayBuffer(4)); try { process.cwd(); } catch { @@ -52,7 +72,8 @@ try { } const require = createRequire(import.meta.url); const here = dirname(fileURLToPath(import.meta.url)); -const installMethod = detectInstallFromPath(here, { exists: existsSync }); +const installOwnership = detectInstallOwnershipFromPath(here, { exists: existsSync }); +const installMethod = installOwnership.installer; const cliPath = join(here, "..", "src", "cli", "index.ts"); const NODE_LAUNCH_CONTEXT_ENV = "OCX_NODE_LAUNCH_CONTEXT"; const NODE_LAUNCH_PROOF_PREFIX = "--ocx-internal-launch-proof="; @@ -188,6 +209,8 @@ function runPackageManagerSelfUpdate(manager) { encoding: "utf8", timeout: 20_000, windowsHide: true, + cwd: PNPM_READ_CWD, + env: pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(process.env)), ...invocation.options, }); }, @@ -201,6 +224,20 @@ function runPackageManagerSelfUpdate(manager) { const managerInvocation = args => manager === "pnpm" ? pnpmOwnerInvocation(owner, args) : npmInvocation(args); + // Read-only pnpm probes run from the installed package directory with project pnpmfiles + // disabled, so an attacker-controlled cwd cannot execute hooks during the update check. + const readProbeOptions = invocation => ({ + encoding: "utf8", + timeout: 12000, + windowsHide: true, + ...(manager === "pnpm" + ? { + cwd: PNPM_READ_CWD, + env: pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(invocation.env ?? process.env)), + } + : invocation.env ? { env: invocation.env } : {}), + ...invocation.options, + }); const latestInvocation = managerInvocation(["view", `${PKG}@${tag}`, "version"]); const installArgs = manager === "pnpm" ? ["add", "-g", "--allow-build=bun", `${PKG}@${tag}`] @@ -210,13 +247,7 @@ function runPackageManagerSelfUpdate(manager) { console.error(`opencodex: could not resolve ${manager} from a trusted absolute PATH entry; aborting before stopping the proxy.`); process.exit(1); } - const latestResult = spawnSync(latestInvocation.file, latestInvocation.args, { - encoding: "utf8", - timeout: 12000, - windowsHide: true, - ...(latestInvocation.env ? { env: latestInvocation.env } : {}), - ...latestInvocation.options, - }); + const latestResult = spawnSync(latestInvocation.file, latestInvocation.args, readProbeOptions(latestInvocation)); const latest = latestResult.status === 0 && typeof latestResult.stdout === "string" ? latestResult.stdout.trim() : ""; console.log(`opencodex v${current} (installed via ${manager}, tag ${tag})`); @@ -228,13 +259,7 @@ function runPackageManagerSelfUpdate(manager) { const integrity = checkRegistryPackageIntegrity(PKG, latest || null, args => { const invocation = managerInvocation(args); if (!invocation) return { status: 1 }; - return spawnSync(invocation.file, invocation.args, { - encoding: "utf8", - timeout: 12000, - windowsHide: true, - ...(invocation.env ? { env: invocation.env } : {}), - ...invocation.options, - }); + return spawnSync(invocation.file, invocation.args, readProbeOptions(invocation)); }); if (integrity.ok === false) { console.error(`opencodex: ${integrity.reason}; aborting before stopping the proxy.`); @@ -256,8 +281,45 @@ function runPackageManagerSelfUpdate(manager) { // Remember whether a background service manages the proxy BEFORE stopping — `ocx stop` // unloads it, so a successful update must refresh and restart it afterwards. - const serviceStatePath = join(configDir(), "service-state.json"); - const serviceWasInstalled = existsSync(serviceStatePath); + const allServiceStatePaths = serviceStateFilesFor(configDir(), join(homedir(), ".opencodex")); + // The test guard's legacy path is the developer's real home. Production always reads the + // same active-home + default-home observations as the Bun resolver. + const serviceStatePaths = process.env.OCX_TEST_HOME_GUARD === "1" + ? allServiceStatePaths.slice(0, 1) + : allServiceStatePaths; + const serviceWasInstalled = serviceStatePaths.some(path => existsSync(path)); + // What this update may do to the runtime. The same rule the Bun updater applies, from the + // same module: a desktop takeover vetoes both the stop and the service refresh below. + const readServiceState = () => selectAuthoritativeServiceState( + serviceStatePaths.map(path => inspectInstallStateBytes(path, at => readFileSync(at, "utf8"))), + ); + const readOwnership = () => { + const selected = readServiceState(); + if (selected.kind === "unknown") return { ownership: null, ownershipUnknown: true, subjectToken: "unknown" }; + if (selected.kind === "none") return { + ownership: null, ownershipUnknown: false, subjectToken: JSON.stringify(["none", selected.revision]), + }; + const ownership = selected.state.ownership ?? null; + return { + ownership, + ownershipUnknown: false, + subjectToken: JSON.stringify(ownership + ? ["owned", selected.revision, ownership] + : ["none", selected.revision]), + }; + }; + const ownershipIdentity = observation => observation.ownershipUnknown + ? null + : JSON.stringify(observation.ownership + ? ["owned", observation.ownership.owner, observation.ownership.installId, observation.ownership.consentGeneration] + : ["none"]); + const initialOwnership = readOwnership(); + let runtimePlan = planUpdateRuntimeHandling({ ...initialOwnership, serviceInstalled: serviceWasInstalled }); + if (runtimePlan.notice) console.log(runtimePlan.notice); + if (!runtimePlan.mayReplacePackage) { + console.error("opencodex: update stopped before tray handoff, runtime stop, or package replacement because runtime ownership is unknown."); + process.exit(1); + } const trayBeforeUpdate = planWindowsTrayUpdate( process.platform === "win32" ? trayInstallState() : { installed: false, running: false }, ); @@ -271,10 +333,11 @@ function runPackageManagerSelfUpdate(manager) { } /** Register from scratch, preserving the recorded backend. Only for a genuinely absent service. */ function serviceInstallArgs() { - try { - const state = JSON.parse(readFileSync(serviceStatePath, "utf8")); - if (state.backend === "native") return [postUpdateLauncher, "service", "install", "--native"]; - } catch { /* missing or corrupt — fall through to default */ } + const selected = readServiceState(); + if (selected.kind === "unknown") throw new Error(`service backend is unknown: ${selected.reason}`); + if (selected.kind === "state" && selected.state.backend === "native") { + return [postUpdateLauncher, "service", "install", "--native"]; + } return [postUpdateLauncher, "service", "install"]; } /** @@ -302,39 +365,45 @@ function runPackageManagerSelfUpdate(manager) { } } - // Capture listen target before stop clears runtime-port.json (mirrors GUI/CLI update worker). - // Do not treat a live runtime port of 10100 as "missing" — track whether the read succeeded. + function readCurrentRuntimeTarget() { + let raw; + try { + raw = readFileSync(join(configDir(), "runtime-port.json"), "utf8"); + } catch (error) { + return error && typeof error === "object" && "code" in error && error.code === "ENOENT" + ? { kind: "absent" } + : { kind: "unknown" }; + } + try { + const rt = JSON.parse(raw); + const pid = Number(rt?.pid); + if (!Number.isFinite(rt?.port) || rt.port <= 0 || rt.port > 65535 + || !Number.isSafeInteger(pid) || pid <= 0) return { kind: "unknown" }; + return { kind: "target", target: { + pid, + port: Math.trunc(rt.port), + hostname: typeof rt.hostname === "string" && rt.hostname.trim() !== "" + ? rt.hostname.trim() + : null, + } }; + } catch { return { kind: "unknown" }; } + } + + // Capture the recovery target before stop clears runtime-port.json. Replacement safety + // re-reads this record under the mutation lease instead of trusting this snapshot. let bakePort = 10100; // The hostname travels with the port: a proxy bound to ::1 or a specific interface is // invisible to a probe that assumes 127.0.0.1, and "no answer" would then read as // "stopped" for exactly the proxy the probe exists to find. let bakeHostname = "127.0.0.1"; - let sawRuntimePort = false; - let sawRuntimeHostname = false; - try { - const rt = JSON.parse(readFileSync(join(configDir(), "runtime-port.json"), "utf8")); - if (Number.isFinite(rt?.port) && rt.port > 0 && rt.port <= 65535) { - // Only trust runtime when its pid still looks alive (stale crash leftovers fall back to config). - const rtPid = Number(rt?.pid); - let runtimeLive = false; - if (Number.isSafeInteger(rtPid) && rtPid > 0) { - try { - process.kill(rtPid, 0); - runtimeLive = true; - } catch (e) { - if (e && typeof e === "object" && "code" in e && e.code === "EPERM") runtimeLive = true; - } - } - if (runtimeLive) { - bakePort = Math.trunc(rt.port); - if (typeof rt?.hostname === "string" && rt.hostname.trim() !== "") { - bakeHostname = rt.hostname.trim(); - sawRuntimeHostname = true; - } - sawRuntimePort = true; - } - } - } catch { /* fall through to config */ } + const initialRuntimeObservation = readCurrentRuntimeTarget(); + const initialRuntimeTarget = initialRuntimeObservation.kind === "target" ? initialRuntimeObservation.target : null; + let sawRuntimePort = initialRuntimeTarget !== null; + let sawRuntimeHostname = initialRuntimeTarget?.hostname !== null && initialRuntimeTarget?.hostname !== undefined; + if (initialRuntimeTarget) { + bakePort = initialRuntimeTarget.port; + if (initialRuntimeTarget.hostname) bakeHostname = initialRuntimeTarget.hostname; + } // Port and hostname resolve INDEPENDENTLY: a legacy runtime record carries a port and no // hostname, and skipping config in that case probed 127.0.0.1 for a proxy bound to ::1. if (!sawRuntimePort || bakeHostname === "127.0.0.1") { @@ -350,6 +419,18 @@ function runPackageManagerSelfUpdate(manager) { } // Wildcard and bracketed-IPv6 normalization lives in probeProxyLiveness, so both lanes // get it from one place. + function currentPackageRuntimeLiveness() { + return inspectPackageRuntimeLiveness({ + capturedTarget: { port: bakePort, hostname: bakeHostname }, + readCurrentTarget: () => { + const current = readCurrentRuntimeTarget(); + return current.kind === "target" + ? { kind: "target", target: { port: current.target.port, hostname: current.target.hostname ?? bakeHostname } } + : current; + }, + probe: target => probeProxyLiveness(target.port, target.hostname), + }).overall; + } const launcher = fileURLToPath(import.meta.url); // The pnpm owner preflight has verified this package tree and global group. Keep that exact @@ -359,14 +440,21 @@ function runPackageManagerSelfUpdate(manager) { ? join(owner.packagePath, "bin", "ocx.mjs") : launcher; let postUpdateLauncherUsable = true; + let delegatedOwnershipMutationToken = null; + const mutationChildEnvironment = () => delegatedOwnershipMutationToken + ? ownershipMutationLeaseChildEnvironment(process.env, delegatedOwnershipMutationToken) + : unprivilegedOwnershipMutationEnvironment(process.env); function startProxyDirectly() { if (!postUpdateLauncherUsable || !existsSync(postUpdateLauncher)) { console.error("opencodex: cannot restart the proxy because the launcher is missing; reinstall opencodex manually."); - return; + return false; } - const env = { ...process.env }; + const env = mutationChildEnvironment(); delete env.OCX_SERVICE; + // The restarted proxy is an ordinary owner; only a sibling's own replacement carries this. + delete env.OCX_SIBLING_OF_PORT; + delete env.OCX_SIBLING_HANDOFF_NONCE; console.log(`Attempting to restart the proxy on port ${bakePort}.`); const child = spawn(process.execPath, [postUpdateLauncher, "start", "--port", String(bakePort)], { detached: true, @@ -378,13 +466,24 @@ function runPackageManagerSelfUpdate(manager) { console.error(`opencodex: direct proxy restart failed: ${error.message}`); }); child.unref(); + const deadline = Date.now() + UPDATE_RECOVERY_READY_MS; + while (Date.now() < deadline) { + const current = readCurrentRuntimeTarget(); + if (current.kind === "target" + && probeProxyLiveness(current.target.port, current.target.hostname ?? bakeHostname) === "live") return true; + Atomics.wait(UPDATE_RECOVERY_SLEEP, 0, 0, UPDATE_RECOVERY_POLL_MS); + } + console.error("opencodex: the recovery proxy did not publish a healthy runtime before the recovery deadline."); + return false; } function refreshBackgroundServiceOrStartDirect() { const prevBake = process.env.OCX_BAKE_PORT; process.env.OCX_BAKE_PORT = String(bakePort); try { - let svc = spawnSync(process.execPath, serviceRefreshArgs(), { stdio: "inherit", windowsHide: true }); + let svc = spawnSync(process.execPath, serviceRefreshArgs(), { + stdio: "inherit", windowsHide: true, env: mutationChildEnvironment(), + }); // `serviceWasInstalled` is inferred from service-state.json alone, which can be // STALE — present while the registration is gone. Repair refuses that case by // design, and its thrown Error is indistinguishable from any other failure at @@ -395,7 +494,9 @@ function runPackageManagerSelfUpdate(manager) { // could re-register a service the user just uninstalled. if (svc.status !== 0 && readServiceInstalledFromStatus(postUpdateLauncher) === false) { console.log("No registered service found — installing it instead."); - svc = spawnSync(process.execPath, serviceInstallArgs(), { stdio: "inherit", windowsHide: true }); + svc = spawnSync(process.execPath, serviceInstallArgs(), { + stdio: "inherit", windowsHide: true, env: mutationChildEnvironment(), + }); } let needDirectStart = svc.status !== 0; if (!needDirectStart) { @@ -421,6 +522,14 @@ function runPackageManagerSelfUpdate(manager) { } } if (needDirectStart) { + // Re-read rather than reuse the plan from before the package install: the app can + // claim the runtime during an update that takes minutes, and the refusal that repair + // just returned is indistinguishable from any other failure at this layer. + const nowOwned = planUpdateRuntimeHandling({ ...readOwnership(), serviceInstalled: true }); + if (!nowOwned.mayStopRuntime) { + console.warn(nowOwned.notice ?? "opencodex: the background runtime is owned elsewhere; not starting a second proxy."); + return; + } // Repair normally avoids elevation for a healthy registration, but a stale Windows // scheduler definition can require it. It can also fail — or exit 0 while leaving // a non-viable manager. Fall back to a direct detached proxy start so the @@ -439,189 +548,280 @@ function runPackageManagerSelfUpdate(manager) { } } - // Never replace package files under a live proxy — stop it first (full `ocx stop` - // semantics: graceful drain, service stop, native Codex restore). Gate on the service - // and the runtime-port record too: a service-managed or orphaned proxy can be live - // while ocx.pid is stale/missing. - if (trayBeforeUpdate.stopBeforeReplacement) { - console.log("⏹ Handing off the Windows tray before updating..."); - try { - handoffWindowsTrayForUpdate(trayBeforeUpdate, { - stop: () => { - const stopped = runTrayLifecycle(launcher, "stop"); - return { exitStatus: stopped.status, running: trayInstallState().running }; - }, - start: () => runTrayLifecycle(launcher, "start"), - }); - } catch { - console.error("opencodex: could not stop the Windows tray; aborting before package replacement."); + const updateLease = acquireOwnershipMutationLease(serviceStatePaths); + delegatedOwnershipMutationToken = updateLease.token; + let updateLeaseReleased = false; + const releaseUpdateLease = () => { + if (updateLeaseReleased) return; + updateLeaseReleased = true; + delegatedOwnershipMutationToken = null; + updateLease.release(); + }; + + let res; + let npmFailure = null; + try { + // Stop authority is decided under the same lease the child joins. A takeover between the + // earlier preflight and this boundary therefore blocks stop before it is sent. + const lockedOwnership = readOwnership(); + const lockedPlan = planUpdateRuntimeHandling({ ...lockedOwnership, serviceInstalled: serviceWasInstalled }); + if (lockedOwnership.subjectToken !== initialOwnership.subjectToken || !lockedPlan.mayReplacePackage) { + releaseUpdateLease(); + console.error(lockedPlan.notice + ?? "opencodex: update stopped because runtime ownership changed before stop authorization; rerun from the beginning."); process.exit(1); } - } - const hasRuntimeState = - existsSync(join(configDir(), "ocx.pid")) || existsSync(join(configDir(), "runtime-port.json")); + runtimePlan = lockedPlan; + const stoppedOwnershipIdentity = ownershipIdentity(lockedOwnership); - function recoverStoppedRuntimeAfterFailure() { - if (!postUpdateLauncherUsable) { - console.error("opencodex: no verified active launcher remains for automatic recovery; reinstall opencodex manually."); - return; + // Never replace package files under a live proxy — stop it first (full `ocx stop` + // semantics: graceful drain, service stop, native Codex restore). Gate on the service + // and the runtime-port record too: a service-managed or orphaned proxy can be live + // while ocx.pid is stale/missing. + if (trayBeforeUpdate.stopBeforeReplacement) { + console.log("⏹ Handing off the Windows tray before updating..."); + try { + handoffWindowsTrayForUpdate(trayBeforeUpdate, { + stop: () => { + const stopped = runTrayLifecycle(launcher, "stop"); + return { exitStatus: stopped.status, running: trayInstallState().running }; + }, + start: () => runTrayLifecycle(launcher, "start"), + }); + } catch { + releaseUpdateLease(); + console.error("opencodex: could not stop the Windows tray; aborting before package replacement."); + process.exit(1); + } } - if (serviceWasInstalled) { - console.warn("opencodex: update failed after stopping the proxy — restoring the previous background service."); - refreshBackgroundServiceOrStartDirect(); - } else if (hasRuntimeState) { - console.warn("opencodex: update failed after stopping the proxy — restarting the previous version directly."); - startProxyDirectly(); + const hasRuntimeState = + existsSync(join(configDir(), "ocx.pid")) || existsSync(join(configDir(), "runtime-port.json")); + let stopAttempted = false; + + function recoverStoppedRuntimeAfterFailure(reason) { + const planRecovery = () => { + const recoveryOwnership = readOwnership(); + const liveness = currentPackageRuntimeLiveness(); + return { + liveness, + plan: planStoppedRuntimeRecovery({ + stopAttempted, + ...recoveryOwnership, + sameOwner: ownershipIdentity(recoveryOwnership) === stoppedOwnershipIdentity, + liveness, + serviceInstalled: serviceWasInstalled, + launcherUsable: postUpdateLauncherUsable, + hadRuntimeState: hasRuntimeState, + }), + }; + }; + let { liveness: recoveryLiveness, plan: recovery } = planRecovery(); + if (recovery.action === "service") { + // The service manager starts the proxy outside this process tree, so it cannot join this + // lease, and holding the lease through the repair's health wait keeps that proxy from + // starting (#5760). Release it as the successful path does, then decide again. + releaseUpdateLease(); + ({ liveness: recoveryLiveness, plan: recovery } = planRecovery()); + } + if (recovery.reason === "ownership-unknown") { + console.error(`opencodex: ${reason}; runtime ownership is unknown, so automatic recovery was refused. Run 'ocx status --json' and repair the service-state record before retrying.`); + } else if (recovery.reason === "ownership-transferred") { + console.log("opencodex: runtime ownership moved to another installation; the stopped CLI runtime was not revived."); + } else if (recovery.reason.startsWith("runtime-")) { + console.error(`opencodex: ${reason}; package runtime liveness is ${recoveryLiveness}, so automatic recovery was refused.`); + } else if (recovery.reason === "launcher-unavailable") { + console.error("opencodex: no verified active launcher remains for automatic recovery; reinstall opencodex manually."); + } else if (recovery.action === "service") { + console.warn(`opencodex: ${reason} after stopping the proxy — restoring the previous background service.`); + refreshBackgroundServiceOrStartDirect(); + } else if (recovery.action === "direct") { + console.warn(`opencodex: ${reason} after stopping the proxy — restarting the previous version directly.`); + startProxyDirectly(); + } + return recovery; } - } - // An outstanding pending-teardown receipt is a fourth reason to run the stop. After a - // parent crashed mid-deferral the service, pid and runtime records can all be absent - // while the shared client config still points at a proxy that is gone; installing over - // that silently skips the recovery the receipt was written to trigger (#3008). Presence - // is the whole test here — the launcher cannot parse it, and `ocx stop` is what decides - // whether the obligation is safe to finish. - const hasPendingTeardown = hasPendingTeardownIn(readdirSync, configDir()); - if (serviceWasInstalled || hasRuntimeState || hasPendingTeardown) { - console.log("⏹ Stopping the running proxy before updating..."); - const stopRes = spawnSync(process.execPath, [launcher, "stop"], { stdio: "inherit", windowsHide: true }); - const stillHasRuntimeState = - existsSync(join(configDir(), "ocx.pid")) || existsSync(join(configDir(), "runtime-port.json")); - // A history-only failure means teardown succeeded and a backup manifest is waiting for - // review: the proxy is down and replacing package files is safe. Every other nonzero - // status is a stop that did not finish, and a signal kill (status null) says nothing - // about whether it did - both abort, because replacing files under a live server - // leaves it running mixed old and new modules (#3008). - // The same decision the Bun updater makes, from the same module (#3008). Absent PID and - // runtime files are weak evidence, so the captured endpoint is asked; "unknown" aborts - // because a silent listener is exactly the state where replacing files is dangerous. - const decision = decidePostStopUpdate({ - status: stopRes.status, - hasRuntimeState: stillHasRuntimeState, - // Re-checked AFTER the stop: a quarantined receipt lets the stop itself succeed - // (there is nothing left to stop), so a pre-stop check alone let the retry install - // over a teardown that never ran. - teardownOutstanding: hasPendingTeardownIn(readdirSync, configDir()), - liveness: probeProxyLiveness(bakePort, bakeHostname), - }); - const historyOnlyStop = decision.reason === "history-only"; - if (!decision.proceed) { - if (trayBeforeUpdate.restoreOnFailure) runTrayLifecycle(launcher, "start"); - if (decision.reason === "teardown-outstanding") { - console.error("opencodex: a shared teardown from an earlier stop is still outstanding and needs manual review; aborting the update."); - console.error("opencodex: confirm no proxy is running, run 'ocx restore', then remove the pending-teardown file in the opencodex home."); - } else console.error(decision.reason === "proxy-unknown" - ? `opencodex: could not confirm the proxy on ${bakeHostname}:${bakePort} is stopped; aborting the update. Run 'ocx stop' and retry.` - : "opencodex: could not stop the running proxy; aborting the update. Run 'ocx stop' and retry."); + // An outstanding pending-teardown receipt is a fourth reason to run the stop. After a + // parent crashed mid-deferral the service, pid and runtime records can all be absent + // while the shared client config still points at a proxy that is gone; installing over + // that silently skips the recovery the receipt was written to trigger (#3008). Presence + // is the whole test here — the launcher cannot parse it, and `ocx stop` is what decides + // whether the obligation is safe to finish. + const hasPendingTeardown = hasPendingTeardownIn(readdirSync, configDir()); + const stopNeeded = serviceWasInstalled || hasRuntimeState || hasPendingTeardown; + if (stopNeeded && !runtimePlan.mayStopRuntime) { + releaseUpdateLease(); + console.error(runtimePlan.notice + ?? "opencodex: update stopped because this installation may not stop the current runtime."); process.exit(1); } - if (historyOnlyStop || historyRestoreIncomplete()) { - console.warn( - "opencodex: WARNING — Codex resume-history metadata restore is incomplete (a backup manifest remains).\n" + - " The DB may be busy or the manifest/target may need review; untracked routed history is intentionally unchanged.\n" + - " After the update: close the Codex app, run 'ocx doctor', then run 'ocx stop' once to retry.", - ); + if (stopNeeded) { + stopAttempted = true; + console.log("⏹ Stopping the running proxy before updating..."); + const stopRes = spawnSync(process.execPath, [launcher, "stop"], { + stdio: "inherit", windowsHide: true, env: mutationChildEnvironment(), + }); + const stillHasRuntimeState = + existsSync(join(configDir(), "ocx.pid")) || existsSync(join(configDir(), "runtime-port.json")); + // A history-only failure means teardown succeeded and a backup manifest is waiting for + // review: the proxy is down and replacing package files is safe. Every other nonzero + // status is a stop that did not finish, and a signal kill (status null) says nothing + // about whether it did - both abort, because replacing files under a live server + // leaves it running mixed old and new modules (#3008). + // The same decision the Bun updater makes, from the same module (#3008). Absent PID and + // runtime files are weak evidence, so the captured endpoint is asked; "unknown" aborts + // because a silent listener is exactly the state where replacing files is dangerous. + const decision = decidePostStopUpdate({ + status: stopRes.status, + hasRuntimeState: stillHasRuntimeState, + // Re-checked AFTER the stop: a quarantined receipt lets the stop itself succeed + // (there is nothing left to stop), so a pre-stop check alone let the retry install + // over a teardown that never ran. + teardownOutstanding: hasPendingTeardownIn(readdirSync, configDir()), + liveness: probeProxyLiveness(bakePort, bakeHostname), + }); + const historyOnlyStop = decision.reason === "history-only"; + if (!decision.proceed) { + if (trayBeforeUpdate.restoreOnFailure) runTrayLifecycle(launcher, "start"); + if (decision.reason === "teardown-outstanding") { + console.error("opencodex: a shared teardown from an earlier stop is still outstanding and needs manual review; aborting the update."); + console.error("opencodex: confirm no proxy is running, run 'ocx restore', then remove the pending-teardown file in the opencodex home."); + } else console.error(decision.reason === "proxy-unknown" + ? `opencodex: could not confirm the proxy on ${bakeHostname}:${bakePort} is stopped; aborting the update. Run 'ocx stop' and retry.` + : "opencodex: could not stop the running proxy; aborting the update. Run 'ocx stop' and retry."); + releaseUpdateLease(); + process.exit(1); + } + if (historyOnlyStop || historyRestoreIncomplete()) { + console.warn( + "opencodex: WARNING — Codex resume-history metadata restore is incomplete (a backup manifest remains).\n" + + " The DB may be busy or the manifest/target may need review; untracked routed history is intentionally unchanged.\n" + + " After the update: close the Codex app, run 'ocx doctor', then run 'ocx stop' once to retry.", + ); + } + if (decision.reason === "history-deferred") { + // The reported #4718 path is this lane. Nothing was restored, so this is a different + // sentence from the manifest warning above: an operator told "history metadata is + // incomplete" would assume config and catalog already came back. + console.warn( + "opencodex: WARNING — the shared teardown was refused by the Codex history preflight and restored nothing.\n" + + " Config, catalog, history and provenance were preserved, and the teardown receipt was kept.\n" + + " The proxy is down, so the update continues; close the Codex app and run 'ocx stop' once afterwards to finish the restore.", + ); + } } - if (decision.reason === "history-deferred") { - // The reported #4718 path is this lane. Nothing was restored, so this is a different - // sentence from the manifest warning above: an operator told "history metadata is - // incomplete" would assume config and catalog already came back. - console.warn( - "opencodex: WARNING — the shared teardown was refused by the Codex history preflight and restored nothing.\n" + - " Config, catalog, history and provenance were preserved, and the teardown receipt was kept.\n" + - " The proxy is down, so the update continues; close the Codex app and run 'ocx stop' once afterwards to finish the restore.", - ); + + const replacementOwnership = readOwnership(); + const replacementPlan = planUpdateRuntimeHandling({ ...replacementOwnership, serviceInstalled: serviceWasInstalled }); + const replacementLiveness = currentPackageRuntimeLiveness(); + if (replacementOwnership.subjectToken !== initialOwnership.subjectToken + || !replacementPlan.mayReplacePackage + || replacementLiveness !== "dead") { + recoverStoppedRuntimeAfterFailure("replacement was refused"); + releaseUpdateLease(); + if (trayBeforeUpdate.restoreOnFailure) runTrayLifecycle(launcher, "start"); + console.error(replacementPlan.notice + ?? "opencodex: update stopped because runtime ownership or liveness changed after the stop decision; rerun from the beginning."); + process.exit(1); } - } - // npm keeps the existing stage -> verify -> swap -> rollback flow. pnpm owns a - // content-addressable store and generated global shims, so its path uses pnpm's own - // global update operation and verifies the active group instead of renaming files. - console.log(`Updating${latest ? ` to v${latest}` : ""} (${manager === "npm" ? "transactional" : "pnpm-managed"})...`); - let res; - try { - if (manager === "npm") { - const packageDir = resolve(here, ".."); - const tx = transactionalNpmUpdate({ - packageDir, - pkgName: PKG, - targetVersion: latest || undefined, - tag, - runNpm: (args) => { - const invocation = npmInvocation(args); - if (!invocation) return { status: 1 }; - return spawnSync(invocation.file, invocation.args, { - stdio: "inherit", - timeout: 180000, - windowsHide: true, - ...invocation.options, - }); - }, - log: (line) => console.log(line), - }); - postUpdateLauncherUsable = tx.ok - || tx.rolledBack === true - || ["stage", "verify", "swap-backup"].includes(tx.phase); - if (tx.ok) { - res = { status: 0 }; - } else if (tx.phase === "stage" || tx.phase === "verify") { - // Live tree untouched: report and stop. Nothing to roll back. - console.error(`opencodex: update aborted before touching the live install (${tx.phase}): ${tx.error}`); - res = { status: 1 }; - } else { - console.error(`opencodex: update failed (${tx.phase}): ${tx.error}${tx.rolledBack ? " — previous version restored." : ""}`); - res = { status: 1 }; - } - } else { - const update = runPnpmGlobalUpdate({ - packageName: PKG, - currentVersion: current, - targetVersion: latest || undefined, - tag, - owner, - runningPackagePath: resolve(here, ".."), - runPnpm: (args, capture = false) => { - const invocation = pnpmOwnerInvocation(owner, args); - if (!invocation) return { status: 1 }; - return spawnSync(invocation.file, invocation.args, { - stdio: capture ? "pipe" : "inherit", - encoding: "utf8", - timeout: 180000, - windowsHide: true, - env: invocation.env, - ...invocation.options, - }); - }, - log: line => console.log(line), - }); - if (update.ok) { - // pnpm switches the active global group and updates its shim. Continue recovery - // through that fresh package tree, not the old group whose launcher is still - // executing this update. - postUpdateLauncher = join(update.path, "bin", "ocx.mjs"); - res = { status: 0 }; + // npm keeps the existing stage -> verify -> swap -> rollback flow. pnpm owns a + // content-addressable store and generated global shims, so its path uses pnpm's own + // global update operation and verifies the active group instead of renaming files. + console.log(`Updating${latest ? ` to v${latest}` : ""} (${manager === "npm" ? "transactional" : "pnpm-managed"})...`); + try { + if (manager === "npm") { + const packageDir = resolve(here, ".."); + const tx = transactionalNpmUpdate({ + packageDir, + pkgName: PKG, + targetVersion: latest || undefined, + tag, + runNpm: (args) => { + const invocation = npmInvocation(args); + if (!invocation) return { status: 1 }; + return spawnSync(invocation.file, invocation.args, { + ...invocation.options, + stdio: "inherit", + timeout: 180000, + windowsHide: true, + env: unprivilegedOwnershipMutationEnvironment(invocation.options?.env ?? process.env), + }); + }, + log: (line) => console.log(line), + }); + postUpdateLauncherUsable = launcherUsableAfterNpmUpdate(tx); + if (!tx.ok) npmFailure = tx; + if (tx.ok) { + res = { status: 0 }; + } else if (tx.phase === "stage" || tx.phase === "verify") { + // Live tree untouched: report and stop. Nothing to roll back. + console.error(`opencodex: update aborted before touching the live install (${tx.phase}): ${tx.error}`); + res = { status: 1 }; + } else { + console.error(`opencodex: update failed (${tx.phase}): ${tx.error}${tx.rolledBack ? " — previous version restored." : ""}`); + res = { status: 1 }; + } } else { - console.error(`opencodex: ${update.error}${update.rolledBack ? "." : " Manual recovery may be required."}`); - postUpdateLauncherUsable = Boolean(update.activePath); - if (update.activePath) postUpdateLauncher = join(update.activePath, "bin", "ocx.mjs"); - res = { status: 1 }; + const update = runPnpmGlobalUpdate({ + packageName: PKG, + currentVersion: current, + targetVersion: latest || undefined, + tag, + owner, + runningPackagePath: resolve(here, ".."), + runPnpm: (args, capture = false) => { + const invocation = pnpmOwnerInvocation(owner, args); + if (!invocation) return { status: 1 }; + return withPnpmCommandCwd(args, cwd => spawnSync(invocation.file, invocation.args, { + ...invocation.options, + stdio: capture ? "pipe" : "inherit", + encoding: "utf8", + timeout: 180000, + windowsHide: true, + // Reads probe from the package dir; mutations (add -g, rollback) must not + // keep a cwd handle inside the package Windows is replacing. + cwd, + env: pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(invocation.env ?? process.env)), + })); + }, + log: line => console.log(line), + }); + if (update.ok) { + // pnpm switches the active global group and updates its shim. Continue recovery + // through that fresh package tree, not the old group whose launcher is still + // executing this update. + postUpdateLauncher = join(update.path, "bin", "ocx.mjs"); + res = { status: 0 }; + } else { + console.error(`opencodex: ${update.error}${update.rolledBack ? "." : " Manual recovery may be required."}`); + postUpdateLauncherUsable = Boolean(update.activePath); + if (update.activePath) postUpdateLauncher = join(update.activePath, "bin", "ocx.mjs"); + res = { status: 1 }; + } } + } catch (error) { + // An unexpected throw means we cannot prove the live tree is untouched, so the + // legacy in-place install (which deletes live first) is exactly the wrong rescue — + // it recreates the #1849 destruction path. Report and stop; the boot probe and the + // recovery marker cover the swap-window states. + const manual = manager === "pnpm" + ? `pnpm add -g --allow-build=bun ${PKG}@${tag}` + : `npm install -g --allow-scripts=bun ${PKG}@${tag}`; + // An unexpected exception leaves the active package path unproven for either manager. + // Do not run service/tray/proxy recovery through a possibly half-swapped tree. + postUpdateLauncherUsable = false; + console.error(`opencodex: ${manager} update failed unexpectedly (${error?.message ?? error}). ` + + `The live install was not knowingly modified; run 'ocx update' again or reinstall with ${manual}.`); + res = { status: 1 }; } - } catch (error) { - // An unexpected throw means we cannot prove the live tree is untouched, so the - // legacy in-place install (which deletes live first) is exactly the wrong rescue — - // it recreates the #1849 destruction path. Report and stop; the boot probe and the - // recovery marker cover the swap-window states. - const manual = manager === "pnpm" - ? `pnpm add -g --allow-build=bun ${PKG}@${tag}` - : `npm install -g --allow-scripts=bun ${PKG}@${tag}`; - // An unexpected exception leaves the active package path unproven for either manager. - // Do not run service/tray/proxy recovery through a possibly half-swapped tree. - postUpdateLauncherUsable = false; - console.error(`opencodex: ${manager} update failed unexpectedly (${error?.message ?? error}). ` + - `The live install was not knowingly modified; run 'ocx update' again or reinstall with ${manual}.`); - res = { status: 1 }; + if (res.status !== 0) recoverStoppedRuntimeAfterFailure("update failed"); + } finally { + // Expected aborts release before process.exit(); this covers every thrown or newly-added + // path and keeps token restoration coupled to the lease itself. + releaseUpdateLease(); } + const postInstallPlan = planUpdateRuntimeHandling({ ...readOwnership(), serviceInstalled: serviceWasInstalled }); if (res.status === 0) { console.log(`\nUpdated${latest ? ` to v${latest}` : ""}.`); repairCodexShimIfNeeded(postUpdateLauncher); @@ -637,16 +837,22 @@ function runPackageManagerSelfUpdate(manager) { } // The stop above unloaded any managed service; refresh via the freshly-installed // launcher so the new files write the baked paths and the service restarts. - if (serviceWasInstalled) { + if (postInstallPlan.mayRestoreService) { console.log("Refreshing the background service with the updated files..."); refreshBackgroundServiceOrStartDirect(); - } else { + } else if (postInstallPlan.mayStopRuntime) { console.log(`Restart the proxy: ${launcherStartHint(postUpdateLauncher, bakePort)}`); } process.exit(0); } if (trayBeforeUpdate.restoreOnFailure && postUpdateLauncherUsable) runTrayLifecycle(postUpdateLauncher, "start"); - recoverStoppedRuntimeAfterFailure(); + if (npmFailure) { + // Phase-specific next step (#5624): whether the previous version is still in place decides + // between "retry" and "restore", and a bare reinstall must follow a stop (#5496). + const guidance = npmUpdateFailureGuidance({ ...npmFailure, pkgName: PKG, version: latest || undefined, tag }); + console.error(`\nUpdate failed (npm ${npmFailure.phase}). ${guidance.lines.join(" ")}`); + process.exit(1); + } const manual = manager === "pnpm" ? `pnpm add -g --allow-build=bun ${PKG}@${tag}` : `npm install -g --allow-scripts=bun ${PKG}@${tag}`; @@ -743,6 +949,19 @@ if (codexCliUpdateInspection && typeof process.versions.bun === "string") { process.exit(1); } +if (process.argv[2] === "update" && installMethod === "mise") { + if (installOwnership.owner) { + console.error( + `opencodex: this installation is externally managed by mise; update it with: mise upgrade ${installOwnership.owner.tool}`, + ); + } else { + console.error( + "opencodex: this installation appears to be managed by mise, but its ownership metadata is unreadable or inconsistent; repair the mise installation metadata before updating.", + ); + } + process.exit(1); +} + if (process.argv[2] === "update" && isNodeModulesInstall() && !isBunGlobalInstall()) { if (installMethod === "npm") runNpmSelfUpdate(); if (installMethod === "pnpm") runPnpmSelfUpdate(); diff --git a/desktop/README.md b/desktop/README.md index 02e793a476..1ef61fa78b 100644 --- a/desktop/README.md +++ b/desktop/README.md @@ -10,7 +10,16 @@ bunx tauri dev ``` The sidecar is generated from the repository's standalone binary build and is -not checked into git. +not checked into git. That build also stages the target-matching native keyring addon +under `keyring/`; Tauri copies it as a resource because Bun cannot load a `.node` addon +from the compiled executable's virtual filesystem. Universal macOS preparation requires +both Darwin optional packages (`bun install --frozen-lockfile --os=darwin --cpu=*`). + +The macOS tray panel is a SwiftUI/AppKit static library built from +`app/Sources/NativeTray` by the Rust build script and linked into this process. +Open `app/Package.swift` in Xcode to build the `NativeTray` and `NativeTrayTests` +schemes alongside the widget. macOS release builds need Xcode 26 or later for +Apple Liquid Glass; the application deployment target remains macOS 13. The CI desktop-shell job performs Rust-only checks. It creates an empty platform-named sidecar stub and a placeholder dashboard resource directory @@ -37,8 +46,14 @@ misleading on a workstation. bun run build:local ``` -This asks for the app and dmg only, so no updater archive is produced and none is expected to be -signed. It prints the bundle path and exits zero. The release path below is unchanged: a published +This asks for the host platform's installable bundles only (app and dmg on macOS, msi and nsis +setup exe on Windows, AppImage and deb on Linux), so no updater archive is produced and none is +expected to be signed. Each format is attempted in its own invocation: a format this machine +cannot bundle (for example an AppImage when a linuxdeploy dependency is missing) fails on its own +line without destroying the formats that do build, the failing format is retried once with +`--verbose` so the bundler's own diagnostics are visible, and the summary prints every format's +outcome beside the artifacts that were produced. The exit code is non-zero if any format failed. +The release path below is unchanged: a published updater artifact still has to be signed. ## Release packaging and updates diff --git a/desktop/package.json b/desktop/package.json index 7f168ff8e8..7e46987309 100644 --- a/desktop/package.json +++ b/desktop/package.json @@ -5,10 +5,13 @@ "dev": "tauri dev", "build": "tauri build", "build:local": "bun scripts/build-local.ts", + "e2e:linux-packaged": "bun scripts/linux-packaged-e2e.ts", + "icons": "bun scripts/generate-icons.ts", + "icons:check": "bun scripts/generate-icons.ts --check", "prepare-sidecar": "bun scripts/prepare-sidecar.ts", "prepare-widget": "bash scripts/build-widget.sh" }, "devDependencies": { - "@tauri-apps/cli": "2.5.0" + "@tauri-apps/cli": "2.11.1" } } diff --git a/desktop/scripts/appimage-patchelf.py b/desktop/scripts/appimage-patchelf.py new file mode 100644 index 0000000000..63e78ef741 --- /dev/null +++ b/desktop/scripts/appimage-patchelf.py @@ -0,0 +1,60 @@ +#!/usr/bin/env python3 +"""Keep the compiled Bun sidecar intact while linuxdeploy patches the host/libs.""" +import os +from pathlib import Path +import sys + + +APPDIR_SIDECAR_TAIL = ( + "release", + "bundle", + "appimage", + "OpenCodex.AppDir", + "usr", + "bin", + "ocx", +) + + +def prepared_sidecar(root, candidate, target_root): + """Return the one prepared Linux CLI that the AppDir sidecar exactly mirrors.""" + try: + relative = candidate.resolve().relative_to(target_root.resolve()) + except ValueError: + return None + if tuple(relative.parts[-len(APPDIR_SIDECAR_TAIL):]) != APPDIR_SIDECAR_TAIL: + return None + prefix = relative.parts[:-len(APPDIR_SIDECAR_TAIL)] + if len(prefix) > 1: + return None + + binaries = root / "desktop/src-tauri/binaries" + candidates = sorted(path for path in binaries.glob("ocx-*-linux-gnu") if path.is_file()) + if prefix: + candidates = [path for path in candidates if path.name == f"ocx-{prefix[0]}"] + matches = [path for path in candidates if path.read_bytes() == candidate.read_bytes()] + return matches[0] if len(matches) == 1 else None + + +def main(args): + root = Path(__file__).resolve().parents[2] + target_root = Path(os.environ.get("CARGO_TARGET_DIR", root / "desktop/src-tauri/target")) + sidecar = Path(args[2]) if len(args) == 3 else None + if ( + sidecar is not None + and args[:2] == ["--set-rpath", "$ORIGIN/../lib"] + and prepared_sidecar(root, sidecar, target_root) is not None + ): + # linuxdeploy's nested GTK pass runs ldd again after patching. Its + # patchelf rewrite breaks the compiled Bun ELF. This sidecar depends + # only on host glibc libraries; it needs no AppDir library search path. + # Never bless an already-modified binary or a different executable. + if sidecar.is_symlink(): + raise RuntimeError("AppImage sidecar differs from the prepared CLI") + print("Preserving compiled ocx bytes (no AppDir RPATH required)", file=sys.stderr) + return + os.execv("/usr/bin/patchelf", ["/usr/bin/patchelf", *args]) + + +if __name__ == "__main__": + main(sys.argv[1:]) diff --git a/desktop/scripts/build-local.ts b/desktop/scripts/build-local.ts index 633a27b6c1..41daf62e28 100644 --- a/desktop/scripts/build-local.ts +++ b/desktop/scripts/build-local.ts @@ -19,16 +19,33 @@ * there is nothing to sign and nothing is skipped unsigned. Selecting bundle targets is not enough: * `createUpdaterArtifacts` is a config flag, so `--bundles app,dmg` still produces * `OpenCodex.app.tar.gz (updater)` and still fails. The override has to reach the config itself. + * + * Two more local-only behaviours, learned from a real GNOME desktop (devlog plan 260921, + * 120_install_verification.md): + * + * - Formats build in SEPARATE invocations. A single `--bundles appimage,deb` call dies on the + * first failing format, so a host that cannot bundle an AppImage (a missing linuxdeploy + * dependency) also lost the deb it could have built. Each format is attempted, and the + * summary at the end names every format's outcome; the exit code is non-zero if any of + * them failed, and the artifacts that DID build are printed either way. + * - A failing format is retried once with `--verbose`. At the bundler's default log level + * the error is a bare "failed to run linuxdeploy" with the tool's own diagnostics + * discarded; the verbose pass is the branch where that stderr actually reaches the + * terminal, so the failure says WHY instead of naming a tool nobody invoked. */ import { spawnSync } from "node:child_process"; -import { existsSync } from "node:fs"; +import { existsSync, readdirSync, statSync } from "node:fs"; import { dirname, join } from "node:path"; import { fileURLToPath } from "node:url"; const desktopDir = dirname(dirname(fileURLToPath(import.meta.url))); -/** Bundle targets that carry no updater archive. */ -const LOCAL_BUNDLES = ["app", "dmg"] as const; +/** Bundle targets per host platform that carry no updater archive. */ +const LOCAL_BUNDLES: Record = { + darwin: ["app", "dmg"], + win32: ["msi", "nsis"], + linux: ["appimage", "deb"], +}; /** * Config merged over `tauri.conf.json` for this invocation only. @@ -37,31 +54,115 @@ const LOCAL_BUNDLES = ["app", "dmg"] as const; * required and unmet. The committed config keeps `createUpdaterArtifacts: true`, so the release * build is untouched. */ -const LOCAL_CONFIG = JSON.stringify({ bundle: { createUpdaterArtifacts: false } }); +function localConfig(platform: string): string { + return JSON.stringify({ bundle: { + createUpdaterArtifacts: false, + // Sign nested executables and the bundle even without a Developer ID. Leaving + // their old linker signatures in place creates an app that macOS kills at launch. + ...(platform === "darwin" ? { macOS: { signingIdentity: "-" } } : {}), + } }); +} + +export interface SpawnResult { + status: number | null; + error?: Error; +} + +export interface ArtifactEntry { + path: string; + mtimeMs: number; +} + +export interface BuildLocalDeps { + spawn(args: string[]): SpawnResult; + log(line: string): void; + error(line: string): void; + listArtifacts(): ArtifactEntry[]; + argv: string[]; + platform: string; +} -function run(): number { - const extra = process.argv.slice(2); - const args = [ - "tauri", "build", "--ci", - "--bundles", LOCAL_BUNDLES.join(","), - "--config", LOCAL_CONFIG, - ...extra, - ]; - const result = spawnSync("bunx", args, { cwd: desktopDir, stdio: "inherit" }); - if (result.error) { - console.error(`[build:local] could not start tauri: ${result.error.message}`); +export interface FormatAttempt { + format: string; + status: number; +} + +export function summarizeAttempts(attempts: FormatAttempt[]): { exitCode: number; lines: string[] } { + const lines = attempts.map( + attempt => `[build:local] ${attempt.format}: ${attempt.status === 0 ? "ok" : `FAILED (exit ${attempt.status})`}`, + ); + return { exitCode: attempts.every(attempt => attempt.status === 0) ? 0 : 1, lines }; +} + +export function runBuildLocal(deps: BuildLocalDeps): number { + const bundles = LOCAL_BUNDLES[deps.platform]; + if (!bundles) { + deps.error(`[build:local] unsupported host platform: ${deps.platform}`); return 1; } - return result.status ?? 1; + // Snapshot before building: a bundle directory that already holds last week's AppImage + // must not be reported as this run's output when this run's AppImage attempt fails. + const baseline = new Map(deps.listArtifacts().map(entry => [entry.path, entry.mtimeMs])); + const attempts: FormatAttempt[] = []; + for (const format of bundles) { + // One invocation per format: a format this host cannot build must not destroy the + // artifacts of formats it can. + const args = ["tauri", "build", "--ci", "--bundles", format, "--config", localConfig(deps.platform), ...deps.argv]; + const first = deps.spawn(args); + let status = first.status ?? 1; + if (first.error) { + deps.error(`[build:local] could not start tauri: ${first.error.message}`); + status = 1; + } else if (status !== 0) { + // The bundler reports a bare "failed to run " at its default log level; the + // verbose pass is where the tool's own stderr reaches the terminal. The retry is + // diagnostics only — the recorded status stands either way. + deps.error(`[build:local] ${format} failed; rerunning with --verbose for the bundler's diagnostics`); + const retry = deps.spawn(["tauri", "--verbose", "build", "--ci", "--bundles", format, "--config", localConfig(deps.platform), ...deps.argv]); + if (retry.error) deps.error(`[build:local] could not start tauri: ${retry.error.message}`); + } + attempts.push({ format, status }); + } + // Name what THIS run produced even when something failed: an error line at the end is + // the least visible place for artifacts that already built. + const produced = deps.listArtifacts().filter( + entry => !baseline.has(entry.path) || baseline.get(entry.path) !== entry.mtimeMs, + ); + for (const entry of produced) deps.log(`[build:local] ${entry.path}`); + const summary = summarizeAttempts(attempts); + for (const line of summary.lines) deps.log(line); + if (summary.exitCode === 0) { + deps.log("[build:local] updater artifacts skipped; release signing is unchanged."); + } + return summary.exitCode; +} + +function main(): void { + const status = runBuildLocal({ + spawn: args => spawnSync("bunx", args, { cwd: desktopDir, stdio: "inherit" }), + log: line => console.log(line), + error: line => console.error(line), + listArtifacts: () => { + const bundleRoot = join(desktopDir, "src-tauri", "target", "release", "bundle"); + const artifacts: ArtifactEntry[] = []; + for (const dir of ["macos", "dmg", "msi", "nsis", "appimage", "deb"]) { + const directory = join(bundleRoot, dir); + if (!existsSync(directory)) continue; + for (const name of readdirSync(directory)) { + if (/\.(app|dmg|msi|exe|AppImage|deb)$/i.test(name)) { + const full = join(directory, name); + artifacts.push({ path: full, mtimeMs: statSync(full).mtimeMs }); + } + } + } + return artifacts; + }, + argv: process.argv.slice(2), + platform: process.platform, + }); + process.exit(status); } -const status = run(); -if (status === 0) { - const bundleRoot = join(desktopDir, "src-tauri", "target", "release", "bundle"); - const app = join(bundleRoot, "macos", "OpenCodex.app"); - // Naming what exists is the point of the script: the previous output ended on an error line, so - // the artifacts it had already written were the least visible thing in it. - if (existsSync(app)) console.log(`[build:local] ${app}`); - console.log("[build:local] updater artifacts skipped; release signing is unchanged."); +if (import.meta.main) { + main(); } -process.exit(status); diff --git a/desktop/scripts/build-widget.sh b/desktop/scripts/build-widget.sh index 59827dcea2..38e0eb9d71 100755 --- a/desktop/scripts/build-widget.sh +++ b/desktop/scripts/build-widget.sh @@ -19,6 +19,25 @@ if [[ "$universal" != "0" && "$universal" != "1" ]]; then exit 1 fi +# A widget extension is loaded by the system, not by the app, so it is validated on its own +# terms: notarization rejects any Mach-O inside it that lacks the hardened runtime, and macOS +# refuses to register an extension whose signature does not chain to the containing app's team. +# An ad-hoc signature satisfies neither, and the ad-hoc branch is the default whenever no +# identity reaches this script. Resolve that before the build so a misconfigured release fails +# in a second rather than after a universal Swift build. +if [[ -n "${MACOS_SIGN_IDENTITY:-}" ]]; then + sign_identity="$MACOS_SIGN_IDENTITY" + timestamp_arg=(--timestamp) +elif [[ "${WIDGET_SIGN_REQUIRED:-0}" == "1" ]]; then + # A release that signs everything else and ad-hoc signs the widget produces an app that + # ships either way and simply has no widget. Refuse instead. + echo "WIDGET_SIGN_REQUIRED=1 but MACOS_SIGN_IDENTITY is empty; refusing to ad-hoc sign a release widget." >&2 + exit 1 +else + sign_identity="-" + timestamp_arg=(--timestamp=none) +fi + build_root="$(mktemp -d "${TMPDIR:-/tmp}/opencodex-widget.XXXXXX")" cleanup() { rm -rf "$build_root"; } trap cleanup EXIT @@ -65,12 +84,41 @@ version_core="${version%%-*}" plutil -replace CFBundleShortVersionString -string "$version_core" "$output_dir/Contents/Info.plist" plutil -replace CFBundleVersion -string "$version_core" "$output_dir/Contents/Info.plist" -if [[ -n "${MACOS_SIGN_IDENTITY:-}" ]]; then - codesign --force --sign "$MACOS_SIGN_IDENTITY" --entitlements "$package_dir/Widget.entitlements" \ - --timestamp "$output_dir" -else - codesign --force --sign - --entitlements "$package_dir/Widget.entitlements" \ - --timestamp=none "$output_dir" -fi +# Sign inside out over every Mach-O the bundle actually contains, chosen by magic bytes rather +# than by name. Today that set is the single widget executable, but a name or extension filter +# is the thing that fails silently when it stops being true: a helper tool or an embedded +# dylib carries no suffix to match, stays unsigned, and the whole submission comes back +# "The binary is not signed with a valid Developer ID certificate" with the bundle itself +# looking perfectly signed. +mach_o_members=() +while IFS= read -r candidate; do + [[ "$(file -b "$candidate")" == *"Mach-O"* ]] || continue + mach_o_members+=("$candidate") +done < <(find "$output_dir" -type f -not -path "*/_CodeSignature/*") + +[[ ${#mach_o_members[@]} -gt 0 ]] || { echo "No Mach-O binary found in $output_dir" >&2; exit 1; } + +for member in "${mach_o_members[@]}"; do + codesign --force --sign "$sign_identity" --options runtime "${timestamp_arg[@]}" "$member" +done + +# The bundle seal goes on last and is the only signature that carries the entitlements. +codesign --force --sign "$sign_identity" --entitlements "$package_dir/Widget.entitlements" \ + --options runtime "${timestamp_arg[@]}" "$output_dir" + +codesign --verify --deep --strict "$output_dir" +# `runtime` is 0x10000 in the code directory flags. Asserting it here is what turns a silently +# unnotarizable widget into a failed build. The output is captured rather than piped into a +# matcher: `set -o pipefail` plus a matcher that exits on its first hit makes codesign die of +# SIGPIPE, and the check then fails on exactly the signatures it was meant to accept. +signature_display="$(codesign --display --verbose=4 "$output_dir" 2>&1)" +case "$signature_display" in + *"flags="*"runtime"*) ;; + *) + echo "Widget signature is missing the hardened runtime:" >&2 + echo "$signature_display" >&2 + exit 1 + ;; +esac echo "$output_dir" diff --git a/desktop/scripts/collect-release-assets.ts b/desktop/scripts/collect-release-assets.ts index f946e1d4a9..1d1266283e 100644 --- a/desktop/scripts/collect-release-assets.ts +++ b/desktop/scripts/collect-release-assets.ts @@ -11,13 +11,13 @@ import { join, resolve } from "node:path"; type BundleKind = "dmg" | "app.tar.gz" | "msi" | "appimage" | "deb"; -interface BundleSpec { +export interface BundleSpec { kind: BundleKind; dir: string; name: string; } -const bundlesByTarget: Record = { +export const bundlesByTarget: Record = { "universal-apple-darwin": [ { kind: "dmg", dir: "dmg", name: "macos.dmg" }, { kind: "app.tar.gz", dir: "macos", name: "macos.app.tar.gz" }, @@ -42,6 +42,7 @@ export interface CollectReleaseAssetsOptions { target: string; out: string; repoRoot?: string; + bundleRoot?: string; } function findBundle(directory: string, kind: BundleKind): string { @@ -61,13 +62,17 @@ export function collectReleaseAssets(options: CollectReleaseAssetsOptions): stri const repoRoot = resolve(options.repoRoot ?? join(import.meta.dir, "../..")); const bundles = bundlesByTarget[options.target]; if (!bundles) throw new Error(`Unsupported desktop target: ${options.target}`); + const bundleRoot = resolve( + options.bundleRoot + ?? join(repoRoot, "desktop", "src-tauri", "target", options.target, "release", "bundle"), + ); const output = resolve(options.out); mkdirSync(output, { recursive: true }); const written: string[] = []; for (const bundle of bundles) { const source = findBundle( - join(repoRoot, "desktop", "src-tauri", "target", options.target, "release", "bundle", bundle.dir), + join(bundleRoot, bundle.dir), bundle.kind, ); const destinationName = `OpenCodex-${options.version}-${bundle.name}`; @@ -98,8 +103,11 @@ if (import.meta.main) { const version = argument("--version"); const target = argument("--target"); const out = argument("--out"); + const bundleRoot = argument("--bundle-root"); if (!version || !target || !out) { throw new Error("Usage: collect-release-assets.ts --version --target --out "); } - for (const path of collectReleaseAssets({ version, target, out })) console.log(`Wrote ${path}`); + const options: CollectReleaseAssetsOptions = { version, target, out }; + if (bundleRoot) options.bundleRoot = bundleRoot; + for (const path of collectReleaseAssets(options)) console.log(`Wrote ${path}`); } diff --git a/desktop/scripts/generate-icons.ts b/desktop/scripts/generate-icons.ts new file mode 100644 index 0000000000..6ce6a00c26 --- /dev/null +++ b/desktop/scripts/generate-icons.ts @@ -0,0 +1,191 @@ +#!/usr/bin/env bun +/** + * Render every app icon from `src-tauri/icons/icon.svg`. + * + * The icon set used to be eighteen independent raster files with no vector source, so each size + * was a separate artifact that could drift from the others and nothing could detect it. This makes + * the sizes derived: one curve, rendered at each dimension the platforms ask for. + * + * The SVG reproduces the raster it replaced to within antialiasing (430 of 262144 pixels at 512), + * measured rather than assumed — the geometry in that file was read off the original bitmap. + * + * `--check` regenerates into a temporary directory and compares, so CI can fail on a hand-edited + * PNG instead of letting the source and the shipped icons disagree quietly. + */ +import { spawnSync } from "node:child_process"; +import { mkdtempSync, mkdirSync, readFileSync, writeFileSync, rmSync, existsSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { buildIco, render as renderSvg } from "../../scripts/lib/icon-render"; + +const desktopDir = dirname(dirname(fileURLToPath(import.meta.url))); +const iconsDir = join(desktopDir, "src-tauri", "icons"); +const source = join(iconsDir, "icon.svg"); + +/** Square PNGs Tauri and the Windows store manifests reference, by output filename. */ +const PNG_SIZES: Record = { + "32x32.png": 32, + "64x64.png": 64, + "128x128.png": 128, + "128x128@2x.png": 256, + "icon.png": 512, + "Square30x30Logo.png": 30, + "Square44x44Logo.png": 44, + "Square71x71Logo.png": 71, + "Square89x89Logo.png": 89, + "Square107x107Logo.png": 107, + "Square142x142Logo.png": 142, + "Square150x150Logo.png": 150, + "Square284x284Logo.png": 284, + "Square310x310Logo.png": 310, + "StoreLogo.png": 50, +}; + +/** Sizes an .icns carries, as iconutil names them. */ +const ICNS_ENTRIES: Array<{ name: string; size: number }> = [ + { name: "icon_16x16.png", size: 16 }, + { name: "icon_16x16@2x.png", size: 32 }, + { name: "icon_32x32.png", size: 32 }, + { name: "icon_32x32@2x.png", size: 64 }, + { name: "icon_128x128.png", size: 128 }, + { name: "icon_128x128@2x.png", size: 256 }, + { name: "icon_256x256.png", size: 256 }, + { name: "icon_256x256@2x.png", size: 512 }, + { name: "icon_512x512.png", size: 512 }, + { name: "icon_512x512@2x.png", size: 1024 }, +]; + +/** Sizes packed into the .ico, which stores each one as an embedded PNG. */ +const ICO_SIZES = [16, 32, 48, 64, 128, 256]; + +/** + * The menu bar image, which is the same mark with no backdrop and the prompt cut through. + * + * It needs its own source because a status item is a template image: macOS reads the alpha as + * coverage and paints it with the menu bar tint, so the backdrop has to be gone rather than + * recoloured. 44px is 22pt at @2x, which is the menu bar working height and the size this asset + * already shipped at. + */ +const TRAY_OUTPUT = "tray/icon.png"; +const TRAY_SIZE = 44; +const traySource = join(iconsDir, "tray", "icon.svg"); +const DOTTED_TRAY_OUTPUT = "tray/icon-update.png"; +const DOTTED_TRAY_SVG = ''; + +function renderDottedTray(target: string): void { + const dottedSvg = join(target, ".tray-update.svg"); + const base = readFileSync(traySource, "utf8"); + if (!base.includes("")) throw new Error("tray icon source is not SVG"); + writeFileSync(dottedSvg, base.replace("", DOTTED_TRAY_SVG + "")); + try { render(TRAY_SIZE, join(target, DOTTED_TRAY_OUTPUT), dottedSvg); } + finally { rmSync(dottedSvg, { force: true }); } +} + +/** Render at `size` from `from`, defaulting to the app icon vector. */ +function render(size: number, out: string, from: string = source): void { + renderSvg(size, out, from); +} + +/** + * Render the whole set into `target`, and report which artifacts were actually produced. + * + * The return value matters: `iconutil` is macOS-only, so on another platform no `.icns` exists to + * compare against. Reporting that is the difference between "the icns matches" and "nothing looked + * at the icns", and the check must not spell the second as the first. + */ +function generateInto(target: string): { produced: string[]; icnsSkipped: boolean } { + mkdirSync(target, { recursive: true }); + const produced: string[] = []; + for (const [name, size] of Object.entries(PNG_SIZES)) { + render(size, join(target, name)); + produced.push(name); + } + + const iconset = join(target, "icon.iconset"); + mkdirSync(iconset, { recursive: true }); + for (const entry of ICNS_ENTRIES) render(entry.size, join(iconset, entry.name)); + const icns = spawnSync("iconutil", ["-c", "icns", iconset, "-o", join(target, "icon.icns")]); + const icnsSkipped = icns.status !== 0; + if (!icnsSkipped) produced.push("icon.icns"); + rmSync(iconset, { recursive: true, force: true }); + + const icoParts: Array<{ size: number; bytes: Buffer }> = []; + for (const size of ICO_SIZES) { + const scratch = join(target, `.ico-${size}.png`); + render(size, scratch); + icoParts.push({ size, bytes: readFileSync(scratch) }); + rmSync(scratch, { force: true }); + } + writeFileSync(join(target, "icon.ico"), buildIco(icoParts)); + produced.push("icon.ico"); + + mkdirSync(join(target, "tray"), { recursive: true }); + render(TRAY_SIZE, join(target, TRAY_OUTPUT), traySource); + produced.push(TRAY_OUTPUT); + renderDottedTray(target); + produced.push(DOTTED_TRAY_OUTPUT); + + return { produced, icnsSkipped }; +} + +function main(): number { + if (!existsSync(source)) { + console.error(`[icons] missing source: ${source}`); + return 1; + } + if (!existsSync(traySource)) { + console.error(`[icons] missing source: ${traySource}`); + return 1; + } + const check = process.argv.includes("--check"); + if (!check) { + // Render into scratch first so a failure half way through cannot leave the committed set + // partly replaced, then move the finished artifacts over in one pass. + const scratch = mkdtempSync(join(tmpdir(), "ocx-icons-")); + try { + const { produced, icnsSkipped } = generateInto(scratch); + if (icnsSkipped) { + // Abort before touching the committed set. Copying the PNGs and the .ico and then + // reporting the missing .icns would leave the icons half regenerated: the rasters new, + // the .icns whatever it was, and no way to tell from the tree which is which. + console.error("[icons] iconutil is unavailable here, so the .icns cannot be regenerated."); + console.error("[icons] nothing was written; run this on a machine with iconutil."); + return 1; + } + for (const name of produced) writeFileSync(join(iconsDir, name), readFileSync(join(scratch, name))); + console.log(`[icons] regenerated ${produced.length} artifacts from ${source}`); + return 0; + } finally { + rmSync(scratch, { recursive: true, force: true }); + } + } + + const scratch = mkdtempSync(join(tmpdir(), "ocx-icons-")); + try { + const { produced, icnsSkipped } = generateInto(scratch); + const drifted: string[] = []; + for (const name of produced) { + const fresh = join(scratch, name); + const committed = join(iconsDir, name); + if (!existsSync(committed) || !readFileSync(fresh).equals(readFileSync(committed))) { + drifted.push(name); + } + } + if (drifted.length > 0) { + console.error(`[icons] these do not match their source: ${drifted.join(", ")}`); + console.error("[icons] regenerate with: bun run icons"); + return 1; + } + console.log(`[icons] ${produced.length} generated icons match the source`); + if (icnsSkipped) { + console.error("[icons] iconutil is unavailable here, so icon.icns was NOT compared."); + return 1; + } + return 0; + } finally { + rmSync(scratch, { recursive: true, force: true }); + } +} + +process.exit(main()); diff --git a/desktop/scripts/installed-gate-platforms.ts b/desktop/scripts/installed-gate-platforms.ts new file mode 100644 index 0000000000..0fd10bf95d --- /dev/null +++ b/desktop/scripts/installed-gate-platforms.ts @@ -0,0 +1,407 @@ +/** + * Platform adapters for the installed-artifact gate (D9, part two). + * + * Adapters turn platform actions into command specs the gate engine runs and records; + * they never hardcode machine detail. Artifact paths, homes and ports arrive as + * arguments. GUI automation the OS cannot reach (the in-page consent dialog, tray + * clicks on some desktops) is supplied by the operator as pre-installed hook files, + * never as dispatch-provided command text — a persistent self-hosted runner must not + * become an arbitrary-execution surface. + * + * The external commands each adapter needs are declared in dependencies() so a runner + * can be audited for readiness before an artifact is ever installed on it. + */ + +export type GatePlatform = "macos" | "windows" | "linux"; +export type GateFormat = "dmg" | "msi" | "deb" | "appimage"; + +import { join } from "node:path"; + +export interface CommandSpec { + file: string; + args: string[]; +} + +export interface ProcessEvidence { + ok: boolean; + exitCode: number | null; + stdout: string; + stderr: string; +} + +export interface InstallResult { + appBinary: string; + packageName?: string; + /** Path scope that identifies THIS install's processes (install dir or binary path). */ + scope: string; + evidence: Record; +} + +export interface PlatformAdapter { + platform: GatePlatform; + /** Every external command this adapter shells out to; the engine preflights them. */ + dependencies(): string[]; + /** The staged npm ocx launcher inside an npm --prefix install. */ + npmLauncher(prefix: string): string; + /** Registers and starts the staged npm runtime as a managed service. */ + serviceInstall(launcher: string): CommandSpec; + serviceUninstall(launcher: string): CommandSpec; + /** + * Three-state registration answer: an existing registration is "present", a clean + * not-found is "absent", and any probe failure that cannot be told apart is + * "unknown" — the engine refuses to mutate on unknown. + */ + registrationState(): Promise<"present" | "absent" | "unknown">; + /** + * On-disk registration files, relative to the runner account's home. A registration + * that is unloaded, disabled or not yet loaded leaves these behind, and the manager + * probes above miss exactly those states. + */ + registrationFiles(): string[]; + /** + * Read-only probe for a dormant installation (MSI registry entry, dpkg record) the + * gate must refuse to overwrite. Null where installs land inside the work dir. + */ + existingInstallation(format: GateFormat, packageName?: string): CommandSpec | null; + /** Installs the real artifact; returns the app executable path. */ + installArtifact(artifact: string, workDir: string, format: GateFormat): Promise; + /** Drives the installed app's window-close gesture. */ + closeGesture(): CommandSpec; + /** Drives the installed app's OS-quit gesture (Cmd+Q, Alt+F4). */ + quitGesture(): CommandSpec; + /** Left-clicks the tray icon (the shell shows the dashboard window), or null. */ + trayClick(): CommandSpec | null; + /** Opens the tray menu and chooses Quit (the drain-then-exit path), or null. */ + trayQuit(): CommandSpec | null; + /** Chooses Check for Updates in the tray menu, or null. */ + trayCheck(): CommandSpec | null; + /** Chooses the enabled Install update item in the tray menu, or null. */ + trayInstall(): CommandSpec | null; + /** Exits zero only when the app currently has a visible window. */ + windowVisible(): CommandSpec; + /** Installed package version probe, where the platform has one (deb), or null. */ + installedVersion(format: GateFormat, packageName?: string): CommandSpec | null; + /** Dismisses exactly the authorization prompts the gate sighted, by pid, or null. */ + cancelElevation(pids: number[]): CommandSpec | null; + /** + * Lists pids of any elevation prompt surface — pkexec, and the zenity/kdialog + * password dialogs the updater plugin falls back to after a pkexec cancel. + * Null where the platform has no package-manager elevation (macOS, Windows). + */ + elevationProbe(): CommandSpec | null; + /** Lists pids whose executable lives under the given install scope. */ + appProcessProbe(scope: string): CommandSpec; + /** Lists pids of ANY installed copy of the app — the preflight's broad probe. */ + appNameProbe(): CommandSpec; + /** Lists direct child pids of the given process. */ + childPids(pid: number): CommandSpec; + /** Removes what installArtifact placed on the machine. */ + uninstall(artifact: string, workDir: string, format: GateFormat, packageName?: string): CommandSpec[]; +} + +export interface AdapterRuntime { + run(spec: CommandSpec): Promise; + mkdir(path: string): void; + fileExists(path: string): boolean; + homeDir(): string; +} + +function requireOk(step: string, result: ProcessEvidence): void { + if (!result.ok) { + throw new Error(`${step} failed (exit ${result.exitCode}): ${result.stderr.trim() || result.stdout.trim()}`); + } +} + +/** macOS: dmg install, AppleScript gestures scoped to the OpenCodex process, launchd. */ +export function macosAdapter(runtime: AdapterRuntime): PlatformAdapter { + const run = runtime.run.bind(runtime); + return { + platform: "macos", + dependencies: () => ["hdiutil", "osascript", "pgrep", "launchctl", "cp", "rm", "/usr/libexec/PlistBuddy"], + npmLauncher: prefix => `${prefix}/node_modules/.bin/ocx`, + serviceInstall: launcher => ({ file: launcher, args: ["service", "install"] }), + serviceUninstall: launcher => ({ file: launcher, args: ["service", "uninstall"] }), + registrationState: async () => { + if (runtime.fileExists(join(runtime.homeDir(), "Library/LaunchAgents/com.opencodex.proxy.plist"))) return "present"; + const probe = await run({ file: "launchctl", args: ["list", "com.opencodex.proxy"] }); + if (probe.ok) return "present"; + // "Could not find service" is a clean absence; anything else is unknowable here. + return /could not find/i.test(probe.stderr) ? "absent" : "unknown"; + }, + registrationFiles: () => ["Library/LaunchAgents/com.opencodex.proxy.plist"], + // A dmg install lands inside the gate's work dir; there is no system-level record. + existingInstallation: () => null, + async installArtifact(artifact, workDir) { + const mount = `${workDir}/dmg-mount`; + const apps = `${workDir}/Applications`; + runtime.mkdir(mount); + runtime.mkdir(apps); + requireOk("dmg attach", await run({ file: "hdiutil", args: ["attach", artifact, "-mountpoint", mount, "-nobrowse", "-readonly"] })); + try { + requireOk("app copy", await run({ file: "cp", args: ["-R", `${mount}/OpenCodex.app`, `${apps}/`] })); + } finally { + await run({ file: "hdiutil", args: ["detach", mount] }); + } + // The executable name is the bundle's own declaration, not a guess: a rename in + // packaging lands here without a driver change (#5351 removed this hardcode once). + const plist = await run({ + file: "/usr/libexec/PlistBuddy", + args: ["-c", "Print :CFBundleExecutable", `${apps}/OpenCodex.app/Contents/Info.plist`], + }); + requireOk("read CFBundleExecutable", plist); + const executable = plist.stdout.trim(); + return { + appBinary: `${apps}/OpenCodex.app/Contents/MacOS/${executable}`, + scope: apps, + evidence: { mounted: mount, copiedTo: apps, executable }, + }; + }, + closeGesture: () => appleScript( + 'tell application "OpenCodex" to activate', + 'tell application "System Events" to keystroke "w" using command down', + ), + quitGesture: () => appleScript( + 'tell application "OpenCodex" to activate', + 'tell application "System Events" to keystroke "q" using command down', + ), + // Menu bar items belong to their owning process; clicking by global index can hit an + // unrelated tray, so every tray action is scoped to the OpenCodex process. + trayClick: () => appleScript( + 'tell application "System Events" to tell process "OpenCodex" to click menu bar item 1 of menu bar 2', + ), + trayQuit: () => appleScript( + 'tell application "System Events" to tell process "OpenCodex" to click menu bar item 1 of menu bar 2', + 'tell application "System Events" to tell process "OpenCodex" to click menu item "Quit" of menu 1 of menu bar item 1 of menu bar 2', + ), + trayCheck: () => appleScript( + 'tell application "System Events" to tell process "OpenCodex" to click menu bar item 1 of menu bar 2', + 'tell application "System Events" to tell process "OpenCodex" to click menu item "Check for Updates…" of menu 1 of menu bar item 1 of menu bar 2', + ), + trayInstall: () => appleScript( + 'tell application "System Events" to tell process "OpenCodex" to click menu bar item 1 of menu bar 2', + 'tell application "System Events" to tell process "OpenCodex" to click (first menu item of menu 1 of menu bar item 1 of menu bar 2 whose name starts with "Install update")', + ), + windowVisible: () => appleScript('tell application "System Events" to count (windows of process "OpenCodex")'), + installedVersion: () => null, + cancelElevation: () => null, + elevationProbe: () => null, + appProcessProbe: scope => ({ file: "pgrep", args: ["-f", scope] }), + appNameProbe: () => ({ file: "pgrep", args: ["-f", "OpenCodex.app/Contents/MacOS"] }), + childPids: pid => ({ file: "pgrep", args: ["-P", String(pid)] }), + uninstall: (_artifact, workDir) => [{ file: "rm", args: ["-rf", `${workDir}/Applications/OpenCodex.app`] }], + }; +} + +/** Windows: msi install, PowerShell gestures, Task Scheduler registration. */ +export function windowsAdapter(runtime: AdapterRuntime): PlatformAdapter { + const run = runtime.run.bind(runtime); + return { + platform: "windows", + dependencies: () => ["msiexec", "powershell", "schtasks", "sc"], + npmLauncher: prefix => `${prefix}\\node_modules\\.bin\\ocx.cmd`, + serviceInstall: launcher => ({ file: launcher, args: ["service", "install"] }), + serviceUninstall: launcher => ({ file: launcher, args: ["service", "uninstall"] }), + // Task Scheduler is the default backend; the native backend registers a WinSW + // service instead, and both count as an existing registration. + registrationState: async () => { + const task = await run({ file: "schtasks", args: ["/Query", "/TN", "opencodex-proxy"] }); + if (task.ok) return "present"; + const taskAbsent = /cannot find|does not exist/i.test(task.stderr + task.stdout); + const service = await run({ file: "sc.exe", args: ["query", "opencodex-proxy-native"] }); + if (service.ok) return "present"; + const serviceAbsent = /does not exist/i.test(service.stderr + service.stdout); + if (taskAbsent && serviceAbsent) return "absent"; + return "unknown"; + }, + registrationFiles: () => [], + // A dormant MSI install shows up in the uninstall registry before any process runs. + existingInstallation: format => + format === "msi" + ? { + file: "powershell", + args: [ + "-NoProfile", + "-Command", + "$key = Get-ItemProperty 'HKLM:\\SOFTWARE\\Microsoft\\Windows\\CurrentVersion\\Uninstall\\*' | Where-Object { $_.DisplayName -eq 'OpenCodex' }; if ($key) { exit 0 } else { exit 1 }", + ], + } + : null, + async installArtifact(artifact, workDir) { + requireOk( + "msi install", + await run({ file: "msiexec", args: ["/i", artifact, "/qn", "/norestart", "/l*v", `${workDir}\\msi-install.log`] }), + ); + const locate = await run({ + file: "powershell", + args: [ + "-NoProfile", + "-Command", + "$key = Get-ItemProperty 'HKLM:\\SOFTWARE\\Microsoft\\Windows\\CurrentVersion\\Uninstall\\*' | Where-Object { $_.DisplayName -eq 'OpenCodex' } | Select-Object -First 1; " + + "$dir = $key.InstallLocation; " + + "@(\"opencodex-desktop.exe\", \"opencodex.exe\") | ForEach-Object { $p = Join-Path $dir $_; if (Test-Path $p) { Write-Output $dir; Write-Output $p; break } }", + ], + }); + const locateLines = locate.stdout.trim().split(/\r?\n/); + const installDir = locateLines[0] ?? ""; + const appBinary = locateLines[1] ?? ""; + if (!appBinary) { + // The MSI may already be installed; a discovery failure must not strand it. + await run({ file: "msiexec", args: ["/x", artifact, "/qn", "/norestart"] }); + throw new Error("MSI installed but no OpenCodex executable was found under its InstallLocation"); + } + return { appBinary, scope: installDir, evidence: { installLog: `${workDir}\\msi-install.log`, located: appBinary } }; + }, + closeGesture: () => ({ + file: "powershell", + args: [ + "-NoProfile", + "-Command", + "$p = Get-Process opencodex-desktop -ErrorAction SilentlyContinue | Where-Object { $_.MainWindowHandle -ne 0 } | Select-Object -First 1; " + + "if (-not $p) { exit 1 }; " + + "$sig = '[DllImport(\"user32.dll\")] public static extern bool PostMessage(IntPtr h, uint m, IntPtr w, IntPtr l);'; " + + "Add-Type -MemberDefinition $sig -Name U32 -Namespace W; [W.U32]::PostMessage($p.MainWindowHandle, 0x0010, [IntPtr]::Zero, [IntPtr]::Zero) | Out-Null", + ], + }), + quitGesture: () => ({ + file: "powershell", + args: [ + "-NoProfile", + "-Command", + "$p = Get-Process opencodex-desktop -ErrorAction SilentlyContinue | Where-Object { $_.MainWindowHandle -ne 0 } | Select-Object -First 1; " + + "if (-not $p) { exit 1 }; " + + "$shell = New-Object -ComObject WScript.Shell; $shell.AppActivate($p.Id) | Out-Null; $shell.SendKeys('%{F4}')", + ], + }), + // The Windows tray lives in the shell's notification area; a pre-installed runner + // hook (UIA) drives it. See --hooks-dir in installed-gate.ts. + trayClick: () => null, + trayQuit: () => null, + trayCheck: () => null, + trayInstall: () => null, + windowVisible: () => ({ + file: "powershell", + args: [ + "-NoProfile", + "-Command", + "$p = Get-Process opencodex-desktop -ErrorAction SilentlyContinue | Where-Object { $_.MainWindowHandle -ne 0 }; if ($p) { exit 0 } else { exit 1 }", + ], + }), + installedVersion: () => ({ + file: "powershell", + args: [ + "-NoProfile", + "-Command", + "(Get-ItemProperty 'HKLM:\\SOFTWARE\\Microsoft\\Windows\\CurrentVersion\\Uninstall\\*' | Where-Object { $_.DisplayName -eq 'OpenCodex' } | Select-Object -First 1).DisplayVersion", + ], + }), + cancelElevation: () => null, + elevationProbe: () => null, + appProcessProbe: scope => ({ + file: "powershell", + args: [ + "-NoProfile", + "-Command", + `(Get-CimInstance Win32_Process -Filter "ExecutablePath LIKE '${scope.replace(/%/g, "")}%'").ProcessId`, + ], + }), + appNameProbe: () => ({ + file: "powershell", + args: ["-NoProfile", "-Command", "(Get-Process opencodex-desktop -ErrorAction SilentlyContinue).Id"], + }), + childPids: pid => ({ + file: "powershell", + args: ["-NoProfile", "-Command", `(Get-CimInstance Win32_Process -Filter "ParentProcessId=${pid}").ProcessId`], + }), + uninstall: artifact => [{ file: "msiexec", args: ["/x", artifact, "/qn", "/norestart"] }], + }; +} + +/** Linux: deb and AppImage installs, xdotool gestures, systemd user registration. */ +export function linuxAdapter(runtime: AdapterRuntime): PlatformAdapter { + const run = runtime.run.bind(runtime); + return { + platform: "linux", + dependencies: () => ["dpkg", "dpkg-deb", "dpkg-query", "xdotool", "pgrep", "systemctl", "sudo", "cp", "chmod", "kill", "rm"], + npmLauncher: prefix => `${prefix}/node_modules/.bin/ocx`, + serviceInstall: launcher => ({ file: launcher, args: ["service", "install"] }), + serviceUninstall: launcher => ({ file: launcher, args: ["service", "uninstall"] }), + registrationState: async () => { + if (runtime.fileExists(join(runtime.homeDir(), ".config/systemd/user/opencodex-proxy.service"))) return "present"; + // is-enabled: "enabled"/"linked" exit 0; "disabled" exits 1 but still means the + // unit file EXISTS. is-active: "active" exits 0; "inactive" exits 3 and also + // means the unit is registered. Absence prints "could not be found". + const enabled = await run({ file: "systemctl", args: ["--user", "is-enabled", "opencodex-proxy"] }); + const enabledOut = (enabled.stdout + enabled.stderr).trim(); + if (enabled.ok || /^\w*enabled$|^linked$/.test(enabled.stdout.trim()) || enabled.stdout.trim() === "disabled") return "present"; + if (!/could not be found|no such file|not found/i.test(enabledOut)) return "unknown"; + const active = await run({ file: "systemctl", args: ["--user", "is-active", "opencodex-proxy"] }); + const activeOut = (active.stdout + active.stderr).trim(); + if (active.ok || active.stdout.trim() === "inactive") return "present"; + if (/could not be found|no such file|not found/i.test(activeOut)) return "absent"; + return "unknown"; + }, + registrationFiles: () => [".config/systemd/user/opencodex-proxy.service"], + existingInstallation: (format, packageName) => + format === "deb" && packageName + ? { file: "dpkg-query", args: ["-W", "-f", "${Status}", packageName] } + : null, + async installArtifact(artifact, workDir, format) { + if (format === "deb") { + const packageName = (await run({ file: "dpkg-deb", args: ["-f", artifact, "Package"] })).stdout.trim(); + requireOk("dpkg install", await run({ file: "sudo", args: ["-n", "dpkg", "-i", artifact] })); + try { + const listing = await run({ file: "dpkg", args: ["-L", packageName] }); + const appBinary = listing.stdout.split(/\r?\n/).find(line => line.startsWith("/usr/bin/")) ?? ""; + if (!appBinary) throw new Error(`No /usr/bin executable found in package ${packageName}`); + return { appBinary, packageName, scope: appBinary, evidence: { packageName } }; + } catch (error) { + // The dpkg install already landed; a discovery failure must not strand it. + await run({ file: "sudo", args: ["-n", "dpkg", "-r", packageName] }); + throw error; + } + } + const appsDir = `${workDir}/apps`; + runtime.mkdir(appsDir); + const destination = `${appsDir}/OpenCodex.AppImage`; + requireOk("AppImage copy", await run({ file: "cp", args: [artifact, destination] })); + requireOk("AppImage chmod", await run({ file: "chmod", args: ["+x", destination] })); + return { appBinary: destination, scope: appsDir, evidence: { staged: destination } }; + }, + closeGesture: () => ({ file: "xdotool", args: ["search", "--name", "OpenCodex", "windowclose"] }), + quitGesture: () => ({ + file: "xdotool", + args: ["search", "--name", "OpenCodex", "windowactivate", "--sync", "key", "alt+F4"], + }), + // A stock GNOME session has no tray; on a runner with a tray extension a + // pre-installed hook drives it. The engine records which path was taken. + trayClick: () => null, + trayQuit: () => null, + trayCheck: () => null, + trayInstall: () => null, + windowVisible: () => ({ file: "xdotool", args: ["search", "--name", "OpenCodex"] }), + installedVersion: (format, packageName) => + format === "deb" && packageName + ? { file: "dpkg-query", args: ["-W", "-f", "${Version}", packageName] } + : null, + // Scoped to the pids the elevation monitor sighted — never a blanket pkill. + cancelElevation: pids => + pids.length > 0 ? { file: "kill", args: pids.map(String) } : null, + // pkexec is the first elevation surface; the updater plugin then falls back to a + // zenity or kdialog password dialog, and a cancel must produce NEITHER. + // The updater's full elevation chain is pkexec -> zenity/kdialog -> terminal sudo. + // The gate never invokes sudo during the monitored update windows, so any sighting + // there is attributable to the updater. + elevationProbe: () => ({ file: "pgrep", args: ["-x", "pkexec|zenity|kdialog|sudo"] }), + appProcessProbe: scope => ({ file: "pgrep", args: ["-f", scope] }), + appNameProbe: () => ({ file: "pgrep", args: ["-f", "opencodex-desktop"] }), + childPids: pid => ({ file: "pgrep", args: ["-P", String(pid)] }), + uninstall: (_artifact, workDir, format, packageName) => + format === "deb" && packageName + ? [{ file: "sudo", args: ["-n", "dpkg", "-r", packageName] }] + : [{ file: "rm", args: ["-f", `${workDir}/apps/OpenCodex.AppImage`] }], + }; +} + +function appleScript(...lines: string[]): CommandSpec { + return { file: "osascript", args: ["-e", lines.join(" ; ")] }; +} diff --git a/desktop/scripts/installed-gate.ts b/desktop/scripts/installed-gate.ts new file mode 100644 index 0000000000..4f02f443d4 --- /dev/null +++ b/desktop/scripts/installed-gate.ts @@ -0,0 +1,1002 @@ +/** + * The installed-artifact gate (D9 part two, R3). + * + * Installs the REAL desktop artifact on the host platform, launches it against a staged + * npm runtime, and drives the ownership contract from devlog plan 260921 + * (080_decisions_round2.md): the staged runtime on a non-default port is drained with + * its registration preserved, the desktop install id becomes the owner with exactly one + * consent-generation increment, healthz on the preserved home and port reports the + * bundled sidecar as a child of the app process, close and the OS quit gesture leave + * both pids alive with the window reopenable, a full quit and relaunch restore + * ownership without re-asking consent, and tray Quit lets an in-flight request finish + * before both pids end. On Linux both update paths are exercised (R3): an AppImage + * updates in place without elevation to the exact target bytes, and a deb install asks + * for authorization only after Install is chosen, never retries a cancelled prompt with + * another elevation mechanism, preserves the old version on cancel, and installs the + * new version on accept. + * + * Safety shape, per the external re-audit (110_reaudit.md): + * - preflight-isolation runs BEFORE any mutation. A run that refuses because it found + * an existing app, service registration or default-home state makes ZERO mutating + * calls, cleanup included — cleanup only ever touches resources this run acquired. + * - GUI automation comes from operator-installed hook FILES under --hooks-dir, never + * from dispatch-provided command text. + * - version inputs are strict semver; the npm package name is derived from this + * repository's own package.json, never accepted as an argument. + * + * Every side effect goes through GateDeps so tests can prove call discipline (see the + * refusal test in tests/ci-workflows/installed-gate-drivers.test.ts). + */ + +import { createHash } from "node:crypto"; +import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs"; +import { homedir } from "node:os"; +import { basename, join } from "node:path"; +import { commandInvocation } from "../../src/lib/win-exec"; +import { + type CommandSpec, + type GateFormat, + type GatePlatform, + type PlatformAdapter, + type ProcessEvidence, + linuxAdapter, + macosAdapter, + windowsAdapter, +} from "./installed-gate-platforms"; + +export interface GateOptions { + platform: GatePlatform; + format: GateFormat; + artifact: string; + olderArtifact?: string; + workDir: string; + toVersion: string; + fromVersion?: string; + reportPath: string; + hooksDir?: string; + consentHook?: string; + trayClickHook?: string; + trayQuitHook?: string; + trayCheckHook?: string; + trayInstallHook?: string; + /** Hook that answers the deb update's elevation prompt (drives the accept path). */ + elevateAcceptHook?: string; + takeoverTimeoutMs: number; +} + +export interface GatePhaseResult { + phase: string; + status: "pass" | "fail"; + detail: string; + evidence: Record; +} + +export interface GateReport { + platform: GatePlatform; + format: GateFormat; + toVersion: string; + startedAt: string; + finishedAt?: string; + phases: GatePhaseResult[]; + ok?: boolean; +} + +export interface SpawnedProcess { + pid: number; + kill: () => void; + exited: Promise; +} + +/** + * Every side effect the engine can perform. Tests inject fakes; production gets the + * real implementations from defaultGateDeps(). + */ +export interface GateDeps { + run(spec: CommandSpec, env?: Record): Promise; + pidAlive(pid?: number): boolean; + killProcess(pid: number): void; + fileExists(path: string): boolean; + readJsonFile(path: string): unknown; + writeTextFile(path: string, content: string): void; + makeDir(path: string): void; + fetchJson(url: string, init?: { method?: string; headers?: Record; body?: string; timeoutMs?: number }): Promise<{ ok: boolean; status: number; body: unknown }>; + spawnLogged(binary: string, logPath: string, errPath: string, env: Record): SpawnedProcess; + serveMockProvider(): MockProvider; + digestFile(path: string): string | null; + homeDir(): string; + sleep(ms: number): Promise; + readTextFile(path: string): string; +} + +/** Hook names are file names inside --hooks-dir, nothing more. */ +const HOOK_NAME = /^[a-z0-9][a-z0-9._-]*$/i; + +/** + * Version inputs become npm dist-tags and artifact URLs. Strict semver shape is the + * whole grammar they are allowed to carry — anything else (an npm alias like + * npm:other@latest, a flag fragment) is rejected before it can reach npm or a shell. + */ +const SEMVER = /^\d+\.\d+\.\d+(?:-[0-9A-Za-z.-]+)?$/; + +/** + * Numeric-triple semver order. A prerelease suffix sorts before the bare release of the + * same triple; two suffixed versions compare lexically. The gate only needs "strictly + * older" answers for well-formed inputs, which parseGateArguments guarantees. + */ +export function compareSemver(a: string, b: string): number { + const parse = (v: string) => { + const [triple, suffix] = v.split("-", 2); + const parts = triple!.split(".").map(Number); + return { parts, suffix }; + }; + const left = parse(a); + const right = parse(b); + for (let i = 0; i < 3; i++) { + const delta = (left.parts[i] ?? 0) - (right.parts[i] ?? 0); + if (delta !== 0) return delta < 0 ? -1 : 1; + } + if (left.suffix === right.suffix) return 0; + if (left.suffix === undefined) return 1; + if (right.suffix === undefined) return -1; + return left.suffix < right.suffix ? -1 : 1; +} + +export function parseGateArguments(argv: string[]): { options?: GateOptions; error?: string } { + const value = (name: string): string | undefined => { + const index = argv.indexOf(`--${name}`); + return index >= 0 ? argv[index + 1] : undefined; + }; + const required = ["platform", "format", "artifact", "work-dir", "to-version", "report"] as const; + const missing = required.filter(name => !value(name)); + if (missing.length > 0) { + return { error: `missing required arguments: ${missing.map(name => `--${name}`).join(", ")}` }; + } + const platform = value("platform"); + const format = value("format"); + if (platform !== "macos" && platform !== "windows" && platform !== "linux") { + return { error: "--platform must be macos, windows or linux" }; + } + const expectedFormat: Record = { + macos: ["dmg"], + windows: ["msi"], + linux: ["deb", "appimage"], + }; + if (!expectedFormat[platform].includes(format as GateFormat)) { + return { error: `--format ${format} is not a ${platform} artifact format` }; + } + // R3: a Linux gate that cannot see one of the two promised update paths is not a gate, + // so the older artifact and its version are required, exactly like every other input. + if (platform === "linux" && !value("older-artifact")) { + return { error: "--older-artifact is required on linux: both update paths are in scope" }; + } + const toVersion = value("to-version")!; + const fromVersion = value("from-version"); + if (!SEMVER.test(toVersion)) return { error: "--to-version must be a strict semver (x.y.z[-suffix])" }; + if (fromVersion !== undefined && !SEMVER.test(fromVersion)) { + return { error: "--from-version must be a strict semver (x.y.z[-suffix])" }; + } + if (platform === "linux") { + if (!fromVersion) return { error: "--from-version is required on linux: the update phases need a proven-older release" }; + if (fromVersion === toVersion) return { error: "--from-version must differ from --to-version" }; + if (compareSemver(fromVersion, toVersion) >= 0) { + return { error: "--from-version must be strictly older than --to-version" }; + } + } + const hooksDir = value("hooks-dir"); + const hooks: Array<[keyof GateOptions, string | undefined]> = [ + ["consentHook", value("consent-hook")], + ["trayClickHook", value("tray-click-hook")], + ["trayQuitHook", value("tray-quit-hook")], + ["trayCheckHook", value("tray-check-hook")], + ["trayInstallHook", value("tray-install-hook")], + ["elevateAcceptHook", value("elevate-accept-hook")], + ]; + for (const [key, name] of hooks) { + if (name === undefined) continue; + if (!hooksDir) return { error: `--${key.replace(/[A-Z]/g, c => "-" + c.toLowerCase())} requires --hooks-dir` }; + if (!HOOK_NAME.test(name)) { + return { error: `hook name \`${name}\` must be a plain file name inside the hooks directory` }; + } + } + const takeoverTimeoutMs = Number(value("takeover-timeout") ?? 180) * 1000; + if (!Number.isFinite(takeoverTimeoutMs) || takeoverTimeoutMs <= 0) { + return { error: "--takeover-timeout must be a positive number of seconds" }; + } + return { + options: { + platform, + format: format as GateFormat, + artifact: value("artifact")!, + olderArtifact: value("older-artifact"), + workDir: value("work-dir")!, + toVersion, + fromVersion, + reportPath: value("report")!, + hooksDir, + consentHook: value("consent-hook"), + trayClickHook: value("tray-click-hook"), + trayQuitHook: value("tray-quit-hook"), + trayCheckHook: value("tray-check-hook"), + trayInstallHook: value("tray-install-hook"), + elevateAcceptHook: value("elevate-accept-hook"), + takeoverTimeoutMs, + }, + }; +} + +/** + * The staged npm runtime's package spec, derived from this repository's own + * package.json — the gate never takes a package spec as an argument. + */ +export function npmPackageSpec(packageName: string, version: string): string { + return `${packageName}@${version}`; +} + +export function readOwnPackageName(packageJsonText: string): string | undefined { + try { + const parsed = JSON.parse(packageJsonText) as { name?: unknown }; + return typeof parsed.name === "string" && parsed.name.length > 0 ? parsed.name : undefined; + } catch { + return undefined; + } +} + +export interface OwnershipObservation { + ownerInstallId?: string; + consentGeneration?: number; + raw: unknown; +} + +/** + * Reads the durable ownership fields from service-state.json using lane C's schema: + * the record root carries an \`ownership\` object with the desktop install id and the + * consent generation. Any other shape is no observation, and the phase fails naming + * the file it read — the gate does not guess at schemas. + */ +export function observeOwnership(state: unknown): OwnershipObservation { + if (typeof state !== "object" || state === null) return { raw: state }; + const ownership = (state as Record).ownership; + if (typeof ownership !== "object" || ownership === null) return { raw: state }; + const record = ownership as Record; + return { + ownerInstallId: typeof record.installId === "string" ? record.installId : undefined, + consentGeneration: typeof record.consentGeneration === "number" ? record.consentGeneration : undefined, + raw: state, + }; +} + +export function evaluateOwnership( + before: OwnershipObservation, + after: OwnershipObservation, +): { ok: boolean; detail: string } { + if (before.ownerInstallId !== undefined) { + return { ok: false, detail: "the staged npm runtime already carried an owner; the takeover precondition is an unowned runtime" }; + } + if (!after.ownerInstallId) { + return { ok: false, detail: "the desktop install id is not recorded as owner in service-state.json" }; + } + const beforeGeneration = before.consentGeneration ?? 0; + if (after.consentGeneration === undefined) { + return { ok: false, detail: "no consent generation is recorded in service-state.json" }; + } + if (after.consentGeneration !== beforeGeneration + 1) { + return { + ok: false, + detail: `consent generation moved from ${beforeGeneration} to ${after.consentGeneration}; the contract allows exactly one increment`, + }; + } + return { ok: true, detail: `owner ${after.ownerInstallId} recorded with consent generation ${after.consentGeneration}` }; +} + +/** + * Parses a pid listing. Empty output is no pids — never pid 0, which on POSIX means + * the caller's own process group and must never be signalled from here. + */ +export function parsePidList(stdout: string): number[] { + return stdout + .split(/\r?\n/) + .map(line => line.trim()) + .filter(line => line.length > 0) + .map(Number) + .filter(value => Number.isSafeInteger(value) && value > 0); +} + +export function describeGatePhases(options: GateOptions): string[] { + const phases = [ + "preflight-isolation", + "runner-readiness", + "stage-npm-runtime", + "install-artifact", + "launch-and-take-over", + "runtime-identity", + "close-gesture", + "quit-gesture", + "relaunch-consent", + "tray-quit-drains", + ]; + if (options.platform === "linux") phases.push("update-verify"); + phases.push("cleanup"); + return phases; +} + +export function summarizeReport(report: GateReport): string { + const lines = report.phases.map( + phase => `${phase.status === "pass" ? "PASS" : "FAIL"} ${phase.phase}${phase.detail ? ` — ${phase.detail}` : ""}`, + ); + return [`installed-artifact gate: ${report.ok ? "PASS" : "FAIL"} (${report.platform}/${report.format} v${report.toVersion})`, ...lines].join("\n"); +} + +interface Healthz { + status: string; + version?: string; + pid?: number; + role?: string; +} + +const NON_DEFAULT_PORT = 10431; +const ELEVATION_POLL_MS = 250; + +export interface MockProvider { + port: number; + /** Resolves when the first chat completion actually reached the mock. */ + reached: Promise; + /** Lets the held request finish. */ + release: () => void; + stop: () => void; +} + +/** The openai-chat compatible mock the drain phase holds an in-flight request against. */ +export function startMockProvider(): MockProvider { + let finish: () => void = () => {}; + let markReached: () => void = () => {}; + const held = new Promise(resolve => { finish = resolve; }); + const reached = new Promise(resolve => { markReached = resolve; }); + const server = Bun.serve({ + port: 0, + fetch: async request => { + if (new URL(request.url).pathname.endsWith("/chat/completions")) { + markReached(); + await held; + return Response.json({ + id: "gate-drain", + object: "chat.completion", + choices: [{ index: 0, message: { role: "assistant", content: "drained" }, finish_reason: "stop" }], + }); + } + return new Response("not found", { status: 404 }); + }, + }); + return { port: server.port, reached, release: finish, stop: () => server.stop(true) }; +} + +export function defaultGateDeps(): GateDeps { + const spawnProcess = (binary: string, env: Record, outPath?: string, errPath?: string): SpawnedProcess => { + const invocation = commandInvocation(binary, []); + const child = Bun.spawn({ + cmd: [invocation.file, ...invocation.args], + env, + // Bun.spawn accepts a BunFile directly; a FileSink is not a valid stdio target. + stdout: outPath ? Bun.file(outPath) : "ignore", + stderr: errPath ? Bun.file(errPath) : "ignore", + stdin: "ignore", + ...invocation.options, + }); + return { pid: child.pid, kill: () => child.kill(), exited: child.exited }; + }; + return { + run: async (spec, env) => { + const invocation = commandInvocation(spec.file, spec.args); + const proc = Bun.spawn({ + cmd: [invocation.file, ...invocation.args], + stdout: "pipe", + stderr: "pipe", + env: env ?? { ...process.env }, + ...invocation.options, + }); + const [stdout, stderr, exitCode] = await Promise.all([ + new Response(proc.stdout).text(), + new Response(proc.stderr).text(), + proc.exited, + ]); + return { ok: exitCode === 0, exitCode, stdout, stderr }; + }, + pidAlive: pid => { + if (typeof pid !== "number") return false; + try { + process.kill(pid, 0); + return true; + } catch { + return false; + } + }, + killProcess: pid => process.kill(pid), + fileExists: path => existsSync(path), + readJsonFile: path => { + try { + return JSON.parse(readFileSync(path, "utf8")); + } catch { + return undefined; + } + }, + writeTextFile: (path, content) => writeFileSync(path, content), + makeDir: path => mkdirSync(path, { recursive: true }), + fetchJson: async (url, init) => { + const response = await fetch(url, { + method: init?.method, + headers: init?.headers, + body: init?.body, + signal: AbortSignal.timeout(init?.timeoutMs ?? 4000), + }); + let body: unknown = undefined; + try { + body = await response.json(); + } catch { + body = undefined; + } + return { ok: response.ok, status: response.status, body }; + }, + spawnLogged: (binary, logPath, errPath, env) => spawnProcess(binary, env, logPath, errPath), + serveMockProvider: () => startMockProvider(), + digestFile: path => (existsSync(path) ? createHash("sha256").update(readFileSync(path)).digest("hex") : null), + homeDir: () => homedir(), + sleep: ms => new Promise(resolve => setTimeout(resolve, ms)), + readTextFile: path => readFileSync(path, "utf8"), + }; +} + +export async function runGate(options: GateOptions, deps: GateDeps = defaultGateDeps()): Promise { + const report: GateReport = { + platform: options.platform, + format: options.format, + toVersion: options.toVersion, + startedAt: new Date().toISOString(), + phases: [], + }; + const workDir = options.workDir; + const home = join(workDir, "preserved-home"); + const codexHome = join(workDir, "codex-home"); + const npmPrefix = join(workDir, "npm-prefix"); + const port = NON_DEFAULT_PORT; + const isolatedEnv = (): Record => ({ + ...process.env, + OPENCODEX_HOME: home, + CODEX_HOME: codexHome, + }); + // Every command the gate drives runs under the staged homes — a probe or a service + // invocation must never read the runner account's real opencodex or codex home. + const run = (spec: CommandSpec): Promise => deps.run(spec, isolatedEnv()); + const adapter: PlatformAdapter = (options.platform === "macos" ? macosAdapter : options.platform === "windows" ? windowsAdapter : linuxAdapter)( + { run, mkdir: path => deps.makeDir(path), fileExists: path => deps.fileExists(path), homeDir: () => deps.homeDir() }, + ); + const launcher = adapter.npmLauncher(npmPrefix); + const spawned: SpawnedProcess[] = []; + let npmPid: number | undefined; + let appPid: number | undefined; + let appBinary: string | undefined; + let packageName: string | undefined; + let takeoverOwnerId: string | undefined; + let takeoverGeneration: number | undefined; + let takeoverRuntimePid: number | undefined; + let installScope: string | undefined; + let mock: MockProvider | undefined; + let stopVerification = false; + + const record = (phase: string, ok: boolean, detail: string, evidence: Record = {}) => { + report.phases.push({ phase, status: ok ? "pass" : "fail", detail, evidence }); + if (!ok) stopVerification = true; + }; + + const listPids = async (spec: CommandSpec): Promise => { + const result = await run(spec); + if (!result.ok) return []; + return parsePidList(result.stdout); + }; + + const healthz = async (): Promise => { + try { + const response = await deps.fetchJson(`http://127.0.0.1:${port}/healthz`, { timeoutMs: 4000 }); + if (!response.ok) return null; + const body = response.body as Record; + if (body?.status !== "ok") return null; + return { + status: "ok", + version: typeof body.version === "string" ? body.version : undefined, + pid: typeof body.pid === "number" ? body.pid : undefined, + role: typeof body.role === "string" ? body.role : undefined, + }; + } catch { + return null; + } + }; + + const waitFor = async (predicate: () => Promise, timeoutMs: number, everyMs = 500): Promise => { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + if (await predicate()) return true; + await deps.sleep(everyMs); + } + return await predicate(); + }; + + /** + * A registration is present when any manager probe exits zero OR any on-disk + * registration file exists — unloaded launchd jobs, disabled systemd units and the + * WinSW native service all leave traces a single manager query would miss. + */ + /** Resources THIS invocation placed on the machine; cleanup touches nothing else. */ + const acquired = { service: false, artifact: false }; + + const resolveHook = (name?: string): string | undefined => + name && options.hooksDir ? join(options.hooksDir, name) : undefined; + + /** A pre-installed hook file runs directly, never through a shell. */ + const runHook = async (name: string | undefined, fallback: () => CommandSpec | null): Promise<{ ok: boolean; via: string }> => { + const hook = resolveHook(name); + if (hook) { + if (!deps.fileExists(hook)) return { ok: false, via: `missing hook ${hook}` }; + return { ok: (await run({ file: hook, args: [] })).ok, via: hook }; + } + const spec = fallback(); + if (!spec) return { ok: false, via: "no hook and no platform default" }; + return { ok: (await run(spec)).ok, via: spec.file }; + }; + + const windowVisible = async (): Promise => { + const probe = await run(adapter.windowVisible()); + // macOS prints a window count, Linux prints one window id per line, Windows + // prints nothing and answers through the exit code. + const firstLine = probe.stdout.trim().split(/\r?\n/)[0] ?? ""; + return probe.ok && (firstLine === "" || Number(firstLine) > 0); + }; + + try { + // ---- preflight-isolation: refuse to touch a machine with real opencodex state. + // This phase completes BEFORE anything mutating; a refusal skips every later phase, + // and the cleanup below touches only resources recorded in `acquired`/`spawned`. + // Registration absence must be PROVEN: an unreadable manager probe is "unknown", + // and unknown refuses the run exactly like present does. + const registration = await adapter.registrationState(); + const defaultState = join(deps.homeDir(), ".opencodex", "service-state.json"); + const defaultStatePresent = deps.fileExists(defaultState); + const runningApp = await listPids(adapter.appNameProbe()); + // A dormant install counts too: the gate would overwrite it, and cleanup could + // then remove something this run never installed. For deb the package name is read + // from the artifact (read-only) before probing dpkg. + let dormantInstall = false; + if (options.format === "deb") { + const nameProbe = await run({ file: "dpkg-deb", args: ["-f", options.artifact, "Package"] }); + if (nameProbe.ok && nameProbe.stdout.trim()) { + const installed = adapter.existingInstallation("deb", nameProbe.stdout.trim()); + dormantInstall = installed !== null && (await run(installed)).ok; + } + } else { + const installed = adapter.existingInstallation(options.format); + dormantInstall = installed !== null && (await run(installed)).ok; + } + const isolated = registration === "absent" && !defaultStatePresent && runningApp.length === 0 && !dormantInstall; + record( + "preflight-isolation", + isolated, + isolated + ? "no existing registration, default-home state or running app" + : "this runner already carries opencodex state; the gate would overwrite or remove it — refusing to run", + { registration, defaultStatePresent, runningApp, dormantInstall }, + ); + + if (!stopVerification) { + deps.makeDir(home); + deps.makeDir(codexHome); + // ---- runner-readiness: every external command the adapter needs must resolve + // before an artifact is installed. + const dependencies = [...adapter.dependencies(), "npm"]; + const missing: string[] = []; + for (const dependency of dependencies) { + const probe = process.platform === "win32" + ? await run({ file: "where.exe", args: [dependency] }) + : await run({ file: "sh", args: ["-c", `command -v ${dependency}`] }); + if (!probe.ok) missing.push(dependency); + } + record("runner-readiness", missing.length === 0, + missing.length === 0 ? `${dependencies.length} external commands resolve` : `missing: ${missing.join(", ")}`, + { missing }); + } + + if (!stopVerification) { + // ---- stage-npm-runtime: register FIRST so the recorded pid is the managed one. + // The package name comes from this repository's own package.json; only the + // semver-validated version is operator input. + const staging: Record = {}; + const packageName_ = readOwnPackageName(deps.readTextFile(join(import.meta.dir, "..", "..", "package.json"))); + if (!packageName_) { + record("stage-npm-runtime", false, "could not read this repository's npm package name", staging); + } else { + const spec = npmPackageSpec(packageName_, options.fromVersion ?? options.toVersion); + staging.npmPackage = spec; + const install = await run({ file: "npm", args: ["install", "--prefix", npmPrefix, spec] }); + staging.npmInstallExit = install.exitCode; + mock = deps.serveMockProvider(); + deps.writeTextFile( + join(home, "config.json"), + JSON.stringify( + { + port, + defaultProvider: "gate-mock", + providers: { + "gate-mock": { + adapter: "openai-chat", + baseUrl: `http://127.0.0.1:${mock.port}/v1`, + apiKey: "gate-mock-key", + }, + }, + }, + null, + 2, + ) + "\n", + ); + let ok = install.ok; + if (ok) { + const registered = await run(adapter.serviceInstall(launcher)); + staging.serviceInstallExit = registered.exitCode; + ok = registered.ok; + // A nonzero exit can still leave a registration behind; if anything is + // registered now, this run owns removing it. + acquired.service = (await adapter.registrationState()) === "present"; + } + if (ok) ok = await waitFor(() => healthz().then(Boolean), 60_000); + const before = await healthz(); + npmPid = before?.pid; + const present = (await adapter.registrationState()) === "present"; + staging.npmRuntime = before; + staging.registrationPresent = present; + ok = ok && present && typeof npmPid === "number"; + record( + "stage-npm-runtime", + ok, + ok ? `managed npm runtime pid ${npmPid} on port ${port}; registration present` : "staging failed; see evidence", + staging, + ); + } + } + + const ownershipBefore = observeOwnership(deps.readJsonFile(join(home, "service-state.json"))); + + if (!stopVerification) { + // ---- install-artifact: the real artifact, installed like a user would. + try { + const installResult = await adapter.installArtifact(options.artifact, workDir, options.format); + appBinary = installResult.appBinary; + packageName = installResult.packageName; + installScope = installResult.scope; + acquired.artifact = true; + record("install-artifact", Boolean(appBinary), `installed ${basename(options.artifact)} -> ${appBinary}`, installResult.evidence); + } catch (error) { + // A partial install is still an acquisition: mark it so cleanup rolls it back. + acquired.artifact = true; + record("install-artifact", false, String(error)); + } + } + + if (!stopVerification && appBinary) { + // ---- launch-and-take-over: consent once, drain the npm runtime, keep the registration. + const launched = deps.spawnLogged(appBinary, join(workDir, "app.log"), join(workDir, "app.err.log"), isolatedEnv()); + spawned.push(launched); + appPid = launched.pid; + if (options.consentHook) await runHook(options.consentHook, () => null); + const taken = await waitFor(async () => { + const now = await healthz(); + return now !== null && typeof now.pid === "number" && now.pid !== npmPid; + }, options.takeoverTimeoutMs); + const after = await healthz(); + const npmDrained = !deps.pidAlive(npmPid); + const registration = (await adapter.registrationState()) === "present"; + const ownershipAfter = observeOwnership(deps.readJsonFile(join(home, "service-state.json"))); + const ownership = evaluateOwnership(ownershipBefore, ownershipAfter); + if (ownership.ok) { + takeoverOwnerId = ownershipAfter.ownerInstallId; + takeoverGeneration = ownershipAfter.consentGeneration; + takeoverRuntimePid = after?.pid; + } + const ok = taken && npmDrained && registration && ownership.ok; + record("launch-and-take-over", ok, [ownership.detail, `npm pid drained: ${npmDrained}`, `registration present: ${registration}`].join("; "), { + before: ownershipBefore.raw, + after: ownershipAfter.raw, + healthzAfter: after, + }); + } + + if (!stopVerification) { + // ---- runtime-identity: healthz on the preserved home+port is the bundled sidecar, + // a child of THIS launched app, reporting THIS release's version. + const now = await healthz(); + const children = appPid !== undefined ? await listPids(adapter.childPids(appPid)) : []; + const portRecord = deps.readJsonFile(join(home, "runtime-port.json")) as { port?: number; pid?: number } | undefined; + const ok = + now?.pid !== undefined && + children.includes(now.pid) && + portRecord?.port === port && + portRecord?.pid === now.pid && + now.version === options.toVersion; + record("runtime-identity", ok, ok + ? `healthz pid ${now?.pid} v${now?.version} is a child of app pid ${appPid} on preserved port ${port}` + : "the answering runtime is not the bundled sidecar of the launched app on the preserved home", + { healthz: now, appPid, sidecarCandidates: children, runtimePortRecord: portRecord }); + } + + const gesture = async (phase: string, spec: CommandSpec) => { + if (stopVerification) return; + const before = await healthz(); + const gestureResult = await run(spec); + // The gesture must actually hide the window; a no-op command exit is not the + // contract. + const windowHidden = await waitFor(async () => !(await windowVisible()), 10_000); + const after = await healthz(); + const runtimeAlive = before?.pid !== undefined && before.pid === after?.pid; + const appAlive = deps.pidAlive(appPid); + const reopen = await runHook(options.trayClickHook, () => adapter.trayClick()); + const visible = await waitFor(windowVisible, 15_000); + const ok = gestureResult.ok && windowHidden && runtimeAlive && appAlive && reopen.ok && visible; + record(phase, ok, + `gesture exit ${gestureResult.exitCode}; window hidden: ${windowHidden}; runtime pid ${after?.pid} alive: ${runtimeAlive}; app pid ${appPid} alive: ${appAlive}; window reopened via ${reopen.via}: ${visible}`, + { before, after }); + }; + + await gesture("close-gesture", adapter.closeGesture()); + await gesture("quit-gesture", adapter.quitGesture()); + + if (!stopVerification && appBinary) { + // ---- relaunch-consent: a FULL quit (tray Quit drains and ends both pids), then a + // cold relaunch must restore ownership WITHOUT asking again — the same install id, + // the same consent generation. Watching a single-instance duplicate exit is not + // this contract. + const quit = await runHook(options.trayQuitHook, () => adapter.trayQuit()); + // Both pids — the app AND the runtime it owned at takeover — must actually end + // before the relaunch means anything. + const previousAppPid = appPid; + const ended = await waitFor(async () => + !deps.pidAlive(previousAppPid) + && !deps.pidAlive(takeoverRuntimePid) + && (installScope === undefined || (await listPids(adapter.appProcessProbe(installScope))).length === 0), + 30_000); + let ownershipRestored = false; + let relaunchHealth: Healthz | null = null; + if (ended) { + const relaunched = deps.spawnLogged(appBinary, join(workDir, "relaunch.log"), join(workDir, "relaunch.err.log"), isolatedEnv()); + spawned.push(relaunched); + appPid = relaunched.pid; + const up = await waitFor(() => healthz().then(Boolean), 60_000); + relaunchHealth = await healthz(); + const ownership = observeOwnership(deps.readJsonFile(join(home, "service-state.json"))); + ownershipRestored = up + && ownership.ownerInstallId !== undefined + && ownership.ownerInstallId === takeoverOwnerId; + // A re-asked consent would move the generation; identical generation is the + // proof that nothing was asked. + ownershipRestored &&= ownership.consentGeneration !== undefined && ownership.consentGeneration === takeoverGeneration; + } + const ok = quit.ok && ended && ownershipRestored; + record("relaunch-consent", ok, ok + ? `full quit and cold relaunch restored owner ${takeoverOwnerId} without re-asking consent` + : "ownership was not restored after a cold relaunch, or consent was asked again", + { quitVia: quit.via, ended, ownerAfterRelaunch: relaunchHealth, takeoverOwnerId }); + } + + if (!stopVerification) { + // ---- tray-quit-drains: the request must be verifiably in flight when Quit fires. + const before = await healthz(); + const request = deps.fetchJson(`http://127.0.0.1:${port}/v1/chat/completions`, { + method: "POST", + headers: { "content-type": "application/json", authorization: "Bearer gate-mock-key" }, + body: JSON.stringify({ model: "gate-mock/gate-model", messages: [{ role: "user", content: "hold" }] }), + timeoutMs: 90_000, + }).then(response => response.status); + const inFlight = await Promise.race([ + mock!.reached.then(() => true), + deps.sleep(15_000).then(() => false), + ]); + const quit = await runHook(options.trayQuitHook, () => adapter.trayQuit()); + await deps.sleep(2000); + mock?.release(); + let requestStatus: number | null = null; + try { + requestStatus = await request; + } catch { + requestStatus = null; + } + const drained = requestStatus === 200; + const bothEnded = await waitFor(async () => !deps.pidAlive(before?.pid) && (installScope === undefined || (await listPids(adapter.appProcessProbe(installScope))).length === 0), 30_000); + const ok = inFlight && quit.ok && drained && bothEnded; + record("tray-quit-drains", ok, + `request in flight at Quit: ${inFlight}; tray Quit driven via ${quit.via}; request finished with ${requestStatus}; both pids ended: ${bothEnded}`, + { runtimePid: before?.pid, requestStatus }); + } + + if (!stopVerification && options.platform === "linux" && options.olderArtifact && options.fromVersion) { + // ---- update-verify (R3): both Linux formats update through their own path, on + // both authorization outcomes. A failed prerequisite stops the phase BEFORE the + // next mutation, never after it. + for (const spec of adapter.uninstall(options.artifact, workDir, options.format, packageName)) { + await run(spec); + } + const older = await adapter.installArtifact(options.olderArtifact, workDir, options.format); + packageName = older.packageName; + installScope = older.scope; + const oldApp = deps.spawnLogged(older.appBinary, join(workDir, "older-app.log"), join(workDir, "older-app.err.log"), isolatedEnv()); + spawned.push(oldApp); + appPid = oldApp.pid; + const oldHealthy = await waitFor(() => healthz().then(Boolean), 60_000); + const preVersion = adapter.installedVersion(options.format, packageName); + const preVersionOutput = preVersion ? await run(preVersion) : null; + const preVersionText = preVersionOutput?.ok ? preVersionOutput.stdout.trim() : ""; + const preDigest = options.format === "appimage" ? deps.digestFile(older.appBinary) : null; + const targetDigest = options.format === "appimage" ? deps.digestFile(options.artifact) : null; + + // One continuous monitor across the whole update operation: a fixed window can + // close before download and signature verification reach the elevation step. + // Sightings are attributed by the timestamp of each driver action. + const sightings: Array<{ pid: number; at: number }> = []; + const elevationProbe = adapter.elevationProbe(); + let monitoring = true; + const monitorTask = (async () => { + while (monitoring && elevationProbe) { + for (const pid of await listPids(elevationProbe)) sightings.push({ pid, at: Date.now() }); + await deps.sleep(ELEVATION_POLL_MS); + } + })(); + const stopMonitor = async () => { monitoring = false; await monitorTask; }; + const sightingsAfter = (timestamp: number): number[] => + [...new Set(sightings.filter(sighting => sighting.at >= timestamp).map(sighting => sighting.pid))]; + + try { + let check = { ok: false, via: "skipped: old app never became healthy" }; + let install = { ok: false, via: "skipped: old app never became healthy" }; + let installStart = Number.POSITIVE_INFINITY; + if (oldHealthy) { + check = await runHook(options.trayCheckHook, () => adapter.trayCheck()); + installStart = Date.now(); + install = await runHook(options.trayInstallHook, () => adapter.trayInstall()); + } + const elevationDuringCheck = [...new Set( + sightings.filter(sighting => sighting.at < installStart).map(sighting => sighting.pid), + )]; + + if (options.format === "appimage") { + // The installed file must become byte-identical to the target artifact — a + // changed digest alone would pass for an update to the wrong version, and a + // missing pre/target digest would make the transition vacuous. + await waitFor(async () => deps.digestFile(older.appBinary) === targetDigest, 120_000); + const postDigest = deps.digestFile(older.appBinary); + const anyElevation = sightingsAfter(0); + const ok = oldHealthy && check.ok && install.ok + && elevationDuringCheck.length === 0 && anyElevation.length === 0 + && preDigest !== null && targetDigest !== null && preDigest !== targetDigest + && postDigest !== null && postDigest === targetDigest; + record("update-verify", ok, + `AppImage updated in place to the exact target artifact (digest match: ${postDigest === targetDigest}); no elevation anywhere (${anyElevation.length} sighted); path kept`, + { preDigest, postDigest, targetDigest, elevation: anyElevation }); + } else { + // Cancel path: wait for the prompt, dismiss exactly the sighted pids, prove + // they exited, then prove NO elevation mechanism retries (the pinned plugin + // otherwise falls back pkexec -> zenity/kdialog -> sudo), and the version + // never moved. + const prompted = await waitFor(async () => sightingsAfter(installStart).length > 0, 90_000); + const elevation = sightingsAfter(installStart); + const cancel = adapter.cancelElevation(elevation); + let cancelOk = elevation.length === 0; + if (cancel && elevation.length > 0) { + cancelOk = (await run(cancel)).ok; + cancelOk &&= await waitFor(async () => elevation.every(pid => !deps.pidAlive(pid)), 10_000); + } + const cancelDoneAt = Date.now(); + await deps.sleep(10_000); + const retriedElevation = sightingsAfter(cancelDoneAt); + const settledVersion = adapter.installedVersion(options.format, packageName); + const settledVersionOutput = settledVersion ? await run(settledVersion) : null; + const settledVersionText = settledVersionOutput?.ok ? settledVersionOutput.stdout.trim() : ""; + const cancelPreserved = preVersionText !== "" && preVersionText === options.fromVersion && settledVersionText === preVersionText; + const cancelOkAll = oldHealthy && check.ok && install.ok + && elevationDuringCheck.length === 0 && prompted + && cancelOk && retriedElevation.length === 0 && cancelPreserved; + + // Accept path: only after the cancel path held. Drive Install again, answer + // through the operator hook, and require the package to reach the target. + let acceptOk = false; + let acceptEvidence: Record = { skipped: "no --elevate-accept-hook" }; + if (options.elevateAcceptHook && cancelOkAll) { + const acceptStart = Date.now(); + const installAgain = await runHook(options.trayInstallHook, () => adapter.trayInstall()); + const promptedAgain = await waitFor(async () => sightingsAfter(acceptStart).length > 0, 90_000); + let hookOk = false; + if (promptedAgain) { + hookOk = (await runHook(options.elevateAcceptHook, () => null)).ok; + } + const accepted = await waitFor(async () => { + const probe = adapter.installedVersion(options.format, packageName); + if (!probe) return false; + const result = await run(probe); + return result.ok && result.stdout.trim() === options.toVersion; + }, 120_000); + acceptOk = installAgain.ok && promptedAgain && hookOk && accepted; + acceptEvidence = { installAgain: installAgain.ok, promptedAgain, hookOk, accepted }; + } else if (options.elevateAcceptHook) { + acceptEvidence = { skipped: "cancel path failed; accept not attempted" }; + } + const ok = cancelOkAll && acceptOk; + record("update-verify", ok, + `deb cancel path preserved ${settledVersionText} with no elevation retry (${retriedElevation.length}); accept path reached ${options.toVersion}: ${acceptOk}`, + { preVersion: preVersionText, postCancelVersion: settledVersionText, cancelOk, prompted, retriedElevation, accept: acceptEvidence }); + } + } finally { + await stopMonitor(); + } + } + } catch (error) { + // A thrown exception is a fatal phase of its own: without this, a crash between + // phases could leave a report whose recorded phases all pass. + record("fatal-error", false, String(error)); + } finally { + // ---- cleanup: rolls back ONLY what this invocation acquired. A preflight refusal + // means nothing here runs against machine state: the gate must never destroy an + // existing installation it detected. Every rollback step runs; a failure fails the + // phase but never stops the remaining steps. + const cleanupEvidence: Record = {}; + let cleanupOk = true; + const fail = (key: string, error: unknown) => { + cleanupOk = false; + cleanupEvidence[key] = String(error); + }; + for (const child of spawned) { + try { child.kill(); } catch (error) { fail(`spawned-${child.pid}`, error); } + } + // Processes the gate no longer owns: an AppImage update restarts detached from the + // original spawn handle. Only swept when this run launched an app at all. + if (appPid !== undefined && installScope !== undefined) { + for (const pid of await listPids(adapter.appProcessProbe(installScope))) { + try { deps.killProcess(pid); } catch (error) { fail(`app-${pid}`, error); } + } + } + if (deps.pidAlive(npmPid)) { + try { deps.killProcess(npmPid!); } catch (error) { fail("npm-runtime", error); } + } + if (acquired.service) { + try { + const result = await run(adapter.serviceUninstall(launcher)); + if (!result.ok) fail("service-uninstall", result.stderr.trim() || `exit ${result.exitCode}`); + } catch (error) { fail("service-uninstall", error); } + } + if (acquired.artifact) { + for (const spec of adapter.uninstall(options.artifact, workDir, options.format, packageName)) { + try { + const result = await run(spec); + if (!result.ok) fail(`uninstall:${spec.args[1] ?? spec.file}`, result.stderr.trim() || `exit ${result.exitCode}`); + } catch (error) { fail("uninstall", error); } + } + } + mock?.stop(); + record("cleanup", cleanupOk, cleanupOk ? "everything the gate installed was rolled back" : "a rollback step failed; see evidence", cleanupEvidence); + report.finishedAt = new Date().toISOString(); + // Green means every phase ran AND passed — a report missing phases (a crash, an + // early refusal) is not green even if everything recorded passed. + const expectedPhases = describeGatePhases(options); + const covered = expectedPhases.every(name => report.phases.some(phase => phase.phase === name)); + report.ok = report.phases.every(phase => phase.status === "pass") && covered; + deps.writeTextFile(options.reportPath, `${JSON.stringify(report, null, 2)}\n`); + } + return report; +} + +if (import.meta.main) { + const parsed = parseGateArguments(Bun.argv.slice(2)); + if (!parsed.options || parsed.error) { + console.error(parsed.error ?? "invalid arguments"); + console.error( + "usage: installed-gate.ts --platform --format --artifact " + + " --older-artifact --work-dir --to-version --from-version --report " + + " [--hooks-dir ] [--consent-hook ] [--tray-click-hook ] [--tray-quit-hook ]" + + " [--tray-check-hook ] [--tray-install-hook ] [--elevate-accept-hook ] [--takeover-timeout ]", + ); + process.exit(2); + } + const report = await runGate(parsed.options); + console.log(summarizeReport(report)); + process.exit(report.ok ? 0 : 1); +} diff --git a/desktop/scripts/linux-packaged-e2e.ts b/desktop/scripts/linux-packaged-e2e.ts new file mode 100644 index 0000000000..9ca9235266 --- /dev/null +++ b/desktop/scripts/linux-packaged-e2e.ts @@ -0,0 +1,566 @@ +#!/usr/bin/env bun +/** + * Hosted Linux packaged-shell acceptance. + * + * This is deliberately narrower than installed-gate.ts. It extracts, rather than + * installs, the AppImage and deb payloads so a hosted runner never mutates its package + * database or the runner account's real OpenCodex home. What it proves is the common + * packaged path: the real application executable and bundled resources can show a + * window in a session with no tray host, start their bundled sidecar, identify that + * runtime, and drain both processes when the only window closes. + * + * Real dpkg/AppImage installation, elevation, takeover, and in-place updates remain the + * responsibility of installed-gate.ts on an approved disposable GUI runner. + */ +import { spawn, spawnSync, type ChildProcess } from "node:child_process"; +import { + closeSync, + existsSync, + mkdirSync, + mkdtempSync, + openSync, + readFileSync, + readdirSync, + rmSync, + statSync, + writeFileSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import { basename, dirname, join, resolve } from "node:path"; +import { createServer } from "node:net"; + +export type LinuxBundleFormat = "appimage" | "deb"; + +export interface LinuxE2eOptions { + bundleRoot: string; + reportPath: string; + version: string; +} + +export interface BundleArtifacts { + appimage: string; + deb: string; +} + +interface RuntimeRecord { + pid: number; + port: number; +} + +interface HealthObservation { + status: number; + body: Record; +} + +interface ReservedLoopbackPort { + port: number; + release: () => Promise; +} + +interface FormatReport { + format: LinuxBundleFormat; + artifact: string; + ok: boolean; + durationMs: number; + windowId?: string; + appPid?: number; + appExitCode?: number | null; + appExitSignal?: string | null; + runtimePid?: number; + runtimeVersion?: string; + configuredPort?: number; + readyMs?: number; + processTreeRssKiB?: number; + error?: string; + stdoutTail?: string[]; + stderrTail?: string[]; +} + +interface AcceptanceReport { + schema: "opencodex-linux-packaged-e2e/1"; + version: string; + startedAt: string; + finishedAt: string; + ok: boolean; + formats: FormatReport[]; +} + +const READY_DEADLINE_MS = 45_000; +const EXIT_DEADLINE_MS = 30_000; +const POLL_MS = 200; +const LOG_TAIL_LINES = 80; +const VERSION = /^\d+\.\d+\.\d+(?:-[0-9A-Za-z.-]+)?$/; + +function argument(argv: string[], name: string): string | undefined { + const index = argv.indexOf(name); + return index >= 0 ? argv[index + 1] : undefined; +} + +export function parseArguments(argv: string[]): LinuxE2eOptions { + const bundleRoot = argument(argv, "--bundle-root"); + const reportPath = argument(argv, "--report"); + const version = argument(argv, "--version"); + if (!bundleRoot || !reportPath || !version) { + throw new Error("--bundle-root, --report and --version are required"); + } + if (!VERSION.test(version)) throw new Error("--version must be a strict semver"); + return { + bundleRoot: resolve(bundleRoot), + reportPath: resolve(reportPath), + version, + }; +} + +function files(directory: string): string[] { + if (!existsSync(directory)) return []; + return readdirSync(directory) + .map(name => join(directory, name)) + .filter(path => statSync(path).isFile()); +} + +function exactlyOne(paths: string[], label: string): string { + if (paths.length !== 1) { + throw new Error(`expected exactly one ${label}, found ${paths.length}`); + } + return paths[0]!; +} + +export function locateArtifacts(bundleRoot: string): BundleArtifacts { + return { + appimage: exactlyOne( + files(join(bundleRoot, "appimage")).filter(path => path.endsWith(".AppImage")), + "AppImage", + ), + deb: exactlyOne( + files(join(bundleRoot, "deb")).filter(path => path.endsWith(".deb")), + "deb", + ), + }; +} + +function command( + file: string, + args: string[], + options: { cwd?: string; env?: NodeJS.ProcessEnv } = {}, +): void { + const result = spawnSync(file, args, { + cwd: options.cwd, + env: options.env, + encoding: "utf8", + maxBuffer: 8 * 1024 * 1024, + }); + if (result.status !== 0) { + const detail = (result.stderr || result.stdout || "no output").trim(); + throw new Error(`${basename(file)} exited ${result.status ?? "without a status"}: ${detail}`); + } +} + +function executableFiles(directory: string): string[] { + if (!existsSync(directory)) return []; + return readdirSync(directory) + .map(name => join(directory, name)) + .filter(path => { + const stat = statSync(path); + return stat.isFile() && (stat.mode & 0o111) !== 0; + }); +} + +export function extractedExecutable( + format: LinuxBundleFormat, + artifact: string, + destination: string, +): string { + mkdirSync(destination, { recursive: true }); + if (format === "appimage") { + command(artifact, ["--appimage-extract"], { cwd: destination }); + const appRun = join(destination, "squashfs-root", "AppRun"); + if (!existsSync(appRun)) throw new Error("AppImage extraction did not produce AppRun"); + return appRun; + } + + command("dpkg-deb", ["--extract", artifact, destination]); + const candidates = executableFiles(join(destination, "usr", "bin")); + return selectDebExecutable(candidates); +} + +export function selectDebExecutable(candidates: string[]): string { + // The package contains the desktop host and its `ocx` sidecar. The sidecar is deliberately + // executable, but it is not the process whose WebView/window lifecycle this acceptance owns. + return exactlyOne( + candidates.filter(candidate => basename(candidate) !== "ocx"), + "deb desktop executable under usr/bin", + ); +} + +function sleep(ms: number): Promise { + return new Promise(resolve => setTimeout(resolve, ms)); +} + +async function reserveLoopbackPort(): Promise { + return await new Promise((resolvePort, reject) => { + const server = createServer(); + server.unref(); + server.once("error", reject); + server.listen(0, "127.0.0.1", () => { + const address = server.address(); + if (!address || typeof address === "string") { + server.close(); + reject(new Error("could not reserve a temporary loopback port")); + return; + } + let released = false; + resolvePort({ + port: address.port, + release: async () => { + if (released) return; + released = true; + await new Promise((resolveClose, rejectClose) => { + server.close(error => error ? rejectClose(error) : resolveClose()); + }); + }, + }); + }); + }); +} + +async function waitFor(read: () => T | undefined | Promise, timeoutMs: number): Promise { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + const value = await read(); + if (value !== undefined) return value; + await sleep(POLL_MS); + } + throw new Error(`condition did not settle within ${timeoutMs}ms`); +} + +function positiveInteger(value: unknown): number | undefined { + return typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : undefined; +} + +export function readRuntimeRecord(path: string): RuntimeRecord | undefined { + try { + const parsed = JSON.parse(readFileSync(path, "utf8")) as Record; + const pid = positiveInteger(parsed.pid); + const port = positiveInteger(parsed.port); + if (pid === undefined || port === undefined || port > 65_535) return undefined; + return { pid, port }; + } catch { + return undefined; + } +} + +export function assertRuntimeRecordPort(record: RuntimeRecord, configuredPort: number): RuntimeRecord { + if (record.port !== configuredPort) { + throw new Error( + `packaged runtime recorded port ${record.port}, expected isolated port ${configuredPort}`, + ); + } + return record; +} + +function processAlive(pid: number | undefined): boolean { + if (pid === undefined) return false; + try { + process.kill(pid, 0); + return true; + } catch (error) { + return typeof error === "object" && error !== null && "code" in error && error.code === "EPERM"; + } +} + +function processRows(): Array<{ pid: number; ppid: number; rssKiB: number }> { + const result = spawnSync("ps", ["-e", "-o", "pid=,ppid=,rss="], { encoding: "utf8" }); + if (result.status !== 0) return []; + return result.stdout + .trim() + .split(/\r?\n/u) + .map(line => line.trim().split(/\s+/u).map(Number)) + .filter(parts => parts.length === 3 && parts.every(Number.isFinite)) + .map(parts => ({ pid: parts[0]!, ppid: parts[1]!, rssKiB: parts[2]! })); +} + +export interface AppExit { + code: number | null; + signal: string | null; +} + +/** + * The close request goes through the window manager (EWMH _NET_CLOSE_WINDOW), the same path a + * person's close button takes. xdotool's windowclose destroys the X window instead, which can end + * the process without ever running Tauri's close/drain handling and still look like a clean exit. + */ +export function windowManagerCloseArgs(windowId: string): string[] { + const id = Number(windowId); + if (!Number.isSafeInteger(id) || id <= 0) throw new Error(`invalid X11 window id: ${windowId}`); + return ["-i", "-c", `0x${id.toString(16)}`]; +} + +/** A graceful close exits 0 on its own; a signal or a nonzero code is a crash, not a drain. */ +export function assertCleanExit(exit: AppExit | undefined): AppExit { + if (!exit) throw new Error("desktop app did not exit after the close request"); + if (exit.signal !== null || exit.code !== 0) { + throw new Error(`desktop app exited with code ${exit.code ?? "none"} and signal ${exit.signal ?? "none"} instead of a clean close`); + } + return exit; +} + +export function processTreeRssKiB(rootPid: number, rows = processRows()): number { + const selected = new Set([rootPid]); + let changed = true; + while (changed) { + changed = false; + for (const row of rows) { + if (selected.has(row.ppid) && !selected.has(row.pid)) { + selected.add(row.pid); + changed = true; + } + } + } + return rows.filter(row => selected.has(row.pid)).reduce((sum, row) => sum + row.rssKiB, 0); +} + +function xdotoolWindow(): string | undefined { + // WebKit exposes an auxiliary `opencodex-desktop` X11 window before the titled top-level + // `OpenCodex` window. A loose match selected that helper and `windowclose` merely destroyed the + // web process surface, never exercising Tauri's close/drain path. + const result = spawnSync( + "xdotool", + ["search", "--onlyvisible", "--name", "^OpenCodex$"], + { encoding: "utf8" }, + ); + if (result.status !== 0) return undefined; + return result.stdout.trim().split(/\r?\n/u).find(Boolean); +} + +async function health(record: RuntimeRecord): Promise { + try { + const response = await fetch(`http://127.0.0.1:${record.port}/healthz`, { + signal: AbortSignal.timeout(1_000), + cache: "no-store", + }); + const body = await response.json(); + return typeof body === "object" && body !== null + ? { status: response.status, body: body as Record } + : undefined; + } catch { + return undefined; + } +} + +function tail(path: string): string[] { + try { + return readFileSync(path, "utf8").split(/\r?\n/u).filter(Boolean).slice(-LOG_TAIL_LINES); + } catch { + return []; + } +} + +async function stopGroup(child: ChildProcess): Promise { + if (!child.pid || !processAlive(child.pid)) return; + try { + process.kill(-child.pid, "SIGTERM"); + } catch { + child.kill("SIGTERM"); + } + try { + await waitFor(() => processAlive(child.pid) ? undefined : true, 5_000); + return; + } catch { + // Escalate only inside the detached process group this test created. + } + try { + process.kill(-child.pid, "SIGKILL"); + } catch { + child.kill("SIGKILL"); + } +} + +async function runFormat( + format: LinuxBundleFormat, + artifact: string, + version: string, + root: string, +): Promise { + const started = Date.now(); + const directory = join(root, format); + const extracted = join(directory, "payload"); + const home = join(directory, "home"); + const opencodexHome = join(home, ".opencodex"); + const codexHome = join(home, ".codex"); + const configHome = join(home, ".config"); + const cacheHome = join(home, ".cache"); + const dataHome = join(home, ".local", "share"); + for (const path of [home, opencodexHome, codexHome, configHome, cacheHome, dataHome]) { + mkdirSync(path, { recursive: true, mode: 0o700 }); + } + const stdoutPath = join(directory, "stdout.log"); + const stderrPath = join(directory, "stderr.log"); + mkdirSync(directory, { recursive: true }); + const stdout = openSync(stdoutPath, "w", 0o600); + const stderr = openSync(stderrPath, "w", 0o600); + let child: ChildProcess | undefined; + let runtimePid: number | undefined; + let configuredPort: number | undefined; + let reservedPort: ReservedLoopbackPort | undefined; + try { + const executable = extractedExecutable(format, artifact, extracted); + reservedPort = await reserveLoopbackPort(); + configuredPort = reservedPort.port; + writeFileSync( + join(opencodexHome, "config.json"), + `${JSON.stringify({ port: configuredPort }, null, 2)}\n`, + { mode: 0o600 }, + ); + const env: NodeJS.ProcessEnv = { + ...process.env, + HOME: home, + USERPROFILE: home, + XDG_CONFIG_HOME: configHome, + XDG_CACHE_HOME: cacheHome, + XDG_DATA_HOME: dataHome, + OPENCODEX_HOME: opencodexHome, + CODEX_HOME: codexHome, + NO_PROXY: "127.0.0.1,localhost", + no_proxy: "127.0.0.1,localhost", + WEBKIT_DISABLE_COMPOSITING_MODE: "1", + }; + // Hold the listener while preparing the isolated home so no unrelated process can claim the + // selected port. Release it only at the spawn boundary; the packaged runtime can then bind it. + await reservedPort.release(); + reservedPort = undefined; + child = spawn(executable, [], { + cwd: dirname(executable), + env, + detached: true, + stdio: ["ignore", stdout, stderr], + }); + if (!child.pid) throw new Error("desktop app did not report a pid"); + const appPid = child.pid; + let appExit: AppExit | undefined; + child.once("exit", (code, signal) => { + appExit = { code, signal }; + }); + const windowId = await waitFor(xdotoolWindow, READY_DEADLINE_MS); + const recordPath = join(opencodexHome, "runtime-port.json"); + const record = assertRuntimeRecordPort( + await waitFor(() => readRuntimeRecord(recordPath), READY_DEADLINE_MS), + configuredPort, + ); + runtimePid = record.pid; + let lastHealth: HealthObservation | undefined; + let ready: Record; + try { + ready = await waitFor(async () => { + const observed = await health(record); + if (!observed) return undefined; + lastHealth = observed; + const body = observed.body; + return observed.status >= 200 && observed.status < 300 + && body.service === "opencodex" + && body.pid === record.pid + && body.port === record.port + && body.version === version + ? body + : undefined; + }, READY_DEADLINE_MS); + } catch { + const observed = lastHealth + ? `status ${lastHealth.status}, body ${JSON.stringify(lastHealth.body)}` + : "no readable /healthz response"; + throw new Error(`packaged runtime health identity did not become ready (${observed})`); + } + const readyMs = Date.now() - started; + const rssKiB = processTreeRssKiB(appPid); + + command("wmctrl", windowManagerCloseArgs(windowId)); + await waitFor( + () => appExit && !processAlive(runtimePid) ? true : undefined, + EXIT_DEADLINE_MS, + ); + const exit = assertCleanExit(appExit); + return { + format, + artifact: basename(artifact), + ok: true, + durationMs: Date.now() - started, + windowId, + appPid, + appExitCode: exit.code, + appExitSignal: exit.signal, + runtimePid, + runtimeVersion: typeof ready.version === "string" ? ready.version : undefined, + configuredPort, + readyMs, + processTreeRssKiB: rssKiB, + stdoutTail: tail(stdoutPath), + stderrTail: tail(stderrPath), + }; + } catch (error) { + return { + format, + artifact: basename(artifact), + ok: false, + durationMs: Date.now() - started, + ...(child?.pid ? { appPid: child.pid } : {}), + ...(runtimePid ? { runtimePid } : {}), + ...(configuredPort ? { configuredPort } : {}), + error: error instanceof Error ? error.message : String(error), + stdoutTail: tail(stdoutPath), + stderrTail: tail(stderrPath), + }; + } finally { + await reservedPort?.release(); + if (child) await stopGroup(child); + closeSync(stdout); + closeSync(stderr); + } +} + +export async function runAcceptance(options: LinuxE2eOptions): Promise { + if (process.platform !== "linux") throw new Error("Linux packaged E2E runs only on Linux"); + for (const dependency of ["dpkg-deb", "ps", "wmctrl", "xdotool"]) { + const probe = spawnSync("sh", ["-c", `command -v ${dependency}`]); + if (probe.status !== 0) throw new Error(`missing required command: ${dependency}`); + } + if (!process.env.DISPLAY) throw new Error("DISPLAY is required; run under Xvfb"); + + const artifacts = locateArtifacts(options.bundleRoot); + const root = mkdtempSync(join(tmpdir(), "opencodex-linux-e2e-")); + const startedAt = new Date().toISOString(); + let formats: FormatReport[] = []; + try { + formats = [ + await runFormat("appimage", artifacts.appimage, options.version, root), + await runFormat("deb", artifacts.deb, options.version, root), + ]; + } finally { + const report: AcceptanceReport = { + schema: "opencodex-linux-packaged-e2e/1", + version: options.version, + startedAt, + finishedAt: new Date().toISOString(), + ok: formats.length === 2 && formats.every(format => format.ok), + formats, + }; + mkdirSync(dirname(options.reportPath), { recursive: true }); + writeFileSync(options.reportPath, `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 }); + rmSync(root, { recursive: true, force: true }); + } + return JSON.parse(readFileSync(options.reportPath, "utf8")) as AcceptanceReport; +} + +async function main(): Promise { + const options = parseArguments(process.argv.slice(2)); + const report = await runAcceptance(options); + for (const format of report.formats) { + console.log(`${format.ok ? "PASS" : "FAIL"} ${format.format}: ${format.error ?? `${format.readyMs}ms ready, ${format.processTreeRssKiB} KiB RSS`}`); + } + process.exitCode = report.ok ? 0 : 1; +} + +if (import.meta.main) { + main().catch(error => { + console.error(error instanceof Error ? error.message : String(error)); + process.exitCode = 1; + }); +} diff --git a/desktop/scripts/prepare-sidecar.ts b/desktop/scripts/prepare-sidecar.ts index 502127870d..20f7eec61f 100644 --- a/desktop/scripts/prepare-sidecar.ts +++ b/desktop/scripts/prepare-sidecar.ts @@ -1,5 +1,6 @@ -import { copyFileSync, cpSync, existsSync, mkdirSync } from "node:fs"; +import { copyFileSync, cpSync, mkdirSync } from "node:fs"; import { join, resolve } from "node:path"; +import { adHocSignSidecar, shouldAdHocSignSidecar } from "./sidecar-signing"; const targetByTriple: Record = { "aarch64-apple-darwin": "bun-darwin-arm64", @@ -37,23 +38,28 @@ if (!triple || !targetByTriple[triple]) { const target = targetByTriple[triple]; const source = join(repoRoot, "dist", "standalone", target); const executable = join(source, target.startsWith("bun-windows-") ? "ocx.exe" : "ocx"); -if (!existsSync(executable)) { - const result = Bun.spawnSync([ - process.execPath, - "run", - "build:standalone", - "--target", - target, - ], { cwd: repoRoot, stdout: "inherit", stderr: "inherit" }); - if (result.exitCode !== 0) process.exit(result.exitCode); -} +// Preparation must consume this checkout, never a stale executable/addon pair left in dist. +const result = Bun.spawnSync([ + process.execPath, + "run", + "build:standalone", + "--target", + target, +], { cwd: repoRoot, stdout: "inherit", stderr: "inherit" }); +if (result.exitCode !== 0) process.exit(result.exitCode); const desktopRoot = resolve(import.meta.dir, ".."); const binaries = join(desktopRoot, "src-tauri", "binaries"); const resources = join(desktopRoot, "src-tauri", "resources", "gui", "dist"); +const keyringResources = join(desktopRoot, "src-tauri", "resources", "keyring"); mkdirSync(binaries, { recursive: true }); mkdirSync(resources, { recursive: true }); const destination = join(binaries, `ocx-${triple}${target.startsWith("bun-windows-") ? ".exe" : ""}`); copyFileSync(executable, destination); +if (shouldAdHocSignSidecar(process.platform, target)) { + const signed = adHocSignSidecar(destination); + if (signed !== 0) process.exit(signed); +} +cpSync(join(source, "keyring"), keyringResources, { recursive: true }); cpSync(join(repoRoot, "gui", "dist"), resources, { recursive: true }); console.log(`Prepared ${destination}`); diff --git a/desktop/scripts/sidecar-signing.ts b/desktop/scripts/sidecar-signing.ts new file mode 100644 index 0000000000..a41642b306 --- /dev/null +++ b/desktop/scripts/sidecar-signing.ts @@ -0,0 +1,31 @@ +// Ad-hoc signing of the prepared desktop sidecar on macOS. +// +// Bun's linker-signed standalone output is killed by macOS page validation +// (CODESIGNING "Invalid Page"), so the copied sidecar is resealed with an +// ad-hoc signature before Tauri bundles it. Only a macOS host preparing a +// bun-darwin-* target signs: a Mac cross-preparing a Linux or Windows sidecar +// must never run codesign on that file. Release builds re-sign the bundled +// binary with Developer ID afterwards; this step only has to leave a runnable +// input. + +export const CODESIGN_PATH = "/usr/bin/codesign"; + +export function shouldAdHocSignSidecar(hostPlatform: string, bunTarget: string): boolean { + return hostPlatform === "darwin" && bunTarget.startsWith("bun-darwin-"); +} + +export function adHocSignArgv(destination: string): string[] { + return [CODESIGN_PATH, "-s", "-", "-f", destination]; +} + +export type SidecarSignSpawn = (argv: string[]) => { exitCode: number | null }; + +const inheritSpawn: SidecarSignSpawn = (argv) => + Bun.spawnSync(argv, { stdout: "inherit", stderr: "inherit" }); + +/** Returns 0 on success, otherwise the nonzero exit code the caller should exit with. */ +export function adHocSignSidecar(destination: string, spawn: SidecarSignSpawn = inheritSpawn): number { + const result = spawn(adHocSignArgv(destination)); + if (result.exitCode === 0) return 0; + return result.exitCode ?? 1; +} diff --git a/desktop/scripts/updater-manifest.ts b/desktop/scripts/updater-manifest.ts index e67459d848..e3fae9a372 100644 --- a/desktop/scripts/updater-manifest.ts +++ b/desktop/scripts/updater-manifest.ts @@ -27,11 +27,17 @@ export interface UpdaterManifest { platforms: Record; } -const platformFiles: Record = { +export const platformFiles: Record = { "darwin-aarch64": "macos.app.tar.gz", "darwin-x86_64": "macos.app.tar.gz", "windows-x86_64": "windows-x64.msi", + // The AppImage is the plugin's default Linux target: it keeps the plain os-arch key so + // AppImage installs from releases before the deb target existed keep resolving updates. "linux-x86_64": "linux-x86_64.AppImage", + // A deb install cannot apply an AppImage payload (the updater validates the downloaded + // bytes as a real .deb before installing), so it must resolve a distinct key. The shell + // selects this key from the bundle type embedded at packaging time; see updater.rs. + "linux-x86_64-deb": "linux-amd64.deb", }; export function buildUpdaterManifest(options: UpdaterManifestOptions): UpdaterManifest { diff --git a/desktop/scripts/verify-linux-sidecar.sh b/desktop/scripts/verify-linux-sidecar.sh new file mode 100644 index 0000000000..c502bb5c8b --- /dev/null +++ b/desktop/scripts/verify-linux-sidecar.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash +# Run only on a Linux packaging runner, against the completed AppImage. +# Usage: verify-linux-sidecar.sh [appimage-bundle-dir] +# The release workflow builds each Linux format in its own Cargo target and stages the AppImage +# into an isolated read-only directory, which it passes here; a local build keeps the default. +set -euo pipefail +root="$(cd "$(dirname "$0")/../.." && pwd)" +bundle="${1:-$root/desktop/src-tauri/target/x86_64-unknown-linux-gnu/release/bundle/appimage}" +original="$root/desktop/src-tauri/binaries/ocx-x86_64-unknown-linux-gnu" +shopt -s nullglob +images=("$bundle"/*.AppImage) +if [ "${#images[@]}" -ne 1 ]; then + echo "Expected exactly one completed AppImage" >&2 + exit 1 +fi +scratch="$(mktemp -d)" +trap 'rm -rf "$scratch"' EXIT +cd "$scratch" +"${images[0]}" --appimage-extract > /dev/null +sidecar="$scratch/squashfs-root/usr/bin/ocx" +keyring="$scratch/squashfs-root/usr/lib/OpenCodex/keyring/keyring.linux-x64-gnu.node" +test ! -L "$sidecar" +test -f "$keyring" +cmp "$original" "$sidecar" +sha256sum "$original" "$sidecar" +mkdir "$scratch/home" +timeout 30s env OPENCODEX_HOME="$scratch/home" "$sidecar" --version +timeout 15s env HOME="$scratch/home" OPENCODEX_HOME="$scratch/home/opencodex" \ + "$sidecar" __keyring-load-check > "$scratch/keyring.json" +python3 - "$scratch/keyring.json" <<'PY' +import json, pathlib, sys +value = json.loads(pathlib.Path(sys.argv[1]).read_text()) +assert value == {"schema": "ocx-keyring-load/1", "available": True}, "Packaged keyring binding is unavailable" +PY diff --git a/desktop/scripts/verify-macos-runtime.sh b/desktop/scripts/verify-macos-runtime.sh new file mode 100755 index 0000000000..43f1a31375 --- /dev/null +++ b/desktop/scripts/verify-macos-runtime.sh @@ -0,0 +1,72 @@ +#!/usr/bin/env bash +set -euo pipefail + +app_input="${1:?usage: verify-macos-runtime.sh /path/to/OpenCodex.app}" +app="$(cd "$(dirname "$app_input")" && pwd)/$(basename "$app_input")" +[[ "$(uname -s)" == Darwin ]] || { echo 'macOS bundle verification requires macOS' >&2; exit 1; } +executable="$(/usr/libexec/PlistBuddy -c 'Print :CFBundleExecutable' "$app/Contents/Info.plist")" +[[ -n "$executable" && "$executable" != */* ]] || { echo 'Invalid app executable name' >&2; exit 1; } +scratch="$(mktemp -d "${TMPDIR:-/tmp}/opencodex-bundle-check.XXXXXX")" +cleanup() { + rm -rf "$scratch" +} +trap cleanup EXIT + +codesign --verify --strict --deep "$app" +verify_member() { + local role="$1" member="$2" + codesign --display --entitlements - --xml "$member" > "$scratch/$role.plist" 2> "$scratch/$role-entitlements.log" + codesign --display --verbose=4 "$member" > "$scratch/$role-signature.log" 2>&1 + python3 - "$role" "$scratch/$role.plist" "$scratch/$role-signature.log" <<'PY' +import pathlib, plistlib, re, sys +role, entitlements, signature = sys.argv[1:] +actual = plistlib.loads(pathlib.Path(entitlements).read_bytes()) +expected = {"com.apple.security.app-sandbox": True} if role == "widget" else {"com.apple.security.cs.allow-jit": True} +if actual != expected: + raise SystemExit(f"Unexpected {role} entitlement dictionary") +text = pathlib.Path(signature).read_text() +if not re.search(r"flags=.*\bruntime\b", text): + raise SystemExit(f"Missing hardened runtime on {role}") +PY +} +verify_member app "$app" +verify_member ocx "$app/Contents/MacOS/ocx" +verify_member widget "$app/Contents/PlugIns/OpenCodexWidget.appex" +# Release stripping removes the nlist symbol table; inspect the loader's bindings. +xcrun llvm-objdump --macho --dyld-info "$app/Contents/MacOS/$executable" > "$scratch/native-symbols.txt" +grep -q NSGlassEffectView "$scratch/native-symbols.txt" || { echo 'Native Liquid Glass code is absent' >&2; exit 1; } +mkdir "$scratch/home" +OPENCODEX_HOME="$scratch/home" "$app/Contents/MacOS/ocx" resolve --json > "$scratch/resolve.json" +python3 - "$scratch/resolve.json" <<'PY' +import json, pathlib, sys +value = json.loads(pathlib.Path(sys.argv[1]).read_text()) +assert value.get("schema") == "ocx-resolve/1", "Unexpected resolve schema" +assert value.get("liveness", {}).get("status") in ("live", "absent-proven"), "Unusable resolve result" +PY + +# Reproduce the packaged-keyring boundary from an unrelated cwd. This is deliberately load-only: +# an ad-hoc CI identity can trigger a Keychain consent dialog, while issue #6139 is module resolution. +mkdir -p "$scratch/home" "$scratch/work" +python3 - "$app/Contents/MacOS/ocx" "$scratch/work" "$scratch/home" "$scratch/keyring.json" <<'PY' +import os, pathlib, subprocess, sys +ocx, work, home, output = sys.argv[1:] +env = os.environ.copy() +env.update(HOME=home, OPENCODEX_HOME=str(pathlib.Path(home) / ".opencodex")) +try: + with open(output, "wb") as stdout: + subprocess.run( + [ocx, "__keyring-load-check"], cwd=work, env=env, stdout=stdout, + stderr=subprocess.PIPE, check=True, timeout=15, + ) +except subprocess.TimeoutExpired as error: + raise SystemExit("Packaged keyring load probe timed out") from error +except subprocess.CalledProcessError as error: + sys.stderr.buffer.write((error.stderr or b"")[-4096:]) + raise SystemExit(f"Packaged keyring load probe exited {error.returncode}") from error +PY +python3 - "$scratch/keyring.json" <<'PY' +import json, pathlib, sys +value = json.loads(pathlib.Path(sys.argv[1]).read_text()) +assert value == {"schema": "ocx-keyring-load/1", "available": True}, "Packaged keyring binding is unavailable" +PY +printf '%s\n' 'PASS: macOS signatures, entitlements, hardened runtime, Liquid Glass, bundled CLI resolve and packaged keyring' diff --git a/desktop/scripts/verify-release-assets.ts b/desktop/scripts/verify-release-assets.ts new file mode 100644 index 0000000000..0c3175791b --- /dev/null +++ b/desktop/scripts/verify-release-assets.ts @@ -0,0 +1,369 @@ +/** + * Pre-publication release asset verification. + * + * Everything a release will publish is checked here, in the verify-release job, + * before any publication step may run: the expected platform file set derived from + * the workflow's own packaging matrices and the producer scripts' tables, every + * recorded checksum against the bytes on disk, every updater signature + * cryptographically against the pinned minisign public key, and the updater + * manifest parsed back against the files it names. The result is a + * machine-readable receipt; attach-release requires the receipt to name the same + * version and commit before it uploads anything, so publication can only ever + * consume the verified bundle. + */ +import { createHash, createPublicKey, verify as ed25519Verify, type KeyObject } from "node:crypto"; +import { existsSync, mkdirSync, readFileSync, readdirSync, renameSync, writeFileSync } from "node:fs"; +import { dirname, join, resolve } from "node:path"; +import { + standaloneArchiveName, + standaloneTargets as sharedStandaloneTargets, +} from "../../scripts/standalone-targets"; +import { bundlesByTarget } from "./collect-release-assets"; +import { platformFiles, writeUpdaterManifest, type UpdaterManifest } from "./updater-manifest"; + +export interface VerifyReleaseAssetsOptions { + version: string; + dir: string; + repo: string; + sha: string; + repoRoot?: string; + manifestOut?: string; + receiptOut?: string; + requireSignatures?: boolean; +} + +export interface ReleaseVerificationReceipt { + version: string; + repo: string; + sha: string; + expectedFiles: number; + checksumsVerified: number; + signaturesVerified: number; + manifestPlatforms: string[]; +} + +/** + * The expected file set, derived from the producer tables rather than restated. + * Signatures are required only for the assets the updater actually signs — the + * unique suffixes in platformFiles — because the DMG and the deb are not updater + * targets and are never signed. + */ +export function expectedReleaseAssets(options: { + version: string; + desktopTargets: string[]; + requireSignatures?: boolean; +}): string[] { + const expected: string[] = []; + for (const target of sharedStandaloneTargets) { + const archive = standaloneArchiveName(options.version, target); + expected.push(archive, `${archive}.sha256`); + } + const updaterSuffixes = new Set(Object.values(platformFiles)); + for (const target of options.desktopTargets) { + const bundles = bundlesByTarget[target]; + if (!bundles) throw new Error(`Unsupported desktop target in release matrix: ${target}`); + for (const bundle of bundles) { + const asset = `OpenCodex-${options.version}-${bundle.name}`; + expected.push(asset, `${asset}.sha256`); + if (options.requireSignatures && updaterSuffixes.has(bundle.name)) { + expected.push(`${asset}.sig`); + } + } + } + return expected; +} + +/** The packaging matrices of the release workflow itself — the source of truth for the set. */ +export function releaseMatrixTargets(workflowText: string): { + standaloneTargets: string[]; + desktopTargets: string[]; +} { + const workflow = Bun.YAML.parse(workflowText) as { + jobs?: Record } } }>; + }; + const read = (job: string): string[] => + (workflow.jobs?.[job]?.strategy?.matrix?.include ?? []) + .map(entry => entry.target) + .filter((target): target is string => typeof target === "string"); + const standaloneTargets = read("package-standalone"); + const desktopTargets = read("package-desktop"); + if (standaloneTargets.length === 0 || desktopTargets.length === 0) { + throw new Error("release.yml packaging matrices are empty or unreadable"); + } + return { standaloneTargets, desktopTargets }; +} + +/** + * Every recorded checksum against the bytes on disk, in exactly the producers' + * format (64 hex, a space, text/binary marker, bare name, newline). The recorded name + * must equal the checksum file's own name minus the suffix: a foo.sha256 naming + * bar would leave foo's bytes unchecked while bar's are checked twice. + */ +export function verifyChecksums(dir: string): number { + const checksumFiles = readdirSync(dir).filter(name => name.endsWith(".sha256")).sort(); + if (checksumFiles.length === 0) throw new Error(`No .sha256 files found in ${dir}`); + for (const checksumFile of checksumFiles) { + const content = readFileSync(join(dir, checksumFile), "utf8"); + const match = /^([0-9a-f]{64}) [ *](\S+)\r?\n$/.exec(content); + if (!match) throw new Error(`Malformed checksum record in ${checksumFile}: ${JSON.stringify(content)}`); + const digest = match[1]!; + const recorded = match[2]!; + const own = checksumFile.slice(0, -".sha256".length); + if (recorded !== own) { + throw new Error(`Checksum ${checksumFile} records ${recorded}; it must record its own payload ${own}`); + } + const payload = join(dir, recorded); + if (!existsSync(payload)) throw new Error(`Checksum ${checksumFile} names ${recorded}, which is missing`); + const actual = createHash("sha256").update(readFileSync(payload)).digest("hex"); + if (actual !== digest) { + throw new Error(`Checksum mismatch for ${recorded}: recorded ${digest}, computed ${actual}`); + } + } + return checksumFiles.length; +} + +const ED25519_SPKI_PREFIX = Buffer.from("302a300506032b6570032100", "hex"); + +export interface MinisignPublicKey { + keyId: string; + publicKey: KeyObject; +} + +function decodeBase64(text: string, what: string, expectedBytes?: number): Buffer { + const payload = Buffer.from(text, "base64"); + // Buffer.from is intentionally permissive; release metadata must be canonical. + if (!text || payload.toString("base64") !== text + || (expectedBytes !== undefined && payload.length !== expectedBytes)) { + throw new Error(`Malformed ${what}: invalid base64 or decoded length`); + } + return payload; +} + +function decodeBox(text: string, what: string): string { + // Transport whitespace is harmless (the updater manifest also trims it). + // The encoded payload itself must still be canonical and valid UTF-8. + const payload = decodeBase64(text.trim(), what); + return new TextDecoder("utf-8", { fatal: true, ignoreBOM: true }).decode(payload); +} + +function boxLines(text: string): string[] { + // Accept minisign text with LF or CRLF and an optional terminal newline; + // signatures authenticate decoded bytes/comments, not transport line endings. + return text.replace(/\r\n/g, "\n").replace(/\n$/, "").split("\n"); +} + +/** minisign public key: base64 of algorithm ("Ed") || key id (8) || raw key (32). */ +export function parseMinisignPublicKey(text: string): MinisignPublicKey { + const lines = boxLines(text); + if (lines.length !== 2 || !lines[0]!.startsWith("untrusted comment: ")) { + throw new Error("Malformed minisign public key box"); + } + const payload = decodeBase64(lines[1]!, "minisign public key", 42); + const algorithm = payload.subarray(0, 2).toString("utf8"); + if (algorithm !== "Ed") { + throw new Error(`Unsupported minisign public key algorithm: ${JSON.stringify(algorithm)}`); + } + return { + keyId: payload.subarray(2, 10).toString("hex"), + publicKey: createPublicKey({ + key: Buffer.concat([ED25519_SPKI_PREFIX, payload.subarray(10, 42)]), + format: "der", + type: "spki", + }), + }; +} + +/** The updater public key pinned in the Tauri configuration. */ +export function loadUpdaterPublicKey(tauriConfPath: string): MinisignPublicKey { + const conf = JSON.parse(readFileSync(tauriConfPath, "utf8")) as { + plugins?: { updater?: { pubkey?: string } }; + }; + const pubkey = conf.plugins?.updater?.pubkey; + if (!pubkey) throw new Error(`No plugins.updater.pubkey in ${tauriConfPath}`); + return parseMinisignPublicKey(decodeBox(pubkey, "Tauri public key")); +} + +/** Tauri CLI 2.11.1 wraps a minisign 0.7.3 prehashed signature box in base64. */ +export function verifyUpdaterSignature(filePath: string, key: MinisignPublicKey): void { + const signaturePath = `${filePath}.sig`; + if (!existsSync(signaturePath)) throw new Error(`Missing signature: ${signaturePath}`); + const lines = boxLines(decodeBox(readFileSync(signaturePath, "utf8"), "Tauri signature")); + const trustedPrefix = "trusted comment: "; + if (lines.length !== 4 || !lines[0]!.startsWith("untrusted comment: ") + || !lines[2]!.startsWith(trustedPrefix)) { + throw new Error(`Malformed signature box in ${signaturePath}`); + } + const payload = decodeBase64(lines[1]!, "signature packet", 74); + const globalSignature = decodeBase64(lines[3]!, "comment signature", 64); + const algorithm = payload.subarray(0, 2).toString("utf8"); + if (algorithm !== "ED") { + throw new Error(`Unsupported signature algorithm in ${signaturePath}: ${JSON.stringify(algorithm)}`); + } + const keyId = payload.subarray(2, 10).toString("hex"); + if (keyId !== key.keyId) { + throw new Error(`Signature ${signaturePath} was made by key ${keyId}, not the pinned updater key ${key.keyId}`); + } + const signature = payload.subarray(10, 74); + // ED is ordinary Ed25519 over the BLAKE2b-512 digest, not Ed25519ph. + const digest = createHash("blake2b512").update(readFileSync(filePath)).digest(); + if (!ed25519Verify(null, digest, key.publicKey, signature)) { + throw new Error(`Signature verification failed for ${filePath}`); + } + // minisign signs the raw signature + trimmed trusted comment, without its + // prefix or line terminator. The original filename may differ after collection. + const trustedComment = lines[2]!.slice(trustedPrefix.length).trim(); + const message = Buffer.concat([signature, Buffer.from(trustedComment, "utf8")]); + if (!ed25519Verify(null, message, key.publicKey, globalSignature)) { + throw new Error(`Comment signature verification failed for ${filePath}`); + } +} + +function parseBackManifest(manifestPath: string, options: VerifyReleaseAssetsOptions): string[] { + const manifest = JSON.parse(readFileSync(manifestPath, "utf8")) as UpdaterManifest; + if (manifest.version !== options.version) { + throw new Error(`Manifest version ${manifest.version} != ${options.version}`); + } + const platforms = Object.keys(manifest.platforms).sort(); + const expectedPlatforms = Object.keys(platformFiles).sort(); + if (JSON.stringify(platforms) !== JSON.stringify(expectedPlatforms)) { + throw new Error( + `Manifest platforms (${platforms.join(", ")}) do not match the updater platform set (${expectedPlatforms.join(", ")})`, + ); + } + for (const [platform, entry] of Object.entries(manifest.platforms)) { + const base = `OpenCodex-${options.version}-${platformFiles[platform]}`; + const expectedUrl = `https://github.com/${options.repo}/releases/download/v${options.version}/${base}`; + if (entry.url !== expectedUrl) { + throw new Error(`Manifest entry ${platform} points at ${entry.url}, expected ${expectedUrl}`); + } + if (!existsSync(join(options.dir, base))) { + throw new Error(`Manifest entry ${platform} names ${base}, which is missing`); + } + // The manifest must carry exactly the signature that was just verified, + // not merely a nonempty string. + const sidecar = readFileSync(join(options.dir, `${base}.sig`), "utf8").trim(); + if (entry.signature !== sidecar) { + throw new Error(`Manifest entry ${platform} signature does not match ${base}.sig`); + } + } + return platforms; +} + +function atomicWrite(path: string, content: string): void { + mkdirSync(dirname(path), { recursive: true }); + const temporary = `${path}.${process.pid}.tmp`; + writeFileSync(temporary, content); + renameSync(temporary, path); +} + +export function verifyReleaseAssets(options: VerifyReleaseAssetsOptions): ReleaseVerificationReceipt { + const repoRoot = resolve(options.repoRoot ?? join(import.meta.dir, "../..")); + const dir = resolve(options.dir); + const { standaloneTargets, desktopTargets } = releaseMatrixTargets( + readFileSync(join(repoRoot, ".github", "workflows", "release.yml"), "utf8"), + ); + // The workflow matrix must describe exactly the shared target set the builder + // uses; a target added to one and not the other fails here, not at release time. + const workflowStandalone = [...standaloneTargets].sort(); + const sharedStandalone = [...sharedStandaloneTargets].sort(); + if (JSON.stringify(workflowStandalone) !== JSON.stringify(sharedStandalone)) { + throw new Error( + `release.yml package-standalone matrix (${workflowStandalone.join(", ")})` + + ` does not match scripts/standalone-targets.ts (${sharedStandalone.join(", ")})`, + ); + } + const expected = expectedReleaseAssets({ + version: options.version, + desktopTargets, + requireSignatures: options.requireSignatures, + }); + const missing = expected.filter(name => !existsSync(join(dir, name))); + if (missing.length > 0) { + throw new Error(`Missing expected release assets:\n${missing.join("\n")}`); + } + + const checksumsVerified = verifyChecksums(dir); + + const updaterKey = loadUpdaterPublicKey( + join(repoRoot, "desktop", "src-tauri", "tauri.conf.json"), + ); + // Every signature present is verified, required or not: a tampered signature in + // an unsigned dry-run bundle must fail, not be skipped. + let signaturesVerified = 0; + for (const name of readdirSync(dir).filter(candidate => candidate.endsWith(".sig")).sort()) { + const payload = join(dir, name.slice(0, -".sig".length)); + if (!existsSync(payload)) throw new Error(`Signature ${name} has no payload beside it`); + verifyUpdaterSignature(payload, updaterKey); + signaturesVerified += 1; + } + + let manifestPlatforms: string[] = []; + if (options.manifestOut) { + writeUpdaterManifest({ + version: options.version, + dir, + repo: options.repo, + out: options.manifestOut, + requireAll: options.requireSignatures, + }); + manifestPlatforms = parseBackManifest(options.manifestOut, options); + } + + // attach-release uploads dist/release/* verbatim, so anything unexpected here + // would be published unchecked. The bundle is exactly the expected set plus + // the manifest this run just generated. + const allowed = new Set(expected); + if (options.manifestOut) allowed.add(options.manifestOut.split(/[\\/]/).pop()!); + const extras = readdirSync(dir).filter(name => !allowed.has(name)); + if (extras.length > 0) { + throw new Error(`Unexpected files in the release bundle (refusing to publish them):\n${extras.join("\n")}`); + } + + const receipt: ReleaseVerificationReceipt = { + version: options.version, + repo: options.repo, + sha: options.sha, + expectedFiles: expected.length, + checksumsVerified, + signaturesVerified, + manifestPlatforms, + }; + if (options.receiptOut) { + atomicWrite(options.receiptOut, `${JSON.stringify(receipt, null, 2)}\n`); + } + return receipt; +} + +function argument(name: string): string | undefined { + const index = Bun.argv.indexOf(name); + return index < 0 ? undefined : Bun.argv[index + 1]; +} + +if (import.meta.main) { + const version = argument("--version"); + const dir = argument("--dir"); + const repo = argument("--repo"); + const sha = argument("--sha"); + if (!version || !dir || !repo || !sha) { + throw new Error( + "Usage: verify-release-assets.ts --version --dir --repo --sha " + + " [--manifest-out ] [--require-signatures] [--receipt-out ]", + ); + } + const receipt = verifyReleaseAssets({ + version, + dir, + repo, + sha, + manifestOut: argument("--manifest-out"), + receiptOut: argument("--receipt-out"), + requireSignatures: Bun.argv.includes("--require-signatures"), + }); + console.log( + `Verified ${receipt.expectedFiles} expected files, ${receipt.checksumsVerified} checksums,` + + ` ${receipt.signaturesVerified} signatures` + + (receipt.manifestPlatforms.length > 0 + ? `, manifest platforms: ${receipt.manifestPlatforms.join(", ")}` + : ""), + ); +} diff --git a/desktop/scripts/windows-installer-config.ts b/desktop/scripts/windows-installer-config.ts new file mode 100644 index 0000000000..0a5f64c2ce --- /dev/null +++ b/desktop/scripts/windows-installer-config.ts @@ -0,0 +1,28 @@ +import { writeFileSync } from "node:fs"; + +/** MSI cannot express SemVer prerelease precedence. Keep public application and + * updater versions intact and override only WiX ProductVersion with the core. + * The pinned Tauri template permits equal-core replacement; manual MSI installs + * therefore do not prevent same-core channel downgrades. */ +export function windowsInstallerVersion(version: string): string { + if (version.length > 128) throw new Error("Public version exceeds the installer metadata bound"); + const match = /^(0|[1-9]\d*)\.(0|[1-9]\d*)\.(0|[1-9]\d*)(?:-([0-9A-Za-z-]+(?:\.[0-9A-Za-z-]+)*))?(?:\+[0-9A-Za-z-]+(?:\.[0-9A-Za-z-]+)*)?$/.exec(version); + if (!match || match[4]?.split(".").some(part => /^\d+$/.test(part) && part.length > 1 && part[0] === "0")) { + throw new Error("A valid public SemVer is required for the Windows installer"); + } + const parts = [match[1]!, match[2]!, match[3]!].map(Number); + if (parts.some((part, index) => !Number.isSafeInteger(part) || part > (index < 2 ? 255 : 65_535))) { + throw new Error("Windows installer version exceeds MSI numeric limits"); + } + return parts.join("."); +} + +export function windowsInstallerConfig(version: string) { + return { bundle: { windows: { wix: { version: windowsInstallerVersion(version) } } } }; +} + +if (import.meta.main) { + const [version, output] = process.argv.slice(2); + if (!version || !output) throw new Error("Usage: windows-installer-config "); + writeFileSync(output, `${JSON.stringify(windowsInstallerConfig(version))}\n`, { mode: 0o600 }); +} diff --git a/desktop/src-tauri/Cargo.lock b/desktop/src-tauri/Cargo.lock index 95bc1a674a..42f4b24ce4 100644 --- a/desktop/src-tauri/Cargo.lock +++ b/desktop/src-tauri/Cargo.lock @@ -1,6 +1,6 @@ # This file is automatically @generated by Cargo. # It is not intended for manual editing. -version = 3 +version = 4 [[package]] name = "addr2line" @@ -363,6 +363,15 @@ dependencies = [ "alloc-stdlib", ] +[[package]] +name = "bs58" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bf88ba1141d185c399bee5288d850d63b8369520c1eafc32a0430b5b6c287bf4" +dependencies = [ + "tinyvec", +] + [[package]] name = "bumpalo" version = "3.20.3" @@ -704,9 +713,9 @@ checksum = "52560adf09603e58c9a7ee1fe1dcb95a16927b17c127f0ac02d6e768a0e25bc1" [[package]] name = "darling" -version = "0.20.11" +version = "0.24.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc7f46116c46ff9ab3eb1597a45688b6715c6e628b5c133e288e709a29bcb4ee" +checksum = "ed17f5901b6630b993ca003def43f2f8ef4014fc13b047b57aad617ff32bc2ec" dependencies = [ "darling_core", "darling_macro", @@ -714,27 +723,26 @@ dependencies = [ [[package]] name = "darling_core" -version = "0.20.11" +version = "0.24.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d00b9596d185e565c2207a0b01f8bd1a135483d02d9b7b0a54b11da8d53412e" +checksum = "6837e2cf7485aaae18f86181d2f0e9a7ed297a025e220aeabf63fdebd3a2ddff" dependencies = [ - "fnv", "ident_case", "proc-macro2", "quote", "strsim", - "syn 2.0.119", + "syn 3.0.6", ] [[package]] name = "darling_macro" -version = "0.20.11" +version = "0.24.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc34b93ccb385b40dc71c6fceac4b2ad23662c7eeb248cf10d529b7e055b6ead" +checksum = "2ac7135c3ef02b2f7833bbeb1be5ba7f966dcde8a87c6b87f65a778d71a02785" dependencies = [ "darling_core", "quote", - "syn 2.0.119", + "syn 3.0.6", ] [[package]] @@ -748,14 +756,44 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "defmt" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e2953bfe4f93bbd20cc71198842756f77d161884c99ebbabc41d80231ded88d1" +dependencies = [ + "bitflags 1.3.2", + "defmt-macros", +] + +[[package]] +name = "defmt-macros" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bad9c72e7ca2137e0dc3813245a0d282fd6daad32fd800af018306a9169b5fe8" +dependencies = [ + "defmt-parser", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "defmt-parser" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" +dependencies = [ + "thiserror 2.0.20", +] + [[package]] name = "deranged" -version = "0.5.3" +version = "0.5.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d630bccd429a5bb5a64b5e94f693bfc48c9f8566418fda4c494cc94f911f87cc" +checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" dependencies = [ - "powerfmt", - "serde", + "serde_core", ] [[package]] @@ -798,6 +836,7 @@ checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" dependencies = [ "block-buffer", "crypto-common", + "subtle", ] [[package]] @@ -1583,6 +1622,15 @@ version = "0.4.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70" +[[package]] +name = "hmac" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6c49c37c09c17a53d937dfbb742eb3a961d65a994e6bcdcf37e7399d0cc8ab5e" +dependencies = [ + "digest", +] + [[package]] name = "html5ever" version = "0.38.0" @@ -1867,6 +1915,8 @@ checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" dependencies = [ "equivalent", "hashbrown 0.17.1", + "serde", + "serde_core", ] [[package]] @@ -1932,6 +1982,60 @@ dependencies = [ "system-deps", ] +[[package]] +name = "jiff" +version = "0.2.37" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ab1baf72f08796de0260609515130699b890ac25f30e610ad894bc5856cafdb" +dependencies = [ + "defmt", + "jiff-core", + "jiff-static", + "jiff-tzdb-platform", + "log", + "portable-atomic", + "portable-atomic-util", + "serde_core", + "windows-link 0.2.1", +] + +[[package]] +name = "jiff-core" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5e52fe76043ccecc9005d2305ebaadf7d7fc0cc89ca6baa10a94d6bc68c7128c" +dependencies = [ + "defmt", + "log", +] + +[[package]] +name = "jiff-static" +version = "0.2.37" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "378268a1116ad67ae6228701118ac9f491d78fda38a40a1f1a9e1348de6f7212" +dependencies = [ + "jiff-core", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "jiff-tzdb" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "142bd39932ad231f10513df9ab62661fead8719872150b7ad02a2df79f4e141e" + +[[package]] +name = "jiff-tzdb-platform" +version = "0.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "875a5a69ac2bab1a891711cf5eccbec1ce0341ea805560dcd90b7a2e925132e8" +dependencies = [ + "jiff-tzdb", +] + [[package]] name = "jni" version = "0.21.1" @@ -2270,9 +2374,9 @@ checksum = "650eef8c711430f1a879fdd01d4745a7deea475becfb90269c06775983bbf086" [[package]] name = "num-conv" -version = "0.1.0" +version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "51d515d32fb182ee37cda2ccdcb92950d6a3c2893aa280e540671c2cd0f3b1d9" +checksum = "521739c6d2bac4aa25192232afe6841231376b2b26d4d9fae5ecf8ca5772e441" [[package]] name = "num-traits" @@ -2541,11 +2645,15 @@ dependencies = [ [[package]] name = "opencodex-desktop" -version = "2.61.0" +version = "2.76.0" dependencies = [ + "base64 0.22.1", + "dbus", + "hmac", "reqwest 0.12.24", "serde", "serde_json", + "sha2", "tauri", "tauri-build", "tauri-plugin-autostart", @@ -2554,6 +2662,7 @@ dependencies = [ "tauri-plugin-shell", "tauri-plugin-single-instance", "tauri-plugin-updater", + "tauri-utils", "tokio", "uuid", ] @@ -2787,6 +2896,21 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "portable-atomic" +version = "1.15.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" + +[[package]] +name = "portable-atomic-util" +version = "0.2.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "10ab3eb7f3becc3a1cbc4f2c6f20267996cfc1a6467a873763411b136a122715" +dependencies = [ + "portable-atomic", +] + [[package]] name = "potential_utf" version = "0.1.6" @@ -3024,6 +3148,26 @@ dependencies = [ "thiserror 2.0.20", ] +[[package]] +name = "ref-cast" +version = "1.0.27" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7e440fb4e4b4147295338efb76001ab9e4efc0e5839df2c47fc5ac2381d365c3" +dependencies = [ + "ref-cast-impl", +] + +[[package]] +name = "ref-cast-impl" +version = "1.0.27" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92ecd8964f8453721699a1ed72037b0db49ce2f5a5138486ee89bed6f67cdf3a" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + [[package]] name = "regex" version = "1.13.1" @@ -3247,6 +3391,30 @@ dependencies = [ "uuid", ] +[[package]] +name = "schemars" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4cd191f9397d57d581cddd31014772520aa448f65ef991055d7f61582c65165f" +dependencies = [ + "dyn-clone", + "ref-cast", + "serde", + "serde_json", +] + +[[package]] +name = "schemars" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "687274d293b6cdc6e73e0fee520bf2049650090d7164f87672d212a3c530cf4a" +dependencies = [ + "dyn-clone", + "ref-cast", + "serde", + "serde_json", +] + [[package]] name = "schemars_derive" version = "0.8.22" @@ -3295,10 +3463,11 @@ dependencies = [ [[package]] name = "serde" -version = "1.0.219" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5f0e2c6ed6606019b4e29e69dbaba95b11854410e5347d525002456dbbb786b6" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" dependencies = [ + "serde_core", "serde_derive", ] @@ -3313,15 +3482,24 @@ dependencies = [ "typeid", ] +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + [[package]] name = "serde_derive" -version = "1.0.219" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5b0276cf7f2c73365f7157c8123c21cd9a50fbbd844757af28ca1f5925fc2a00" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.6", ] [[package]] @@ -3337,14 +3515,15 @@ dependencies = [ [[package]] name = "serde_json" -version = "1.0.140" +version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "20068b6e96dc6c9bd23e01df8827e6c7e1f2fddd43c21810382803c136b99373" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ "itoa", "memchr", - "ryu", "serde", + "serde_core", + "zmij", ] [[package]] @@ -3390,15 +3569,20 @@ dependencies = [ [[package]] name = "serde_with" -version = "3.1.0" +version = "3.23.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "21e47d95bc83ed33b2ecf84f4187ad1ab9685d18ff28db000c99deac8ce180e3" +checksum = "935177bb8c0cd8ca1a4e6d1a2ac8988bea69cab4f9d3a31311e012ad27868ea4" dependencies = [ - "base64 0.21.7", + "base64 0.23.1", + "bs58", "chrono", "hex", "indexmap 1.9.3", - "serde", + "indexmap 2.14.2", + "jiff", + "schemars 0.9.0", + "schemars 1.2.2", + "serde_core", "serde_json", "serde_with_macros", "time", @@ -3406,14 +3590,14 @@ dependencies = [ [[package]] name = "serde_with_macros" -version = "3.1.0" +version = "3.23.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ea3cee93715c2e266b9338b7544da68a9f24e227722ba482bd1c024367c77c65" +checksum = "1d607aa01a3cb0ad757d6fd216136910db3c97b102fe686585689615a02dbcdc" dependencies = [ "darling", "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.6", ] [[package]] @@ -3854,7 +4038,7 @@ dependencies = [ "glob", "heck 0.5.0", "json-patch", - "schemars", + "schemars 0.8.22", "semver", "serde", "serde_json", @@ -3913,7 +4097,7 @@ dependencies = [ "anyhow", "glob", "plist", - "schemars", + "schemars 0.8.22", "serde", "serde_json", "tauri-utils", @@ -3945,7 +4129,7 @@ dependencies = [ "objc2-app-kit", "objc2-foundation", "open", - "schemars", + "schemars 0.8.22", "serde", "serde_json", "tauri", @@ -3968,16 +4152,16 @@ dependencies = [ [[package]] name = "tauri-plugin-shell" -version = "2.2.0" +version = "2.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb2c50a63e60fb8925956cc5b7569f4b750ac197a4d39f13b8dd46ea8e2bad79" +checksum = "69d5eb3368b959937ad2aeaf6ef9a8f5d11e01ffe03629d3530707bbcb27ff5d" dependencies = [ "encoding_rs", "log", "open", "os_pipe", "regex", - "schemars", + "schemars 0.8.22", "serde", "serde_json", "shared_child", @@ -4108,7 +4292,7 @@ dependencies = [ "proc-macro2", "quote", "regex", - "schemars", + "schemars 0.8.22", "semver", "serde", "serde-untagged", @@ -4198,30 +4382,29 @@ dependencies = [ [[package]] name = "time" -version = "0.3.44" +version = "0.3.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "91e7d9e3bb61134e77bde20dd4825b97c010155709965fedf0f49bb138e52a9d" +checksum = "cdb87b95ec50ddfa440816d227a17b2ccbdda963a316a727fda0fc4334f7d134" dependencies = [ "deranged", - "itoa", "num-conv", "powerfmt", - "serde", + "serde_core", "time-core", "time-macros", ] [[package]] name = "time-core" -version = "0.1.6" +version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "40868e7c1d2f0b8d73e4a8c7f0ff63af4f6d19be117e90bd73eb1d62cf831c6b" +checksum = "9e1c906769ad99c88eaa54e728060edef082f8e358ff32030cb7c7d315e81109" [[package]] name = "time-macros" -version = "0.2.24" +version = "0.2.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30cfb0125f12d9c277f35663a0a33f8c30190f4e4574868a330595412d34ebf3" +checksum = "7e689342a48d2ea927c87ea50cabf8594854bf940e9310208848d680d668ed85" dependencies = [ "num-conv", "time-core", @@ -5612,6 +5795,12 @@ version = "0.6.8" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112" +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" + [[package]] name = "zvariant" version = "5.15.0" diff --git a/desktop/src-tauri/Cargo.toml b/desktop/src-tauri/Cargo.toml index 2548b000cd..c788432f9f 100644 --- a/desktop/src-tauri/Cargo.toml +++ b/desktop/src-tauri/Cargo.toml @@ -1,11 +1,11 @@ [package] name = "opencodex-desktop" -version = "2.61.0" +version = "2.76.0" description = "OpenCodex desktop shell" authors = ["OpenCodex contributors"] license = "MIT" edition = "2021" -rust-version = "1.77" +rust-version = "1.88" [lib] name = "opencodex_desktop_lib" @@ -15,19 +15,29 @@ crate-type = ["staticlib", "cdylib", "rlib"] tauri-build = { version = "=2.6.3", features = [] } [dependencies] +base64 = "=0.22.1" +hmac = "=0.12.1" reqwest = { version = "=0.12.24", default-features = false, features = ["json", "rustls-tls"] } -serde = { version = "=1.0.219", features = ["derive"] } -serde_json = "=1.0.140" +serde = { version = "=1.0.229", features = ["derive"] } +serde_json = "=1.0.151" +sha2 = "=0.10.9" uuid = { version = "=1.18.1", features = ["v4"] } -tauri = { version = "=2.11.6", features = ["tray-icon", "image-png"] } +tauri = { version = "=2.11.6", features = ["tray-icon", "image-png", "macos-private-api"] } +tauri-utils = "=2.9.3" tauri-plugin-autostart = "=2.5.0" tauri-plugin-opener = "=2.5.3" tauri-plugin-process = "=2.3.0" -tauri-plugin-shell = "=2.2.0" +tauri-plugin-shell = "=2.2.1" tauri-plugin-single-instance = "=2.4.0" tauri-plugin-updater = "=2.9.0" tokio = { version = "=1.45.1", features = ["sync", "time"] } +# Linux only, and already in this graph: `tao` enables its own `dbus` feature by default, so +# `libdbus-sys` is compiled for every Linux build of this shell today. Naming it here adds a +# session-bus probe for the StatusNotifier watcher without adding a package or a system library. +[target.'cfg(target_os = "linux")'.dependencies] +dbus = "=0.9.12" + [profile.release] codegen-units = 1 lto = "thin" diff --git a/desktop/src-tauri/Entitlements.plist b/desktop/src-tauri/Entitlements.plist new file mode 100644 index 0000000000..705971e888 --- /dev/null +++ b/desktop/src-tauri/Entitlements.plist @@ -0,0 +1,9 @@ + + + + + + com.apple.security.cs.allow-jit + + + diff --git a/desktop/src-tauri/build.rs b/desktop/src-tauri/build.rs index d860e1e6a7..f86b827eff 100644 --- a/desktop/src-tauri/build.rs +++ b/desktop/src-tauri/build.rs @@ -1,3 +1,70 @@ fn main() { - tauri_build::build() + if std::env::var("CARGO_CFG_TARGET_OS").as_deref() == Ok("macos") { + build_native_tray(); + } + tauri_build::build(); +} + +fn build_native_tray() { + use std::{env, fs, path::PathBuf, process::Command}; + let manifest = PathBuf::from(env::var_os("CARGO_MANIFEST_DIR").unwrap()); + let sources = manifest.join("../../app/Sources/NativeTray"); + println!("cargo:rerun-if-changed={}", sources.display()); + let mut files: Vec<_> = fs::read_dir(&sources) + .expect("NativeTray source directory is missing") + .map(|entry| entry.expect("cannot read native tray source").path()) + .filter(|path| { + path.extension() + .is_some_and(|extension| extension == "swift") + }) + .collect(); + files.sort(); + assert!(!files.is_empty(), "NativeTray source set is empty"); + let arch = match env::var("CARGO_CFG_TARGET_ARCH").unwrap().as_str() { + "aarch64" => "arm64", + "x86_64" => "x86_64", + other => panic!("unsupported macOS native tray architecture: {other}"), + }; + let out = PathBuf::from(env::var_os("OUT_DIR").unwrap()); + let archive = out.join("libNativeTray.a"); + let status = Command::new("xcrun") + .args([ + "--sdk", + "macosx", + "swiftc", + "-parse-as-library", + "-emit-library", + "-static", + ]) + .args([ + "-module-name", + "NativeTray", + "-target", + &format!("{arch}-apple-macos13.0"), + ]) + .arg(if env::var("PROFILE").as_deref() == Ok("release") { + "-O" + } else { + "-Onone" + }) + .args(&files) + .arg("-o") + .arg(&archive) + .status() + .expect("cannot run swiftc; install the macOS developer tools"); + assert!(status.success(), "NativeTray Swift compilation failed"); + if env::var("PROFILE").as_deref() == Ok("release") { + let symbols = Command::new("xcrun") + .args(["nm", "-u"]) + .arg(&archive) + .output() + .expect("cannot inspect NativeTray archive"); + assert!(symbols.status.success() && String::from_utf8_lossy(&symbols.stdout).contains("NSGlassEffectView"), + "macOS release builds require Xcode 26+ so supported systems receive Apple Liquid Glass"); + } + println!("cargo:rustc-link-search=native={}", out.display()); + println!("cargo:rustc-link-lib=static=NativeTray"); + // Darwin object autolinking supplies the system frameworks used by SwiftUI/Charts. + println!("cargo:rustc-link-search=native=/usr/lib/swift"); + println!("cargo:rustc-link-arg=-Wl,-rpath,/usr/lib/swift"); } diff --git a/desktop/src-tauri/capabilities/dashboard-titlebar.json b/desktop/src-tauri/capabilities/dashboard-titlebar.json new file mode 100644 index 0000000000..09eb19e711 --- /dev/null +++ b/desktop/src-tauri/capabilities/dashboard-titlebar.json @@ -0,0 +1,14 @@ +{ + "$schema": "../gen/schemas/desktop-schema.json", + "identifier": "dashboard-titlebar", + "description": "Integrated title bar: the loopback dashboard's top strips drag and double-click-zoom the main window", + "windows": ["main"], + "remote": { + "urls": ["http://127.0.0.1:*"] + }, + "permissions": [ + "core:window:allow-start-dragging", + "core:window:allow-toggle-maximize", + "core:window:allow-scale-factor" + ] +} diff --git a/desktop/src-tauri/capabilities/dashboard-zoom.json b/desktop/src-tauri/capabilities/dashboard-zoom.json new file mode 100644 index 0000000000..dc2b073424 --- /dev/null +++ b/desktop/src-tauri/capabilities/dashboard-zoom.json @@ -0,0 +1,10 @@ +{ + "$schema": "../gen/schemas/desktop-schema.json", + "identifier": "dashboard-zoom", + "description": "Page zoom hotkeys for the main window, including the loopback dashboard", + "windows": ["main"], + "remote": { + "urls": ["http://127.0.0.1:*"] + }, + "permissions": ["core:webview:allow-set-webview-zoom"] +} diff --git a/desktop/src-tauri/capabilities/default.json b/desktop/src-tauri/capabilities/default.json index 622143b1fa..cb83863e40 100644 --- a/desktop/src-tauri/capabilities/default.json +++ b/desktop/src-tauri/capabilities/default.json @@ -8,6 +8,8 @@ "core:window:allow-show", "core:window:allow-hide", "core:window:allow-set-title", + "core:window:allow-start-dragging", + "core:window:allow-toggle-maximize", "opener:default", "autostart:default" ] diff --git a/desktop/src-tauri/icons/128x128.png b/desktop/src-tauri/icons/128x128.png index e77df1e28f..8827b1fdc2 100644 Binary files a/desktop/src-tauri/icons/128x128.png and b/desktop/src-tauri/icons/128x128.png differ diff --git a/desktop/src-tauri/icons/128x128@2x.png b/desktop/src-tauri/icons/128x128@2x.png index c6a3f4f321..21c507c776 100644 Binary files a/desktop/src-tauri/icons/128x128@2x.png and b/desktop/src-tauri/icons/128x128@2x.png differ diff --git a/desktop/src-tauri/icons/32x32.png b/desktop/src-tauri/icons/32x32.png index 9299f36004..47a19b8131 100644 Binary files a/desktop/src-tauri/icons/32x32.png and b/desktop/src-tauri/icons/32x32.png differ diff --git a/desktop/src-tauri/icons/64x64.png b/desktop/src-tauri/icons/64x64.png index 0a93d6825e..5f7a0713c8 100644 Binary files a/desktop/src-tauri/icons/64x64.png and b/desktop/src-tauri/icons/64x64.png differ diff --git a/desktop/src-tauri/icons/Square107x107Logo.png b/desktop/src-tauri/icons/Square107x107Logo.png index 5437a9844d..9436cf61e1 100644 Binary files a/desktop/src-tauri/icons/Square107x107Logo.png and b/desktop/src-tauri/icons/Square107x107Logo.png differ diff --git a/desktop/src-tauri/icons/Square142x142Logo.png b/desktop/src-tauri/icons/Square142x142Logo.png index 6cb3949d35..80d163ad59 100644 Binary files a/desktop/src-tauri/icons/Square142x142Logo.png and b/desktop/src-tauri/icons/Square142x142Logo.png differ diff --git a/desktop/src-tauri/icons/Square150x150Logo.png b/desktop/src-tauri/icons/Square150x150Logo.png index b2ba7ef594..5148c874dc 100644 Binary files a/desktop/src-tauri/icons/Square150x150Logo.png and b/desktop/src-tauri/icons/Square150x150Logo.png differ diff --git a/desktop/src-tauri/icons/Square284x284Logo.png b/desktop/src-tauri/icons/Square284x284Logo.png index 7c578ed553..d4ccde8b23 100644 Binary files a/desktop/src-tauri/icons/Square284x284Logo.png and b/desktop/src-tauri/icons/Square284x284Logo.png differ diff --git a/desktop/src-tauri/icons/Square30x30Logo.png b/desktop/src-tauri/icons/Square30x30Logo.png index 46537b6bfb..acbbb84ceb 100644 Binary files a/desktop/src-tauri/icons/Square30x30Logo.png and b/desktop/src-tauri/icons/Square30x30Logo.png differ diff --git a/desktop/src-tauri/icons/Square310x310Logo.png b/desktop/src-tauri/icons/Square310x310Logo.png index dba929dbcd..1b9c14d45a 100644 Binary files a/desktop/src-tauri/icons/Square310x310Logo.png and b/desktop/src-tauri/icons/Square310x310Logo.png differ diff --git a/desktop/src-tauri/icons/Square44x44Logo.png b/desktop/src-tauri/icons/Square44x44Logo.png index 98c1c62f4f..5ce2668bf2 100644 Binary files a/desktop/src-tauri/icons/Square44x44Logo.png and b/desktop/src-tauri/icons/Square44x44Logo.png differ diff --git a/desktop/src-tauri/icons/Square71x71Logo.png b/desktop/src-tauri/icons/Square71x71Logo.png index 812f7a506e..f9a361ad93 100644 Binary files a/desktop/src-tauri/icons/Square71x71Logo.png and b/desktop/src-tauri/icons/Square71x71Logo.png differ diff --git a/desktop/src-tauri/icons/Square89x89Logo.png b/desktop/src-tauri/icons/Square89x89Logo.png index 90d34e7435..eafa6f2087 100644 Binary files a/desktop/src-tauri/icons/Square89x89Logo.png and b/desktop/src-tauri/icons/Square89x89Logo.png differ diff --git a/desktop/src-tauri/icons/StoreLogo.png b/desktop/src-tauri/icons/StoreLogo.png index 03afa12a7b..8621824587 100644 Binary files a/desktop/src-tauri/icons/StoreLogo.png and b/desktop/src-tauri/icons/StoreLogo.png differ diff --git a/desktop/src-tauri/icons/icon.icns b/desktop/src-tauri/icons/icon.icns index 3ca608c923..12ddb09d39 100644 Binary files a/desktop/src-tauri/icons/icon.icns and b/desktop/src-tauri/icons/icon.icns differ diff --git a/desktop/src-tauri/icons/icon.ico b/desktop/src-tauri/icons/icon.ico index 78c4c78335..80ded4a940 100644 Binary files a/desktop/src-tauri/icons/icon.ico and b/desktop/src-tauri/icons/icon.ico differ diff --git a/desktop/src-tauri/icons/icon.png b/desktop/src-tauri/icons/icon.png index 94ac887238..5c8fa696c6 100644 Binary files a/desktop/src-tauri/icons/icon.png and b/desktop/src-tauri/icons/icon.png differ diff --git a/desktop/src-tauri/icons/icon.svg b/desktop/src-tauri/icons/icon.svg new file mode 100644 index 0000000000..7070d673ad --- /dev/null +++ b/desktop/src-tauri/icons/icon.svg @@ -0,0 +1,43 @@ + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/desktop/src-tauri/icons/tray/icon-update.png b/desktop/src-tauri/icons/tray/icon-update.png new file mode 100644 index 0000000000..9cd0fc910e Binary files /dev/null and b/desktop/src-tauri/icons/tray/icon-update.png differ diff --git a/desktop/src-tauri/icons/tray/icon.png b/desktop/src-tauri/icons/tray/icon.png index f475a9e9c7..1eec4cccaf 100644 Binary files a/desktop/src-tauri/icons/tray/icon.png and b/desktop/src-tauri/icons/tray/icon.png differ diff --git a/desktop/src-tauri/icons/tray/icon.svg b/desktop/src-tauri/icons/tray/icon.svg new file mode 100644 index 0000000000..1f917f1f01 --- /dev/null +++ b/desktop/src-tauri/icons/tray/icon.svg @@ -0,0 +1,28 @@ + + + + + + + + + + + + + + + + + diff --git a/desktop/src-tauri/src/auth.rs b/desktop/src-tauri/src/auth.rs index 81a16467b0..71175823da 100644 --- a/desktop/src-tauri/src/auth.rs +++ b/desktop/src-tauri/src/auth.rs @@ -1,28 +1,45 @@ use std::path::PathBuf; +use serde::Deserialize; + +/// The runtime record the server publishes in `runtime-port.json`. +/// +/// The attestation secret is what lets this client tell the instance it was bound to apart from a +/// foreign process that later takes the port over: only the real runtime can answer an attestation +/// challenge with a proof keyed by it. +#[derive(Clone, Debug, Deserialize, PartialEq, Eq)] +#[serde(rename_all = "camelCase")] +pub struct RecordedRuntime { + pub pid: u32, + pub port: u16, + pub attestation_secret: String, +} + #[derive(Clone, Debug)] pub struct Auth { home: PathBuf, - environment_token: Option, } impl Auth { pub fn new(home: PathBuf) -> Self { - Self { - home, - environment_token: std::env::var("OPENCODEX_ADMIN_AUTH_TOKEN") - .ok() - .filter(|value| !value.is_empty()), - } + Self { home } } - pub fn token(&self) -> Option { - self.environment_token.clone().or_else(|| { - std::fs::read_to_string(self.home.join("admin-api-token")) - .ok() - .map(|value| value.trim().to_owned()) - .filter(|value| !value.is_empty()) - }) + /// The runtime record, or `None` when it is missing, malformed, or carries no usable + /// attestation secret — all of which mean the peer cannot prove the identity this client was + /// bound to. + pub fn runtime_identity(&self) -> Option { + let value = std::fs::read(self.home.join("runtime-port.json")).ok()?; + let identity: RecordedRuntime = serde_json::from_slice(&value).ok()?; + let secret_ok = identity.attestation_secret.len() == 43 + && identity + .attestation_secret + .bytes() + .all(|b| b.is_ascii_alphanumeric() || b == b'-' || b == b'_'); + if identity.pid == 0 || !secret_ok { + return None; + } + Some(identity) } pub fn user_agent() -> &'static str { diff --git a/desktop/src-tauri/src/claim.rs b/desktop/src-tauri/src/claim.rs new file mode 100644 index 0000000000..22d0cfcbc2 --- /dev/null +++ b/desktop/src-tauri/src/claim.rs @@ -0,0 +1,289 @@ +//! Recording this installation as the runtime owner, through the bundled CLI. +//! +//! The claim is only valid against the exact answer the consent prompt was approved from, +//! so this is a subprocess with expectations on argv rather than an in-process write: the +//! ownership mutation lease, the subject revalidation and the managing-CLI re-observation +//! all live in the CLI's `recordServiceOwner`, and re-running them here would be a second +//! implementation of a rule that has to be identical. +//! +//! Like `runtime_stop`, the result is a document, not a guess: `ocx service claim --json` +//! puts one summary on stdout and this consumes `ok` and the exit code rather than +//! inferring them. A claim that did not end in exit 0 with `ok:true` is a claim that did +//! not happen — and a takeover that reached here already stopped the foreign runtime, so +//! the caller's failure is a stopped runtime with no owner recorded, which the next launch +//! resolves as an ordinary absence. + +use serde::Deserialize; +use tauri::AppHandle; +use tauri_plugin_shell::ShellExt; +use tokio::time::{timeout_at, Instant}; + +/// The wire version this shell understands. +pub const SCHEMA: &str = "ocx-service-claim/1"; + +#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct ClaimOwnership { + pub owner: String, + pub install_id: String, + pub consent_generation: u64, +} + +#[derive(Clone, Debug, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct ClaimSummary { + pub schema: String, + pub ok: bool, + /// Present on success. + pub ownership: Option, + /// Present on failure: the CLI's machine-readable error code. + pub code: Option, + /// Present on failure. + pub message: Option, +} + +/// What the shell concluded. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum ClaimResult { + /// The CLI recorded the claim and named the generation it landed at. + Recorded(ClaimOwnership), + /// It reported anything else, or the run could not be read at all. + Failed(String), +} + +impl ClaimResult { + #[cfg(test)] + pub fn is_recorded(&self) -> bool { + matches!(self, Self::Recorded(_)) + } +} + +/// The arguments a takeover builds from the resolve answer it was approved against. +/// +/// `Recorded::Unknown` gets no argv: fabricating `--expect-none --expect-revision 0` would claim +/// against a subject nobody approved, so the answer is None and the caller refuses. +pub fn args( + install_id: &str, + recorded: &crate::ownership::Recorded, + token: &str, +) -> Option> { + let mut argv = vec![ + "service".to_owned(), + "claim".to_owned(), + "--owner".to_owned(), + "desktop".to_owned(), + "--install-id".to_owned(), + install_id.to_owned(), + ]; + match recorded { + crate::ownership::Recorded::None { revision } => { + argv.push("--expect-none".to_owned()); + argv.push("--expect-revision".to_owned()); + argv.push(revision.to_string()); + } + crate::ownership::Recorded::Owned { + ownership, + revision, + } => { + argv.extend([ + "--expect-owner".to_owned(), + match ownership.owner { + crate::ownership::Owner::Cli => "cli".to_owned(), + crate::ownership::Owner::Desktop => "desktop".to_owned(), + }, + "--expect-install-id".to_owned(), + ownership.install_id.clone(), + "--expect-generation".to_owned(), + ownership.consent_generation.to_string(), + "--expect-revision".to_owned(), + revision.to_string(), + ]); + } + // A takeover is only offered when the record was read; unknown never reaches here, + // and refusing beats inventing an approval. + crate::ownership::Recorded::Unknown { .. } => return None, + } + argv.extend([ + "--expect-compatibility-token".to_owned(), + token.to_owned(), + "--json".to_owned(), + ]); + Some(argv) +} + +/// Read one claim summary. +/// +/// Exit 0 with `ok:true` is the only success — the claim path uses exit 1 with a +/// machine-readable `code` for subject mismatches and changed compatibility, and both of +/// those are refusals to re-ask from, not partial writes. +pub fn read(exit_code: Option, stdout: &[u8], stderr: &[u8]) -> ClaimResult { + let text = String::from_utf8_lossy(stdout); + let summary: ClaimSummary = match serde_json::from_str(text.trim()) { + Ok(summary) => summary, + Err(error) => { + let detail = String::from_utf8_lossy(stderr); + let detail = detail.trim(); + let code = exit_code + .map(|code| code.to_string()) + .unwrap_or_else(|| "no exit code".to_owned()); + return ClaimResult::Failed(if detail.is_empty() { + format!("the bundled CLI's claim output could not be read (exit {code}: {error})") + } else { + format!("the bundled CLI's claim output could not be read (exit {code}): {detail}") + }); + } + }; + if summary.schema != SCHEMA { + return ClaimResult::Failed(format!( + "the bundled CLI answered with schema {} and this app understands {SCHEMA}", + summary.schema + )); + } + if exit_code != Some(0) || !summary.ok { + return ClaimResult::Failed(summary.message.unwrap_or_else(|| { + format!( + "the claim was refused ({})", + summary.code.unwrap_or_else(|| "no code".to_owned()) + ) + })); + } + match summary.ownership { + Some(ownership) => ClaimResult::Recorded(ownership), + None => ClaimResult::Failed( + "the claim reported success but carried no ownership record".to_owned(), + ), + } +} + +/// Run the bundled `ocx service claim`, under the caller's deadline. +pub async fn run(app: &AppHandle, argv: Vec, deadline: Instant) -> ClaimResult { + let command = match app.shell().sidecar("ocx") { + Ok(command) => command.args(argv), + Err(error) => { + return ClaimResult::Failed(format!("the bundled CLI could not be started ({error})")) + } + }; + match timeout_at(deadline, command.output()).await { + Ok(Ok(output)) => read(output.status.code(), &output.stdout, &output.stderr), + Ok(Err(error)) => { + ClaimResult::Failed(format!("the bundled CLI could not be run ({error})")) + } + Err(_) => ClaimResult::Failed( + "the bundled CLI did not finish the claim before the deadline".to_owned(), + ), + } +} + +#[cfg(test)] +mod tests { + use super::{args, read, ClaimResult}; + use crate::ownership::{Owner, Recorded}; + + fn document(ok: bool, extra: &str) -> String { + format!(r#"{{"schema":"ocx-service-claim/1","ok":{ok}{extra}}}"#) + } + + #[test] + fn the_arguments_carry_the_exact_approved_subject() { + let none = args("install-a", &Recorded::None { revision: 0 }, "tok"); + assert_eq!( + none.expect("argv for a read record"), + vec![ + "service", + "claim", + "--owner", + "desktop", + "--install-id", + "install-a", + "--expect-none", + "--expect-revision", + "0", + "--expect-compatibility-token", + "tok", + "--json", + ] + ); + let owned = Recorded::Owned { + ownership: crate::ownership::Claim { + owner: Owner::Cli, + install_id: "npm-1".to_owned(), + consent_generation: 2, + }, + revision: 9, + }; + let argv = args("install-a", &owned, "tok").expect("a claim against a read record"); + assert!(argv + .windows(2) + .any(|pair| pair == ["--expect-owner", "cli"])); + assert!(argv + .windows(2) + .any(|pair| pair == ["--expect-install-id", "npm-1"])); + assert!(argv + .windows(2) + .any(|pair| pair == ["--expect-generation", "2"])); + assert!(argv + .windows(2) + .any(|pair| pair == ["--expect-revision", "9"])); + } + + #[test] + fn an_unread_record_gets_no_claim_rather_than_a_fabricated_one() { + // Nobody approved a subject the resolve could not read, so there is nothing to claim + // against -- and "expect none, revision 0" would be that approval invented. + assert!(args( + "install-a", + &Recorded::Unknown { + reason: "why".to_owned() + }, + "tok" + ) + .is_none()); + } + + #[test] + fn a_recorded_claim_is_the_only_success() { + let ok = document( + true, + r#","ownership":{"owner":"desktop","installId":"install-a","consentGeneration":1},"revision":3"#, + ); + let result = read(Some(0), ok.as_bytes(), b""); + match result { + ClaimResult::Recorded(ownership) => { + assert_eq!(ownership.install_id, "install-a"); + assert_eq!(ownership.consent_generation, 1); + } + ClaimResult::Failed(reason) => panic!("{reason}"), + } + // Success has to arrive with exit 0 and the record it wrote. + assert!(!read(Some(1), ok.as_bytes(), b"").is_recorded()); + assert!(!read(Some(0), document(true, "").as_bytes(), b"").is_recorded()); + } + + #[test] + fn a_refusal_carries_the_clis_own_message() { + let refused = document( + false, + r#","code":"service-ownership-subject-mismatch","message":"ownership changed""#, + ); + let result = read(Some(1), refused.as_bytes(), b""); + match result { + ClaimResult::Failed(reason) => assert!(reason.contains("ownership changed")), + ClaimResult::Recorded(_) => panic!("a refused claim is not recorded"), + } + } + + #[test] + fn output_that_cannot_be_read_is_a_failure_not_a_claim() { + assert!(!read(Some(0), b"", b"boom").is_recorded()); + assert!(!read(Some(0), b"not json", b"").is_recorded()); + assert!(!read(None, b"", b"").is_recorded()); + let future = document(true, r#","ownership":{"owner":"desktop","installId":"i","consentGeneration":1},"revision":1"#) + .replace("ocx-service-claim/1", "ocx-service-claim/2"); + let result = read(Some(0), future.as_bytes(), b""); + assert!(!result.is_recorded()); + match result { + ClaimResult::Failed(reason) => assert!(reason.contains("ocx-service-claim/2")), + _ => unreachable!(), + } + } +} diff --git a/desktop/src-tauri/src/companion_query.rs b/desktop/src-tauri/src/companion_query.rs new file mode 100644 index 0000000000..8dd450ef40 --- /dev/null +++ b/desktop/src-tauri/src/companion_query.rs @@ -0,0 +1,147 @@ +//! Shared request encoding and compatibility projection for companion timelines. +use crate::companion_usage::{has_filters, selected, text}; +use serde_json::Value; +use std::collections::BTreeSet; + +pub fn timeline_query(settings: &Value) -> String { + let settings = settings.get("settings").unwrap_or(settings); + let mut url = + reqwest::Url::parse("http://127.0.0.1/api/usage/timeline").expect("constant loopback URL"); + { + let mut query = url.query_pairs_mut(); + for (query_key, key, fallback) in [ + ("hours", "chartHours", "24"), + ("bucketMinutes", "bucketMinutes", "60"), + ("metric", "tokenMetric", "total"), + ("aggregation", "aggregation", "sum"), + ("grouping", "chartGrouping", "model"), + ] { + let value = settings[key] + .as_str() + .map(str::to_owned) + .or_else(|| settings[key].as_i64().map(|value| value.to_string())) + .unwrap_or_else(|| fallback.into()); + query.append_pair(query_key, &value); + } + if let Some(models) = settings["models"] + .as_array() + .filter(|rows| !rows.is_empty()) + { + query.append_pair( + "models", + &models + .iter() + .filter_map(Value::as_str) + .collect::>() + .join(","), + ); + } + if let Some(providers) = settings["hiddenProviders"].as_array() { + for provider in providers.iter().filter_map(Value::as_str) { + query.append_pair("hiddenProvider", provider); + } + } + } + url.query().unwrap_or_default().into() +} + +fn canonical(value: &Value) -> Option> { + let rows = value.as_array()?; + if rows.len() > 100 { + return None; + } + rows.iter().map(Value::as_str).collect() +} + +pub fn timeline_rows<'a>(body: &'a Value, settings: &Value) -> Option<(Vec<&'a Value>, bool)> { + let settings = settings.get("settings").unwrap_or(settings); + let rows = body["series"].as_array()?; + let incomplete = body["truncated"].as_bool() == Some(true) + || body["missingMeasurements"] + .as_f64() + .is_some_and(|n| n > 0.0); + if settings["models"] + .as_array() + .is_some_and(|rows| rows.is_empty()) + { + return Some((vec![], incomplete)); + } + let active = has_filters(settings); + let echo = &body["appliedFilters"]; + let hidden = settings + .get("hiddenProviders") + .cloned() + .unwrap_or_else(|| serde_json::json!([])); + let matches = echo.is_object() + && echo.get("models").is_some() + && canonical(&echo["hiddenProviders"]).is_some() + && canonical(&echo["hiddenProviders"]) == canonical(&hidden) + && (if settings["models"].is_null() { + echo["models"].is_null() + } else { + canonical(&echo["models"]).is_some() + && canonical(&echo["models"]) == canonical(&settings["models"]) + }); + let visible = rows + .iter() + .filter(|row| { + if text(row, "id") == "other" && text(row, "provider").is_empty() { + return !active || matches; + } + selected(settings, row) + }) + .collect(); + Some(( + visible, + incomplete || ((active || !echo.is_null()) && !matches), + )) +} + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + #[test] + fn query_encodes_nested_models_and_repeated_provider_exclusions() { + let query = timeline_query( + &json!({"models":["p/vendor/model+one"],"hiddenProviders":["a+b","hidden"]}), + ); + let url = reqwest::Url::parse(&format!("http://127.0.0.1/?{query}")).unwrap(); + let pairs: Vec<_> = url.query_pairs().collect(); + assert!(pairs + .iter() + .any(|(key, value)| key == "models" && value == "p/vendor/model+one")); + assert_eq!( + pairs + .iter() + .filter(|(key, _)| key == "hiddenProvider") + .count(), + 2 + ); + assert!(pairs + .iter() + .any(|(key, value)| key == "hiddenProvider" && value == "a+b")); + } + #[test] + fn a_matching_echo_preserves_folded_rows_and_older_responses_stay_incomplete() { + let mut body = json!({"series":[{"id":"visible/m","provider":"visible","model":"m"}, + {"id":"hidden/m","provider":"hidden","model":"m"},{"id":"other","provider":"","model":"other"}]}); + let settings = json!({"models":null,"hiddenProviders":["hidden"]}); + assert_eq!(timeline_rows(&body, &settings).unwrap().0.len(), 1); + assert!(timeline_rows(&body, &settings).unwrap().1); + body["appliedFilters"] = settings.clone(); + assert_eq!(timeline_rows(&body, &settings).unwrap().0.len(), 2); + assert!(!timeline_rows(&body, &settings).unwrap().1); + let models = json!({"models":["visible/m"],"hiddenProviders":[]}); + body["appliedFilters"] = models.clone(); + assert_eq!(timeline_rows(&body, &models).unwrap().0.len(), 2); + assert_eq!( + timeline_rows(&body, &json!({"models":[]})).unwrap().0.len(), + 0 + ); + body.as_object_mut().unwrap().remove("appliedFilters"); + let all = timeline_rows(&body, &json!({})).unwrap(); + assert_eq!(all.0.len(), 3); + assert!(!all.1); + } +} diff --git a/desktop/src-tauri/src/companion_usage.rs b/desktop/src-tauri/src/companion_usage.rs new file mode 100644 index 0000000000..a6fd6e7561 --- /dev/null +++ b/desktop/src-tauri/src/companion_usage.rs @@ -0,0 +1,188 @@ +//! Platform-neutral companion usage filtering; no application, transport or credential owner. +use serde_json::{json, Map, Value}; + +pub fn number(value: &Value) -> Option { + value.as_f64().filter(|n| n.is_finite() && *n >= 0.0) +} + +pub fn text<'a>(value: &'a Value, key: &str) -> &'a str { + value.get(key).and_then(Value::as_str).unwrap_or("") +} + +pub fn hidden(settings: &Value, provider: &str) -> bool { + settings["hiddenProviders"] + .as_array() + .is_some_and(|rows| rows.iter().any(|row| row.as_str() == Some(provider))) +} + +pub fn selected(settings: &Value, row: &Value) -> bool { + let provider = text(row, "provider"); + let model = text(row, "model"); + !hidden(settings, provider) + && settings["models"].as_array().is_none_or(|models| { + models.iter().any(|item| { + item.as_str() + .is_some_and(|item| item == model || item == format!("{provider}/{model}")) + }) + }) +} + +const TOTAL_KEYS: [&str; 9] = [ + "requests", + "totalTokens", + "inputTokens", + "outputTokens", + "cachedInputTokens", + "cacheReadInputTokens", + "estimatedCostUsd", + "measuredRequests", + "pricedRequests", +]; + +pub fn usage(body: &Value, settings: &Value) -> Option<(Value, Vec)> { + let source = body["summary"].as_object()?; + if body.get("error").is_some() { + return None; + } + if body.get("models").is_some_and(|value| !value.is_array()) { + return None; + } + let filtering = has_filters(settings); + let empty_selection = settings["models"] + .as_array() + .is_some_and(|rows| rows.is_empty()); + let all = match body["models"].as_array() { + _ if empty_selection => &[], + Some(rows) => rows.as_slice(), + None if !filtering => &[], + None => return None, + }; + if filtering + && all.iter().any(|row| { + text(row, "provider").is_empty() + || text(row, "model").is_empty() + || (text(row, "provider") == "other" && text(row, "model") == "other") + }) + { + return None; + } + let rows: Vec<_> = all.iter().filter(|row| selected(settings, row)).collect(); + let mut totals = Map::new(); + for key in TOTAL_KEYS { + let value = if filtering { + if rows.is_empty() { + None + } else { + rows.iter() + .map(|row| number(&row[key])) + .collect::>>() + .map(|values| values.iter().sum()) + } + } else { + source.get(key).and_then(number) + }; + totals.insert(key.into(), json!(value)); + } + // Both spellings exist on management projections; prefer the exact cache-read field. + if !totals["cacheReadInputTokens"].is_null() { + totals.insert( + "cachedInputTokens".into(), + totals["cacheReadInputTokens"].clone(), + ); + } + if number(&body["summary"]["coverageRatio"]) == Some(0.0) && !filtering { + totals.insert("measuredRequests".into(), json!(0)); + } + totals.remove("cacheReadInputTokens"); + totals.insert( + "incomplete".into(), + json!(["usageIncomplete", "historyTruncated", "entriesTruncated"] + .iter() + .any(|key| body[key].as_bool() == Some(true))), + ); + let models = rows + .iter() + .enumerate() + .map(|(index, row)| { + let unmeasured = number(&row["requests"]).is_some_and(|n| n > 0.0) + && (number(&row["measuredRequests"]) == Some(0.0) + || number(&row["coverageRatio"]) == Some(0.0)); + json!({"id":format!("{}/{}/{}",text(row,"provider"),text(row,"model"),index), + "label":text(row,"model"),"requests":number(&row["requests"]), + "tokens":if unmeasured { None } else { number(&row["totalTokens"]) }}) + }) + .collect(); + Some((Value::Object(totals), models)) +} + +pub fn has_filters(settings: &Value) -> bool { + settings["models"].is_array() + || settings["hiddenProviders"] + .as_array() + .is_some_and(|rows| !rows.is_empty()) +} + +pub fn filtered_summary(body: &Value, settings: &Value) -> Option { + let settings = settings.get("settings").unwrap_or(settings); + if body.get("error").is_some() { + return None; + } + if has_filters(settings) { + return usage(body, settings).map(|(summary, _)| summary); + } + let summary = body.get("summary").unwrap_or(body); + summary.as_object().map(|_| summary.clone()) +} + +pub fn integer(value: &Value) -> Option { + value.as_i64().filter(|n| *n >= 0).or_else(|| { + number(value) + .filter(|n| n.fract() == 0.0 && *n < 9_223_372_036_854_775_808.0) + .map(|n| n as i64) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn counts_accept_whole_json_doubles_without_truncation_or_overflow() { + for value in [json!(0), json!(2), json!(2.0)] { + assert!(integer(&value).is_some()); + } + for value in [ + json!(-1), + json!(2.5), + json!(9_223_372_036_854_775_808_u64), + json!("2"), + Value::Null, + ] { + assert_eq!(integer(&value), None); + } + assert_eq!(integer(&json!(i64::MAX)), Some(i64::MAX)); + } + #[test] + fn filtering_preserves_missingness_and_refuses_unrecoverable_folded_attribution() { + let settings = json!({"models":null,"hiddenProviders":["hidden"]}); + let mut body = json!({"summary":{"requests":9,"totalTokens":99},"models":[ + {"provider":"hidden","model":"m","requests":7,"totalTokens":94}, + {"provider":"visible","model":"m","requests":2,"totalTokens":5,"estimatedCostUsd":0.25} + ]}); + let filtered = filtered_summary(&body, &settings).unwrap(); + assert_eq!(integer(&filtered["requests"]), Some(2)); + assert_eq!(integer(&filtered["totalTokens"]), Some(5)); + assert_eq!(filtered["estimatedCostUsd"], json!(0.25)); + assert!(filtered["inputTokens"].is_null()); + body["models"] + .as_array_mut() + .unwrap() + .push(json!({"provider":"other","model":"other","totalTokens":1})); + assert!(filtered_summary(&body, &settings).is_none()); + assert!(filtered_summary(&json!({"summary":{"totalTokens":99}}), &settings).is_none()); + assert_eq!( + filtered_summary(&json!({"summary":{"totalTokens":0}}), &json!({})), + Some(json!({"totalTokens":0})) + ); + assert!(usage(&json!({"summary":{"totalTokens":0}}), &json!({})).is_some()); + } +} diff --git a/desktop/src-tauri/src/discovery.rs b/desktop/src-tauri/src/discovery.rs deleted file mode 100644 index 2eaded758d..0000000000 --- a/desktop/src-tauri/src/discovery.rs +++ /dev/null @@ -1,116 +0,0 @@ -use serde::Deserialize; -use std::path::{Path, PathBuf}; - -pub const DEFAULT_PORT: u16 = 10100; -const HOME_ENV: &str = "OPENCODEX_HOME"; - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub struct ProxyEndpoint { - pub host: &'static str, - pub port: u16, -} - -impl ProxyEndpoint { - pub fn url(&self, path: &str) -> String { - format!("http://{}:{}{}", self.host, self.port, path) - } -} - -#[derive(Debug, Deserialize)] -struct RuntimePort { - port: u16, -} - -pub fn config_directory(environment: impl Fn(&str) -> Option, home: &Path) -> PathBuf { - environment(HOME_ENV) - .map(|value| value.trim().to_owned()) - .filter(|value| !value.is_empty()) - .map(|value| expand_tilde(PathBuf::from(value), home)) - .unwrap_or_else(|| home.join(".opencodex")) -} - -pub fn resolve(environment: impl Fn(&str) -> Option, home: &Path) -> ProxyEndpoint { - let directory = config_directory(environment, home); - let path = directory.join("runtime-port.json"); - let port = std::fs::read(&path) - .ok() - .and_then(|bytes| serde_json::from_slice::(&bytes).ok()) - .map(|record| record.port) - .filter(|port| (1..=u16::MAX).contains(port)) - .unwrap_or(DEFAULT_PORT); - ProxyEndpoint { - host: "127.0.0.1", - port, - } -} - -fn expand_tilde(path: PathBuf, home: &Path) -> PathBuf { - if path == Path::new("~") { - return home.to_path_buf(); - } - path.strip_prefix("~/") - .map(|rest| home.join(rest)) - .unwrap_or(path) -} - -pub fn current() -> (ProxyEndpoint, PathBuf) { - let home = dirs_home(); - let directory = config_directory(|key| std::env::var(key).ok(), &home); - let endpoint = resolve(|key| std::env::var(key).ok(), &home); - (endpoint, directory) -} - -fn dirs_home() -> PathBuf { - std::env::var_os("HOME") - .map(PathBuf::from) - .or_else(|| std::env::var_os("USERPROFILE").map(PathBuf::from)) - .unwrap_or_else(|| PathBuf::from(".")) -} - -#[cfg(test)] -mod tests { - use super::*; - use std::fs::{create_dir_all, write}; - - #[test] - fn resolves_home_override_and_runtime_port() { - let root = std::env::temp_dir().join(format!("ocx-discovery-{}", std::process::id())); - let home = root.join("home"); - let custom = home.join("custom"); - create_dir_all(&custom).unwrap(); - write( - custom.join("runtime-port.json"), - r#"{"pid":1,"port":12345}"#, - ) - .unwrap(); - let endpoint = resolve(|key| (key == HOME_ENV).then(|| "~/custom".into()), &home); - assert_eq!( - endpoint, - ProxyEndpoint { - host: "127.0.0.1", - port: 12345 - } - ); - let _ = std::fs::remove_dir_all(root); - } - - #[test] - fn any_failure_falls_back_to_default() { - let home = std::env::temp_dir().join("ocx-missing-home"); - let endpoint = resolve(|_| None, &home); - assert_eq!(endpoint.port, DEFAULT_PORT); - } - - #[test] - fn empty_override_and_invalid_port_use_default() { - let root = - std::env::temp_dir().join(format!("ocx-discovery-invalid-{}", std::process::id())); - let home = root.join("home"); - let directory = root.join("custom"); - create_dir_all(&directory).unwrap(); - write(directory.join("runtime-port.json"), r#"{"port":0}"#).unwrap(); - let endpoint = resolve(|key| (key == HOME_ENV).then(|| " ".into()), &home); - assert_eq!(endpoint.port, DEFAULT_PORT); - let _ = std::fs::remove_dir_all(root); - } -} diff --git a/desktop/src-tauri/src/endpoint.rs b/desktop/src-tauri/src/endpoint.rs new file mode 100644 index 0000000000..09a78e22fb --- /dev/null +++ b/desktop/src-tauri/src/endpoint.rs @@ -0,0 +1,33 @@ +//! The loopback endpoint the shell talks to. +//! +//! This file was `discovery.rs`, and it resolved the endpoint itself: it read `runtime-port.json`, +//! fell back to 10100 and let the shell start there, so a user with a configured `config.port` was +//! started on a port they had not chosen. Resolution belongs to the bundled CLI now — see +//! `resolve.rs` — and what is left here is the value it hands back. + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct ProxyEndpoint { + pub host: &'static str, + pub port: u16, +} + +impl ProxyEndpoint { + pub fn url(&self, path: &str) -> String { + format!("http://{}:{}{}", self.host, self.port, path) + } +} + +#[cfg(test)] +mod tests { + use super::ProxyEndpoint; + + #[test] + fn the_endpoint_is_loopback_and_carries_its_port() { + let endpoint = ProxyEndpoint { + host: "127.0.0.1", + port: 12345, + }; + assert_eq!(endpoint.url("/healthz"), "http://127.0.0.1:12345/healthz"); + assert_eq!(endpoint.url(""), "http://127.0.0.1:12345"); + } +} diff --git a/desktop/src-tauri/src/exit.rs b/desktop/src-tauri/src/exit.rs new file mode 100644 index 0000000000..8950c33863 --- /dev/null +++ b/desktop/src-tauri/src/exit.rs @@ -0,0 +1,1060 @@ +//! Who is allowed to end the process, and what has to happen first. +//! +//! Three gestures arrive looking like an exit: closing the window, the platform's own quit gesture +//! (Cmd+Q on macOS, Alt+F4 on Windows, the window manager's close on Linux), and the tray's Quit +//! item. Only the last one means "end the runtime". Until this module existed the shell had no +//! `ExitRequested` handler, so the quit gesture went straight to `RunEvent::Exit`, which called +//! `CommandChild::kill()` — a SIGKILL on Unix — on a keystroke the user reads as "hide". +//! +//! Tauri separates a user gesture from a programmatic exit: `RunEvent::ExitRequested` carries +//! `code: None` for the gesture and `Some(_)` for `AppHandle::exit` or `AppHandle::restart`. What +//! it cannot tell apart is the tray's Quit from an update's restart, and those end differently. +//! [`ExitReason`] records which one asked. macOS needs one more thing on top, because its menu Quit +//! never raises the event at all; see [`crate::menu`]. +//! +//! Everything that stops the runtime funnels through one phase here — the tray's Quit, the tray's +//! Stop, an update, and a window close on a session with no tray. Two of them running at once is +//! two stops racing over one child, so a second one waits rather than starting its own. +//! +//! A drain that does not complete is **not** recorded as a drain. A quit may still proceed on one, +//! because refusing to close is the worse answer and a standing runtime is recoverable. A restart +//! may not: coming back onto a runtime that was never stopped puts the user on the old version +//! while they believe they are on the new one. + +use crate::{ + proxy::ProxyClient, runtime_stop, sidecar, tray_availability::TrayAvailability, window, + AppState, +}; +use std::sync::{Mutex, MutexGuard, PoisonError}; +use tauri::{AppHandle, ExitRequestApi, Manager}; +use tokio::time::Instant; + +/// Why the process has been asked to end. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ExitReason { + /// The tray's Quit item, or a window close on a session with no tray to hide into. + UserQuit, + /// An installed update restarting the app. It drains exactly as a quit does and then comes + /// back, which is why it is a coordinated restart rather than an exception to the quit rule. + CoordinatedRestart, +} + +/// How far the one drain sequence has got. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ExitPhase { + /// Nothing is in flight. + Idle, + /// A runtime is being started. An exit arriving now is held until the child exists and is + /// recorded, because the alternative is a process nobody owns and nobody will stop. + Spawning, + /// The runtime is being stopped without ending the app: the tray's Stop. + Stopping, + /// The drain that ends the app is running. + Draining, + /// The runtime this app owned is confirmed stopped, or was never ours to stop. + Drained, + /// The stop was refused, or the runtime still answered after the deadline. + DrainFailed, + /// Who owns the runtime could not be established, so nothing was stopped and nothing may be + /// replaced on the assumption that it was. + OwnershipUnknown, +} + +/// What the event loop should do with an exit request. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ExitDecision { + /// Nothing asked the app to end and there is a tray to come back from: hide instead. + Hide, + /// Hold the exit and drain for this reason; the exit is requested again when the drain reports. + Drain(ExitReason), + /// Something else is already draining. Hold the exit and let that one finish. + Wait, + /// The drain has reported success. Let the process end. + Proceed, + /// The drain did not complete, and this reason is one that may not proceed on that. + Refuse, +} + +/// What a drain established. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum DrainVerdict { + /// The runtime this app owned is gone, or there was never one of ours to stop. + Drained, + /// The stop was refused, or the endpoint still answered after the deadline. + Failed, + /// The process answering could not be identified, so nothing was stopped. + OwnershipUnknown, +} + +/// Whether an update may start replacing files. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum RestartReadiness { + /// The runtime is confirmed stopped. The install may proceed. + Ready, + /// The drain did not complete, so the install must not start. + DrainFailed, + /// Who owns the runtime could not be established. + OwnershipUnknown, + /// Something else is already ending or stopping the runtime. + Busy, +} + +impl RestartReadiness { + pub fn describe(self) -> &'static str { + match self { + Self::Ready => "the runtime is stopped", + Self::DrainFailed => "the runtime did not stop", + Self::OwnershipUnknown => "the running proxy could not be identified", + Self::Busy => "the app is already stopping its runtime", + } + } +} + +/// Decide what an exit request means. +/// +/// `reason` is what the app itself asked for and is `None` for a bare user gesture. +/// `hides_to_tray` is D6: on a session with no usable tray there is nowhere to hide, so a close is +/// a quit and takes the same graceful drain rather than leaving a running process unreachable. +pub fn decide(phase: ExitPhase, reason: Option, hides_to_tray: bool) -> ExitDecision { + match phase { + ExitPhase::Spawning | ExitPhase::Stopping | ExitPhase::Draining => ExitDecision::Wait, + ExitPhase::Drained => ExitDecision::Proceed, + // A quit that could not drain still closes the app: refusing to close is the worse answer + // and the runtime is recoverable. A restart is a different judgement — it would come back + // attached to a runtime that was never stopped, under a user who believes they upgraded. + ExitPhase::DrainFailed | ExitPhase::OwnershipUnknown => match reason { + Some(ExitReason::CoordinatedRestart) => ExitDecision::Refuse, + _ => ExitDecision::Proceed, + }, + ExitPhase::Idle => match reason { + Some(reason) => ExitDecision::Drain(reason), + None if hides_to_tray => ExitDecision::Hide, + None => ExitDecision::Drain(ExitReason::UserQuit), + }, + } +} + +/// What the runtime supervisor needs to know before it brings a runtime back. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Supervision { + pub phase: ExitPhase, + /// Whether the app still wants a runtime. True from launch; the tray's Stop, a quit's drain and + /// an update's drain clear it before the runtime's exit can arrive, so the supervisor never + /// undoes any of them. + pub wanted: bool, + /// Something has claimed the app's ending: a quit or a restart. + pub reason_set: bool, +} + +/// State restored when an update's coordinated restart is abandoned. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct AbortedRestart { + pub phase: ExitPhase, + /// Whether the person wanted a runtime before the drain or requested one while it ran. + pub runtime_was_wanted: bool, +} + +impl Supervision { + /// Nothing is in flight, nobody asked for the runtime to stop, and the app is not ending. + pub fn allowed(self) -> bool { + self.phase == ExitPhase::Idle && self.wanted && !self.reason_set + } +} + +struct Inner { + phase: ExitPhase, + reason: Option, + hides_to_tray: bool, + /// An exit that arrived while a runtime was being started or stopped, and still has to happen. + deferred: bool, + /// See [`Supervision::wanted`]. Sticky: finishing a stop does not restore it. + wanted: bool, + /// The intent an update temporarily replaced with `wanted=false`; consumed if it aborts. + restart_wanted: Option, +} + +fn remember_restart_intent(inner: &mut Inner, reason: ExitReason, prior_wanted: bool) { + if reason == ExitReason::CoordinatedRestart { + inner.restart_wanted.get_or_insert(prior_wanted); + } +} + +/// The exit sequence's state, managed by the app. +pub struct ExitCoordinator { + inner: Mutex, +} + +impl ExitCoordinator { + pub fn new() -> Self { + Self { + inner: Mutex::new(Inner { + phase: ExitPhase::Idle, + reason: None, + // Until the probe answers, assume only what the platform guarantees. Assuming a + // tray that turns out not to exist is the exact failure D6 is about. + hides_to_tray: TrayAvailability::assumed().hides_to_tray(), + deferred: false, + wanted: true, + restart_wanted: None, + }), + } + } + + fn inner(&self) -> MutexGuard<'_, Inner> { + self.inner.lock().unwrap_or_else(PoisonError::into_inner) + } + + /// Record the session's tray verdict once the probe has answered and an icon exists. + pub fn set_tray(&self, tray: TrayAvailability) { + self.inner().hides_to_tray = tray.hides_to_tray(); + } + + pub fn decision(&self) -> ExitDecision { + let inner = self.inner(); + decide(inner.phase, inner.reason, inner.hides_to_tray) + } + + /// Record why the app is ending, without starting anything. The first reason wins. + pub fn claim(&self, reason: ExitReason) { + let mut inner = self.inner(); + inner.reason.get_or_insert(reason); + } + + /// Take ownership of the drain, and with it the reason the app is ending. + /// + /// Claiming the reason and moving out of `Idle` is one step on purpose. Split apart, a Quit + /// that claimed first could still be overtaken by an update that started the drain, and the app + /// would restart under a user who asked it to stop. `fallback` is only used when nothing has + /// claimed a reason yet. + /// + /// While a runtime is being started or stopped the answer is "not yet": the reason is recorded + /// and the drain is handed to whichever of [`ExitCoordinator::finish_spawn`] or + /// [`ExitCoordinator::finish_stop`] is holding the phase. + pub fn claim_drain(&self, fallback: ExitReason) -> Option { + let mut inner = self.inner(); + let prior_wanted = inner.wanted; + // A quit or an update is on its way, whoever ends up running the drain: the runtime it + // stops is not one to bring back. + inner.wanted = false; + match inner.phase { + ExitPhase::Idle => { + let reason = *inner.reason.get_or_insert(fallback); + remember_restart_intent(&mut inner, reason, prior_wanted); + inner.phase = ExitPhase::Draining; + Some(reason) + } + ExitPhase::Spawning | ExitPhase::Stopping => { + let reason = *inner.reason.get_or_insert(fallback); + remember_restart_intent(&mut inner, reason, prior_wanted); + inner.deferred = true; + None + } + // A failed drain is a terminal failure, not work in flight, and retrying it is the + // recovery: the update stayed pending, so the next attempt runs the stop again. Without + // this the first refusal would be permanent until the app was restarted by hand — which + // is the one thing a user with a runtime that would not stop cannot easily do. + ExitPhase::DrainFailed | ExitPhase::OwnershipUnknown => { + let reason = *inner.reason.get_or_insert(fallback); + remember_restart_intent(&mut inner, reason, prior_wanted); + inner.phase = ExitPhase::Draining; + Some(reason) + } + ExitPhase::Draining | ExitPhase::Drained => None, + } + } + + /// Record what the drain established. A failure is not a drain. + pub fn finish_drain(&self, verdict: DrainVerdict) { + self.inner().phase = match verdict { + DrainVerdict::Drained => ExitPhase::Drained, + DrainVerdict::Failed => ExitPhase::DrainFailed, + DrainVerdict::OwnershipUnknown => ExitPhase::OwnershipUnknown, + }; + } + + /// Give a coordinated restart that is not going to happen back to a running app. + /// + /// An update drains before it installs. When the install then fails, or the drain itself did, + /// the drain's phase used to be the end of the road: `Drained` is terminal, so no runtime could + /// be started again and a bare window close quit the app. This returns the app to `Idle` with + /// no claimed reason and restores the runtime intent the update temporarily suppressed. A + /// retry requested while the drain was in flight wins too: aborting an older update must not + /// overwrite newer user intent. It touches nothing unless an update's restart holds the phase: + /// a quit is never aborted, and a + /// drain still running belongs to whoever runs it. Returns the phase and restored intent. + pub fn abort_restart(&self) -> Option { + let mut inner = self.inner(); + let left = inner.phase; + let restart = inner.reason == Some(ExitReason::CoordinatedRestart); + let settled = matches!( + left, + ExitPhase::Drained | ExitPhase::DrainFailed | ExitPhase::OwnershipUnknown + ); + if !restart || !settled { + return None; + } + inner.phase = ExitPhase::Idle; + inner.reason = None; + inner.deferred = false; + // `resume` can arrive after the update captured its original intent. Preserve that newer + // request as well as the older snapshot; otherwise the abort races the startup retry and + // can leave a runtime stopped even though the person just asked for it. + let runtime_was_wanted = inner.wanted || inner.restart_wanted.take().unwrap_or(false); + inner.wanted = runtime_was_wanted; + Some(AbortedRestart { + phase: left, + runtime_was_wanted, + }) + } + + /// What the runtime supervisor reads before it acts. + pub fn supervision(&self) -> Supervision { + let inner = self.inner(); + Supervision { + phase: inner.phase, + wanted: inner.wanted, + reason_set: inner.reason.is_some(), + } + } + + pub fn supervision_allowed(&self) -> bool { + self.supervision().allowed() + } + + /// A person asked for a runtime again (the startup page's retry). + pub fn resume(&self) { + let mut inner = self.inner(); + inner.wanted = true; + // A pending update snapshot must learn about the request too. A retried install that finds + // the drain already settled clears `wanted` again without replacing the snapshot, so a + // retry recorded only in `wanted` would be lost when that install fails and aborts. + if let Some(snapshot) = inner.restart_wanted.as_mut() { + *snapshot = true; + } + } + + /// Reserve the right to start a runtime. False once something else owns the phase. + /// + /// The reservation exists instead of holding the lock across the spawn. Holding it would make + /// the main thread's exit handler wait on process creation, so a wedged spawn would be a Quit + /// that never responds. An exit arriving in between is deferred rather than lost — which is the + /// thing that must not happen, because a quit that reads "we own nothing" leaves the child it + /// just missed running forever. + pub fn begin_spawn(&self) -> bool { + self.begin(ExitPhase::Spawning) + } + + /// Release the spawn reservation, handing back a reason that arrived meanwhile. + pub fn finish_spawn(&self) -> Option { + self.finish(ExitPhase::Spawning) + } + + /// Reserve the runtime for a stop that does not end the app. + /// + /// A stop that takes the phase also stops the app wanting a runtime, in the same step, so the + /// runtime's exit can never reach the supervisor first. One that cannot take it stops nothing + /// and changes nothing: the runtime it would have stopped stays supervised. + pub fn begin_stop(&self) -> bool { + let mut inner = self.inner(); + if inner.phase != ExitPhase::Idle { + return false; + } + inner.phase = ExitPhase::Stopping; + inner.wanted = false; + true + } + + /// Release the stop, handing back a reason that arrived meanwhile. + pub fn finish_stop(&self) -> Option { + self.finish(ExitPhase::Stopping) + } + + fn begin(&self, phase: ExitPhase) -> bool { + let mut inner = self.inner(); + if inner.phase != ExitPhase::Idle { + return false; + } + inner.phase = phase; + true + } + + fn finish(&self, phase: ExitPhase) -> Option { + let mut inner = self.inner(); + if inner.phase != phase { + return None; + } + if inner.deferred { + let reason = *inner.reason.get_or_insert(ExitReason::UserQuit); + inner.phase = ExitPhase::Draining; + inner.deferred = false; + return Some(reason); + } + inner.phase = ExitPhase::Idle; + None + } + + #[cfg(test)] + fn phase(&self) -> ExitPhase { + self.inner().phase + } + + #[cfg(test)] + fn hides_to_tray(&self) -> bool { + self.inner().hides_to_tray + } +} + +impl Default for ExitCoordinator { + fn default() -> Self { + Self::new() + } +} + +/// The platform's quit gesture, and what closing the window means. +/// +/// It is not a request to end. D2 makes it mean the same thing on all three platforms: hide where +/// there is a tray to come back from, and a graceful quit where there is not. Closing the window +/// arrives here too: one decision point, so the two gestures cannot drift apart. +pub fn gesture(app: &AppHandle) { + let Some(coordinator) = app.try_state::() else { + return; + }; + match coordinator.decision() { + ExitDecision::Hide => hide_windows(app), + ExitDecision::Drain(reason) => start_drain(app, reason), + ExitDecision::Wait | ExitDecision::Proceed | ExitDecision::Refuse => {} + } +} + +/// Ask the app to end for a stated reason. This is the only way the shell ends itself. +pub fn request(app: &AppHandle, reason: ExitReason) { + if let Some(coordinator) = app.try_state::() { + coordinator.claim(reason); + } + app.exit(0); +} + +/// Stop the runtime without ending the app: the tray's Stop item. +/// +/// It takes the same phase the quit path takes, so pressing Stop twice, or Stop and then Quit, or +/// Stop during an update, is one execution rather than two racing over one child. Unlike a quit it +/// returns the coordinator to idle, because the app is still running and may start a runtime again. +pub fn request_stop(app: &AppHandle) { + let app = app.clone(); + tauri::async_runtime::spawn(async move { + let Some(claimed) = app + .try_state::() + .map(|coordinator| coordinator.begin_stop()) + else { + return; + }; + if !claimed { + return; + } + let verdict = drain_current(&app).await; + if verdict == DrainVerdict::Drained { + if let Some(state) = app.try_state::() { + state.release(); + } + crate::tray::set_owned(&app, false); + } else { + crate::logging::log_once("the runtime could not be stopped", verdict.describe()); + } + let deferred = app + .try_state::() + .and_then(|coordinator| coordinator.finish_stop()); + if let Some(reason) = deferred { + drain_now(&app, reason); + } + }); +} + +impl DrainVerdict { + pub fn describe(self) -> &'static str { + match self { + Self::Drained => "the runtime is stopped", + Self::Failed => "the stop was refused or the runtime still answered", + Self::OwnershipUnknown => "the running proxy could not be identified", + } + } +} + +/// Handle `RunEvent::ExitRequested`. +pub fn on_exit_requested(app: &AppHandle, code: Option, api: &ExitRequestApi) { + // `AppHandle::restart` documents that `prevent_exit` is ignored for its own exit code, so a + // restart cannot be held here even to drain. The update path therefore drains before it + // restarts, and this branch only records the reason so nothing reads the restart as a quit. + if code == Some(tauri::RESTART_EXIT_CODE) { + if let Some(coordinator) = app.try_state::() { + coordinator.claim(ExitReason::CoordinatedRestart); + } + return; + } + let Some(coordinator) = app.try_state::() else { + return; + }; + match coordinator.decision() { + ExitDecision::Hide => { + api.prevent_exit(); + hide_windows(app); + } + ExitDecision::Wait => api.prevent_exit(), + ExitDecision::Refuse => api.prevent_exit(), + ExitDecision::Drain(reason) => { + api.prevent_exit(); + start_drain(app, reason); + } + ExitDecision::Proceed => {} + } +} + +/// Drain and then ask to end again. +pub fn start_drain(app: &AppHandle, reason: ExitReason) { + let Some(coordinator) = app.try_state::() else { + return; + }; + let Some(reason) = coordinator.claim_drain(reason) else { + return; + }; + drain_now(app, reason); +} + +/// Run the drain for a reason the coordinator has already moved to draining for. +pub fn drain_now(app: &AppHandle, reason: ExitReason) { + let app = app.clone(); + tauri::async_runtime::spawn(async move { + finish_and_exit_after(&app, reason).await; + }); +} + +async fn finish_and_exit_after(app: &AppHandle, reason: ExitReason) { + let verdict = drain_current(app).await; + if let Some(coordinator) = app.try_state::() { + coordinator.finish_drain(verdict); + } + match (reason, verdict) { + (ExitReason::UserQuit, _) => app.exit(0), + (ExitReason::CoordinatedRestart, DrainVerdict::Drained) => { + app.restart(); + } + (ExitReason::CoordinatedRestart, _) => { + crate::logging::log_once("the update restart was refused", verdict.describe()); + } + } +} + +/// Prepare for an update's restart: confirm ownership, drain, and confirm the child is gone. +/// +/// This is awaited rather than fired and forgotten, because the pinned updater's Windows install +/// ends the process itself. A restart asked for after `install` returns is a restart that never +/// happens there, so the stop has to be finished before the installer is started at all. +pub async fn prepare_restart(app: &AppHandle) -> RestartReadiness { + let Some(reason) = app + .try_state::() + .and_then(|coordinator| coordinator.claim_drain(ExitReason::CoordinatedRestart)) + else { + return RestartReadiness::Busy; + }; + if reason != ExitReason::CoordinatedRestart { + // A quit claimed the exit first. It owns the drain now, and the update does not install + // into an app that is on its way out. + drain_now(app, reason); + return RestartReadiness::Busy; + } + let verdict = drain_current(app).await; + if let Some(coordinator) = app.try_state::() { + coordinator.finish_drain(verdict); + } + match verdict { + DrainVerdict::Drained => RestartReadiness::Ready, + DrainVerdict::Failed => RestartReadiness::DrainFailed, + DrainVerdict::OwnershipUnknown => RestartReadiness::OwnershipUnknown, + } +} + +/// Come back, once the installer has finished and returned. +pub fn complete_restart(app: &AppHandle) -> ! { + app.restart() +} + +/// Stop the runtime this app owns and confirm it is gone. +/// +/// Ownership is re-established here rather than read off a flag. A flag set when the child was +/// spawned says nothing about the process answering the endpoint now: the child can have exited and +/// a service can have taken the port back. Sending an owner's stop to that listener is sending it +/// to somebody else's runtime, so the pid is checked first and a listener that cannot be identified +/// is left alone. +/// +/// The stop itself is the bundled `ocx stop` (D4), not a management call from inside this process. +/// The CLI's stop owns the receipt-backed teardown, the drain, the Windows respawn verification and +/// the client-configuration restore, and an in-process endpoint cannot own its own teardown: launchd +/// and systemd can terminate the request handler during self-unload. Nothing kills the child either +/// — the original path did, with `CommandChild::kill()`, a SIGKILL on Unix that cut off exactly the +/// work the CLI's stop exists to finish. +pub async fn drain_current(app: &AppHandle) -> DrainVerdict { + let Some((proxy, child_pid, watch)) = app + .try_state::() + .map(|state| (state.proxy(), state.child_pid(), state.watch.clone())) + else { + return DrainVerdict::OwnershipUnknown; + }; + let Some(proxy) = proxy else { + // Nothing resolved, so there is nothing of ours listening anywhere. + return DrainVerdict::Drained; + }; + let Some(child_pid) = child_pid else { + // This app never started a runtime, so it does not stop one. + return DrainVerdict::Drained; + }; + match confirm(&proxy, child_pid, &watch).await { + Ownership::Gone => DrainVerdict::Drained, + Ownership::Foreign => DrainVerdict::Drained, + Ownership::Unknown => DrainVerdict::OwnershipUnknown, + Ownership::Ours => { + let deadline = Instant::now() + runtime_stop::DEADLINE; + let result = runtime_stop::run(app, deadline).await; + if result.is_stopped() { + DrainVerdict::Drained + } else { + // The CLI's own outcome and exit code, carried rather than reinterpreted. A stop + // that did not end in exit 0 with the runtime down is a stop that did not happen. + crate::logging::log_once("the bundled stop did not complete", &result.describe()); + DrainVerdict::Failed + } + } + } +} + +enum Ownership { + /// The process answering is the child this app started. + Ours, + /// Something else holds the port. + Foreign, + /// Nothing is listening, and the child has reported its own exit. + Gone, + /// The listener could not be identified. + Unknown, +} + +async fn confirm(proxy: &ProxyClient, child_pid: u32, watch: &sidecar::SidecarWatch) -> Ownership { + match proxy.identify().await { + Ok(identity) if identity.pid == child_pid => Ownership::Ours, + Ok(_) => Ownership::Foreign, + Err(error) if error.is_unreachable() => { + // Nothing is listening. That is only proof the child is gone if the child said so. + if watch.exit().is_some() { + Ownership::Gone + } else { + Ownership::Unknown + } + } + Err(_) => Ownership::Unknown, + } +} + +fn hide_windows(app: &AppHandle) { + for window in app.webview_windows().values() { + window::hide(window); + } +} + +#[cfg(test)] +mod tests { + use super::{ + decide, AbortedRestart, DrainVerdict, ExitCoordinator, ExitDecision, ExitPhase, ExitReason, + RestartReadiness, Supervision, + }; + use crate::tray_availability::TrayAvailability; + + #[test] + fn stop_quit_and_update_stop_wanting_a_runtime_and_only_a_resume_restores_it() { + let coordinator = ExitCoordinator::new(); + assert!(coordinator.supervision_allowed()); + assert!(coordinator.begin_stop()); + assert!(!coordinator.supervision().wanted); + assert_eq!(coordinator.finish_stop(), None); + // Back to Idle and still not wanted: the stop was the person's intent, not a phase. + assert_eq!(coordinator.phase(), ExitPhase::Idle); + assert!(!coordinator.supervision_allowed()); + coordinator.resume(); + assert!(coordinator.supervision_allowed()); + + for reason in [ExitReason::UserQuit, ExitReason::CoordinatedRestart] { + let coordinator = ExitCoordinator::new(); + assert_eq!(coordinator.claim_drain(reason), Some(reason)); + assert!(!coordinator.supervision().wanted); + } + // A stop that could not take the phase stopped nothing, so it changes nothing either. + let coordinator = ExitCoordinator::new(); + assert!(coordinator.begin_spawn()); + assert!(!coordinator.begin_stop()); + assert!(coordinator.supervision().wanted); + } + + #[test] + fn supervision_is_allowed_only_idle_wanted_and_with_no_ending_claimed() { + for phase in [ + ExitPhase::Spawning, + ExitPhase::Stopping, + ExitPhase::Draining, + ExitPhase::Drained, + ExitPhase::DrainFailed, + ExitPhase::OwnershipUnknown, + ] { + let supervision = Supervision { + phase, + wanted: true, + reason_set: false, + }; + assert!(!supervision.allowed(), "{phase:?}"); + } + let idle = Supervision { + phase: ExitPhase::Idle, + wanted: true, + reason_set: false, + }; + assert!(idle.allowed()); + assert!(!Supervision { + wanted: false, + ..idle + } + .allowed()); + assert!(!Supervision { + reason_set: true, + ..idle + } + .allowed()); + let coordinator = ExitCoordinator::new(); + coordinator.claim(ExitReason::UserQuit); + assert!(!coordinator.supervision_allowed()); + } + + #[test] + fn a_failed_update_returns_the_app_to_a_running_runtime() { + for verdict in [ + DrainVerdict::Drained, + DrainVerdict::Failed, + DrainVerdict::OwnershipUnknown, + ] { + let coordinator = ExitCoordinator::new(); + coordinator.set_tray(TrayAvailability::Available); + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + Some(ExitReason::CoordinatedRestart) + ); + coordinator.finish_drain(verdict); + let left = coordinator.phase(); + assert_eq!( + coordinator.abort_restart(), + Some(AbortedRestart { + phase: left, + runtime_was_wanted: true, + }) + ); + assert_eq!(coordinator.phase(), ExitPhase::Idle); + // A bare close hides again instead of quitting out of a terminal phase. + assert_eq!(coordinator.decision(), ExitDecision::Hide); + assert!(coordinator.supervision_allowed()); + assert!(coordinator.begin_spawn()); + } + } + + #[test] + fn a_failed_update_preserves_a_completed_tray_stop() { + let coordinator = ExitCoordinator::new(); + coordinator.set_tray(TrayAvailability::Available); + assert!(coordinator.begin_stop()); + assert_eq!(coordinator.finish_stop(), None); + assert!(!coordinator.supervision().wanted); + + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + Some(ExitReason::CoordinatedRestart) + ); + coordinator.finish_drain(DrainVerdict::Drained); + assert_eq!( + coordinator.abort_restart(), + Some(AbortedRestart { + phase: ExitPhase::Drained, + runtime_was_wanted: false, + }) + ); + assert_eq!(coordinator.phase(), ExitPhase::Idle); + assert_eq!(coordinator.decision(), ExitDecision::Hide); + assert!(!coordinator.supervision_allowed()); + } + + #[test] + fn a_startup_retry_during_an_update_drain_is_not_overwritten_by_abort() { + let coordinator = ExitCoordinator::new(); + coordinator.set_tray(TrayAvailability::Available); + assert!(coordinator.begin_stop()); + assert_eq!(coordinator.finish_stop(), None); + assert!(!coordinator.supervision().wanted); + + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + Some(ExitReason::CoordinatedRestart) + ); + coordinator.resume(); + // The ending claim still prevents supervision until the failed update is handed back. + assert!(!coordinator.supervision_allowed()); + coordinator.finish_drain(DrainVerdict::Drained); + assert_eq!( + coordinator.abort_restart(), + Some(AbortedRestart { + phase: ExitPhase::Drained, + runtime_was_wanted: true, + }) + ); + assert!(coordinator.supervision_allowed()); + } + + #[test] + fn a_retry_between_update_attempts_survives_the_second_claim() { + let coordinator = ExitCoordinator::new(); + coordinator.set_tray(TrayAvailability::Available); + assert!(coordinator.begin_stop()); + assert_eq!(coordinator.finish_stop(), None); + + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + Some(ExitReason::CoordinatedRestart) + ); + coordinator.finish_drain(DrainVerdict::Drained); + coordinator.resume(); + // The next install attempt finds the drain settled and clears `wanted` again. + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + None + ); + assert_eq!( + coordinator.abort_restart(), + Some(AbortedRestart { + phase: ExitPhase::Drained, + runtime_was_wanted: true, + }) + ); + assert!(coordinator.supervision_allowed()); + } + + #[test] + fn a_quit_or_a_drain_in_flight_is_never_aborted() { + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::UserQuit); + coordinator.finish_drain(DrainVerdict::Drained); + assert_eq!(coordinator.abort_restart(), None); + assert_eq!(coordinator.phase(), ExitPhase::Drained); + assert_eq!(coordinator.decision(), ExitDecision::Proceed); + // A restart still draining belongs to whoever runs it. + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::CoordinatedRestart); + assert_eq!(coordinator.abort_restart(), None); + assert_eq!(coordinator.phase(), ExitPhase::Draining); + } + + #[test] + fn a_bare_gesture_hides_when_there_is_a_tray_to_come_back_from() { + assert_eq!(decide(ExitPhase::Idle, None, true), ExitDecision::Hide); + } + + #[test] + fn a_bare_gesture_quits_through_the_drain_when_there_is_no_tray() { + assert_eq!( + decide(ExitPhase::Idle, None, false), + ExitDecision::Drain(ExitReason::UserQuit) + ); + } + + #[test] + fn an_explicit_quit_drains_even_though_a_tray_exists() { + assert_eq!( + decide(ExitPhase::Idle, Some(ExitReason::UserQuit), true), + ExitDecision::Drain(ExitReason::UserQuit) + ); + } + + #[test] + fn every_in_flight_phase_holds_the_exit() { + for phase in [ + ExitPhase::Spawning, + ExitPhase::Stopping, + ExitPhase::Draining, + ] { + assert_eq!( + decide(phase, Some(ExitReason::UserQuit), true), + ExitDecision::Wait + ); + assert_eq!(decide(phase, None, false), ExitDecision::Wait); + } + } + + #[test] + fn only_a_reported_drain_lets_the_process_end() { + assert_eq!( + decide(ExitPhase::Drained, Some(ExitReason::UserQuit), true), + ExitDecision::Proceed + ); + } + + #[test] + fn a_quit_tolerates_a_failed_drain_and_a_restart_refuses_it() { + for phase in [ExitPhase::DrainFailed, ExitPhase::OwnershipUnknown] { + assert_eq!( + decide(phase, Some(ExitReason::UserQuit), true), + ExitDecision::Proceed + ); + assert_eq!( + decide(phase, Some(ExitReason::CoordinatedRestart), true), + ExitDecision::Refuse + ); + } + } + + #[test] + fn a_failed_drain_is_not_recorded_as_a_drain() { + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.finish_drain(DrainVerdict::Failed); + assert_eq!(coordinator.phase(), ExitPhase::DrainFailed); + // The restart that asked for it does not get to proceed on that. + assert_eq!(coordinator.decision(), ExitDecision::Refuse); + } + + #[test] + fn an_unidentified_runtime_is_its_own_state() { + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.finish_drain(DrainVerdict::OwnershipUnknown); + assert_eq!(coordinator.phase(), ExitPhase::OwnershipUnknown); + assert_eq!(coordinator.decision(), ExitDecision::Refuse); + } + + #[test] + fn a_refused_restart_can_be_tried_again() { + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.finish_drain(DrainVerdict::Failed); + // The update stayed pending, so pressing Install again runs the stop again rather than + // finding the app permanently unable to try. + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + Some(ExitReason::CoordinatedRestart) + ); + assert_eq!(coordinator.phase(), ExitPhase::Draining); + coordinator.finish_drain(DrainVerdict::Drained); + assert_eq!(coordinator.decision(), ExitDecision::Proceed); + } + + #[test] + fn a_successful_drain_is_not_re_entered() { + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::UserQuit); + coordinator.finish_drain(DrainVerdict::Drained); + assert_eq!(coordinator.claim_drain(ExitReason::UserQuit), None); + } + + #[test] + fn the_first_claimed_reason_wins_the_drain() { + let coordinator = ExitCoordinator::new(); + coordinator.claim(ExitReason::UserQuit); + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + Some(ExitReason::UserQuit) + ); + assert_eq!(coordinator.phase(), ExitPhase::Draining); + } + + #[test] + fn only_one_caller_owns_the_drain() { + let coordinator = ExitCoordinator::new(); + assert_eq!( + coordinator.claim_drain(ExitReason::UserQuit), + Some(ExitReason::UserQuit) + ); + assert_eq!(coordinator.claim_drain(ExitReason::UserQuit), None); + coordinator.finish_drain(DrainVerdict::Drained); + assert_eq!(coordinator.phase(), ExitPhase::Drained); + assert_eq!(coordinator.claim_drain(ExitReason::UserQuit), None); + } + + #[test] + fn a_stop_and_a_spawn_both_hold_the_runtime_alone() { + for begin in [ExitPhase::Spawning, ExitPhase::Stopping] { + let coordinator = ExitCoordinator::new(); + let started = match begin { + ExitPhase::Spawning => coordinator.begin_spawn(), + _ => coordinator.begin_stop(), + }; + assert!(started); + assert!(!coordinator.begin_spawn()); + assert!(!coordinator.begin_stop()); + assert_eq!(coordinator.decision(), ExitDecision::Wait); + } + } + + #[test] + fn a_quit_during_a_stop_is_deferred_rather_than_lost() { + let coordinator = ExitCoordinator::new(); + assert!(coordinator.begin_stop()); + assert_eq!(coordinator.claim_drain(ExitReason::UserQuit), None); + assert_eq!(coordinator.phase(), ExitPhase::Stopping); + assert_eq!(coordinator.finish_stop(), Some(ExitReason::UserQuit)); + assert_eq!(coordinator.phase(), ExitPhase::Draining); + assert_eq!(coordinator.finish_stop(), None); + } + + #[test] + fn a_quit_during_a_spawn_is_deferred_rather_than_lost() { + let coordinator = ExitCoordinator::new(); + assert!(coordinator.begin_spawn()); + assert_eq!(coordinator.decision(), ExitDecision::Wait); + assert_eq!(coordinator.claim_drain(ExitReason::UserQuit), None); + assert_eq!(coordinator.finish_spawn(), Some(ExitReason::UserQuit)); + assert_eq!(coordinator.phase(), ExitPhase::Draining); + } + + #[test] + fn a_deferred_update_restart_keeps_its_own_reason() { + let coordinator = ExitCoordinator::new(); + assert!(coordinator.begin_spawn()); + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + None + ); + assert_eq!( + coordinator.finish_spawn(), + Some(ExitReason::CoordinatedRestart) + ); + } + + #[test] + fn an_undrained_runtime_is_never_reported_as_ready_to_install_over() { + assert_eq!(RestartReadiness::Ready.describe(), "the runtime is stopped"); + for refused in [ + RestartReadiness::DrainFailed, + RestartReadiness::OwnershipUnknown, + RestartReadiness::Busy, + ] { + assert_ne!(refused, RestartReadiness::Ready); + assert!(!refused.describe().is_empty()); + } + } + + #[test] + fn the_tray_verdict_replaces_the_platform_assumption() { + let coordinator = ExitCoordinator::new(); + assert_eq!( + coordinator.hides_to_tray(), + TrayAvailability::assumed().hides_to_tray() + ); + coordinator.set_tray(TrayAvailability::Unavailable); + assert!(!coordinator.hides_to_tray()); + assert_eq!( + coordinator.decision(), + ExitDecision::Drain(ExitReason::UserQuit) + ); + coordinator.set_tray(TrayAvailability::Available); + assert_eq!(coordinator.decision(), ExitDecision::Hide); + } +} diff --git a/desktop/src-tauri/src/first_run.rs b/desktop/src-tauri/src/first_run.rs new file mode 100644 index 0000000000..3fc7616b93 --- /dev/null +++ b/desktop/src-tauri/src/first_run.rs @@ -0,0 +1,111 @@ +use std::fs; +use tauri::{AppHandle, Manager}; +use tauri_plugin_autostart::ManagerExt; + +/// Marker file recording that the one-time Start at Login default has already been applied. +const MARKER: &str = "start-at-login-claimed"; + +/// Marker file recording that the login item names the launch-origin argument. +const ORIGIN_MARKER: &str = "start-at-login-origin-flag"; + +/// What the one-time Start at Login decision did on this launch. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum StartAtLogin { + /// This installation had already decided, so whatever the user set is left alone. + AlreadyDecided, + /// Turned on for the first time on this installation. + Enabled, + /// The default could not be applied. The app still starts and the tray item still toggles it. + Unavailable, +} + +impl StartAtLogin { + pub fn describe(self) -> &'static str { + match self { + Self::AlreadyDecided => "already decided on this installation, left as the user set it", + Self::Enabled => "turned on for this installation", + Self::Unavailable => "could not be registered; the tray item still toggles it", + } + } +} + +/// Turn Start at Login on once, the first time this installation runs. +/// +/// A menu bar app that is not running has no menu bar item. Leaving autostart off by default +/// therefore means that after the next reboot an installed app is simply absent, with nothing on +/// screen to explain why — which is not a neutral default for an app whose main surface *is* the +/// menu bar. +/// +/// This runs exactly once per installation. The marker is written **before** the login item is +/// touched, and is never removed, so a user who turns Start at Login back off keeps it off: the +/// next launch sees the marker and does nothing. Writing afterwards instead would mean that a +/// failed or partial enable retries on every launch, and would eventually flip the setting back on +/// under a user who had deliberately turned it off in between. +/// +/// No failure stops the app. Not being able to write a marker or register a login item is not a +/// reason to refuse to start, and the user can still toggle the menu item. What has changed is that +/// the outcome is returned rather than swallowed: D7's startup sequence reports this decision as +/// one of its named states, so a registration that did not happen is visible instead of silent. +pub fn apply_start_at_login_default(app: &AppHandle) -> StartAtLogin { + let Ok(dir) = app.path().app_config_dir() else { + return StartAtLogin::Unavailable; + }; + let marker = dir.join(MARKER); + if marker.exists() { + return StartAtLogin::AlreadyDecided; + } + if fs::create_dir_all(&dir).is_err() { + return StartAtLogin::Unavailable; + } + if fs::write(&marker, b"").is_err() { + return StartAtLogin::Unavailable; + } + if app.autolaunch().is_enabled().unwrap_or(false) { + return StartAtLogin::AlreadyDecided; + } + match app.autolaunch().enable() { + Ok(()) => StartAtLogin::Enabled, + Err(_) => StartAtLogin::Unavailable, + } +} + +/// Rewrite an existing login item so a launch from it can be recognised as one. +/// +/// The autostart entry is written once, carrying whatever arguments the plugin was configured with +/// at the time. An installation that registered before the launch-origin argument existed has an +/// entry without it, and a bare launch carries nothing else that distinguishes login from manual — +/// so D7's hidden login start would quietly never happen for exactly the users who already had +/// autostart on. Re-registering rewrites the entry with the current arguments. +/// +/// It runs once, behind its own marker, and only where autostart is already on. It never turns the +/// setting on and never turns it off; the worst case is an entry that keeps its old arguments and a +/// login launch that shows its window, which is the visible failure rather than the silent one. +pub fn adopt_launch_origin_argument(app: &AppHandle) { + let Ok(dir) = app.path().app_config_dir() else { + return; + }; + let claimed = dir.join(ORIGIN_MARKER); + if claimed.exists() { + return; + } + if fs::create_dir_all(&dir).is_err() { + return; + } + // Unlike the default above, the marker is written *after* the work, and that difference is + // deliberate. Writing first exists there to stop a failed enable from flipping a setting the + // user turned off. Here there is no setting to flip: the rewrite only ever runs on an entry + // that is already enabled, so retrying after a transient registry, LaunchAgent or desktop-file + // error is free — and claiming the marker first would suppress the migration permanently and + // leave a login launch showing its window forever. + match app.autolaunch().is_enabled() { + Ok(true) => { + if app.autolaunch().enable().is_err() { + return; + } + } + // Nothing registered to migrate. A later enable writes the current arguments anyway. + Ok(false) => {} + Err(_) => return, + } + let _ = fs::write(&claimed, b""); +} diff --git a/desktop/src-tauri/src/formatting.rs b/desktop/src-tauri/src/formatting.rs index 5f940b68fe..1d0406257a 100644 --- a/desktop/src-tauri/src/formatting.rs +++ b/desktop/src-tauri/src/formatting.rs @@ -47,11 +47,12 @@ fn abbreviate_float(value: f64, integer: bool) -> String { 2 }; let rendered = format!("{scaled:.decimals$}"); - return format!( - "{}{}", - rendered.trim_end_matches('0').trim_end_matches('.'), - suffix - ); + let rendered = if rendered.contains('.') { + rendered.trim_end_matches('0').trim_end_matches('.') + } else { + &rendered + }; + return format!("{rendered}{suffix}"); } } format!("{value:.0}") @@ -76,4 +77,33 @@ mod tests { assert_eq!(cost(Some(12.345)), "$12.35"); assert_eq!(cost(Some(1_234.0)), "$1.23K"); } + + #[test] + fn abbreviations_preserve_integer_trailing_zeros() { + for (unit, suffix) in [ + (1_000, "K"), + (1_000_000, "M"), + (1_000_000_000, "B"), + (1_000_000_000_000, "T"), + ] { + for multiple in [10, 100, 110] { + let value = unit * multiple; + let expected = format!("{multiple}{suffix}"); + assert_eq!(tokens(Some(value)), expected); + assert_eq!(count(Some(value)), expected); + assert_eq!(cost(Some(value as f64)), format!("${expected}")); + } + } + assert_eq!(tokens(Some(9_600_000)), "10M"); + assert_eq!(count(Some(99_960_000)), "100M"); + } + + #[test] + fn fractional_trailing_zeros_are_still_trimmed() { + assert_eq!(count(Some(1_000_000)), "1M"); + assert_eq!(count(Some(1_200_000)), "1.2M"); + assert_eq!(count(Some(1_234_000)), "1.23M"); + assert_eq!(count(Some(12_340_000)), "12.3M"); + assert_eq!(cost(Some(1_200.0)), "$1.2K"); + } } diff --git a/desktop/src-tauri/src/identity.rs b/desktop/src-tauri/src/identity.rs new file mode 100644 index 0000000000..9deb1c32e0 --- /dev/null +++ b/desktop/src-tauri/src/identity.rs @@ -0,0 +1,111 @@ +//! This installation's own identity. +//! +//! The shared service install state records who owns the running proxy, and the claim names the +//! owning *installation* rather than the user or the machine. So the app has to hold a value of its +//! own to compare against, which is this: an opaque id written once into the app's config directory +//! and never rewritten. +//! +//! Two records rather than one is D3, and its cost is recorded there. An id stored only in the +//! shared record would be whoever wrote it last, which gives a reinstalled app no way to tell its +//! own prior consent from another installation's. The price is that a reinstall which keeps this +//! directory keeps its consent, and one that loses it has to ask again. + +use std::{ + fs, + io::{ErrorKind, Write}, + path::Path, +}; +use tauri::{AppHandle, Manager}; +use uuid::Uuid; + +/// The file holding this installation's id, in the app's own config directory. +const FILE: &str = "install-id"; + +pub fn install_id(app: &AppHandle) -> Option { + let directory = app.path().app_config_dir().ok()?; + install_id_in(&directory) +} + +/// Read this installation's id, minting it the first time. +/// +/// The mint is exclusive and the value is read back afterwards, so two launches racing each other +/// both answer to the id that won rather than to two different ones. Two ids would be two +/// installations as far as the recorded claim is concerned, and the second would find a claim that +/// is not its own and ask again for consent the user had already given. +pub fn install_id_in(directory: &Path) -> Option { + let path = directory.join(FILE); + if let Some(existing) = read(&path) { + return Some(existing); + } + fs::create_dir_all(directory).ok()?; + match mint(&path) { + Ok(()) => {} + // The file is there and says nothing: a blank or truncated write from an interrupted first + // run. An empty id matches nothing, so every comparison against the recorded claim would + // quietly be false and the app would ask for consent it already had. Replace it. + Err(ErrorKind::AlreadyExists) => { + if read(&path).is_none() { + fs::write(&path, Uuid::new_v4().to_string()).ok()?; + } + } + Err(_) => return None, + } + read(&path) +} + +fn mint(path: &Path) -> Result<(), ErrorKind> { + fs::OpenOptions::new() + .write(true) + .create_new(true) + .open(path) + .and_then(|mut file| file.write_all(Uuid::new_v4().to_string().as_bytes())) + .map_err(|error| error.kind()) +} + +fn read(path: &Path) -> Option { + let value = fs::read_to_string(path).ok()?; + let trimmed = value.trim(); + (!trimmed.is_empty()).then(|| trimmed.to_owned()) +} + +#[cfg(test)] +mod tests { + use super::{install_id_in, FILE}; + use std::fs; + + fn scratch(name: &str) -> std::path::PathBuf { + let directory = + std::env::temp_dir().join(format!("ocx-identity-{name}-{}", std::process::id())); + let _ = fs::remove_dir_all(&directory); + directory + } + + #[test] + fn the_id_is_minted_once_and_then_read_back() { + let directory = scratch("mint"); + let first = install_id_in(&directory).expect("an id"); + assert!(!first.is_empty()); + assert_eq!(install_id_in(&directory).as_deref(), Some(first.as_str())); + let _ = fs::remove_dir_all(&directory); + } + + #[test] + fn two_installations_do_not_share_an_id() { + let one = scratch("one"); + let two = scratch("two"); + assert_ne!(install_id_in(&one), install_id_in(&two)); + let _ = fs::remove_dir_all(&one); + let _ = fs::remove_dir_all(&two); + } + + #[test] + fn a_blank_record_is_replaced_rather_than_answered_with() { + let directory = scratch("blank"); + fs::create_dir_all(&directory).unwrap(); + fs::write(directory.join(FILE), " \n").unwrap(); + let minted = install_id_in(&directory).expect("an id"); + assert!(!minted.trim().is_empty()); + assert_eq!(install_id_in(&directory).as_deref(), Some(minted.as_str())); + let _ = fs::remove_dir_all(&directory); + } +} diff --git a/desktop/src-tauri/src/lib.rs b/desktop/src-tauri/src/lib.rs index aa688daca9..b8e6e9fe7d 100644 --- a/desktop/src-tauri/src/lib.rs +++ b/desktop/src-tauri/src/lib.rs @@ -1,46 +1,162 @@ mod auth; -mod discovery; +mod claim; +#[cfg(target_os = "macos")] +mod companion_query; +mod companion_usage; +mod endpoint; +mod exit; +mod first_run; mod formatting; +mod identity; mod logging; +// macOS only: it exists to replace one item in a menu no other platform installs. Compiling it +// elsewhere would leave its contents unreachable, which -D warnings rejects. +#[cfg(target_os = "macos")] +mod menu; +#[cfg(target_os = "macos")] +mod native_tray_accounts; +#[cfg(target_os = "macos")] +mod native_tray_data; +#[cfg(target_os = "macos")] +mod native_tray_snapshot; +mod ownership; +#[cfg(not(target_os = "macos"))] +mod popup; +#[cfg(target_os = "macos")] +#[path = "native_tray.rs"] +mod popup; +#[cfg(target_os = "macos")] +mod provider_icons; +// The macOS build selects native_tray.rs as the popup module; compile the portable popup +// module's tests on macOS too so its navigation rules run on the maintainers' platform. +#[cfg(all(test, target_os = "macos"))] +#[allow(dead_code)] +#[path = "popup.rs"] +mod popup_portable_test; mod proxy; +mod resolve; +mod runtime_stop; mod sidecar; +mod startup; +mod supervisor; mod tray; +mod tray_availability; mod updater; mod widget; mod window; use std::sync::{ atomic::{AtomicBool, Ordering}, - Mutex, + Mutex, MutexGuard, PoisonError, }; use tauri::{Manager, WebviewUrl, WebviewWindowBuilder}; use tauri_plugin_autostart::MacosLauncher; use tauri_plugin_shell::process::CommandChild; pub struct AppState { - pub proxy: proxy::ProxyClient, - pub spawned_by_us: AtomicBool, - pub child: Mutex>, + /// Absent until the startup sequence has resolved a home and a port. Nothing guesses an + /// endpoint any more, so there is no client to hand out before that. + proxy: Mutex>, + child: Mutex>, + /// The pid of the child this app started, if it started one. + child_pid: Mutex>, + /// When that child was spawned, so a later run can tell a child still starting from one that + /// will never answer (`startup::waits_on_child`). + child_spawned: Mutex>, + /// Whether the process answering the endpoint has been confirmed to be that child. + /// + /// Durable consent and current process ownership are different facts. Consent is a recorded + /// claim that survives restarts; this is a statement about the process on the other end of the + /// endpoint right now, and it has to be re-established whenever the endpoint or the answering + /// process can have changed. Carrying a bool across an attach is how a retry that lands on a + /// foreign runtime would still send it an owner's stop. + confirmed: AtomicBool, + /// The consumed spawn event stream of the child, if this app started one. + pub watch: sidecar::SidecarWatch, } impl AppState { - pub fn shutdown_child(&self) { - if !self.spawned_by_us.swap(false, Ordering::AcqRel) { - return; - } - if let Ok(mut child) = self.child.lock() { - if let Some(child) = child.take() { - let _ = child.kill(); - } + pub fn new() -> Self { + Self { + proxy: Mutex::new(None), + child: Mutex::new(None), + child_pid: Mutex::new(None), + child_spawned: Mutex::new(None), + confirmed: AtomicBool::new(false), + watch: sidecar::SidecarWatch::default(), } } + + fn slot(lock: &Mutex) -> MutexGuard<'_, T> { + lock.lock().unwrap_or_else(PoisonError::into_inner) + } + + pub fn proxy(&self) -> Option { + Self::slot(&self.proxy).clone() + } + + /// Point at a runtime. Nothing is owned until it is confirmed again. + pub fn attach(&self, proxy: proxy::ProxyClient) { + self.confirmed.store(false, Ordering::Release); + *Self::slot(&self.proxy) = Some(proxy); + } + + pub fn owns_runtime(&self) -> bool { + self.confirmed.load(Ordering::Acquire) + } + + pub fn child_pid(&self) -> Option { + *Self::slot(&self.child_pid) + } + + /// How long ago the child this app tracks was spawned; nothing when it tracks none. + pub fn child_age(&self) -> Option { + self.child_pid()?; + Self::slot(&self.child_spawned).map(|spawned| spawned.elapsed()) + } + + /// Confirm that the instance answering is the child this app started. + /// + /// This is the only thing that grants ownership. A spawn records a pid; it does not record that + /// the pid is what holds the port, because between the two the child can exit and a service can + /// take the port back. + pub fn confirm_ownership(&self, identity: proxy::RuntimeIdentity) -> bool { + let ours = self.child_pid() == Some(identity.pid); + self.confirmed.store(ours, Ordering::Release); + ours + } + + pub fn adopt(&self, child: CommandChild) { + *Self::slot(&self.child_pid) = Some(child.pid()); + *Self::slot(&self.child_spawned) = Some(std::time::Instant::now()); + *Self::slot(&self.child) = Some(child); + // Spawned, not yet confirmed: the health probe is what establishes that this pid is the + // one answering. + self.confirmed.store(false, Ordering::Release); + } + + /// Let go of a runtime that has already been drained. + /// + /// Dropping the handle does not signal the process — the shell plugin installs no `Drop` — so + /// this releases ownership without reintroducing the `kill()` that D2 removed. + pub fn release(&self) { + self.confirmed.store(false, Ordering::Release); + let _ = Self::slot(&self.child_pid).take(); + let _ = Self::slot(&self.child_spawned).take(); + let _ = Self::slot(&self.child).take(); + } +} + +impl Default for AppState { + fn default() -> Self { + Self::new() + } } #[tauri::command] fn show_dashboard(app: tauri::AppHandle) { - if let Some(window) = app.get_webview_window("main") { - window::show(&window); - } + popup::hide(&app); + startup::open_dashboard(&app); } #[tauri::command] @@ -50,58 +166,202 @@ fn hide_dashboard(app: tauri::AppHandle) { } } +/// Everything the startup sequence has said so far, including the states it has already finished. +/// +/// The page asks for this when it loads rather than relying only on the event stream: the first +/// states finish in milliseconds and an event emitted before the listener exists is simply gone. +/// +/// It always answers with a state. Answering `None` put the one case the page cannot render — a +/// shell with no startup state — behind a value the page silently discards, which is a frozen +/// window with no diagnostic and no way to tell it from a slow start. +#[tauri::command] +fn startup_snapshot(app: tauri::AppHandle) -> startup::Progress { + app.try_state::() + .map(|startup| startup.latest()) + .unwrap_or_else(startup::unavailable) +} + +/// The named states the startup sequence moves through, in order. +/// +/// The page asks for them instead of restating them, so a state added in the shell appears in the +/// UI and one removed cannot leave a row behind. +#[tauri::command] +fn startup_phases() -> Vec { + startup::phase_list() +} + +/// Run the startup sequence again. A run already in flight is left alone. +/// +/// A person asking for a runtime again also resumes supervision, even after the tray's Stop. +#[tauri::command] +fn retry_startup(app: tauri::AppHandle) { + supervisor::resume(&app); + startup::begin(&app); +} + +/// The user's answer to the takeover prompt the startup sequence is waiting on. +/// +/// The sequence holds a oneshot for exactly the duration of the prompt; a decision arriving +/// with nothing pending is a click after the fact, and it changes nothing. +#[tauri::command] +fn decide_takeover(app: tauri::AppHandle, approved: bool) { + if let Some(startup) = app.try_state::() { + startup.decide_takeover(approved); + } +} + +#[tauri::command] +async fn update_status( + window: tauri::WebviewWindow, + app: tauri::AppHandle, +) -> Result { + window::require_update_page(&window)?; + Ok(updater::page_status(&app)) +} + +#[tauri::command] +async fn update_check( + window: tauri::WebviewWindow, + app: tauri::AppHandle, +) -> Result { + window::require_update_page(&window)?; + let check_result = updater::check_and_show(&app).await; + check_result.map_err(|_| "the update check failed; try again".to_owned())?; + Ok(updater::page_status(&app)) +} + +#[tauri::command] +async fn update_install( + window: tauri::WebviewWindow, + app: tauri::AppHandle, +) -> Result { + window::require_update_page(&window)?; + updater::install_pending(&app).await.map_err(|error| { + logging::log_once("updater install failed", &error); + "the update could not be installed; try again".to_owned() + }) +} + +#[tauri::command] +fn return_to_dashboard(window: tauri::WebviewWindow, app: tauri::AppHandle) -> Result<(), String> { + window::require_update_page(&window)?; + startup::return_to_dashboard(&app) +} + pub fn run() { - tauri::Builder::default() + let builder = tauri::Builder::default() .plugin(tauri_plugin_single_instance::init(|app, _args, _cwd| { - if let Some(window) = app.get_webview_window("main") { - window::show(&window); - } + popup::hide(app); + startup::open_dashboard(app); })) - .plugin(tauri_plugin_opener::init()) + // Links are opened by the webviews' own new-window handler (`window.rs`), not by the + // plugin's injected click interceptor, which calls an IPC command the loopback dashboard + // is not granted and so swallowed every `target="_blank"` click. + .plugin( + tauri_plugin_opener::Builder::new() + .open_js_links_on_click(false) + .build(), + ) .plugin(tauri_plugin_process::init()) + // The argument is what makes a login launch recognisable. Nothing else in a bare launch + // distinguishes it from a person opening the app, and D7 needs the difference. .plugin(tauri_plugin_autostart::init( MacosLauncher::LaunchAgent, - None, + Some(vec![startup::AUTOSTART_FLAG]), )) .plugin(tauri_plugin_shell::init()) - .plugin(tauri_plugin_updater::Builder::new().build()) - .invoke_handler(tauri::generate_handler![show_dashboard, hide_dashboard]) + .plugin(tauri_plugin_updater::Builder::new().build()); + + // macOS is the one platform where the event loop cannot enforce D2 on its own: Tauri's default + // menu carries a predefined Quit wired to Cocoa's terminate:, and the pinned tao raises no + // cancellable event for it. Replacing that one item is what lets Cmd+Q mean hide. + #[cfg(target_os = "macos")] + let builder = builder + .menu(menu::build) + .on_menu_event(|app, event| menu::on_event(app, event.id().as_ref())); + + builder + .invoke_handler(tauri::generate_handler![ + show_dashboard, + hide_dashboard, + startup_snapshot, + startup_phases, + retry_startup, + decide_takeover, + update_status, + update_check, + update_install, + return_to_dashboard + ]) .setup(|app| { - let (endpoint, home) = discovery::current(); - let proxy = proxy::ProxyClient::new(endpoint, auth::Auth::new(home)) - .map_err(|error| error.to_string())?; - let child = tauri::async_runtime::block_on(sidecar::ensure_proxy( - app.handle(), - &proxy, - endpoint, - )) - .map_err(std::io::Error::other)?; - app.manage(AppState { - proxy: proxy.clone(), - spawned_by_us: AtomicBool::new(child.is_some()), - child: Mutex::new(child), - }); + app.manage(AppState::new()); app.manage(updater::PendingUpdate(Mutex::new(None))); + app.manage(updater::DesktopUpdateState::new( + app.package_info().version.to_string(), + )); + app.manage(updater::CheckGeneration::default()); + updater::start_ui_projection_worker(app.handle().clone()); + updater::start_snapshot_publisher(app.handle().clone()); app.manage(tray::TrayState::default()); + app.manage(exit::ExitCoordinator::new()); + app.manage(startup::Startup::new()); + // Registers the exit hook and an idle watchdog; it starts no runtime of its own. + supervisor::watch_runtime(app.handle()); - let window = WebviewWindowBuilder::new( - app, - "main", - WebviewUrl::App(format!("index.html?port={}", endpoint.port).into()), - ) - .title("OpenCodex") - .inner_size(1100.0, 720.0) - .visible(false) - .user_agent(&window::webview_user_agent()) - .on_navigation(window::navigation_allowed(endpoint)) - .build()?; + // D7: the window is created and shown before anything is registered, resolved, probed + // or started, so every state below has somewhere to be reported. A login launch stays + // hidden until the tray verdict, because R1 shows it after all when there turns out to + // be nowhere to hide. + let builder = + WebviewWindowBuilder::new(app, "main", WebviewUrl::App("index.html".into())) + .title("OpenCodex") + .inner_size(1100.0, 720.0) + .visible(false) + .user_agent(&window::webview_user_agent()) + // Cmd on macOS, Ctrl elsewhere, with + / - / 0. WebView2 zooms natively; on + // macOS and Linux Tauri injects a keydown polyfill whose one IPC call is granted + // to the loopback dashboard by `capabilities/dashboard-zoom.json`. + .zoom_hotkeys_enabled(true) + .on_navigation(window::navigation_allowed(app.handle().clone())) + .on_new_window(window::open_new_windows_in_default_browser()) + // A hidden window still loads pages: wry builds this one with WebView2 + // IsVisible=false, and the bootstrap page navigates to the dashboard URL + // afterwards, so the eval that a later show or hide would rely on has nowhere + // to land during a reload. Re-sending the current state here is what keeps the + // GUI's answer correct across navigation. + .on_page_load(|window, payload| { + if matches!(payload.event(), tauri::webview::PageLoadEvent::Finished) { + window::report_visibility( + &window, + window.is_visible().unwrap_or(false), + ); + } + }); + // The integrated title bar: macOS keeps its traffic lights but draws them over the + // webview, so the dashboard's sidebar top strip reserves the space they land in + // (`app-titlebar.css` keeps it aligned with this position) and the strips move or + // zoom the window through `plugin:window` commands granted by + // `capabilities/dashboard-titlebar.json`. Windows and Linux keep the native title + // bar — the shell ships no min/max/close widgets of its own — while the sidebar-top + // layout applies unchanged. + #[cfg(target_os = "macos")] + let builder = builder + .title_bar_style(tauri::TitleBarStyle::Overlay) + .hidden_title(true) + .min_inner_size(360.0, 320.0) + .traffic_light_position(tauri::Position::Logical(tauri::LogicalPosition::new( + 18.0, 22.0, + ))); + let window = builder.build()?; window::configure(&window); - window::set_tray_policy(app.handle(), false); - let dashboard = endpoint.url("/#/usage"); - if tauri::async_runtime::block_on(proxy.is_alive()).is_ok() { - let _ = window.eval(format!("window.location.replace({dashboard:?})")); + if startup::LaunchOrigin::detect() == startup::LaunchOrigin::User { + window::show(&window); + } else { + window::set_tray_policy(app.handle(), false); } - tray::install(app.handle(), proxy)?; + + startup::begin(app.handle()); + if !cfg!(debug_assertions) { updater::start_background_checks(app.handle().clone()); } @@ -110,10 +370,16 @@ pub fn run() { .build(tauri::generate_context!()) .expect("error while building OpenCodex desktop shell") .run(|app, event| { - if matches!(event, tauri::RunEvent::Exit) { - if let Some(state) = app.try_state::() { - state.shutdown_child(); - } + // Dock/Finder reopening an existing macOS app does not launch a second instance. + #[cfg(target_os = "macos")] + if let tauri::RunEvent::Reopen { .. } = event { + show_dashboard(app.clone()); + } + // Window close and the platform quit gesture arrive here as an exit request, and until + // this handler existed they went straight through to a SIGKILL of the runtime. D2 makes + // them hide; only the tray's Quit, and an update's coordinated restart, get past. + if let tauri::RunEvent::ExitRequested { code, api, .. } = event { + exit::on_exit_requested(app, code, &api); } }); } diff --git a/desktop/src-tauri/src/main.rs b/desktop/src-tauri/src/main.rs index c0e7716829..8d5174cf6f 100644 --- a/desktop/src-tauri/src/main.rs +++ b/desktop/src-tauri/src/main.rs @@ -1,3 +1,5 @@ +#![cfg_attr(not(debug_assertions), windows_subsystem = "windows")] + fn main() { opencodex_desktop_lib::run(); } diff --git a/desktop/src-tauri/src/menu.rs b/desktop/src-tauri/src/menu.rs new file mode 100644 index 0000000000..49c193d024 --- /dev/null +++ b/desktop/src-tauri/src/menu.rs @@ -0,0 +1,141 @@ +//! The macOS application menu. +//! +//! Tauri installs a default menu when the app sets none, and that menu's Quit is a predefined item +//! wired straight to Cocoa's `terminate:`. The pinned tao implements only +//! `applicationWillTerminate`, never the cancellable `applicationShouldTerminate`, so a Cmd+Q +//! through that item reaches `RunEvent::Exit` without ever raising `RunEvent::ExitRequested`. +//! Nothing can hold it, which means D2's rule — the quit gesture hides, only the tray's Quit ends +//! the app — cannot be enforced from the event loop alone on macOS. The one item is replaced here +//! with an ordinary item on the same accelerator, routed through the same gesture path as closing +//! the window. +//! +//! The rest is reproduced rather than mutated: `Menu::default` is not decomposable, and dropping it +//! would take Cut, Copy, Paste and Select All with it — which the startup diagnostic needs the user +//! to be able to use. This mirrors `tauri::menu::Menu::default` for the pinned version, minus that +//! item. + +/// The id of the replacement Quit item. Nothing else in the app uses it, so a menu event carrying +/// it is unambiguously this one. +pub const QUIT_ID: &str = "app-menu-quit"; +pub const USAGE_ID: &str = "app-menu-show-usage"; + +pub fn build(app: &tauri::AppHandle) -> tauri::Result> { + use tauri::menu::{ + AboutMetadata, Menu, MenuItem, PredefinedMenuItem, Submenu, HELP_SUBMENU_ID, + WINDOW_SUBMENU_ID, + }; + + let package = app.package_info(); + let config = app.config(); + let about = AboutMetadata { + name: Some(package.name.clone()), + version: Some(package.version.to_string()), + copyright: config.bundle.copyright.clone(), + authors: config + .bundle + .publisher + .clone() + .map(|publisher| vec![publisher]), + ..Default::default() + }; + + // Labelled as a quit because that is the gesture the user is making. What it means here is + // D2's answer to that gesture: the window goes away and the runtime keeps serving. + let quit = MenuItem::with_id( + app, + QUIT_ID, + format!("Quit {}", package.name), + true, + Some("CmdOrCtrl+Q"), + )?; + + Menu::with_items( + app, + &[ + &Submenu::with_items( + app, + package.name.clone(), + true, + &[ + &PredefinedMenuItem::about(app, None, Some(about))?, + &PredefinedMenuItem::separator(app)?, + &PredefinedMenuItem::services(app, None)?, + &PredefinedMenuItem::separator(app)?, + &PredefinedMenuItem::hide(app, None)?, + &PredefinedMenuItem::hide_others(app, None)?, + &PredefinedMenuItem::separator(app)?, + &quit, + ], + )?, + &Submenu::with_items( + app, + "File", + true, + &[&PredefinedMenuItem::close_window(app, None)?], + )?, + &Submenu::with_items( + app, + "Edit", + true, + &[ + &PredefinedMenuItem::undo(app, None)?, + &PredefinedMenuItem::redo(app, None)?, + &PredefinedMenuItem::separator(app)?, + &PredefinedMenuItem::cut(app, None)?, + &PredefinedMenuItem::copy(app, None)?, + &PredefinedMenuItem::paste(app, None)?, + &PredefinedMenuItem::select_all(app, None)?, + ], + )?, + &Submenu::with_items( + app, + "View", + true, + &[ + &MenuItem::with_id( + app, + USAGE_ID, + "Show Usage", + true, + Some("CmdOrCtrl+Shift+U"), + )?, + &PredefinedMenuItem::separator(app)?, + &PredefinedMenuItem::fullscreen(app, None)?, + ], + )?, + &Submenu::with_id_and_items( + app, + WINDOW_SUBMENU_ID, + "Window", + true, + &[ + &PredefinedMenuItem::minimize(app, None)?, + &PredefinedMenuItem::maximize(app, None)?, + &PredefinedMenuItem::separator(app)?, + &PredefinedMenuItem::close_window(app, None)?, + ], + )?, + &Submenu::with_id_and_items(app, HELP_SUBMENU_ID, "Help", true, &[])?, + ], + ) +} + +/// Route an application-menu event. +/// +/// Tray menu events carry different ids and retain their existing owner. +pub fn on_event(app: &tauri::AppHandle, id: &str) { + use tauri::Manager; + if id == QUIT_ID { + crate::exit::gesture(app); + } else if id == USAGE_ID { + if let Some(proxy) = app.state::().proxy() { + let _ = crate::popup::show( + app, + proxy.endpoint(), + tauri::PhysicalPosition::new(0.0, 0.0), + ); + } else if let Some(main) = app.get_webview_window("main") { + crate::window::show(&main); + } + } +} diff --git a/desktop/src-tauri/src/native_tray.rs b/desktop/src-tauri/src/native_tray.rs new file mode 100644 index 0000000000..583c337831 --- /dev/null +++ b/desktop/src-tauri/src/native_tray.rs @@ -0,0 +1,443 @@ +//! macOS usage popup: one native panel in the existing Tauri process. +use crate::{ + endpoint::ProxyEndpoint, native_tray_data, native_tray_snapshot, proxy::RuntimeBinding, window, + AppState, +}; +use serde_json::{json, Value}; +use std::{ + ffi::{c_char, c_void, CStr}, + sync::{ + atomic::{AtomicU64, Ordering}, + Mutex, OnceLock, + }, + time::Duration, +}; +use tauri::{AppHandle, Manager, PhysicalPosition}; + +extern "C" { + fn ocx_native_tray_show(item: *mut c_void, toggle: i32, callback: extern "C" fn(i32)); + fn ocx_native_tray_hide(); + fn ocx_native_tray_visible() -> i32; + fn ocx_native_tray_update(bytes: *const u8, count: isize); + fn ocx_native_tray_update_dot(item: *mut c_void, show: i32); + fn ocx_native_tray_set_switch_handler(callback: extern "C" fn(*const c_char, *const c_char)); +} + +static HOST: OnceLock = OnceLock::new(); + +struct NativeTrayState { + generation: AtomicU64, + task: Mutex>>, + cache: Mutex<(Option, Value)>, +} + +impl Default for NativeTrayState { + fn default() -> Self { + Self { + generation: AtomicU64::new(0), + task: Mutex::new(None), + cache: Mutex::new((None, native_tray_snapshot::empty())), + } + } +} + +pub fn show( + app: &AppHandle, + _endpoint: ProxyEndpoint, + _anchor: PhysicalPosition, +) -> tauri::Result<()> { + present(app, false) +} + +pub fn toggle( + app: &AppHandle, + _endpoint: ProxyEndpoint, + _anchor: PhysicalPosition, +) -> tauri::Result<()> { + present(app, true) +} + +fn present(app: &AppHandle, toggle: bool) -> tauri::Result<()> { + let _ = HOST.set(app.clone()); + if app.try_state::().is_none() { + app.manage(NativeTrayState::default()); + } + let Some(tray) = app.tray_by_id("main") else { + return Ok(()); + }; + // Tauri guarantees this closure runs on AppKit's main thread. The tray retains + // its status item; Swift only borrows it for the synchronous presentation call. + tray.with_inner_tray_icon(move |tray| { + if let Some(item) = tray.ns_status_item() { + let pointer = (&*item as *const _ as *mut c_void).cast(); + unsafe { + ocx_native_tray_set_switch_handler(native_switch); + ocx_native_tray_show(pointer, i32::from(toggle), native_event); + } + } + }) +} + +pub fn set_update_dot(app: &AppHandle, _show: bool) { + let app = app.clone(); + let target = app.clone(); + let _ = target.run_on_main_thread(move || { + let Some(tray) = app.tray_by_id("main") else { + return; + }; + let pending = app + .try_state::() + .is_some_and(|state| state.update_pending.load(Ordering::Acquire)); + let _ = tray.with_inner_tray_icon(move |inner| { + if let Some(item) = inner.ns_status_item() { + let pointer = (&*item as *const _ as *mut c_void).cast(); + unsafe { + ocx_native_tray_update_dot(pointer, i32::from(pending)); + } + } + }); + }); +} + +pub fn hide(app: &AppHandle) { + stop_refresh(app); + let _ = app.run_on_main_thread(|| unsafe { ocx_native_tray_hide() }); +} + +extern "C" fn native_event(event: i32) { + let Some(app) = HOST.get() else { + return; + }; + match event { + 1 => start_refresh(app), + 2 => stop_refresh(app), + 3 | 4 => { + stop_refresh(app); + let Some(proxy) = app.state::().proxy() else { + return; + }; + if let Some(main) = app.get_webview_window("main") { + let session = app + .state::() + .session_id() + .to_string(); + let path = if event == 4 { + format!("/?desktop=open&desktop_session={session}#/usage/companion") + } else { + format!("/?desktop=open&desktop_session={session}#/usage") + }; + if let Ok(url) = proxy.endpoint().url(&path).parse() { + let _ = main.navigate(url); + window::show(&main); + } + } + } + _ => {} + } +} + +/// A provider name or account id the panel sends back. Anything empty, oversized, not UTF-8 or +/// carrying control characters did not come from a snapshot this host published. +fn switch_argument(value: Option<&CStr>) -> Option { + let value = value?.to_str().ok()?; + (!value.is_empty() && value.len() <= 256 && !value.chars().any(char::is_control)) + .then(|| value.to_owned()) +} + +/// The panel's "Use" action. Swift calls this on the main thread with borrowed C strings; they are +/// copied before returning and the switch runs on the async runtime. +extern "C" fn native_switch(provider: *const c_char, account: *const c_char) { + // SAFETY: Swift passes NUL-terminated buffers that stay valid for the duration of this call. + let borrow = + |pointer: *const c_char| (!pointer.is_null()).then(|| unsafe { CStr::from_ptr(pointer) }); + let (Some(provider), Some(account)) = ( + switch_argument(borrow(provider)), + switch_argument(borrow(account)), + ) else { + return; + }; + let Some(app) = HOST.get().cloned() else { + return; + }; + tauri::async_runtime::spawn(async move { + match switch_account(&app, &provider, &account).await { + Ok(()) => start_refresh(&app), + Err(message) => report_switch_failure(&app, message), + } + }); +} + +/// Resolve the route from the host's own provider sources, never from the panel, then send the +/// single-use, body-bound switch request. +async fn switch_account( + app: &AppHandle, + provider: &str, + account: &str, +) -> Result<(), &'static str> { + let proxy = app + .state::() + .proxy() + .ok_or("The local runtime is not connected.")?; + let config = proxy + .get("/api/config") + .await + .map_err(|_| "Could not read the provider list to switch accounts.")?; + let sources = crate::native_tray_accounts::sources(&config) + .ok_or("Could not read the provider list to switch accounts.")?; + let source = sources + .iter() + .find(|source| source.name == provider) + .ok_or("That provider is no longer configured.")?; + let (kind, body) = crate::native_tray_accounts::switch_request(source, account) + .ok_or("This provider has no account to switch.")?; + proxy + .put_account_switch(kind, &body) + .await + .map(|_| ()) + .map_err(|error| switch_error_message(&error)) +} + +fn switch_error_message(error: &crate::proxy::ProxyError) -> &'static str { + use crate::proxy::ProxyError; + match error { + ProxyError::Http(status) if status.as_u16() == 409 => { + "The runtime refused that account right now (paused or still validating)." + } + ProxyError::Http(status) if matches!(status.as_u16(), 400 | 404) => { + "That account no longer exists. Refresh and try again." + } + ProxyError::Unauthorized | ProxyError::Foreign => { + "The runtime did not accept the desktop app's switch request." + } + ProxyError::Unreachable => "The local runtime is not reachable.", + _ => "The account could not be switched. Open the dashboard to try again.", + } +} + +/// Show a failed switch in the open panel without waiting for the next refresh. +fn report_switch_failure(app: &AppHandle, message: &str) { + let Some(state) = app.try_state::() else { + return; + }; + let (binding, mut snapshot) = state + .cache + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .clone(); + snapshot["refreshing"] = json!(false); + snapshot["errors"] = json!([message]); + // Tells the panel this publish answers its switch, so the row spinner stops even when the + // failure arrives immediately. + snapshot["switchFailed"] = json!(true); + publish( + app, + state.generation.load(Ordering::Acquire), + binding, + snapshot, + ); +} + +fn stop_refresh(app: &AppHandle) { + let Some(state) = app.try_state::() else { + return; + }; + let mut slot = state + .task + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + state.generation.fetch_add(1, Ordering::AcqRel); + if let Some(task) = slot.take() { + task.abort(); + } +} + +fn start_refresh(app: &AppHandle) { + let state = app.state::(); + let mut slot = state + .task + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + if let Some(task) = slot.take() { + task.abort(); + } + let generation = state.generation.fetch_add(1, Ordering::AcqRel) + 1; + let app = app.clone(); + *slot = Some(tauri::async_runtime::spawn(async move { + loop { + let proxy = app.state::().proxy(); + let binding = proxy.as_ref().and_then(|p| p.binding()); + let mut loading = { + let state = app.state::(); + let cache = state + .cache + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + if cache.0 == binding { + cache.1.clone() + } else { + native_tray_snapshot::empty() + } + }; + loading["refreshing"] = json!(true); + publish(&app, generation, binding, loading.clone()); + let snapshot = if let Some(proxy) = proxy.filter(|_| binding.is_some()) { + let target = app.clone(); + native_tray_data::load(&proxy, loading, move |partial| { + publish(&target, generation, binding, partial); + }) + .await + } else { + failed(loading, "The local runtime is not connected.") + }; + publish(&app, generation, binding, snapshot); + tokio::time::sleep(Duration::from_secs(60)).await; + if app + .state::() + .generation + .load(Ordering::Acquire) + != generation + { + break; + } + } + })); +} + +fn failed(mut value: Value, message: &str) -> Value { + value["refreshing"] = json!(false); + value["errors"] = json!([message]); + value +} + +fn publish(app: &AppHandle, generation: u64, binding: Option, snapshot: Value) { + let app = app.clone(); + let target = app.clone(); + let _ = target.run_on_main_thread(move || { + let state = app.state::(); + let current = app.state::().proxy().and_then(|p| p.binding()); + if !may_publish( + generation, + state.generation.load(Ordering::Acquire), + binding, + current, + ) { + return; + } + // This callback, unlike the network task, is guaranteed to be on the main thread. + if unsafe { ocx_native_tray_visible() } == 0 { + return; + } + if let Some((snapshot, bytes)) = display_payload(snapshot) { + unsafe { + ocx_native_tray_update(bytes.as_ptr(), bytes.len() as isize); + } + *state + .cache + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) = (binding, cached(snapshot)); + } + }); +} + +/// The cached copy every later refresh starts from. `switchFailed` answers one switch, so it is +/// delivered once and never cached: carried forward, it would settle the next switch's spinner on +/// that switch's first loading publish. +fn cached(mut snapshot: Value) -> Value { + if let Some(fields) = snapshot.as_object_mut() { + fields.remove("switchFailed"); + } + snapshot +} + +fn display_payload(snapshot: Value) -> Option<(Value, Vec)> { + let bytes = serde_json::to_vec(&snapshot).ok()?; + if bytes.len() <= 8 * 1024 * 1024 { + return Some((snapshot, bytes)); + } + // A rejected payload must settle the spinner instead of leaving Refresh disabled. + let error = failed(native_tray_snapshot::empty(), "Usage data is too large for this panel. Open the dashboard to narrow the visible sections."); + let bytes = serde_json::to_vec(&error).ok()?; + Some((error, bytes)) +} + +fn may_publish( + start: u64, + now: u64, + binding: Option, + current: Option, +) -> bool { + start == now && binding == current +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::proxy::RuntimeIdentity; + #[test] + fn switch_arguments_are_bounded_copies_of_what_the_panel_can_send() { + let arg = |bytes: &[u8]| switch_argument(Some(CStr::from_bytes_with_nul(bytes).unwrap())); + assert_eq!(arg(b"anthropic\0").as_deref(), Some("anthropic")); + assert_eq!(arg("계정-1\0".as_bytes()).as_deref(), Some("계정-1")); + assert_eq!(switch_argument(None), None); + assert_eq!(arg(b"\0"), None); + assert_eq!(arg(b"line\nbreak\0"), None); + assert_eq!(arg(b"\xff\xfe\0"), None); + let long = [vec![b'a'; 257], vec![0]].concat(); + assert_eq!(arg(&long), None); + let limit = [vec![b'a'; 256], vec![0]].concat(); + assert_eq!(arg(&limit).map(|value| value.len()), Some(256)); + } + #[test] + fn switch_failures_name_the_cause_without_server_text() { + use crate::proxy::ProxyError; + use reqwest::StatusCode; + assert!(switch_error_message(&ProxyError::Http(StatusCode::CONFLICT)).contains("paused")); + assert!( + switch_error_message(&ProxyError::Http(StatusCode::NOT_FOUND)) + .contains("no longer exists") + ); + assert!(switch_error_message(&ProxyError::Unauthorized).contains("did not accept")); + assert!(switch_error_message(&ProxyError::Unreachable).contains("not reachable")); + } + #[test] + fn closed_refresh_or_rebound_runtime_cannot_overwrite_visible_state() { + let a = RuntimeBinding { + identity: RuntimeIdentity { + pid: 1, + port: 10100, + }, + generation: 1, + }; + let b = RuntimeBinding { generation: 2, ..a }; + assert!(may_publish(7, 7, Some(a), Some(a))); + assert!(!may_publish(7, 8, Some(a), Some(a))); + assert!(!may_publish(7, 7, Some(a), Some(b))); + assert!(!may_publish(7, 7, Some(a), None)); + } + #[test] + fn refresh_failure_preserves_age_and_clears_busy_state() { + let reported = json!({"refreshing":false,"errors":["refused"],"switchFailed":true}); + let kept = cached(reported); + assert!(kept.get("switchFailed").is_none()); + assert_eq!(kept["errors"], json!(["refused"])); + + let before = json!({"updatedAt":12,"refreshing":true,"today":{"totalTokens":30}}); + let after = failed(before, "Unavailable"); + assert_eq!(after["updatedAt"], 12); + assert_eq!(after["today"]["totalTokens"], 30); + assert_eq!(after["refreshing"], false); + assert_eq!(after["errors"], json!(["Unavailable"])); + } + + #[test] + fn oversized_display_data_settles_with_a_small_readable_error() { + let mut snapshot = native_tray_snapshot::empty(); + snapshot["models"] = + json!([{"id":"large","label":"x".repeat(8*1024*1024),"tokens":null,"requests":null}]); + let (result, bytes) = display_payload(snapshot).unwrap(); + assert_eq!(result["schemaVersion"], 1); + assert_eq!(result["refreshing"], false); + assert_eq!(result["errors"].as_array().unwrap().len(), 1); + assert!(bytes.len() < 1024); + } +} diff --git a/desktop/src-tauri/src/native_tray_accounts.rs b/desktop/src-tauri/src/native_tray_accounts.rs new file mode 100644 index 0000000000..54309ce674 --- /dev/null +++ b/desktop/src-tauri/src/native_tray_accounts.rs @@ -0,0 +1,395 @@ +use crate::native_tray_snapshot::{number, text}; +use crate::proxy::AccountSwitchKind; +use serde_json::{json, Value}; + +pub struct Source { + pub name: String, + pub label: String, + pub path: Option, + /// The route that switches this provider's active account, when it has one. Chosen here, from + /// the host's own view of the config; the panel never names a route. + pub switch: Option, +} + +pub fn sources(config: &Value) -> Option> { + Some( + config["providers"] + .as_object()? + .iter() + .filter(|(_, row)| row["disabled"].as_bool() != Some(true)) + .map(|(name, row)| { + let (path, switch) = if name == "openai" { + ( + Some("/api/codex-auth/accounts".into()), + Some(AccountSwitchKind::Codex), + ) + } else if text(row, "authMode") == "oauth" { + ( + Some(query("/api/oauth/accounts", "provider", name)), + Some(AccountSwitchKind::OAuth), + ) + } else if row["hasApiKey"].as_bool() == Some(true) + && text(row, "authMode") != "forward" + { + ( + Some(query("/api/providers/keys", "name", name)), + Some(AccountSwitchKind::ApiKey), + ) + } else { + (None, None) + }; + let label = if !text(row, "label").is_empty() { + text(row, "label") + } else { + match name.as_str() { + "openai" => "OpenAI (Codex login)", + "anthropic" => "Anthropic Claude", + "xai" => "xAI Grok", + "google" => "Google Gemini", + "google-antigravity" => "Google Antigravity", + _ => name, + } + }; + Source { + name: name.clone(), + label: label.into(), + path, + switch, + } + }) + .collect(), + ) +} + +/// The management request that makes `account_id` the active account of `source`, or `None` +/// when the provider has no switch route. Bodies match the dashboard's own switch calls. +pub fn switch_request(source: &Source, account_id: &str) -> Option<(AccountSwitchKind, Value)> { + let kind = source.switch?; + let body = match kind { + AccountSwitchKind::Codex => json!({ "accountId": account_id }), + AccountSwitchKind::OAuth => json!({ "provider": source.name, "accountId": account_id }), + AccountSwitchKind::ApiKey => json!({ "name": source.name, "id": account_id }), + }; + Some((kind, body)) +} + +fn query(path: &str, key: &str, provider: &str) -> String { + let mut url = + reqwest::Url::parse(&format!("http://127.0.0.1{path}")).expect("constant loopback URL"); + url.query_pairs_mut() + .append_pair(key, provider) + .append_pair("quota", "1"); + format!("{}?{}", url.path(), url.query().unwrap_or_default()) +} + +fn mask_email(email: &str) -> String { + let Some((local, domain)) = email.split_once('@') else { + return "•••".into(); + }; + let suffix = domain.rfind('.').map(|i| &domain[i..]).unwrap_or_default(); + format!( + "{}•••@{}•••{suffix}", + local.chars().next().unwrap_or('•'), + domain.chars().next().unwrap_or('•') + ) +} + +fn reset(value: &Value) -> Option { + number(value) + .filter(|n| *n > 0.0) + .map(|n| if n >= 1e12 { n / 1000.0 } else { n }) + .filter(|n| *n < 253_402_300_800.0) +} + +fn windows(quota: &Value, plan: &str) -> Vec { + let monthly_only = matches!(plan.trim().to_lowercase().as_str(), "go" | "free"); + let mut rows = Vec::new(); + // A window is listed only when it reports something: a finite percentage (zero included) or + // a valid reset time. Plans differ in which windows they have, so an absent 5-hour window is + // not a 5-hour window with unknown usage; an account that reports nothing keeps the panel's + // "No quota data" line instead of a row of dashes. + let mut push = |id: &str, label: &str, percent: &Value, at: &Value| { + let percent = number(percent); + let at = reset(at); + if percent.is_some() || at.is_some() { + rows.push(json!({"id":format!("{id}:{}",rows.len()),"label":label,"percent":percent,"resetAt":at})); + } + }; + if !monthly_only { + let short = quota + .get("fiveHourPercent") + .filter(|v| !v.is_null()) + .unwrap_or("a["shortPercent"]); + let short_reset = quota + .get("fiveHourResetAt") + .filter(|v| !v.is_null()) + .unwrap_or("a["shortResetAt"]); + push("short", "5-hour limit", short, short_reset); + push( + "weekly", + "Weekly limit", + "a["weeklyPercent"], + "a["weeklyResetAt"], + ); + } + push( + "monthly", + "30-day limit", + "a["monthlyPercent"], + "a["monthlyResetAt"], + ); + if !monthly_only { + if let Some(custom) = quota["customWindows"].as_array() { + for window in custom { + if let Some(label) = window["label"].as_str() { + push(label, label, &window["percent"], &window["resetAt"]); + } + } + } + } + rows +} + +pub fn provider(source: &Source, body: Option<&Value>, unavailable: bool) -> Value { + let parsed = body.and_then(parse_accounts); + let malformed = body.is_some() && parsed.is_none(); + let accounts = parsed.unwrap_or_default(); + let mut row = json!({"id":source.name,"label":source.label, + "unavailable":unavailable || malformed,"accounts":accounts, + "switchable":source.switch.is_some()}); + // Optional: older panels ignore unknown keys, and a provider without a mark simply has none. + if let Some(icon) = crate::provider_icons::icon(&source.name) { + row["iconSvg"] = json!(icon.svg); + row["iconPaint"] = json!(icon.paint); + } + row +} + +fn parse_accounts(body: &Value) -> Option> { + let rows = body + .get("accounts") + .or_else(|| body.get("keys"))? + .as_array()?; + let active = ["activeAccountId", "activeId", "activeCodexAccountId"] + .iter() + .find_map(|key| body[key].as_str()); + rows.iter().enumerate().map(|(index,row)| { + let id = row["id"].as_str()?; + let email = row["email"].as_str().map(mask_email); + let label = [row["alias"].as_str(),row["label"].as_str(),email.as_deref(),row["logLabel"].as_str(),Some(id)] + .into_iter().flatten().find(|v| !v.is_empty()).unwrap_or(id); + let unavailable = row["quotaUnavailable"].as_bool()==Some(true) + || text(row,"quotaMode")=="unsupported" || !row["quota"].is_object(); + let is_active = active.map_or(row["active"].as_bool()==Some(true),|selected|selected==id); + let (switch_state, blocked_reason) = switch_state(row, is_active); + Some(json!({"id":format!("{id}:{index}"),"accountId":id,"label":label,"email":email,"plan":row["plan"].as_str(), + "active":is_active, + "switchState":switch_state,"blockedReason":blocked_reason, + "exhausted":!unavailable && exhausted(&row["quota"],text(row,"plan")), + "unavailable":unavailable, + "windows":if unavailable {vec![]} else {windows(&row["quota"],text(row,"plan"))}})) + }).collect() +} + +/// Mirrors what the server itself refuses or drains, and nothing more. The 98% hard lock exists +/// only on the main Codex account and is reported by the roster's `mainAccountHardLock`; a paused +/// account and a Codex account whose validation is pending (`health.reason`) are refused by the +/// switch route with 409. Everything else stays switchable, exhausted or not. +fn switch_state(row: &Value, active: bool) -> (&'static str, Option<&'static str>) { + if active { + ("active", None) + } else if row["mainAccountHardLock"]["state"].as_str() == Some("blocked") { + ("blocked", Some("mainHardLock")) + } else if row["paused"].as_bool() == Some(true) { + ("blocked", Some("paused")) + } else if row["health"]["reason"].as_str() == Some("validation_pending") { + ("blocked", Some("validationPending")) + } else { + ("available", None) + } +} + +/// Same windows as `isCodexQuotaExhausted` (src/codex/quota.ts): a reading at 100% in any window +/// the plan is governed by, plus the burst window on every plan. +fn exhausted(quota: &Value, plan: &str) -> bool { + let monthly_only = matches!(plan.trim().to_lowercase().as_str(), "go" | "free"); + let short = quota + .get("fiveHourPercent") + .filter(|v| !v.is_null()) + .unwrap_or("a["shortPercent"]); + let windows: &[&Value] = if monthly_only { + &["a["monthlyPercent"], short] + } else { + &["a["weeklyPercent"], "a["monthlyPercent"], short] + }; + windows + .iter() + .any(|value| number(value).is_some_and(|percent| percent >= 100.0)) +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn source_routes_are_encoded_and_never_forward_credentials() { + let rows = sources(&json!({"providers":{ + "off":{"disabled":true},"oauth&x":{"authMode":"oauth","apiKey":"secret"}, + "forward":{"hasApiKey":true,"authMode":"forward"},"openai":{}}})) + .unwrap(); + assert_eq!(rows.len(), 3); + assert!(rows + .iter() + .find(|s| s.name == "oauth&x") + .unwrap() + .path + .as_ref() + .unwrap() + .contains("provider=oauth%26x")); + assert!(rows + .iter() + .find(|s| s.name == "forward") + .unwrap() + .path + .is_none()); + } + #[test] + fn account_projection_masks_email_preserves_selection_and_deduplicates_windows() { + let rows=parse_accounts(&json!({"activeId":"a","keys":[{"id":"a","email":"example@example.com","key":"secret", + "quota":{"shortPercent":12,"shortResetAt":1900000000000_u64,"customWindows":[{"label":"same","percent":1},{"label":"same","percent":2}]}}]})).unwrap(); + assert_eq!(rows[0]["label"], "e•••@e•••.com"); + assert_eq!(rows[0]["active"], true); + assert!(!rows[0].to_string().contains("secret")); + assert_eq!(rows[0]["windows"][0]["resetAt"], 1900000000.0); + assert_ne!(rows[0]["windows"][1]["id"], rows[0]["windows"][2]["id"]); + } + #[test] + fn monthly_plans_and_unavailable_quota_do_not_reuse_short_windows() { + let rows = parse_accounts(&json!({"accounts":[ + {"id":"a","plan":" Go ","quota":{"fiveHourPercent":99,"monthlyPercent":0}}, + {"id":"b","quotaUnavailable":true,"quota":{"weeklyPercent":50}}]})) + .unwrap(); + assert_eq!(rows[0]["windows"].as_array().unwrap().len(), 1); + assert_eq!(rows[0]["windows"][0]["percent"], 0.0); + assert!(rows[1]["windows"].as_array().unwrap().is_empty()); + assert!(parse_accounts(&json!({"accounts":[{"quota":{}}]})).is_none()); + } + #[test] + fn only_windows_that_report_data_are_listed() { + let labels = |quota: Value| -> Vec { + windows("a, "pro") + .iter() + .map(|w| w["label"].as_str().unwrap().to_owned()) + .collect() + }; + // Weekly-only plan: no 5-hour row at all, with or without an explicit null. + assert_eq!( + labels(json!({"weeklyPercent":49,"weeklyResetAt":1900000000})), + ["Weekly limit"] + ); + assert_eq!( + labels(json!({"fiveHourPercent":null,"fiveHourResetAt":null,"weeklyPercent":1})), + ["Weekly limit"] + ); + // Zero is a measurement and a reset time alone still identifies a live window. + assert_eq!( + labels(json!({"fiveHourPercent":0,"weeklyPercent":0})), + ["5-hour limit", "Weekly limit"] + ); + assert_eq!( + labels(json!({"shortResetAt":1900000000,"weeklyPercent":3})), + ["5-hour limit", "Weekly limit"] + ); + // Nothing reported means no rows; the view shows its "No quota data" line. + assert!(labels(json!({})).is_empty()); + assert!(labels(json!({"fiveHourPercent":"n/a","weeklyPercent":-1})).is_empty()); + } + #[test] + fn providers_carry_their_mark_when_one_exists() { + let source = |name: &str| Source { + name: name.into(), + label: name.into(), + path: None, + switch: None, + }; + let openai = provider(&source("openai"), None, false); + assert!(openai["iconSvg"].as_str().unwrap().contains(", + models: Option>, + error: &'static str, +} + +async fn bounded(future: F, budget: Duration) -> Option { + tokio::time::timeout(budget, future).await.ok() +} + +fn merge(result: &mut Value, section: Section) { + if let Some(value) = section.value { + result[section.key] = value; + if let Some(models) = section.models { + result["models"] = json!(models); + } + result["updatedAt"] = json!(SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_secs_f64()); + } else { + result[section.key] = if section.key == "providers" { + json!([]) + } else { + Value::Null + }; + if section.key == "today" { + result["models"] = json!([]); + } + result["errors"] + .as_array_mut() + .expect("snapshot error array") + .push(json!(section.error)); + } +} + +pub async fn load( + proxy: &ProxyClient, + mut result: Value, + publish: impl Fn(Value) + Send + Sync, +) -> Value { + result["refreshing"] = json!(true); + result["errors"] = json!([]); + let settings = bounded(proxy.companion_settings(), READ_BUDGET) + .await + .and_then(Result::ok) + .and_then(|v| { + let s = &v["settings"]; + (s.is_object() + && s["hiddenProviders"].is_array() + && (s["models"].is_null() || s["models"].is_array())) + .then(|| s.clone()) + }); + result["settings"] = snapshot::display_settings(settings.as_ref()); + let mut tasks = JoinSet::new(); + if let Some(settings) = &settings { + for (range, key, error) in [ + ("today", "today", "Today's usage is unavailable."), + ("30d", "month", "30-day usage is unavailable."), + ] { + let proxy = proxy.clone(); + let settings = settings.clone(); + tasks.spawn(async move { + let projected = + bounded(proxy.get(&format!("/api/usage?range={range}")), READ_BUDGET) + .await + .and_then(Result::ok) + .and_then(|body| snapshot::usage(&body, &settings)); + match projected { + Some((totals, models)) => Section { + key, + value: Some(totals), + models: (key == "today").then_some(models), + error, + }, + None => Section { + key, + value: None, + models: None, + error, + }, + } + }); + } + if settings["showChart"].as_bool() == Some(true) { + let proxy = proxy.clone(); + let settings = settings.clone(); + tasks.spawn(async move { + let value = bounded(proxy.timeline(&timeline_query(&settings)), READ_BUDGET) + .await + .and_then(Result::ok) + .and_then(|body| snapshot::chart(&body, &settings)); + Section { + key: "chart", + value, + models: None, + error: "Usage chart is unavailable.", + } + }); + } + } else { + result["errors"] = json!(["Display settings are unavailable."]); + } + if result["settings"]["showAccounts"].as_bool() == Some(true) { + let proxy = proxy.clone(); + let settings = settings.clone(); + tasks.spawn(async move { + let sources = bounded(proxy.get("/api/config"), READ_BUDGET) + .await + .and_then(Result::ok) + .and_then(|v| accounts::sources(&v)); + let value = if let Some(sources) = sources { + let selected = sources + .into_iter() + .filter(|source| { + settings + .as_ref() + .is_none_or(|s| !snapshot::hidden(s, &source.name)) + }) + .collect(); + Some(json!(load_providers(&proxy, selected).await)) + } else { + None + }; + Section { + key: "providers", + value, + models: None, + error: "Account limits are unavailable.", + } + }); + } + publish(result.clone()); + while let Some(section) = tasks.join_next().await { + match section { + Ok(section) => merge(&mut result, section), + Err(_) => result["errors"] + .as_array_mut() + .expect("snapshot error array") + .push(json!("A usage section could not be loaded.")), + } + publish(result.clone()); + } + result["refreshing"] = json!(false); + result +} + +async fn load_providers(proxy: &ProxyClient, sources: Vec) -> Vec { + // Preserve successful rows at the deadline; untouched rows already say unavailable. + let mut rows: Vec<_> = sources + .iter() + .map(|s| accounts::provider(s, None, s.path.is_some())) + .collect(); + let mut pending = sources.into_iter().enumerate(); + let mut tasks = JoinSet::new(); + let deadline = Instant::now() + ACCOUNT_BUDGET; + loop { + while tasks.len() < 4 { + let Some((index, source)) = pending.next() else { + break; + }; + let proxy = proxy.clone(); + tasks.spawn(async move { + let Some(path) = &source.path else { + return (index, accounts::provider(&source, None, false)); + }; + let mut body = bounded(proxy.get(path), READ_BUDGET) + .await + .and_then(Result::ok); + if source.name == "openai" { + if let (Some(body), Some(Ok(active))) = ( + body.as_mut().and_then(Value::as_object_mut), + bounded(proxy.get("/api/codex-auth/active"), READ_BUDGET).await, + ) { + body.insert( + "activeCodexAccountId".into(), + json!(active["activeCodexAccountId"] + .as_str() + .unwrap_or("__main__")), + ); + } + } + ( + index, + accounts::provider(&source, body.as_ref(), body.is_none()), + ) + }); + } + match tokio::time::timeout_at(deadline, tasks.join_next()).await { + Ok(Some(Ok((index, row)))) => rows[index] = row, + Ok(Some(Err(_))) => {} + Ok(None) | Err(_) => break, + } + } + // Dropping JoinSet aborts in-flight requests, including when the root task is cancelled. + rows +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn timeline_query_uses_canonical_settings_and_escapes_model_names() { + let query = timeline_query( + &json!({"chartHours":24,"bucketMinutes":60,"tokenMetric":"input","aggregation":"max","chartGrouping":"modelAccount","models":["provider/model&x"]}), + ); + assert!(query.contains("metric=input")); + assert!(query.contains("aggregation=max")); + assert!(query.contains("models=provider%2Fmodel%26x")); + } + #[test] + fn successful_usage_survives_a_stalled_quota_section() { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_time() + .build() + .unwrap(); + runtime.block_on(async { + let mut result = snapshot::empty(); + merge( + &mut result, + Section { + key: "today", + value: Some(json!({"totalTokens":321})), + models: Some(vec![]), + error: "Usage failed", + }, + ); + let value = bounded(std::future::pending::(), Duration::ZERO).await; + merge( + &mut result, + Section { + key: "providers", + value, + models: None, + error: "Account limits are unavailable.", + }, + ); + assert_eq!(result["today"]["totalTokens"], 321); + assert_eq!(result["errors"], json!(["Account limits are unavailable."])); + assert!(result["updatedAt"].as_f64().is_some()); + }); + } + + #[test] + fn real_collector_publishes_usage_before_a_nonresponsive_account_endpoint() { + use crate::{auth::Auth, endpoint::ProxyEndpoint}; + use std::{ + io::{Read, Write}, + net::{TcpListener, TcpStream}, + sync::{Arc, Condvar, Mutex}, + thread, + }; + struct Fixture { + port: u16, + stop: Arc<(Mutex, Condvar)>, + worker: Option>, + } + impl Drop for Fixture { + fn drop(&mut self) { + *self.stop.0.lock().unwrap() = true; + self.stop.1.notify_all(); + let _ = TcpStream::connect(("127.0.0.1", self.port)); + if let Some(worker) = self.worker.take() { + worker.join().unwrap(); + } + } + } + let listener = TcpListener::bind("127.0.0.1:0").unwrap(); + let port = listener.local_addr().unwrap().port(); + let stop = Arc::new((Mutex::new(false), Condvar::new())); + let stopped = stop.clone(); + let worker = thread::spawn(move || { + let mut requests = Vec::new(); + for connection in listener.incoming() { + if *stopped.0.lock().unwrap() { + break; + } + let mut stream = connection.unwrap(); + let stop = stopped.clone(); + requests.push(thread::spawn(move || { + stream.set_read_timeout(Some(Duration::from_secs(2))).unwrap(); + let mut bytes = [0; 4096]; let count = stream.read(&mut bytes).unwrap(); + let request = String::from_utf8_lossy(&bytes[..count]); + let path = request.split_whitespace().nth(1).unwrap_or(""); + if path.starts_with("/api/oauth/accounts") { + let _guard = stop.1.wait_while(stop.0.lock().unwrap(), |stop| !*stop).unwrap(); + return; + } + let body = if path == "/api/companion/settings" { + json!({"settings":{"showToday":true,"showChart":false,"showModels":true,"showAccounts":true,"showCost":true,"chartStyle":"line","models":null,"hiddenProviders":[]}}) + } else if path == "/api/config" { + json!({"providers":{"test-oauth":{"authMode":"oauth"}}}) + } else { + json!({"summary":{"requests":1,"measuredRequests":1,"totalTokens":321},"models":[]}) + }.to_string(); + let response = format!("HTTP/1.1 200 OK\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}",body.len()); + let _ = stream.write_all(response.as_bytes()); + })); + } + for request in requests { + request.join().unwrap(); + } + }); + let _fixture = Fixture { + port, + stop, + worker: Some(worker), + }; + let proxy = ProxyClient::new( + ProxyEndpoint { + host: "127.0.0.1", + port, + }, + Auth::new(std::env::temp_dir().join("native-tray-no-credentials")), + ) + .unwrap(); + let snapshots = Mutex::new(Vec::new()); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .unwrap(); + let result = runtime.block_on(load(&proxy, snapshot::empty(), |value| { + snapshots.lock().unwrap().push(value) + })); + assert_eq!(result["today"]["totalTokens"], 321.0); + assert_eq!(result["providers"][0]["unavailable"], true); + assert_eq!(result["refreshing"], false); + assert!(snapshots + .lock() + .unwrap() + .iter() + .any(|value: &Value| value["today"]["totalTokens"] == 321.0 + && value["refreshing"] == true + && value["providers"].as_array().unwrap().is_empty())); + } +} diff --git a/desktop/src-tauri/src/native_tray_snapshot.rs b/desktop/src-tauri/src/native_tray_snapshot.rs new file mode 100644 index 0000000000..c81247fa30 --- /dev/null +++ b/desktop/src-tauri/src/native_tray_snapshot.rs @@ -0,0 +1,101 @@ +//! Native display settings and timeline projection. +pub use crate::companion_usage::{hidden, number, text, usage}; +use serde_json::{json, Value}; + +pub fn display_settings(settings: Option<&Value>) -> Value { + let enabled = |key| settings.is_some_and(|s| s[key].as_bool() == Some(true)); + json!({ + "showToday": enabled("showToday"), "show30Days": settings.is_some(), + "showChart": enabled("showChart"), "showModels": enabled("showModels"), + "showAccounts": settings.is_none_or(|s| s["showAccounts"].as_bool() != Some(false)), + "showCost": enabled("showCost"), + "chartStyle": settings.map(|s| text(s, "chartStyle")).unwrap_or("line") + }) +} + +pub fn empty() -> Value { + json!({"schemaVersion":1,"refreshing":true,"errors":[],"updatedAt":null, + "settings":display_settings(None),"today":null,"month":null,"models":[],"chart":null,"providers":[]}) +} + +pub fn chart(body: &Value, settings: &Value) -> Option { + let start = number(&body["start"])?; + let bucket = number(&body["bucketSeconds"])?; + if bucket == 0.0 || start >= 253_402_300_800.0 { + return None; + } + let (rows, incomplete) = crate::companion_query::timeline_rows(body, settings)?; + let series: Option> = rows + .into_iter() + .enumerate() + .map(|(index, row)| { + let points: Option> = row["points"].as_array()?.iter().map(number).collect(); + let label = row["id"].as_str()?; + Some(json!({"id":format!("{}:{index}",text(row,"id")),"label":label,"points":points?})) + }) + .collect(); + Some( + json!({"start":start,"bucketSeconds":bucket,"series":series?, + "incomplete":incomplete}), + ) +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn projection_filters_and_does_not_invent_measurements_or_copy_secrets() { + let settings = json!({"models":["a/kept"],"hiddenProviders":[]}); + let body = json!({"apiKey":"do-not-copy", "summary":{"totalTokens":100},"models":[ + {"provider":"a","model":"kept","requests":2,"measuredRequests":0,"totalTokens":0,"apiKey":"hidden"}, + {"provider":"b","model":"dropped","requests":3,"totalTokens":100}]}); + let (totals, models) = usage(&body, &settings).unwrap(); + assert_eq!(totals["requests"], 2.0); + assert!(totals["inputTokens"].is_null()); + assert!(models[0]["tokens"].is_null()); + assert_eq!(models.len(), 1); + assert!(!json!([totals, models]).to_string().contains("apiKey")); + } + #[test] + fn no_matches_and_malformed_reports_remain_unknown() { + let settings = json!({"models":[],"hiddenProviders":[]}); + let (totals, models) = + usage(&json!({"summary":{"requests":4},"models":[]}), &settings).unwrap(); + assert!(totals["requests"].is_null()); + assert!(models.is_empty()); + assert!(usage(&json!({"summary":{},"models":"bad"}), &settings).is_none()); + } + #[test] + fn cache_alias_and_incomplete_chart_are_preserved() { + let settings = json!({"models":null,"hiddenProviders":[]}); + let (totals,_)=usage(&json!({"summary":{"cacheReadInputTokens":9,"cachedInputTokens":2},"models":[],"historyTruncated":true}),&settings).unwrap(); + assert_eq!(totals["cachedInputTokens"], 9.0); + assert_eq!(totals["incomplete"], true); + let c = chart( + &json!({"start":1000,"bucketSeconds":60,"series":[],"missingMeasurements":1}), + &settings, + ) + .unwrap(); + assert_eq!(c["incomplete"], true); + assert!(chart( + &json!({"start":1000,"bucketSeconds":0,"series":[]}), + &settings + ) + .is_none()); + } + + #[test] + fn same_model_from_two_providers_keeps_distinct_series_labels() { + let settings = json!({"models":null,"hiddenProviders":[]}); + let value = chart( + &json!({"start":1000,"bucketSeconds":60,"series":[ + {"id":"first/shared","provider":"first","model":"shared","points":[1,2]}, + {"id":"second/shared","provider":"second","model":"shared","points":[3,4]}]}), + &settings, + ) + .unwrap(); + assert_eq!(value["series"][0]["label"], "first/shared"); + assert_eq!(value["series"][1]["label"], "second/shared"); + assert_ne!(value["series"][0]["id"], value["series"][1]["id"]); + } +} diff --git a/desktop/src-tauri/src/ownership.rs b/desktop/src-tauri/src/ownership.rs new file mode 100644 index 0000000000..5b30eb366f --- /dev/null +++ b/desktop/src-tauri/src/ownership.rs @@ -0,0 +1,303 @@ +//! Who owns the running proxy, as the shared service install state records it. +//! +//! The rule is not this lane's to invent. `src/service/state.ts` defines the claim — an owner, an +//! opaque install id naming the owning installation, and a consent generation — and +//! `ownershipGrantedTo` defines the comparison an installation applies to its own locally stored +//! install id. This is that comparison, and the three-valued reading it is applied to, so the shell +//! reaches the same verdict the CLI does instead of a weaker one of its own. +//! +//! What the shell deliberately does not do is read the record itself. Resolving a claim means +//! reading every state path and failing closed on an unreadable one, on a corrupt anchor record and +//! on paths that name different owners; absence is the only thing that means nobody owns the +//! runtime. Reimplementing that here is how `discovery.rs` ended up asking a weaker liveness +//! question than the one core already answered. The bundled CLI answers it: see [`resolve`]. +//! +//! The types below are the CLI's own answer as it will arrive on the wire, field for field, so the +//! contract that lands fills a hole rather than reshaping this file. + +use serde::Deserialize; + +/// Who a claim names. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum Owner { + Cli, + Desktop, +} + +/// A recorded ownership claim. +#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct Claim { + pub owner: Owner, + pub install_id: String, + /// Moves once per grant, and never back: the recorded ceiling survives a release so a later + /// grant cannot reuse a number an app-local record may still be holding. + /// + /// The record accepts any non-negative integer, and this accepts the ones it can represent. A + /// generation outside that range fails to parse, which makes the whole answer unreadable and + /// so refuses a takeover — the safe direction, and unreachable in practice by a counter that + /// moves by one per grant. + pub consent_generation: u64, +} + +/// What the recorded state says, in the CLI's own three answers. +#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[serde(tag = "kind", rename_all = "lowercase")] +pub enum Recorded { + /// No claim. The CLI install that registered the service owns the runtime, which is also what + /// every record written before the field existed says. The revision is the record's own + /// sequence, and a later `service claim` carries it as `expect-revision`. + None { revision: u64 }, + /// A claim, whoever it names. + Owned { ownership: Claim, revision: u64 }, + /// The claim could not be read for a decision. This is not "nobody owns it": an unreadable + /// path, a corrupt anchor record and paths naming different owners all land here. + Unknown { reason: String }, +} + +impl Default for Recorded { + /// A resolve document that carries no ownership field at all did not answer the question — + /// the older bundled CLI predates it — and an unanswered question is not a claim. + fn default() -> Self { + Self::Unknown { + reason: "the bundled CLI did not report ownership".to_owned(), + } + } +} + +/// The comparison `ownershipGrantedTo` defines: same owner, same install id. +/// +/// True means this installation already holds consent. False against a recorded claim means a +/// different installation owns the runtime and consent has to be asked again. No claim means the +/// CLI install still owns it. The generation is not part of the comparison. +pub fn granted_to(claim: Option<&Claim>, owner: Owner, install_id: &str) -> bool { + matches!(claim, Some(claim) if claim.owner == owner && claim.install_id == install_id) +} + +/// What this installation should do about consent. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Consent { + /// This installation already holds consent. Asked once, owned permanently — so a second launch + /// does not ask again. + Held, + /// Nothing is recorded, so the CLI install still owns the runtime. This is the first discovery + /// of an existing installation, and the one time the user is asked. + AskFirstTime, + /// A claim exists and it is not ours: another installation, or the CLI explicitly. + AskAgain, + /// The record could not be read for a decision, so nothing is taken over on a guess. + Refuse, +} + +pub fn consent(recorded: &Recorded, install_id: &str) -> Consent { + match recorded { + Recorded::Unknown { .. } => Consent::Refuse, + Recorded::None { .. } => Consent::AskFirstTime, + Recorded::Owned { ownership, .. } => { + if granted_to(Some(ownership), Owner::Desktop, install_id) { + Consent::Held + } else { + Consent::AskAgain + } + } + } +} + +/// One line for the startup state and for the diagnostic. +pub fn describe(recorded: &Recorded, install_id: Option<&str>) -> String { + let installation = match install_id { + Some(id) => format!("installation {id}"), + None => "installation id unavailable".to_owned(), + }; + let verdict = match (recorded, install_id) { + (Recorded::Unknown { reason }, _) => { + format!("recorded owner could not be read ({reason}), so nothing is claimed") + } + (_, None) => "recorded owner read, but this installation has no id to compare".to_owned(), + (_, Some(id)) => match (consent(recorded, id), recorded) { + (Consent::Held, Recorded::Owned { ownership, .. }) => format!( + "this installation owns the runtime (consent generation {})", + ownership.consent_generation + ), + (Consent::AskFirstTime, _) => "the CLI install owns the runtime".to_owned(), + (Consent::AskAgain, _) => "another installation owns the runtime".to_owned(), + _ => "recorded owner could not be read, so nothing is claimed".to_owned(), + }, + }; + format!("{installation}; {verdict}") +} + +/// Who the recorded claim names, for the consent panel. +pub fn owner_label(recorded: &Recorded) -> String { + match recorded { + Recorded::None { .. } => "no recorded owner (an npm or standalone ocx install)".to_owned(), + Recorded::Owned { ownership, .. } => match ownership.owner { + Owner::Cli => format!( + "the OpenCodex CLI install (installation {})", + ownership.install_id + ), + Owner::Desktop => format!( + "another OpenCodex desktop installation (installation {})", + ownership.install_id + ), + }, + Recorded::Unknown { reason } => format!("unknown ({reason})"), + } +} + +#[cfg(test)] +mod tests { + use super::{consent, describe, granted_to, owner_label, Claim, Consent, Owner, Recorded}; + + fn owned(owner: Owner, install_id: &str, generation: u64) -> Recorded { + Recorded::Owned { + ownership: Claim { + owner, + install_id: install_id.to_owned(), + consent_generation: generation, + }, + revision: 4, + } + } + + fn claim(owner: Owner, install_id: &str) -> Claim { + Claim { + owner, + install_id: install_id.to_owned(), + consent_generation: 1, + } + } + + #[test] + fn the_comparison_is_the_owner_and_the_install_id_together() { + let ours = claim(Owner::Desktop, "abc"); + assert!(granted_to(Some(&ours), Owner::Desktop, "abc")); + assert!(!granted_to(Some(&ours), Owner::Desktop, "def")); + assert!(!granted_to(Some(&ours), Owner::Cli, "abc")); + assert!(!granted_to(None, Owner::Desktop, "abc")); + } + + #[test] + fn the_generation_is_not_part_of_the_comparison() { + let mut later = claim(Owner::Desktop, "abc"); + later.consent_generation = 9; + assert!(granted_to(Some(&later), Owner::Desktop, "abc")); + } + + #[test] + fn consent_is_asked_once_and_then_held() { + assert_eq!( + consent(&owned(Owner::Desktop, "abc", 1), "abc"), + Consent::Held + ); + assert_eq!( + consent(&Recorded::None { revision: 0 }, "abc"), + Consent::AskFirstTime + ); + } + + #[test] + fn a_claim_that_is_not_ours_asks_again() { + assert_eq!( + consent(&owned(Owner::Desktop, "other", 2), "abc"), + Consent::AskAgain + ); + assert_eq!( + consent(&owned(Owner::Cli, "abc", 1), "abc"), + Consent::AskAgain + ); + } + + #[test] + fn an_unreadable_record_refuses_instead_of_reading_as_unowned() { + let unknown = Recorded::Unknown { + reason: "a service state path could not be read".to_owned(), + }; + assert_eq!(consent(&unknown, "abc"), Consent::Refuse); + assert!(describe(&unknown, Some("abc")).contains("could not be read")); + } + + #[test] + fn the_wire_shape_is_the_one_the_cli_records() { + let resolution: Recorded = serde_json::from_str( + r#"{"kind":"owned","ownership":{"owner":"desktop","installId":"abc","consentGeneration":3},"revision":4}"#, + ) + .expect("the recorded resolution"); + assert_eq!(resolution, owned(Owner::Desktop, "abc", 3)); + assert_eq!(consent(&resolution, "abc"), Consent::Held); + assert!(describe(&resolution, Some("abc")).contains("consent generation 3")); + assert_eq!( + serde_json::from_str::(r#"{"kind":"none","revision":0}"#).expect("no claim"), + Recorded::None { revision: 0 } + ); + assert_eq!( + serde_json::from_str::(r#"{"kind":"unknown","reason":"why"}"#) + .expect("a refusal"), + Recorded::Unknown { + reason: "why".to_owned() + } + ); + } + + #[test] + fn a_resolve_document_without_an_ownership_answer_reads_unknown() { + assert_eq!( + Recorded::default(), + Recorded::Unknown { + reason: "the bundled CLI did not report ownership".to_owned() + } + ); + assert!(serde_json::from_str::(r#"{"kind":"none"}"#).is_err()); + } + + #[test] + fn the_description_separates_not_read_from_nobody_owns_it() { + let unread = describe( + &Recorded::Unknown { + reason: "why".to_owned(), + }, + Some("abc"), + ); + let unowned = describe(&Recorded::None { revision: 0 }, Some("abc")); + assert!(unread.contains("abc")); + assert_ne!(unread, unowned); + assert!(describe(&Recorded::None { revision: 0 }, None).contains("unavailable")); + } + + #[test] + fn owner_label_names_who_the_claim_is_for() { + assert_eq!( + owner_label(&Recorded::None { revision: 0 }), + "no recorded owner (an npm or standalone ocx install)" + ); + assert_eq!( + owner_label(&owned(Owner::Cli, "npm-1", 1)), + "the OpenCodex CLI install (installation npm-1)" + ); + assert_eq!( + owner_label(&owned(Owner::Desktop, "other", 2)), + "another OpenCodex desktop installation (installation other)" + ); + assert_eq!( + owner_label(&Recorded::Unknown { + reason: "why".to_owned() + }), + "unknown (why)" + ); + } + + #[test] + fn a_generation_this_cannot_represent_is_not_read_as_a_claim() { + // Refusing beats granting on a number we cannot compare, and the claim is what a takeover + // would be authorised against. + assert!(serde_json::from_str::( + r#"{"kind":"owned","ownership":{"owner":"desktop","installId":"abc","consentGeneration":-1}}"# + ) + .is_err()); + assert!(serde_json::from_str::( + r#"{"kind":"owned","ownership":{"owner":"cli","installId":"abc","consentGeneration":"3"}}"# + ) + .is_err()); + } +} diff --git a/desktop/src-tauri/src/popup.rs b/desktop/src-tauri/src/popup.rs new file mode 100644 index 0000000000..5eac99d8d3 --- /dev/null +++ b/desktop/src-tauri/src/popup.rs @@ -0,0 +1,424 @@ +use crate::{endpoint::ProxyEndpoint, window}; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::{Duration, Instant}; +use tauri::webview::PageLoadEvent; +#[cfg(target_os = "macos")] +use tauri::window::EffectState; +#[cfg(any(target_os = "macos", target_os = "windows"))] +use tauri::window::{Effect, EffectsBuilder}; +use tauri::{ + AppHandle, Manager, PhysicalPosition, PhysicalRect, PhysicalSize, Url, WebviewUrl, + WebviewWindow, WebviewWindowBuilder, WindowEvent, +}; + +pub const LABEL: &str = "usage-popup"; +pub const TRAY_PATH: &str = "/#/tray"; +pub const DASHBOARD_PATH: &str = "/?desktop=open#/usage"; +pub const CLOSE_PATH: &str = "/?desktop=popup-close#/tray-close"; +#[cfg(any(target_os = "macos", target_os = "windows"))] +pub const VIBRANT_SURFACE: bool = true; +#[cfg(not(any(target_os = "macos", target_os = "windows")))] +pub const VIBRANT_SURFACE: bool = false; +const TRAY_VIBRANCY_DATASET: &str = "document.documentElement.dataset.trayVibrancy"; +pub const ESCAPE_INITIALIZATION_SCRIPT: &str = r#" +(() => { + window.__OPENCODEX_TRAY_VISIBLE__ = false; + document.addEventListener("keydown", (event) => { + if (event.key !== "Escape") return; + event.preventDefault(); + event.stopPropagation(); + window.location.replace("/?desktop=popup-close#/tray-close"); + }, true); +})(); +"#; + +fn initialization_script() -> String { + let tray_vibrancy = if VIBRANT_SURFACE { "on" } else { "off" }; + format!( + r#"{TRAY_VIBRANCY_DATASET} = "{tray_vibrancy}"; +{ESCAPE_INITIALIZATION_SCRIPT}"# + ) +} + +/// How long after being shown the popup ignores losing focus. +/// +/// Closing on focus loss is what makes this feel like a menu rather than a window. The cost is +/// that a platform which hands focus back to the tray, the shell, or nothing at all right after +/// the click closes the popup in the same gesture that opened it -- the user sees a flash and no +/// window. A short grace period keeps the dismiss behaviour while making that race unreachable; +/// it is deliberately shorter than a deliberate click elsewhere. +const FOCUS_GRACE: Duration = Duration::from_millis(400); + +/// Monotonic milliseconds since process start, written when the popup is shown. +static SHOWN_AT_MS: AtomicU64 = AtomicU64::new(0); + +fn process_start() -> Instant { + use std::sync::OnceLock; + static START: OnceLock = OnceLock::new(); + *START.get_or_init(Instant::now) +} + +fn mark_shown() { + let elapsed = process_start().elapsed().as_millis() as u64; + SHOWN_AT_MS.store(elapsed, Ordering::Release); +} + +fn within_focus_grace() -> bool { + let shown = SHOWN_AT_MS.load(Ordering::Acquire); + if shown == 0 { + return false; + } + let now = process_start().elapsed().as_millis() as u64; + now.saturating_sub(shown) < FOCUS_GRACE.as_millis() as u64 +} + +const WIDTH_LOGICAL: f64 = 440.0; +const HEIGHT_LOGICAL: f64 = 700.0; +const EDGE_PHYSICAL: i64 = 8; +const GAP_PHYSICAL: i64 = 6; + +#[derive(Clone, Copy, Debug, PartialEq)] +pub struct PopupGeometry { + pub position: PhysicalPosition, + pub size: PhysicalSize, +} + +/// Calculates a tray-anchored physical rectangle. `anchor` and `work_area` are physical pixels; +/// `scale_factor` only converts the logical 440x700 design size, so mixed-DPI monitors stay exact. +pub fn geometry( + anchor: PhysicalPosition, + work_area: PhysicalRect, + scale_factor: f64, +) -> PopupGeometry { + let scale = if scale_factor.is_finite() && scale_factor > 0.0 { + scale_factor + } else { + 1.0 + }; + let edge = EDGE_PHYSICAL; + let gap = GAP_PHYSICAL; + let left = work_area.position.x as i64; + let top = work_area.position.y as i64; + let right = left + work_area.size.width as i64; + let bottom = top + work_area.size.height as i64; + let available_width = (right - left - edge * 2).max(1) as u32; + let available_height = (bottom - top - edge * 2).max(1) as u32; + let width = ((WIDTH_LOGICAL * scale).round() as u32).min(available_width); + let height = ((HEIGHT_LOGICAL * scale).round() as u32).min(available_height); + let width_i = width as i64; + let height_i = height as i64; + let min_x = left + edge; + let max_x = (right - edge - width_i).max(min_x); + let min_y = top + edge; + let max_y = (bottom - edge - height_i).max(min_y); + let anchor_x = anchor.x.round() as i64; + let anchor_y = anchor.y.round() as i64; + let x = (anchor_x - width_i / 2).clamp(min_x, max_x); + let below = anchor_y + gap; + let above = anchor_y - gap - height_i; + let y = if below <= max_y { below } else { above }.clamp(min_y, max_y); + + PopupGeometry { + position: PhysicalPosition::new(x as i32, y as i32), + size: PhysicalSize::new(width, height), + } +} + +pub fn show( + app: &AppHandle, + endpoint: ProxyEndpoint, + anchor: PhysicalPosition, +) -> tauri::Result<()> { + let popup = ensure(app, endpoint)?; + if let Some(monitor) = popup + .monitor_from_point(anchor.x, anchor.y) + .ok() + .flatten() + .or_else(|| popup.primary_monitor().ok().flatten()) + { + let layout = geometry(anchor, *monitor.work_area(), monitor.scale_factor()); + let _ = popup.set_size(layout.size); + let _ = popup.set_position(layout.position); + } + if popup + .url() + .map(|url| !is_tray_url(&url, endpoint)) + .unwrap_or(true) + { + popup.navigate(proxy_url(endpoint, TRAY_PATH))?; + } + let was_visible = popup.is_visible().unwrap_or(false); + mark_shown(); + popup.show()?; + popup.set_focus()?; + if !was_visible { + set_visibility(&popup, true); + } + Ok(()) +} + +pub fn toggle( + app: &AppHandle, + endpoint: ProxyEndpoint, + anchor: PhysicalPosition, +) -> tauri::Result<()> { + if app + .get_webview_window(LABEL) + .and_then(|popup| popup.is_visible().ok()) + .unwrap_or(false) + { + hide(app); + Ok(()) + } else { + show(app, endpoint, anchor) + } +} + +pub fn hide(app: &AppHandle) { + if let Some(popup) = app.get_webview_window(LABEL) { + if popup.is_visible().unwrap_or(false) { + let _ = popup.hide(); + set_visibility(&popup, false); + } + } +} + +fn ensure(app: &AppHandle, endpoint: ProxyEndpoint) -> tauri::Result { + if let Some(popup) = app.get_webview_window(LABEL) { + return Ok(popup); + } + + let app_handle = app.clone(); + let mut builder = WebviewWindowBuilder::new( + app, + LABEL, + WebviewUrl::External(proxy_url(endpoint, TRAY_PATH)), + ) + .title("OpenCodex Usage") + .inner_size(WIDTH_LOGICAL, HEIGHT_LOGICAL) + .max_inner_size(WIDTH_LOGICAL, HEIGHT_LOGICAL) + .decorations(false) + .resizable(false) + .always_on_top(true) + .skip_taskbar(true) + .visible(false) + .user_agent(&window::webview_user_agent()) + .initialization_script(initialization_script()) + .on_navigation(popup_navigation_allowed(endpoint, app_handle.clone())) + .on_new_window(window::open_new_windows_in_default_browser()) + .on_page_load(|popup, payload| { + if matches!(payload.event(), PageLoadEvent::Finished) { + set_visibility(&popup, popup.is_visible().unwrap_or(false)); + } + }); + if VIBRANT_SURFACE { + builder = builder.transparent(true); + #[cfg(target_os = "macos")] + { + builder = builder.effects( + EffectsBuilder::new() + .effect(Effect::HudWindow) + .state(EffectState::Active) + .radius(12.0) + .build(), + ); + } + #[cfg(target_os = "windows")] + { + builder = builder.effects(EffectsBuilder::new().effect(Effect::Acrylic).build()); + } + } + let popup = builder.build()?; + popup.on_window_event(move |event| match event { + WindowEvent::Focused(false) if !within_focus_grace() => { + hide(&app_handle); + } + WindowEvent::CloseRequested { api, .. } => { + api.prevent_close(); + hide(&app_handle); + } + _ => {} + }); + Ok(popup) +} + +fn popup_navigation_allowed( + endpoint: ProxyEndpoint, + app: AppHandle, +) -> impl Fn(&Url) -> bool + Send + 'static { + move |url| { + if !same_origin(url, endpoint) { + // WKWebView asks this policy before it would create a window for a `_blank` link, so + // refusing here without opening is what left the popup's external links dead on macOS. + window::open_in_default_browser(url); + return false; + } + if is_close_url(url, endpoint) { + hide(&app); + return false; + } + if is_dashboard_url(url, endpoint) { + hide(&app); + if let Some(main) = app.get_webview_window("main") { + window::show(&main); + let session = app + .state::() + .session_id() + .to_string(); + let _ = main.navigate(dashboard_destination(url, &session)); + } + return false; + } + is_tray_url(url, endpoint) + } +} + +fn proxy_url(endpoint: ProxyEndpoint, path: &str) -> Url { + endpoint + .url(path) + .parse() + .expect("proxy endpoint URL is valid") +} + +fn same_origin(url: &Url, endpoint: ProxyEndpoint) -> bool { + url.scheme() == "http" + && url.host_str() == Some(endpoint.host) + && url.port_or_known_default() == Some(endpoint.port) +} + +/// Split one of the paths above into the query and fragment a navigation must carry. +/// +/// The matchers read the constant instead of restating it. A matcher that restated it would +/// keep answering yes after the page it names moved, and these three decide what the popup is +/// allowed to navigate to, so a stale yes is the failure that matters. +fn parts(path: &str) -> (Option<&str>, Option<&str>) { + let (before_fragment, fragment) = match path.split_once('#') { + Some((before, fragment)) => (before, Some(fragment)), + None => (path, None), + }; + ( + before_fragment.split_once('?').map(|(_, query)| query), + fragment, + ) +} + +fn matches(url: &Url, endpoint: ProxyEndpoint, path: &str) -> bool { + let (query, fragment) = parts(path); + same_origin(url, endpoint) + && url.path() == "/" + && url.query() == query + && url.fragment() == fragment +} + +fn is_tray_url(url: &Url, endpoint: ProxyEndpoint) -> bool { + matches(url, endpoint, TRAY_PATH) +} + +fn is_close_url(url: &Url, endpoint: ProxyEndpoint) -> bool { + matches(url, endpoint, CLOSE_PATH) +} + +fn is_dashboard_url(url: &Url, endpoint: ProxyEndpoint) -> bool { + // The dashboard accepts the usage page and its companion view under the same query. + let (query, _) = parts(DASHBOARD_PATH); + same_origin(url, endpoint) + && url.path() == "/" + && url.query() == query + && matches!(url.fragment(), Some("/usage") | Some("/usage/companion")) +} + +fn dashboard_destination(url: &Url, session: &str) -> Url { + let mut destination = url.clone(); + destination + .query_pairs_mut() + .append_pair("desktop_session", session); + destination +} + +fn set_visibility(popup: &WebviewWindow, visible: bool) { + let script = format!( + "window.__OPENCODEX_TRAY_VISIBLE__ = {visible}; window.dispatchEvent(new CustomEvent('opencodex:tray-visibility', {{detail: {visible}}}));" + ); + let _ = popup.eval(script); +} + +#[cfg(test)] +mod tests { + use super::*; + + const ENDPOINT: ProxyEndpoint = ProxyEndpoint { + host: "127.0.0.1", + port: 53998, + }; + + #[test] + fn geometry_uses_physical_dpi_and_clamps_to_work_area() { + let layout = geometry( + PhysicalPosition::new(1_900.0, 1_050.0), + PhysicalRect { + position: PhysicalPosition::new(0, 0), + size: PhysicalSize::new(2_560, 1_440), + }, + 2.0, + ); + assert_eq!(layout.size, PhysicalSize::new(880, 1400)); + assert_eq!(layout.position.x, 1_460); + assert_eq!(layout.position.y, 8); + } + + #[test] + fn geometry_keeps_top_tray_below_and_clamps_left() { + let layout = geometry( + PhysicalPosition::new(-20.0, 20.0), + PhysicalRect { + position: PhysicalPosition::new(-1_280, 0), + size: PhysicalSize::new(1_280, 800), + }, + 1.0, + ); + assert_eq!(layout.position.x, -448); + assert_eq!(layout.position.y, 26); + assert_eq!(layout.size, PhysicalSize::new(440, 700)); + } + + #[test] + fn navigation_accepts_only_tray_close_and_dashboard_sentinels() { + let tray: Url = ENDPOINT.url(TRAY_PATH).parse().unwrap(); + let close: Url = ENDPOINT.url(CLOSE_PATH).parse().unwrap(); + let dashboard: Url = ENDPOINT.url(DASHBOARD_PATH).parse().unwrap(); + let external: Url = "https://example.com/#/tray".parse().unwrap(); + assert!(is_tray_url(&tray, ENDPOINT)); + assert!(is_close_url(&close, ENDPOINT)); + assert!(is_dashboard_url(&dashboard, ENDPOINT)); + assert!(!is_tray_url(&external, ENDPOINT)); + assert!(!is_tray_url( + &ENDPOINT.url("/#/usage").parse().unwrap(), + ENDPOINT + )); + } + + #[test] + fn dashboard_navigation_keeps_the_validated_fragment() { + for fragment in ["/usage", "/usage/companion"] { + let source: Url = ENDPOINT + .url(&format!("/?desktop=open#{fragment}")) + .parse() + .unwrap(); + assert!(is_dashboard_url(&source, ENDPOINT)); + let destination = dashboard_destination(&source, "session-123"); + assert_eq!( + destination.as_str(), + ENDPOINT.url(&format!( + "/?desktop=open&desktop_session=session-123#{fragment}" + )) + ); + } + } + + #[test] + fn initialization_script_matches_native_surface() { + let expected_value = if VIBRANT_SURFACE { "on" } else { "off" }; + let expected = format!(r#"{TRAY_VIBRANCY_DATASET} = "{expected_value}";"#); + assert!(initialization_script().contains(&expected)); + } +} diff --git a/desktop/src-tauri/src/provider_icons.rs b/desktop/src-tauri/src/provider_icons.rs new file mode 100644 index 0000000000..750bd5cb7b --- /dev/null +++ b/desktop/src-tauri/src/provider_icons.rs @@ -0,0 +1,248 @@ +//! Provider marks for the native menu bar panel. +//! +//! The SVG files are the dashboard's own (`gui/public/provider-icons`), embedded at build time so +//! the panel needs no web view or file access. `ALIASES` and the paint sets mirror +//! `PROVIDER_ICON_ALIASES` and `providerIconPaint` in `gui/src/provider-icons.ts`; +//! `gui/tests/provider-icons-native.test.ts` fails when the two drift apart. + +/// One provider's mark and how to paint it: `image` as drawn, `mask` as a template tinted with +/// the label color, `plate` / `dark-plate` on a constant light or dark plate. +pub struct ProviderIcon { + pub svg: &'static str, + pub paint: &'static str, +} + +macro_rules! svg { + ($file:literal) => { + include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../gui/public/provider-icons/", + $file + )) + }; +} + +const ALIASES: &[(&str, &str)] = &[ + ("anthropic", "claude-color.svg"), + ("anthropic-apikey", "claude-color.svg"), + ("claude-cli", "claude-color.svg"), + ("azure-openai", "openai.svg"), + ("chatgpt", "openai.svg"), + ("cloudflare-ai-gateway", "cloudflare-ai-gateway-color.svg"), + ("cloudflare-workers-ai", "cloudflare-ai-gateway-color.svg"), + ("cline", "cline-color.svg"), + ("cline-pass", "cline-color.svg"), + ("command-code", "commandcode-color.svg"), + ("commandcode", "commandcode-color.svg"), + ("cursor", "cursor-color.svg"), + ("deepseek", "deepseek-color.svg"), + ("devin", "devin.svg"), + ("firepass", "firepass-color.svg"), + ("fireworks", "fireworks-color.svg"), + ("github", "github-copilot-color.svg"), + ("github-copilot", "copilot-color.svg"), + ("gitlab-duo", "gitlab-duo-color.svg"), + ("google", "gemini-color.svg"), + ("google-antigravity", "antigravity-color.svg"), + ("google-vertex", "gemini-color.svg"), + ("groq", "groq-color.svg"), + ("huggingface", "huggingface-color.svg"), + ("kimi", "kimi-color.svg"), + ("kimi-code", "kimi-color.svg"), + ("kimi-responses", "kimi-color.svg"), + ("kiro", "kiro-color.svg"), + ("lm-studio", "lm-studio-color.svg"), + ("meta-model", "meta.svg"), + ("meta-muse", "meta.svg"), + ("mistral", "mistral-color.svg"), + ("minimax", "minimax.svg"), + ("minimax-cn", "minimax.svg"), + ("moonshot", "moonshot-color.svg"), + ("nvidia", "nvidia-color.svg"), + ("ollama", "ollama-color.svg"), + ("ollama-cloud", "ollama-color.svg"), + ("openai", "openai.svg"), + ("openai-apikey", "openai.svg"), + ("opencode-free", "opencode.svg"), + ("opencode-go", "opencode.svg"), + ("opencode-zen", "opencode.svg"), + ("openrouter", "openrouter-color.svg"), + ("opper", "opper.svg"), + ("qianfan", "qianfan-color.svg"), + ("qoder", "qoder.svg"), + ("qoder-cn", "qoder.svg"), + ("alibaba", "alibaba-color.svg"), + ("alibaba-token-plan", "alibaba-color.svg"), + ("alibaba-token-plan-intl", "alibaba-color.svg"), + ("baseten", "baseten.svg"), + ("bizrouter", "bizrouter.svg"), + ("cerebras", "cerebras.svg"), + ("crusoe", "crusoe.svg"), + ("deepinfra", "deepinfra.svg"), + ("digitalocean", "digitalocean.svg"), + ("featherless", "featherless.svg"), + ("hyperbolic", "hyperbolic.svg"), + ("kilo", "kilo.svg"), + ("nanogpt", "nanogpt.svg"), + ("nebius", "nebius.svg"), + ("neuralwatt", "neuralwatt.svg"), + ("nous", "nous.svg"), + ("novita", "novita.svg"), + ("orcarouter", "orcarouter.svg"), + ("orcarouter-oauth", "orcarouter.svg"), + ("packycode", "packycode.svg"), + ("tokenlab", "tokenlab.svg"), + ("parallel", "parallel.svg"), + ("sambanova", "sambanova.svg"), + ("scaleway", "scaleway.svg"), + ("stepfun", "stepfun-color.svg"), + ("siliconflow", "siliconflow.svg"), + ("synthetic", "synthetic.svg"), + ("together", "together.svg"), + ("umans", "umans.svg"), + ("venice", "venice.svg"), + ("vultr", "vultr.svg"), + ("litellm", "litellm.svg"), + ("zenmux", "zenmux.svg"), + ("zai", "zai.svg"), + ("zhipu-bigmodel", "zai.svg"), + ("zhipu-bigmodel-coding", "zai.svg"), + ("qwen-cloud", "qwen-portal-color.svg"), + ("vercel-ai-gateway", "vercel-ai-gateway-color.svg"), + ("vllm", "vllm-color.svg"), + ("xai", "grok.svg"), + ("mimo-free", "xiaomi-color.svg"), + ("mimo", "xiaomi-color.svg"), + ("xiaomi", "xiaomi-color.svg"), + ("xiaomi-mimo", "xiaomi-color.svg"), +]; + +fn paint(file: &str) -> &'static str { + match file { + "cerebras.svg" + | "deepinfra.svg" + | "grok.svg" + | "kimi-color.svg" + | "neuralwatt.svg" + | "nous.svg" + | "novita.svg" + | "ollama-color.svg" + | "opencode.svg" + | "opper.svg" + | "packycode.svg" + | "siliconflow.svg" + | "synthetic.svg" + | "tokenlab.svg" + | "vercel-ai-gateway-color.svg" + | "zenmux.svg" => "mask", + "baseten.svg" | "kilo.svg" | "sambanova.svg" | "venice.svg" | "zai.svg" => "plate", + "bizrouter.svg" | "featherless.svg" | "hyperbolic.svg" | "nebius.svg" | "parallel.svg" + | "umans.svg" => "dark-plate", + _ => "image", + } +} + +fn svg(file: &str) -> Option<&'static str> { + Some(match file { + "alibaba-color.svg" => svg!("alibaba-color.svg"), + "antigravity-color.svg" => svg!("antigravity-color.svg"), + "baseten.svg" => svg!("baseten.svg"), + "bizrouter.svg" => svg!("bizrouter.svg"), + "cerebras.svg" => svg!("cerebras.svg"), + "claude-color.svg" => svg!("claude-color.svg"), + "cline-color.svg" => svg!("cline-color.svg"), + "cloudflare-ai-gateway-color.svg" => svg!("cloudflare-ai-gateway-color.svg"), + "commandcode-color.svg" => svg!("commandcode-color.svg"), + "copilot-color.svg" => svg!("copilot-color.svg"), + "crusoe.svg" => svg!("crusoe.svg"), + "cursor-color.svg" => svg!("cursor-color.svg"), + "deepinfra.svg" => svg!("deepinfra.svg"), + "deepseek-color.svg" => svg!("deepseek-color.svg"), + "devin.svg" => svg!("devin.svg"), + "digitalocean.svg" => svg!("digitalocean.svg"), + "featherless.svg" => svg!("featherless.svg"), + "firepass-color.svg" => svg!("firepass-color.svg"), + "fireworks-color.svg" => svg!("fireworks-color.svg"), + "gemini-color.svg" => svg!("gemini-color.svg"), + "github-copilot-color.svg" => svg!("github-copilot-color.svg"), + "gitlab-duo-color.svg" => svg!("gitlab-duo-color.svg"), + "grok.svg" => svg!("grok.svg"), + "groq-color.svg" => svg!("groq-color.svg"), + "huggingface-color.svg" => svg!("huggingface-color.svg"), + "hyperbolic.svg" => svg!("hyperbolic.svg"), + "kilo.svg" => svg!("kilo.svg"), + "kimi-color.svg" => svg!("kimi-color.svg"), + "kiro-color.svg" => svg!("kiro-color.svg"), + "litellm.svg" => svg!("litellm.svg"), + "lm-studio-color.svg" => svg!("lm-studio-color.svg"), + "meta.svg" => svg!("meta.svg"), + "minimax.svg" => svg!("minimax.svg"), + "mistral-color.svg" => svg!("mistral-color.svg"), + "moonshot-color.svg" => svg!("moonshot-color.svg"), + "nanogpt.svg" => svg!("nanogpt.svg"), + "nebius.svg" => svg!("nebius.svg"), + "neuralwatt.svg" => svg!("neuralwatt.svg"), + "nous.svg" => svg!("nous.svg"), + "novita.svg" => svg!("novita.svg"), + "nvidia-color.svg" => svg!("nvidia-color.svg"), + "ollama-color.svg" => svg!("ollama-color.svg"), + "openai.svg" => svg!("openai.svg"), + "opencode.svg" => svg!("opencode.svg"), + "openrouter-color.svg" => svg!("openrouter-color.svg"), + "opper.svg" => svg!("opper.svg"), + "orcarouter.svg" => svg!("orcarouter.svg"), + "packycode.svg" => svg!("packycode.svg"), + "parallel.svg" => svg!("parallel.svg"), + "qianfan-color.svg" => svg!("qianfan-color.svg"), + "qoder.svg" => svg!("qoder.svg"), + "qwen-portal-color.svg" => svg!("qwen-portal-color.svg"), + "sambanova.svg" => svg!("sambanova.svg"), + "scaleway.svg" => svg!("scaleway.svg"), + "siliconflow.svg" => svg!("siliconflow.svg"), + "stepfun-color.svg" => svg!("stepfun-color.svg"), + "synthetic.svg" => svg!("synthetic.svg"), + "together.svg" => svg!("together.svg"), + "tokenlab.svg" => svg!("tokenlab.svg"), + "umans.svg" => svg!("umans.svg"), + "venice.svg" => svg!("venice.svg"), + "vercel-ai-gateway-color.svg" => svg!("vercel-ai-gateway-color.svg"), + "vllm-color.svg" => svg!("vllm-color.svg"), + "vultr.svg" => svg!("vultr.svg"), + "xiaomi-color.svg" => svg!("xiaomi-color.svg"), + "zai.svg" => svg!("zai.svg"), + "zenmux.svg" => svg!("zenmux.svg"), + _ => return None, + }) +} + +/// The mark for a provider id, matched case-insensitively like the dashboard. +pub fn icon(provider: &str) -> Option { + let key = provider.to_ascii_lowercase(); + let (_, file) = ALIASES.iter().find(|(alias, _)| *alias == key)?; + Some(ProviderIcon { + svg: svg(file)?, + paint: paint(file), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn every_alias_resolves_to_an_embedded_mark() { + for (alias, _) in ALIASES { + let icon = icon(alias).unwrap_or_else(|| panic!("{alias} has no embedded file")); + assert!(icon.svg.contains(" &'static str { + match self { + Self::Codex => "/api/codex-auth/active", + Self::OAuth => "/api/oauth/accounts/active", + Self::ApiKey => "/api/providers/keys/active", + } + } +} + +/// Which instance answered, taken from the unauthenticated health body. +/// +/// This is a discovery hint, not cryptographic proof of who holds the port. Management requests +/// carry a scoped capability keyed by the recorded runtime secret, never the reusable admin token. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RuntimeIdentity { + pub pid: u32, + pub port: u16, +} + +/// The instance this client is bound to, and the binding it was bound under. +/// +/// The generation moves every time the shell binds to a runtime. A request authorised under an +/// earlier binding is not authorised under this one, which is what stops an in-flight management +/// call from landing on a runtime the shell rebound to in between. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RuntimeBinding { + pub identity: RuntimeIdentity, + pub generation: u64, +} #[derive(Clone)] pub struct ProxyClient { client: Client, endpoint: ProxyEndpoint, auth: Auth, + binding: Arc>>, + generations: Arc>, } #[derive(Debug)] @@ -16,6 +81,38 @@ pub enum ProxyError { Unauthorized, Http(StatusCode), Decode(reqwest::Error), + /// The listener answered, but not as the instance this client is bound to — a foreign service + /// on the port, or a different process than the one the shell confirmed. + Foreign, +} + +impl ProxyError { + /// Whether nothing is listening on the endpoint at all. + /// + /// This is the only error that says anything about the process behind the port. A timeout, an + /// unauthorized reply or a body that will not parse all mean the listener answered or might + /// still be there, and a stop that reads any of them as "gone" reports a drain that did not + /// happen. + pub fn is_unreachable(&self) -> bool { + matches!(self, Self::Unreachable) + } +} + +/// Read an identity out of a health body. +/// +/// The marker is required: a 200 from something else on the port is not this proxy. The port is +/// required to be the one addressed, so a body describing a different listener cannot authorise a +/// credential for this one. +pub fn identity_from(body: &Value, addressed_port: u16) -> Option { + if body.get("service").and_then(Value::as_str) != Some("opencodex") { + return None; + } + let pid = u32::try_from(body.get("pid").and_then(Value::as_u64)?).ok()?; + let port = u16::try_from(body.get("port").and_then(Value::as_u64)?).ok()?; + if port != addressed_port { + return None; + } + Some(RuntimeIdentity { pid, port }) } impl ProxyClient { @@ -24,9 +121,23 @@ impl ProxyClient { client: Client::builder() .timeout(Duration::from_secs(4)) .user_agent(Auth::user_agent()) + // The capability attached to these requests is for the loopback endpoint and + // nowhere else. Two defaults would carry it off that endpoint, so both are turned + // off here rather than re-checked anywhere in the request path. + // + // A redirect is the first: the pinned client does not treat this custom credential + // header as sensitive, so it would follow the hop to wherever it pointed. + .redirect(redirect::Policy::none()) + // System proxy resolution is the second: reqwest honours system proxy + // configuration by default, which would route the credential through whatever + // proxy the machine declares and put another process between the shell and its + // own runtime. + .no_proxy() .build()?, endpoint, auth, + binding: Arc::new(Mutex::new(None)), + generations: Arc::new(Mutex::new(0)), }) } @@ -34,10 +145,47 @@ impl ProxyClient { self.endpoint } + fn slot(lock: &Mutex) -> MutexGuard<'_, T> { + lock.lock().unwrap_or_else(PoisonError::into_inner) + } + + /// Bind this client to an instance, and return the binding it is now on. + pub fn bind(&self, identity: RuntimeIdentity) -> RuntimeBinding { + let mut generations = Self::slot(&self.generations); + *generations += 1; + let binding = RuntimeBinding { + identity, + generation: *generations, + }; + *Self::slot(&self.binding) = Some(binding); + binding + } + + pub fn binding(&self) -> Option { + *Self::slot(&self.binding) + } + + /// Ask the endpoint who it is, without sending anything secret. + pub async fn identify(&self) -> Result { + let response = self.send(&Method::GET, "/healthz", None).await?; + let body = decode(response).await?; + identity_from(&body, self.endpoint.port).ok_or(ProxyError::Foreign) + } + pub async fn is_alive(&self) -> Result { self.get("/healthz").await } + /// A health probe that cannot outlive the caller's deadline. + /// + /// The client's own timeout is per request and knows nothing about the budget the caller is + /// working to. A probe started a moment before a deadline would otherwise overrun it by that + /// whole timeout, which is how a stated 30-second startup ceiling quietly becomes 34. + /// `None` means the deadline arrived first. + pub async fn alive_within(&self, deadline: Instant) -> Option> { + timeout_at(deadline, self.is_alive()).await.ok() + } + pub async fn companion_settings(&self) -> Result { self.get("/api/companion/settings").await } @@ -62,42 +210,301 @@ impl ProxyClient { self.get(&format!("/api/usage/timeline?{query}")).await } - pub async fn stop(&self) -> Result { - self.request(Method::POST, "/api/stop").await + pub(crate) async fn get(&self, path: &str) -> Result { + self.request(Method::GET, path).await } - async fn get(&self, path: &str) -> Result { - self.request(Method::GET, path).await + /// Publish only these exact display-state bytes, without exposing the reusable admin token. + pub async fn post_desktop_snapshot(&self, body: &Value) -> Result<(), ProxyError> { + let body = + serde_json::to_vec(body).map_err(|_| ProxyError::Http(StatusCode::BAD_REQUEST))?; + if body.len() > 1024 { + return Err(ProxyError::Http(StatusCode::PAYLOAD_TOO_LARGE)); + } + let recorded = self.authorised_runtime()?; + let headers = + CapabilityHeaders::mint_snapshot(&recorded, &body).ok_or(ProxyError::Unauthorized)?; + // Serialize once: the bytes hashed by mint_snapshot are the bytes reqwest sends. + let request = self + .client + .post(self.endpoint.url(DESKTOP_SNAPSHOT_PATH)) + .header("content-type", "application/json") + .body(body); + let response = headers + .apply(request) + .send() + .await + .map_err(map_request_error)?; + let _ = decode(response).await?; + Ok(()) } async fn request(&self, method: Method, path: &str) -> Result { let response = self.send(&method, path, None).await?; if response.status() == StatusCode::UNAUTHORIZED { - let token = self.auth.token().ok_or(ProxyError::Unauthorized)?; - let response = self.send(&method, path, Some(token)).await?; + let signed = signed_target(&self.endpoint.url(path)).ok_or(ProxyError::Unauthorized)?; + let headers = self.authorised_capability(&method, &signed)?; + let response = self.send(&method, path, Some(headers)).await?; return decode(response).await; } decode(response).await } + #[cfg(target_os = "macos")] + /// Switch a provider's active account through its existing management route, signed by a + /// single-use grant bound to this exact body. The grant cannot authorize any other request. + pub async fn put_account_switch( + &self, + kind: AccountSwitchKind, + body: &Value, + ) -> Result { + let body = + serde_json::to_vec(body).map_err(|_| ProxyError::Http(StatusCode::BAD_REQUEST))?; + if body.len() > ACCOUNT_SWITCH_BODY_LIMIT { + return Err(ProxyError::Http(StatusCode::PAYLOAD_TOO_LARGE)); + } + let recorded = self.authorised_runtime()?; + let headers = CapabilityHeaders::mint_account_switch(&recorded, kind, &body) + .ok_or(ProxyError::Unauthorized)?; + // Serialize once: the bytes hashed by mint_account_switch are the bytes reqwest sends. + let request = self + .client + .put(self.endpoint.url(kind.path())) + .header("content-type", "application/json") + .body(body); + let response = headers + .apply(request) + .send() + .await + .map_err(map_request_error)?; + decode(response).await + } + + /// Mint one read grant, preserving the existing v1 method/path/query contract. + fn authorised_capability( + &self, + method: &Method, + path: &str, + ) -> Result { + let recorded = self.authorised_runtime()?; + CapabilityHeaders::mint(&recorded, method, path).ok_or(ProxyError::Unauthorized) + } + + /// Re-confirm the recorded runtime for both read and snapshot grants. + /// + /// A replacement listener can observe only a short-lived proof, not the secret. The server + /// consumes each proof once; a captured, unused proof is limited to its exact signed request + /// until expiry. Snapshot grants additionally bind the body and cannot authorize other writes. + fn authorised_runtime(&self) -> Result { + let Some(binding) = self.binding() else { + return Err(ProxyError::Unauthorized); + }; + let recorded = self + .auth + .runtime_identity() + .ok_or(ProxyError::Unauthorized)?; + if recorded.port != self.endpoint.port { + return Err(ProxyError::Unauthorized); + } + if recorded.pid != binding.identity.pid || recorded.port != binding.identity.port { + return Err(ProxyError::Foreign); + } + if self.binding() != Some(binding) { + return Err(ProxyError::Foreign); + } + Ok(recorded) + } + async fn send( &self, method: &Method, path: &str, - token: Option, + capability: Option, ) -> Result { let mut request = self.client.request(method.clone(), self.endpoint.url(path)); - if let Some(value) = token { - request = request.header("X-OpenCodex-API-Key", value); + if let Some(headers) = capability { + request = headers.apply(request); + } + request.send().await.map_err(map_request_error) + } +} + +/// A single-use read or body-bound snapshot grant, never a reusable management credential. +struct CapabilityHeaders { + expected_pid: String, + nonce: String, + expires_at: String, + capability: String, + body_digest: Option<(&'static str, String)>, +} + +impl CapabilityHeaders { + /// Mint a GET grant using the unchanged local-management-read-v1 wire format. + fn mint(recorded: &RecordedRuntime, method: &Method, path: &str) -> Option { + if method != Method::GET { + return None; } - request.send().await.map_err(|error| { - if error.is_connect() { - ProxyError::Unreachable - } else { - ProxyError::Decode(error) - } + let (nonce, expires_at) = fresh_capability_fields()?; + Some(Self { + expected_pid: recorded.pid.to_string(), + capability: capability_mac(recorded, path, &nonce, expires_at)?, + nonce, + expires_at: expires_at.to_string(), + body_digest: None, }) } + + /// Mint only the bounded snapshot POST; its domain is distinct from every read grant. + fn mint_snapshot(recorded: &RecordedRuntime, body: &[u8]) -> Option { + if body.len() > 1024 { + return None; + } + let (nonce, expires_at) = fresh_capability_fields()?; + let body_digest = URL_SAFE_NO_PAD.encode(Sha256::digest(body)); + Some(Self { + expected_pid: recorded.pid.to_string(), + capability: snapshot_capability_mac(recorded, &nonce, expires_at, &body_digest)?, + nonce, + expires_at: expires_at.to_string(), + body_digest: Some((SNAPSHOT_DIGEST_HEADER, body_digest)), + }) + } + + #[cfg(target_os = "macos")] + /// Mint only one account-switch PUT; its domain is distinct from read and snapshot grants. + fn mint_account_switch( + recorded: &RecordedRuntime, + kind: AccountSwitchKind, + body: &[u8], + ) -> Option { + if body.len() > ACCOUNT_SWITCH_BODY_LIMIT { + return None; + } + let (nonce, expires_at) = fresh_capability_fields()?; + let body_digest = URL_SAFE_NO_PAD.encode(Sha256::digest(body)); + Some(Self { + expected_pid: recorded.pid.to_string(), + capability: account_switch_capability_mac( + recorded, + kind, + &nonce, + expires_at, + &body_digest, + )?, + nonce, + expires_at: expires_at.to_string(), + body_digest: Some((ACCOUNT_SWITCH_DIGEST_HEADER, body_digest)), + }) + } + + /// Attach only scoped proof headers, shared by both transport paths. + fn apply(self, request: RequestBuilder) -> RequestBuilder { + let mut request = request + .header("x-opencodex-local-expected-pid", self.expected_pid) + .header("x-opencodex-local-nonce", self.nonce) + .header("x-opencodex-local-expires-at", self.expires_at) + .header("x-opencodex-local-capability", self.capability); + if let Some((header, digest)) = self.body_digest { + request = request.header(header, digest); + } + request + } +} + +/// Fresh randomness and a ten-second expiry for either scoped capability. +fn fresh_capability_fields() -> Option<(String, u64)> { + let mut nonce_bytes = [0_u8; 32]; + nonce_bytes[..16].copy_from_slice(uuid::Uuid::new_v4().as_bytes()); + nonce_bytes[16..].copy_from_slice(uuid::Uuid::new_v4().as_bytes()); + let expires_at = u64::try_from( + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .ok()? + .as_millis(), + ) + .ok()? + .checked_add(10_000)?; + Some((URL_SAFE_NO_PAD.encode(nonce_bytes), expires_at)) +} + +/// The signed half of a capability, split out so the wire format can be tested against a fixed +/// vector from the TypeScript implementation. `None` means the inputs cannot form a valid grant. +fn capability_mac( + recorded: &RecordedRuntime, + path: &str, + nonce: &str, + expires_at: u64, +) -> Option { + // The server keys the MAC with the Base64URL text's UTF-8 bytes, not the decoded secret. + let mut mac = Hmac::::new_from_slice(recorded.attestation_secret.as_bytes()).ok()?; + mac.update( + format!( + "opencodex-local-management-read-v1\n{nonce}\nGET\n{path}\n{}\n{}\n{expires_at}", + recorded.pid, recorded.port + ) + .as_bytes(), + ); + Some(URL_SAFE_NO_PAD.encode(mac.finalize().into_bytes())) +} + +/// The snapshot wire contract includes the exact body's SHA-256 digest. +fn snapshot_capability_mac( + recorded: &RecordedRuntime, + nonce: &str, + expires_at: u64, + body_digest: &str, +) -> Option { + let mut mac = Hmac::::new_from_slice(recorded.attestation_secret.as_bytes()).ok()?; + mac.update( + format!( + "opencodex-local-desktop-snapshot-v1\n{nonce}\nPOST\n{DESKTOP_SNAPSHOT_PATH}\n{}\n{}\n{expires_at}\n{body_digest}", + recorded.pid, recorded.port + ) + .as_bytes(), + ); + Some(URL_SAFE_NO_PAD.encode(mac.finalize().into_bytes())) +} + +#[cfg(target_os = "macos")] +/// The account-switch wire contract binds the method, one of three constant paths, and the exact +/// body's SHA-256 digest (`src/lib/local-account-switch-capability.ts`). +fn account_switch_capability_mac( + recorded: &RecordedRuntime, + kind: AccountSwitchKind, + nonce: &str, + expires_at: u64, + body_digest: &str, +) -> Option { + let mut mac = Hmac::::new_from_slice(recorded.attestation_secret.as_bytes()).ok()?; + mac.update( + format!( + "opencodex-local-account-switch-v1\n{nonce}\nPUT\n{}\n{}\n{}\n{expires_at}\n{body_digest}", + kind.path(), + recorded.pid, + recorded.port + ) + .as_bytes(), + ); + Some(URL_SAFE_NO_PAD.encode(mac.finalize().into_bytes())) +} + +fn map_request_error(error: reqwest::Error) -> ProxyError { + if error.is_connect() { + ProxyError::Unreachable + } else { + ProxyError::Decode(error) + } +} + +/// The request target the capability signs, derived from the parsed URL rather than the raw path +/// string. The server verifies `pathname + url.search`, which drops a bare `?` and keeps the +/// percent-encoding reqwest applies on send; signing the raw path would mismatch on both. +fn signed_target(url: &str) -> Option { + let url = reqwest::Url::parse(url).ok()?; + match url.query().filter(|query| !query.is_empty()) { + Some(query) => Some(format!("{}?{query}", url.path())), + None => Some(url.path().to_owned()), + } } async fn decode(response: reqwest::Response) -> Result { @@ -109,3 +516,240 @@ async fn decode(response: reqwest::Response) -> Result { } response.json().await.map_err(ProxyError::Decode) } + +#[cfg(test)] +mod tests { + #[cfg(target_os = "macos")] + use super::{account_switch_capability_mac, AccountSwitchKind}; + use super::{ + capability_mac, identity_from, signed_target, snapshot_capability_mac, CapabilityHeaders, + RuntimeIdentity, + }; + use crate::auth::RecordedRuntime; + use reqwest::Method; + use serde_json::json; + + fn recorded_runtime() -> RecordedRuntime { + RecordedRuntime { + pid: 4242, + port: 10100, + attestation_secret: "BwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwc".into(), + } + } + + #[test] + fn a_capability_matches_the_server_contract() { + // The fixed nonce and expiry make the signature reproducible against the TypeScript + // implementation: this vector is createLocalManagementReadCapability over the same + // inputs, so a drift on either side fails here before it fails on the wire. + let headers = + CapabilityHeaders::mint(&recorded_runtime(), &Method::GET, "/api/usage?range=7d") + .expect("a mintable grant"); + assert_eq!(headers.expected_pid, "4242"); + assert_eq!(headers.nonce.len(), 43); + assert_eq!(headers.capability.len(), 43); + assert!(headers.expires_at.parse::().unwrap() > 0); + // A write method cannot mint a read grant. The literal Method::POST is avoided because an + // exit-ownership source assertion forbids it in this file. + let write = Method::from_bytes(b"POST").expect("a write method"); + assert!(CapabilityHeaders::mint(&recorded_runtime(), &write, "/api/usage").is_none()); + + // The fixed nonce and expiry pin the exact wire signature to the TypeScript vector. + let capability = capability_mac( + &recorded_runtime(), + "/api/usage?range=7d", + "AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA", + 1_700_000_010_000, + ) + .expect("a signable grant"); + assert_eq!(capability, "oGyWOCGZsICYctxQv-mPK0gCiDocvOVHQG5plyjYCUg"); + assert_eq!( + capability_mac( + &recorded_runtime(), + "/api/system/memory", + "AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA", + 1_700_000_010_000, + ) + .as_deref(), + Some("_a3HS292KKaMcXsDx0owmWr3zRFTYjH6vhUdpunRW28") + ); + } + + #[test] + fn a_snapshot_grant_binds_the_body_and_matches_the_server_contract() { + let body = br#"{"sessionId":"test"}"#; + let headers = CapabilityHeaders::mint_snapshot(&recorded_runtime(), body).unwrap(); + let digest = "5pREWDDMbj42QHj3DvVNrC54yVF7Vpd8cNj5c-z3rQ4"; + assert_eq!( + headers + .body_digest + .as_ref() + .map(|(_, digest)| digest.as_str()), + Some(digest) + ); + assert_eq!(headers.expected_pid, "4242"); + assert_eq!( + snapshot_capability_mac( + &recorded_runtime(), + "AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA", + 1_700_000_010_000, + digest, + ) + .as_deref(), + Some("yEkTQtyXzQsi_kJGmzmUO1wyJIjc_G4-iiNNoyB4Zcw") + ); + assert!(CapabilityHeaders::mint_snapshot(&recorded_runtime(), &[0; 1025]).is_none()); + let request = headers + .apply( + reqwest::Client::new().post("http://127.0.0.1:10100/api/update/desktop-snapshot"), + ) + .body(body.to_vec()) + .build() + .unwrap(); + assert!(!request.headers().contains_key("x-opencodex-api-key")); + assert!(!request.headers().contains_key("authorization")); + assert_eq!( + request.headers()["x-opencodex-desktop-snapshot-sha256"], + digest + ); + assert_eq!(request.body().unwrap().as_bytes(), Some(body.as_slice())); + } + + #[cfg(target_os = "macos")] + #[test] + fn an_account_switch_grant_binds_route_and_body_and_matches_the_server_contract() { + // Same fixed inputs as the TypeScript vectors in + // tests/server/local-account-switch-capability.test.ts. + let nonce = "AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA"; + let vectors = [ + ( + AccountSwitchKind::Codex, + r#"{"accountId":"acct-1"}"#, + "TAvg-9rOZVklyLUY6FQ1afEc-hJ80OLD8qSKdKSrjh4", + "oaJMrRckk98viNXA3nUO6y7nHDkehgcPIGjbEhgDX4M", + ), + ( + AccountSwitchKind::OAuth, + r#"{"provider":"anthropic","accountId":"acct-1"}"#, + "AkfVPiY6XR2cClHqqApykOKZcdiTqwpkeVuBQDg3u5A", + "5_u4nzIUGl32US-V7eftK1r4iV90gS4WSdtEgimYtFQ", + ), + ( + AccountSwitchKind::ApiKey, + r#"{"name":"xai","id":"key-1"}"#, + "TwpvGp8TFleVkY5xRWSl_UMujQUbQMsk-P2oFvMvjjg", + "kdKDBHHEYD9E56oKxS2NHSU5DIuhdLEHv5XLDh9KM9Q", + ), + ]; + for (kind, body, digest, mac) in vectors { + let headers = + CapabilityHeaders::mint_account_switch(&recorded_runtime(), kind, body.as_bytes()) + .unwrap(); + assert_eq!( + headers.body_digest, + Some(("x-opencodex-account-switch-sha256", digest.to_owned())) + ); + assert_eq!( + account_switch_capability_mac( + &recorded_runtime(), + kind, + nonce, + 1_700_000_010_000, + digest + ) + .as_deref(), + Some(mac) + ); + } + // The switch domain never reproduces a snapshot proof over the same inputs. + assert_ne!( + account_switch_capability_mac( + &recorded_runtime(), + AccountSwitchKind::Codex, + nonce, + 1_700_000_010_000, + "5pREWDDMbj42QHj3DvVNrC54yVF7Vpd8cNj5c-z3rQ4" + ), + snapshot_capability_mac( + &recorded_runtime(), + nonce, + 1_700_000_010_000, + "5pREWDDMbj42QHj3DvVNrC54yVF7Vpd8cNj5c-z3rQ4" + ) + ); + assert!(CapabilityHeaders::mint_account_switch( + &recorded_runtime(), + AccountSwitchKind::Codex, + &[b' '; 1025] + ) + .is_none()); + let body = br#"{"accountId":"acct-1"}"#; + let request = CapabilityHeaders::mint_account_switch( + &recorded_runtime(), + AccountSwitchKind::Codex, + body, + ) + .unwrap() + .apply(reqwest::Client::new().put("http://127.0.0.1:10100/api/codex-auth/active")) + .body(body.to_vec()) + .build() + .unwrap(); + assert!(!request.headers().contains_key("x-opencodex-api-key")); + assert!(!request.headers().contains_key("authorization")); + assert!(!request + .headers() + .contains_key("x-opencodex-desktop-snapshot-sha256")); + assert_eq!(request.body().unwrap().as_bytes(), Some(body.as_slice())); + } + + #[test] + fn a_health_body_without_the_marker_is_not_this_proxy() { + let body = json!({ "status": "ok", "pid": 42, "port": 10100 }); + assert!(identity_from(&body, 10100).is_none()); + let foreign = json!({ "service": "something-else", "pid": 42, "port": 10100 }); + assert!(identity_from(&foreign, 10100).is_none()); + } + + #[test] + fn the_body_has_to_describe_the_listener_that_was_addressed() { + let body = json!({ "service": "opencodex", "pid": 42, "port": 10101 }); + assert!(identity_from(&body, 10100).is_none()); + } + + #[test] + fn a_complete_body_identifies_the_instance() { + let body = json!({ "service": "opencodex", "version": "2.61.0", "pid": 42, "port": 10100 }); + assert_eq!( + identity_from(&body, 10100), + Some(RuntimeIdentity { + pid: 42, + port: 10100 + }) + ); + } + + #[test] + fn a_body_missing_the_instance_facts_identifies_nothing() { + assert!(identity_from(&json!({ "service": "opencodex", "port": 10100 }), 10100).is_none()); + assert!(identity_from(&json!({ "service": "opencodex", "pid": 42 }), 10100).is_none()); + } + + #[test] + fn the_signed_target_matches_what_the_server_reconstructs() { + // A bare `?` has an empty search on the server, so it must not be signed. + assert_eq!( + signed_target("http://127.0.0.1:10100/api/usage/timeline?").as_deref(), + Some("/api/usage/timeline") + ); + // A populated query is signed verbatim, including percent-encoding reqwest applies. + assert_eq!( + signed_target("http://127.0.0.1:10100/api/usage/timeline?range=7d").as_deref(), + Some("/api/usage/timeline?range=7d") + ); + assert_eq!( + signed_target("http://127.0.0.1:10100/api/usage/timeline?model=a b").as_deref(), + Some("/api/usage/timeline?model=a%20b") + ); + assert_eq!(signed_target("not a url"), None); + } +} diff --git a/desktop/src-tauri/src/resolve.rs b/desktop/src-tauri/src/resolve.rs new file mode 100644 index 0000000000..1f9ddcfdda --- /dev/null +++ b/desktop/src-tauri/src/resolve.rs @@ -0,0 +1,532 @@ +//! What the bundled CLI says about this machine's runtime. +//! +//! D5: the shell stops resolving the configuration home, the port and liveness itself. It used to, +//! in a file called `discovery.rs` that read `runtime-port.json`, fell back to 10100 and started on +//! that port — so a user with a configured `config.port` was started somewhere else. The tuned probe +//! budgets it should have been using exist because a shell-side reimplementation answered "nobody is +//! listening" twice and started duplicate proxies. This asks instead. +//! +//! Liveness has three answers and the third one is the point. `live` means attach. `absent-proven` +//! means every recorded and configured endpoint was definitively dead, and only that authorises +//! starting a runtime. Anything else is unknown, and the CLI exits 1 rather than putting absence on +//! the wire. Everything that can go wrong on this side — a missing binary, a timeout, output that +//! will not parse, a schema this shell does not know — folds into the same unknown, because the one +//! reading that must never happen is "the resolve failed, so nobody must be listening". + +use crate::endpoint::ProxyEndpoint; +use crate::ownership::Recorded; +use serde::Deserialize; +use std::path::PathBuf; +use tauri::AppHandle; +use tauri_plugin_shell::ShellExt; +use tokio::time::{timeout_at, Instant}; + +/// The wire version this shell understands. A document announcing anything else is unknown. +pub const SCHEMA: &str = "ocx-resolve/1"; + +/// The CLI's liveness verdict. Only two reach the wire; the third exits 1. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum Status { + Live, + AbsentProven, +} + +#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct Liveness { + pub status: Status, + pub pid: Option, + pub port: Option, + /// The bind address that answered. Absent on a proven absence, because nothing answered. + pub hostname: Option, + pub version: Option, + pub role: Option, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct Port { + /// The port a client should use: the live listener's, or the configured one. + pub effective: u16, + /// What a start would prefer. + pub configured: u16, +} + +/// Whether the CLI says a desktop takeover can be offered. +/// +/// The token is the binding a later `ocx service claim` repeats back: it covers the exact +/// subject and managing-CLI observations the consent was approved against, so a claim made +/// after either moved is refused rather than recorded. +#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[serde(tag = "kind", rename_all = "camelCase")] +pub enum Takeover { + #[serde(rename_all = "camelCase")] + Supported { + protocol_version: u64, + minimum_cli_version: String, + token: String, + }, + Blocked { + reason: String, + detail: String, + }, +} + +impl Default for Takeover { + /// An older bundled CLI carries no takeover answer at all; silence is not approval. + fn default() -> Self { + Self::Blocked { + reason: "unreported".to_owned(), + detail: "the bundled CLI did not report takeover compatibility".to_owned(), + } + } +} + +/// Which side of the CLI-versus-runtime comparison runs newer. +/// +/// The strings are the wire values the CLI emits; `Unknown` also stands in for an +/// absent `versionSkew` document or a future relation string. Unknown display metadata +/// must not discard an otherwise valid live-runtime answer. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum VersionRelation { + Match, + CliNewer, + ProxyNewer, + Incomparable, + #[serde(other)] + Unknown, +} + +/// The bundled CLI's version laid next to the live runtime's, as the CLI computed it. +/// The warning is the operator-facing sentence `ocx status` already prints; this shell +/// repeats it verbatim so two surfaces never describe the same skew differently. +#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct VersionSkew { + pub cli_version: String, + pub proxy_version: Option, + pub skewed: bool, + pub relation: VersionRelation, + pub warning: Option, +} + +#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct Resolved { + pub schema: String, + pub cli_version: String, + pub config_home: String, + pub port: Port, + pub liveness: Liveness, + /// The recorded runtime owner, already in the CLI's three answers. Absent on older + /// documents, which read as unknown rather than as nobody owning the runtime. + #[serde(default)] + pub ownership: Recorded, + #[serde(default)] + pub takeover: Takeover, + /// The version comparison, absent on every CLI older than this field and on a + /// proven absence. Absent reads as unknown, never as a match. + #[serde(default)] + pub version_skew: Option, +} + +impl Resolved { + pub fn endpoint(&self) -> ProxyEndpoint { + ProxyEndpoint { + host: "127.0.0.1", + port: self.port.effective, + } + } + + pub fn home(&self) -> PathBuf { + PathBuf::from(&self.config_home) + } + + /// The runtime's version relative to the bundled CLI's, or Unknown when the + /// document does not say — including every CLI that predates the field. + pub fn runtime_relation(&self) -> VersionRelation { + self.version_skew + .as_ref() + .map_or(VersionRelation::Unknown, |skew| skew.relation) + } + + /// The skew warning when there is a confirmed difference worth surfacing. + pub fn skew_warning(&self) -> Option<&str> { + self.version_skew + .as_ref() + .and_then(|skew| skew.warning.as_deref()) + } +} + +/// What the shell got back. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Resolution { + /// The CLI produced a verdict this shell trusts. + Answered(Box), + /// It did not, for whatever reason. Never read as absence. + Unknown(String), +} + +impl Resolution { + pub fn resolved(&self) -> Option<&Resolved> { + match self { + Self::Answered(resolved) => Some(resolved.as_ref()), + Self::Unknown(_) => None, + } + } + + pub fn reason(&self) -> Option<&str> { + match self { + Self::Unknown(reason) => Some(reason), + Self::Answered(_) => None, + } + } +} + +/// Whether the shell may start a runtime of its own. +/// +/// Proven absence and nothing else. `live` means attach to what is there, and unknown means refuse: +/// a resolution that could not be trusted must never read as "nobody is listening", which is the +/// reading that puts a second proxy next to the one already running. +pub fn may_start(resolution: &Resolution) -> bool { + matches!( + resolution + .resolved() + .map(|resolved| resolved.liveness.status), + Some(Status::AbsentProven) + ) +} + +/// What the shell may do with a listener the CLI found alive. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum LiveVerdict { + /// Nothing is listening; this verdict does not apply. + NotLive, + /// It is a proxy, on an address this shell can reach. Attach as a guest. + Attach, + /// A Child's client runtime, on loopback. It serves Codex through its Home and the Child's own + /// dashboard, not the management plane, and it is never taken over: the shell attaches to it as + /// a guest and asks nothing. When it is the child this app started, that attach is ownership. + Client, + /// Something is listening and this shell cannot use it. Never a reason to start a second one. + Unusable(String), +} + +/// Whether the address the CLI reported is one this shell can reach on loopback. +/// +/// The shell speaks to loopback and nothing else — that is what makes sending the management token +/// to it safe. A proxy bound to either loopback spelling, or to every interface, is reachable at +/// 127.0.0.1. One bound to the IPv6 loopback or to a specific external address is not, and +/// addressing 127.0.0.1 anyway would turn a running proxy into a health wait that times out. +pub fn loopback_reachable(hostname: Option<&str>) -> bool { + matches!( + hostname, + None | Some("127.0.0.1") | Some("localhost") | Some("0.0.0.0") + ) +} + +/// Read a live verdict. +/// +/// Liveness answers "is something there", and core's predicate accepts a connected client's +/// listener on purpose so duplicate-start avoidance can see it. The shell discriminates on the role +/// the CLI carried: a client listener serves machine routes and the Child's dashboard, not `/api/*`, +/// and the takeover a proxy can be offered does not apply to it. It is also what a Child runs, +/// including this app's own sidecar after Connect as Child, so it is attached to +/// ([`LiveVerdict::Client`]) rather than refused. Refusing it failed every recovery on a Child whose +/// runtime restarted outside the app, and each failure scheduled the next. +pub fn live_verdict(resolution: &Resolution) -> LiveVerdict { + let Some(resolved) = resolution.resolved() else { + return LiveVerdict::NotLive; + }; + if resolved.liveness.status != Status::Live { + return LiveVerdict::NotLive; + } + let client = resolved.liveness.role.as_deref() == Some("client"); + if !loopback_reachable(resolved.liveness.hostname.as_deref()) { + return LiveVerdict::Unusable(format!( + "the runtime is bound to {} and this app only speaks to loopback", + resolved + .liveness + .hostname + .as_deref() + .unwrap_or("an unknown address") + )); + } + if client { + return LiveVerdict::Client; + } + LiveVerdict::Attach +} + +/// Read one resolve document, refusing anything that is not exactly one. +/// +/// The CLI puts the document on stdout and its human output on stderr, so stdout is parsed whole. +/// A non-zero exit is the CLI's own refusal — including the exit 1 it uses for unknown liveness and +/// for a config it will not guess at — and is carried through rather than reinterpreted here. +pub fn read(exit_code: Option, stdout: &[u8], stderr: &[u8]) -> Resolution { + if exit_code != Some(0) { + let detail = String::from_utf8_lossy(stderr); + let detail = detail.trim(); + let code = exit_code + .map(|code| code.to_string()) + .unwrap_or_else(|| "no exit code".to_owned()); + return Resolution::Unknown(if detail.is_empty() { + format!("the bundled CLI could not resolve the runtime (exit {code})") + } else { + format!("the bundled CLI could not resolve the runtime (exit {code}): {detail}") + }); + } + let text = String::from_utf8_lossy(stdout); + let resolved: Resolved = match serde_json::from_str(text.trim()) { + Ok(resolved) => resolved, + Err(error) => { + return Resolution::Unknown(format!( + "the bundled CLI's resolve output could not be read ({error})" + )) + } + }; + if resolved.schema != SCHEMA { + return Resolution::Unknown(format!( + "the bundled CLI answered with schema {} and this app understands {SCHEMA}", + resolved.schema + )); + } + Resolution::Answered(Box::new(resolved)) +} + +/// Ask the bundled CLI, under the caller's deadline. +pub async fn run(app: &AppHandle, deadline: Instant) -> Resolution { + let command = match app.shell().sidecar("ocx") { + Ok(command) => command.args(["resolve", "--json"]), + Err(error) => { + return Resolution::Unknown(format!("the bundled CLI could not be started ({error})")) + } + }; + match timeout_at(deadline, command.output()).await { + Ok(Ok(output)) => read(output.status.code(), &output.stdout, &output.stderr), + Ok(Err(error)) => { + Resolution::Unknown(format!("the bundled CLI could not be run ({error})")) + } + Err(_) => Resolution::Unknown( + "the bundled CLI did not answer before the startup deadline".to_owned(), + ), + } +} + +#[cfg(test)] +mod tests { + use super::{ + live_verdict, loopback_reachable, may_start, read, LiveVerdict, Resolution, Status, + Takeover, VersionRelation, SCHEMA, + }; + use crate::ownership::{Owner, Recorded}; + + const LIVE: &str = r#"{"schema":"ocx-resolve/1","cliVersion":"2.61.0","configHome":"/h", + "port":{"effective":10100,"configured":10100,"source":"runtime-record"}, + "liveness":{"status":"live","pid":42,"port":10100,"source":"runtime-record","version":"2.61.0"}}"#; + const ABSENT: &str = r#"{"schema":"ocx-resolve/1","cliVersion":"2.61.0","configHome":"/h", + "port":{"effective":10100,"configured":10100,"source":"config"}, + "liveness":{"status":"absent-proven","pid":null,"port":null,"source":null}}"#; + + #[test] + fn a_live_verdict_is_read_whole() { + let resolution = read(Some(0), LIVE.as_bytes(), b""); + let resolved = resolution.resolved().expect("a document"); + assert_eq!(resolved.schema, SCHEMA); + assert_eq!(resolved.liveness.status, Status::Live); + assert_eq!(resolved.liveness.pid, Some(42)); + assert_eq!(resolved.endpoint().port, 10100); + assert_eq!(resolved.home().display().to_string(), "/h"); + assert_eq!(live_verdict(&resolution), LiveVerdict::Attach); + assert!(!may_start(&resolution)); + } + + #[test] + fn only_a_proven_absence_authorises_a_start() { + let resolution = read(Some(0), ABSENT.as_bytes(), b""); + assert_eq!( + resolution.resolved().map(|r| r.liveness.status), + Some(Status::AbsentProven) + ); + assert!(may_start(&resolution)); + assert_eq!(live_verdict(&resolution), LiveVerdict::NotLive); + } + + #[test] + fn a_connected_client_is_attached_to_and_never_started_beside() { + let client = LIVE.replace( + r#""version":"2.61.0""#, + r#""version":"2.61.0","role":"client""#, + ); + let resolution = read(Some(0), client.as_bytes(), b""); + // A Child's runtime: attached to as a guest, never taken over and never refused. + assert_eq!(live_verdict(&resolution), LiveVerdict::Client); + // Live is still live: it is never a reason to start a second one. + assert!(!may_start(&resolution)); + // Off loopback it is as unusable as any other listener there. + let elsewhere = client.replace(r#""pid":42"#, r#""pid":42,"hostname":"::1""#); + let resolution = read(Some(0), elsewhere.as_bytes(), b""); + assert!(matches!( + live_verdict(&resolution), + LiveVerdict::Unusable(_) + )); + assert!(!may_start(&resolution)); + } + + #[test] + fn only_a_loopback_bind_is_addressed_as_loopback() { + for reachable in [None, Some("127.0.0.1"), Some("localhost"), Some("0.0.0.0")] { + assert!(loopback_reachable(reachable), "{reachable:?}"); + } + for elsewhere in [Some("::1"), Some("192.168.1.10"), Some("example.internal")] { + assert!(!loopback_reachable(elsewhere), "{elsewhere:?}"); + } + let bound = LIVE.replace(r#""pid":42"#, r#""pid":42,"hostname":"::1""#); + let resolution = read(Some(0), bound.as_bytes(), b""); + assert!(matches!( + live_verdict(&resolution), + LiveVerdict::Unusable(_) + )); + assert!(!may_start(&resolution)); + } + + #[test] + fn the_clis_own_refusal_is_unknown_and_never_authorises_a_start() { + // Exit 1 is what the CLI uses for unknown liveness and for a config it will not guess at. + let resolution = read(Some(1), b"", b"resolve: liveness is unknown"); + assert!(matches!(resolution, Resolution::Unknown(_))); + assert!(resolution.reason().unwrap().contains("liveness is unknown")); + assert!(!may_start(&resolution)); + assert_eq!(live_verdict(&resolution), LiveVerdict::NotLive); + } + + #[test] + fn everything_that_can_go_wrong_here_folds_into_unknown() { + for (code, out) in [ + (Some(64), &b""[..]), + (None, &b""[..]), + (Some(0), &b"not json"[..]), + (Some(0), &b"{}"[..]), + ] { + let resolution = read(code, out, b""); + assert!(matches!(resolution, Resolution::Unknown(_)), "{code:?}"); + assert!(!may_start(&resolution)); + } + } + + #[test] + fn ownership_and_takeover_answers_are_read_whole() { + let document = format!( + "{}{}}}", + LIVE.strip_suffix('}').unwrap(), + r#","ownership":{"kind":"owned","ownership":{"owner":"cli","installId":"npm-1","consentGeneration":2},"revision":9},"takeover":{"kind":"supported","protocolVersion":1,"minimumCliVersion":"2.61.0","token":"abc"}"# + ); + let resolution = read(Some(0), document.as_bytes(), b""); + let resolved = match resolution.resolved() { + Some(resolved) => resolved.clone(), + None => panic!("{}", resolution.reason().unwrap()), + }; + assert_eq!( + resolved.ownership, + Recorded::Owned { + ownership: crate::ownership::Claim { + owner: Owner::Cli, + install_id: "npm-1".to_owned(), + consent_generation: 2, + }, + revision: 9, + } + ); + assert!(matches!( + resolved.takeover, + Takeover::Supported { ref token, .. } if token == "abc" + )); + } + + #[test] + fn a_missing_ownership_or_takeover_answer_is_not_consent() { + // Older bundled CLIs carry neither field; silence must read unknown/blocked, never + // "nobody owns it" or "takeover supported". + let resolved = read(Some(0), LIVE.as_bytes(), b"") + .resolved() + .expect("a document") + .clone(); + assert!(matches!(resolved.ownership, Recorded::Unknown { .. })); + assert!(matches!(resolved.takeover, Takeover::Blocked { .. })); + assert_eq!(resolved.takeover, Takeover::default()); + } + + #[test] + fn a_blocked_takeover_carries_its_reason() { + let document = format!( + "{}{}}}", + LIVE.strip_suffix('}').unwrap(), + r#","ownership":{"kind":"none","revision":0},"takeover":{"kind":"blocked","reason":"managing-cli-unsupported","detail":"path uses 2.59.0","minimumCliVersion":"2.61.0"}"# + ); + let resolved = read(Some(0), document.as_bytes(), b"") + .resolved() + .expect("a document") + .clone(); + assert_eq!(resolved.ownership, Recorded::None { revision: 0 }); + assert_eq!( + resolved.takeover, + Takeover::Blocked { + reason: "managing-cli-unsupported".to_owned(), + detail: "path uses 2.59.0".to_owned(), + } + ); + } + + #[test] + fn the_version_skew_is_read_whole_and_defaults_to_unknown() { + // A document carrying the comparison hands the shell both the direction and the + // operator-facing warning verbatim. + let document = format!( + "{}{}}}", + LIVE.strip_suffix('}').unwrap(), + r#","versionSkew":{"cliVersion":"2.61.0","proxyVersion":"2.62.0","skewed":true,"relation":"proxy-newer","warning":"CLI 2.61.0 does not match the running proxy 2.62.0"}"# + ); + let resolved = read(Some(0), document.as_bytes(), b"") + .resolved() + .expect("a document") + .clone(); + assert_eq!(resolved.runtime_relation(), VersionRelation::ProxyNewer); + assert_eq!( + resolved.skew_warning(), + Some("CLI 2.61.0 does not match the running proxy 2.62.0") + ); + // An older CLI sends nothing; absent must read unknown, not a match. + let resolved = read(Some(0), LIVE.as_bytes(), b"") + .resolved() + .expect("a document") + .clone(); + assert_eq!(resolved.runtime_relation(), VersionRelation::Unknown); + assert_eq!(resolved.skew_warning(), None); + } + + #[test] + fn a_future_version_relation_keeps_the_live_answer() { + let document = format!( + "{}{}}}", + LIVE.strip_suffix('}').unwrap(), + r#","versionSkew":{"cliVersion":"2.61.0","proxyVersion":"2.62.0","skewed":true,"relation":"future-comparison","warning":"upgrade the CLI"}"# + ); + let resolution = read(Some(0), document.as_bytes(), b""); + let resolved = resolution.resolved().expect("a live answer"); + assert_eq!(resolved.runtime_relation(), VersionRelation::Unknown); + assert_eq!(resolved.skew_warning(), Some("upgrade the CLI")); + assert!(matches!(live_verdict(&resolution), LiveVerdict::Attach)); + assert!(!may_start(&resolution)); + } + + #[test] + fn a_schema_this_app_does_not_know_is_unknown() { + let future = LIVE.replace("ocx-resolve/1", "ocx-resolve/2"); + let resolution = read(Some(0), future.as_bytes(), b""); + assert!(matches!(resolution, Resolution::Unknown(_))); + assert!(resolution.reason().unwrap().contains("ocx-resolve/2")); + assert!(!may_start(&resolution)); + } +} diff --git a/desktop/src-tauri/src/runtime_stop.rs b/desktop/src-tauri/src/runtime_stop.rs new file mode 100644 index 0000000000..c92528a5f4 --- /dev/null +++ b/desktop/src-tauri/src/runtime_stop.rs @@ -0,0 +1,428 @@ +//! Stopping a runtime through the bundled CLI. +//! +//! D4: the shell drives the real `ocx stop` as a child process, so the receipt-backed teardown, the +//! drain, the Windows respawn verification and the client-configuration restore all run exactly as +//! they do from a terminal. An in-process management call cannot own that teardown — launchd and +//! systemd can terminate the request handler during self-unload, and the Windows respawn window can +//! only be verified after the process exits — so the shell reads the run's result instead of +//! performing it. +//! +//! The result is a document, not a guess. `ocx stop --json` puts one summary on stdout and its +//! human output on stderr, and this consumes the outcome and the exit code rather than inferring +//! either. Only an exact exit-0 stop or validated history-only completion may lead to takeover; +//! approval and manager refusals stay terminal even when the endpoint becomes quiet. + +use serde::Deserialize; +use std::time::Duration; +use tauri::AppHandle; +use tauri_plugin_shell::ShellExt; +use tokio::time::{timeout_at, Instant}; + +/// The wire version this shell understands. +pub const SCHEMA: &str = "ocx-stop/1"; +const HISTORY_INCOMPLETE_EXIT_CODE: i32 = 79; + +/// How long the stop may take. +/// +/// The CLI's stop drains in-flight requests, restores client configuration and verifies the Windows +/// respawn window, so this is generous on purpose: it bounds a hang, it does not pace a healthy +/// stop. Overrunning it is a failure, not a stop, because the caller's next step is to end the app +/// or replace the files the runtime is serving out of. +pub const DEADLINE: Duration = Duration::from_secs(30); + +/// The outcomes the CLI can report. An outcome this shell does not know fails to parse, which is +/// the same answer as a stop that did not happen. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum Outcome { + Stopped, + NotRunning, + HistoryIncomplete, + HistoryDeferred, + Failed, + ApprovalChanged, + ManagerStillActive, +} + +impl Outcome { + pub fn as_str(self) -> &'static str { + match self { + Self::Stopped => "stopped", + Self::NotRunning => "not-running", + Self::HistoryIncomplete => "history-incomplete", + Self::HistoryDeferred => "history-deferred", + Self::Failed => "failed", + Self::ApprovalChanged => "approval-changed", + Self::ManagerStillActive => "manager-still-active", + } + } +} + +/// How the proxy half of the stop ended. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum Proxy { + Stopped, + StoppedOrphan, + NotRunning, + StopFailed, + OwnershipRefused, + UnresolvablePid, + Respawned, + Unknown, +} + +impl Proxy { + pub fn as_str(self) -> &'static str { + match self { + Self::Stopped => "stopped", + Self::StoppedOrphan => "stopped-orphan", + Self::NotRunning => "not-running", + Self::StopFailed => "stop-failed", + Self::OwnershipRefused => "ownership-refused", + Self::UnresolvablePid => "unresolvable-pid", + Self::Respawned => "respawned", + Self::Unknown => "unknown", + } + } +} + +#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct StopSummary { + pub schema: String, + /// Strict exit-code view: true only for exit 0. + pub ok: bool, + pub outcome: Outcome, + pub exit_code: i32, + /// True when this stop left no proxy of this home running by its own paths. + pub runtime_down: bool, + pub proxy: Proxy, + pub message: String, +} + +/// What the shell concluded. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum StopResult { + /// The CLI reported a clean stop and a runtime that is down. + Stopped(Box), + ApprovalChanged(String), + ManagerStillActive(String), + HistoryIncomplete(String), + /// It reported anything else, or the run could not be read at all. + Failed(String), +} + +impl StopResult { + pub fn is_stopped(&self) -> bool { + matches!(self, Self::Stopped(_)) + } + + pub fn is_approval_changed(&self) -> bool { + matches!(self, Self::ApprovalChanged(_)) + } + + pub fn may_check_silence(&self) -> bool { + matches!(self, Self::Stopped(summary) if summary.outcome == Outcome::Stopped) + || matches!(self, Self::HistoryIncomplete(_)) + } + + pub fn describe(&self) -> String { + match self { + Self::Stopped(summary) => summary.message.clone(), + Self::ApprovalChanged(reason) => reason.clone(), + Self::ManagerStillActive(reason) => reason.clone(), + Self::HistoryIncomplete(reason) => reason.clone(), + Self::Failed(reason) => reason.clone(), + } + } +} + +/// Read one stop summary. +/// +/// Five facts have to hold together, and no four of them are enough. +/// +/// The process has to have exited 0, and the document has to say so too: `ok` is the strict +/// exit-code view and `exitCode` is the number behind it, so 1, 79 and 80 are refusals however the +/// rest of the document reads. Reading only the document would take a run's word for its own exit +/// status; reading only the status would accept a summary that disagrees with it. And `runtimeDown` +/// is the CLI's own statement that no proxy of this home is left running — a service that failed +/// while the proxy happened to stop satisfies that and not the others, and it is exactly the case +/// that may respawn the runtime a moment later. +/// +/// The fifth is that the document agrees with itself. The CLI's own summarizer cannot emit a +/// `failed` outcome beside a `stopped` proxy, but a reader that assumes that is trusting a +/// document to be self-consistent rather than checking. Only the two shapes that mean a runtime is +/// down are accepted, and an outcome or a proxy state this shell does not know fails to parse — +/// which is the same answer as a stop that did not happen. +pub fn read(exit_code: Option, stdout: &[u8], stderr: &[u8]) -> StopResult { + let text = String::from_utf8_lossy(stdout); + let summary: StopSummary = match serde_json::from_str(text.trim()) { + Ok(summary) => summary, + Err(error) => { + let detail = String::from_utf8_lossy(stderr); + let detail = detail.trim(); + let code = exit_code + .map(|code| code.to_string()) + .unwrap_or_else(|| "no exit code".to_owned()); + return StopResult::Failed(if detail.is_empty() { + format!("the bundled CLI's stop output could not be read (exit {code}: {error})") + } else { + format!("the bundled CLI's stop output could not be read (exit {code}): {detail}") + }); + } + }; + if summary.schema != SCHEMA { + return StopResult::Failed(format!( + "the bundled CLI answered with schema {} and this app understands {SCHEMA}", + summary.schema + )); + } + if summary.outcome == Outcome::ApprovalChanged { + return StopResult::ApprovalChanged(summary.message); + } + if summary.outcome == Outcome::ManagerStillActive { + return StopResult::ManagerStillActive(summary.message); + } + if summary.outcome == Outcome::HistoryIncomplete + && exit_code == Some(HISTORY_INCOMPLETE_EXIT_CODE) + && summary.exit_code == HISTORY_INCOMPLETE_EXIT_CODE + && !summary.ok + && summary.runtime_down + && matches!(summary.proxy, Proxy::Stopped | Proxy::StoppedOrphan) + { + return StopResult::HistoryIncomplete(format!( + "{} (outcome {})", + summary.message, + summary.outcome.as_str() + )); + } + let agrees = matches!( + (summary.outcome, summary.proxy), + (Outcome::Stopped, Proxy::Stopped) + | (Outcome::Stopped, Proxy::StoppedOrphan) + | (Outcome::NotRunning, Proxy::NotRunning) + ); + if exit_code != Some(0) + || !summary.ok + || summary.exit_code != 0 + || !summary.runtime_down + || !agrees + { + return StopResult::Failed(format!( + "{} (outcome {}, proxy {}, exit {}, process exit {})", + summary.message, + summary.outcome.as_str(), + summary.proxy.as_str(), + summary.exit_code, + exit_code + .map(|code| code.to_string()) + .unwrap_or_else(|| "none".to_owned()) + )); + } + StopResult::Stopped(Box::new(summary)) +} + +/// Run the ordinary bundled stop used when this app exits its own runtime. +pub async fn run(app: &AppHandle, deadline: Instant) -> StopResult { + run_with_args(app, deadline, vec!["stop".to_owned(), "--json".to_owned()]).await +} + +/// Run the bundled stop bound to the approved runtime, under the caller's deadline. +pub async fn run_approved( + app: &AppHandle, + deadline: Instant, + approved: &crate::resolve::Resolved, +) -> StopResult { + let (Some(pid), Some(port), crate::resolve::Takeover::Supported { token, .. }) = ( + approved.liveness.pid, + approved.liveness.port, + &approved.takeover, + ) else { + return StopResult::Failed("the approved runtime could not be identified".to_owned()); + }; + if pid == 0 || port == 0 { + return StopResult::Failed("the approved runtime could not be identified".to_owned()); + } + let argv = vec![ + "stop".to_owned(), + "--json".to_owned(), + "--expect-pid".to_owned(), + pid.to_string(), + "--expect-port".to_owned(), + port.to_string(), + "--expect-hostname".to_owned(), + approved.liveness.hostname.clone().unwrap_or_default(), + "--expect-config-home".to_owned(), + approved.config_home.clone(), + "--expect-cli-version".to_owned(), + approved.cli_version.clone(), + "--expect-compatibility-token".to_owned(), + token.clone(), + ]; + run_with_args(app, deadline, argv).await +} + +async fn run_with_args(app: &AppHandle, deadline: Instant, argv: Vec) -> StopResult { + let command = match app.shell().sidecar("ocx") { + Ok(command) => command.args(argv), + Err(error) => { + return StopResult::Failed(format!("the bundled CLI could not be started ({error})")) + } + }; + match timeout_at(deadline, command.output()).await { + Ok(Ok(output)) => read(output.status.code(), &output.stdout, &output.stderr), + Ok(Err(error)) => StopResult::Failed(format!("the bundled CLI could not be run ({error})")), + Err(_) => { + StopResult::Failed("the bundled CLI did not finish stopping before the deadline".into()) + } + } +} + +#[cfg(test)] +mod tests { + use super::{read, Outcome, Proxy, StopResult, SCHEMA}; + + fn document(ok: bool, outcome: &str, exit: i32, down: bool, proxy: &str) -> String { + format!( + r#"{{"schema":"ocx-stop/1","ok":{ok},"outcome":"{outcome}","exitCode":{exit}, + "runtimeDown":{down},"service":"absent","proxy":"{proxy}", + "sharedTeardown":"restored","message":"a message"}}"# + ) + } + + #[test] + fn a_clean_stop_with_the_runtime_down_is_the_only_success() { + let ok = document(true, "stopped", 0, true, "stopped"); + let result = read(Some(0), ok.as_bytes(), b""); + assert!(result.is_stopped()); + match result { + StopResult::Stopped(summary) => { + assert_eq!(summary.schema, SCHEMA); + assert_eq!(summary.outcome, Outcome::Stopped); + assert_eq!(summary.proxy, Proxy::Stopped); + assert!(summary.runtime_down); + } + StopResult::ApprovalChanged(reason) + | StopResult::ManagerStillActive(reason) + | StopResult::HistoryIncomplete(reason) + | StopResult::Failed(reason) => panic!("{reason}"), + } + // Nothing was running is equally a runtime that is down. + assert!(read( + Some(0), + document(true, "not-running", 0, true, "not-running").as_bytes(), + b"" + ) + .is_stopped()); + } + + #[test] + fn a_non_zero_exit_is_never_folded_into_success() { + // 79 and 80 report a proxy that went down with an obligation still owed. The runtime may + // be down, but the run did not succeed, and an update must not install over it. + for (outcome, exit) in [ + ("history-incomplete", 79), + ("history-deferred", 80), + ("failed", 1), + ] { + let document = document(false, outcome, exit, true, "stopped"); + let result = read(Some(exit), document.as_bytes(), b""); + assert!(!result.is_stopped(), "{outcome}"); + assert!(result.describe().contains(outcome)); + } + } + + #[test] + fn the_process_status_and_the_document_have_to_agree() { + let clean = document(true, "stopped", 0, true, "stopped"); + // A run that exited non-zero is a refusal even when its summary reads clean: taking the + // document's word for its own exit status is taking one claim as evidence of itself. + assert!(!read(Some(1), clean.as_bytes(), b"").is_stopped()); + assert!(!read(None, clean.as_bytes(), b"").is_stopped()); + // And a summary that contradicts its own exit code is not a stop either. + let contradictory = document(true, "stopped", 1, true, "stopped"); + assert!(!read(Some(0), contradictory.as_bytes(), b"").is_stopped()); + } + + #[test] + fn a_document_that_contradicts_itself_is_not_a_stop() { + // The CLI's summarizer cannot emit this, and the reader does not assume that. + let mixed = document(true, "failed", 0, true, "respawned"); + assert!(!read(Some(0), mixed.as_bytes(), b"").is_stopped()); + let orphan = document(true, "stopped", 0, true, "stopped-orphan"); + assert!(read(Some(0), orphan.as_bytes(), b"").is_stopped()); + // An outcome or a proxy state this shell does not know is not read at all. + let future = document(true, "stopped", 0, true, "stopped") + .replace("\"proxy\":\"stopped\"", "\"proxy\":\"parked\""); + assert!(!read(Some(0), future.as_bytes(), b"").is_stopped()); + } + + #[test] + fn a_runtime_still_up_is_a_failure_however_the_exit_reads() { + for proxy in [ + "respawned", + "stop-failed", + "ownership-refused", + "unresolvable-pid", + ] { + let document = document(true, "stopped", 0, false, proxy); + assert!( + !read(Some(0), document.as_bytes(), b"").is_stopped(), + "{proxy}" + ); + } + } + + #[test] + fn output_that_cannot_be_read_is_a_failure_not_a_stop() { + assert!(!read(Some(0), b"", b"boom").is_stopped()); + assert!(!read(Some(0), b"not json", b"").is_stopped()); + assert!(!read(None, b"", b"").is_stopped()); + let future = + document(true, "stopped", 0, true, "stopped").replace("ocx-stop/1", "ocx-stop/2"); + let result = read(Some(0), future.as_bytes(), b""); + assert!(!result.is_stopped()); + assert!(result.describe().contains("ocx-stop/2")); + } + + #[test] + fn guarded_refusals_parse_as_terminal_results() { + for (outcome, expected) in [ + ("approval-changed", Outcome::ApprovalChanged), + ("manager-still-active", Outcome::ManagerStillActive), + ] { + let output = document(false, outcome, 1, false, "unknown"); + let result = read(Some(1), output.as_bytes(), b""); + assert!(!result.may_check_silence()); + match result { + StopResult::ApprovalChanged(_) if expected == Outcome::ApprovalChanged => {} + StopResult::ManagerStillActive(_) if expected == Outcome::ManagerStillActive => {} + other => panic!("unexpected result: {other:?}"), + } + assert_eq!(expected.as_str(), outcome); + } + } + + #[test] + fn only_proven_history_incomplete_may_continue_to_silence() { + let complete = document(false, "history-incomplete", 79, true, "stopped"); + let result = read(Some(79), complete.as_bytes(), b""); + assert!(matches!(&result, StopResult::HistoryIncomplete(_))); + assert!(!result.is_stopped()); + assert!(result.may_check_silence()); + + let wrong_exit = read(Some(80), complete.as_bytes(), b""); + assert!(matches!(wrong_exit, StopResult::Failed(_))); + let unproven = document(false, "history-incomplete", 79, false, "stopped"); + assert!(matches!( + read(Some(79), unproven.as_bytes(), b""), + StopResult::Failed(_) + )); + let deferred = document(false, "history-deferred", 80, true, "stopped"); + assert!(!read(Some(80), deferred.as_bytes(), b"").may_check_silence()); + let not_running = document(true, "not-running", 0, true, "not-running"); + assert!(!read(Some(0), not_running.as_bytes(), b"").may_check_silence()); + assert!(!read(Some(1), b"{", b"").may_check_silence()); + } +} diff --git a/desktop/src-tauri/src/sidecar.rs b/desktop/src-tauri/src/sidecar.rs index 434c1d6288..71f16db8f0 100644 --- a/desktop/src-tauri/src/sidecar.rs +++ b/desktop/src-tauri/src/sidecar.rs @@ -1,27 +1,197 @@ -use crate::{discovery::ProxyEndpoint, proxy::ProxyClient}; -use tauri::{AppHandle, Manager}; -use tauri_plugin_shell::{process::CommandChild, ShellExt}; -use tokio::time::{sleep, timeout, Duration, Instant}; +//! Starting and watching the runtime this app owns. +//! +//! The spawn event stream used to be discarded into `_events`, which is why a sidecar that exited +//! immediately — a binary built for a CPU instruction set this machine does not have, a port +//! already taken, a corrupt install — presented as the same generic health failure as a slow start. +//! The child's exit code and its last output were both available and both thrown away. They are +//! consumed here instead, and they are what the startup diagnostic is made of. +//! +//! Stopping it is not here. D4 gives that to the bundled `ocx stop`, which owns the receipt-backed +//! teardown this process cannot perform on itself; see `runtime_stop.rs`. +//! +//! The exit is also where supervision starts: once it is recorded, the hook `supervisor.rs` +//! registered hears which child ended and how, so a runtime that went away is noticed at once +//! instead of only when the next startup run reads the record. -pub async fn ensure_proxy( - app: &AppHandle, - proxy: &ProxyClient, - endpoint: ProxyEndpoint, -) -> Result, String> { - let deadline = Instant::now() + Duration::from_secs(2); - loop { - if matches!( - timeout(Duration::from_millis(250), proxy.is_alive()).await, - Ok(Ok(_)) - ) { - return Ok(None); +use crate::endpoint::ProxyEndpoint; +use std::{ + collections::VecDeque, + sync::{Arc, Mutex}, +}; +use tauri::{async_runtime::Receiver, AppHandle, Manager}; +use tauri_plugin_shell::{ + process::{CommandChild, CommandEvent}, + ShellExt, +}; +/// How much sidecar output the diagnostic keeps. Enough to carry a stack trace or a startup +/// refusal, bounded so a chatty runtime cannot grow the buffer for the life of the process. +const MAX_LINES: usize = 40; + +/// The marker that tells the runtime this app waits on it (`DESKTOP_SUPERVISED_ENV` in +/// `src/lib/system-restart-contract.ts`). Under it a restart exits with +/// [`crate::supervisor::REQUESTED_RESTART_EXIT_CODE`] instead of spawning a detached replacement +/// this app could neither see nor stop, and the supervisor starts the replacement. +pub const SUPERVISED_ENV: &str = "OCX_DESKTOP_SUPERVISED"; + +/// Told which child ended, and how, once its exit is recorded. +pub type ExitHook = Arc; + +/// How the sidecar process ended. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct SidecarExit { + pub code: Option, + pub signal: Option, +} + +impl SidecarExit { + pub fn describe(&self) -> String { + match (self.code, self.signal) { + (Some(code), _) => format!("exit code {code}"), + (None, Some(signal)) => format!("terminated by signal {signal}"), + (None, None) => "exited without reporting a code".to_owned(), + } + } +} + +/// One thing the spawned child told us. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum SidecarEvent { + Line(String), + Exited(SidecarExit), +} + +#[derive(Default)] +struct WatchInner { + lines: VecDeque, + exit: Option, +} + +impl WatchInner { + fn record(&mut self, event: SidecarEvent) { + match event { + SidecarEvent::Line(line) => { + let line = line.trim_end().to_owned(); + if line.is_empty() { + return; + } + if self.lines.len() == MAX_LINES { + self.lines.pop_front(); + } + self.lines.push_back(line); + } + SidecarEvent::Exited(exit) => self.exit = Some(exit), + } + } +} + +/// The consumed spawn event stream of the child this app started. +#[derive(Clone, Default)] +pub struct SidecarWatch { + inner: Arc>, + hook: Arc>>, + /// The child this view follows. Only a view made by [`SidecarWatch::for_child`] has one, and + /// only such a view reports an exit to the hook: the record is shared across attempts, the pid + /// is what tells the supervisor which attempt ended. + pid: Option, +} + +impl SidecarWatch { + pub fn record(&self, event: SidecarEvent) { + if let Ok(mut inner) = self.inner.lock() { + inner.record(event); + } + } + + /// Register what hears about a child's exit. One hook for the life of the app. + pub fn on_exit(&self, hook: impl Fn(u32, SidecarExit) + Send + Sync + 'static) { + if let Ok(mut slot) = self.hook.lock() { + *slot = Some(Arc::new(hook)); + } + } + + /// The same record, following one spawned child. + pub fn for_child(&self, pid: u32) -> Self { + Self { + inner: self.inner.clone(), + hook: self.hook.clone(), + pid: Some(pid), } - if Instant::now() >= deadline { - break; + } + + /// Record one event, and report an exit to the hook after it is recorded, so whatever the hook + /// starts reads the exit it was told about. + fn deliver(&self, event: SidecarEvent) { + let exit = match &event { + SidecarEvent::Exited(exit) => Some(*exit), + SidecarEvent::Line(_) => None, + }; + self.record(event); + let (Some(exit), Some(pid)) = (exit, self.pid) else { + return; + }; + // Cloned out so the hook never runs under this lock. + let hook = self.hook.lock().ok().and_then(|slot| slot.clone()); + if let Some(hook) = hook { + hook(pid, exit); + } + } + + pub fn exit(&self) -> Option { + self.inner.lock().ok().and_then(|inner| inner.exit) + } + + pub fn lines(&self) -> Vec { + self.inner + .lock() + .map(|inner| inner.lines.iter().cloned().collect()) + .unwrap_or_default() + } + + /// Forget the previous attempt so a retry's diagnostic describes the retry. + pub fn reset(&self) { + if let Ok(mut inner) = self.inner.lock() { + inner.lines.clear(); + inner.exit = None; } - sleep(Duration::from_millis(150)).await; } + /// Drain the spawn event stream into this record for as long as the child lives. + pub fn follow(&self, mut events: Receiver) { + let watch = self.clone(); + tauri::async_runtime::spawn(async move { + while let Some(event) = events.recv().await { + if let Some(event) = translate(event) { + watch.deliver(event); + } + } + }); + } +} + +fn translate(event: CommandEvent) -> Option { + match event { + CommandEvent::Stdout(bytes) | CommandEvent::Stderr(bytes) => Some(SidecarEvent::Line( + String::from_utf8_lossy(&bytes).into_owned(), + )), + CommandEvent::Error(message) => Some(SidecarEvent::Line(format!("error: {message}"))), + CommandEvent::Terminated(payload) => Some(SidecarEvent::Exited(SidecarExit { + code: payload.code, + signal: payload.signal, + })), + _ => None, + } +} + +/// Start the bundled runtime and begin consuming what it says. +/// +/// The port is still passed explicitly. D5 hands that resolution to the bundled CLI so a user on a +/// custom `config.port` is not started on a different one; this is the call site that changes when +/// lane A's resolve verb lands, and nothing else here depends on where the number came from. +pub fn start( + app: &AppHandle, + endpoint: ProxyEndpoint, + watch: &SidecarWatch, +) -> Result { let gui_dist = app .path() .resource_dir() @@ -33,15 +203,105 @@ pub async fn ensure_proxy( .sidecar("ocx") .map_err(|error| error.to_string())? .args(["start", "--port", &endpoint.port.to_string()]) - .env("OPENCODEX_GUI_DIST", gui_dist); - let (_events, child) = command.spawn().map_err(|error| error.to_string())?; + .env("OPENCODEX_GUI_DIST", gui_dist) + .env(SUPERVISED_ENV, "1"); + let (events, child) = command.spawn().map_err(|error| error.to_string())?; + let watch = watch.for_child(child.pid()); + watch.follow(events); + Ok(child) +} + +#[cfg(test)] +mod tests { + use super::{SidecarEvent, SidecarExit, SidecarWatch, MAX_LINES}; + use std::sync::{Arc, Mutex}; + + #[test] + fn an_exit_reaches_the_hook_with_the_childs_pid_after_it_is_recorded() { + let watch = SidecarWatch::default(); + let seen = Arc::new(Mutex::new(Vec::new())); + let record = watch.clone(); + let sink = seen.clone(); + watch.on_exit(move |pid, exit| { + // Whatever the hook starts reads the exit it was told about. + let recorded = record.exit().and_then(|exit| exit.code); + sink.lock().unwrap().push((pid, exit.code, recorded)); + }); + // The shared record follows no child, so it names nobody to the hook. + watch.deliver(SidecarEvent::Exited(SidecarExit { + code: Some(1), + signal: None, + })); + let child = watch.for_child(4242); + child.deliver(SidecarEvent::Line("listening on 10100".into())); + child.deliver(SidecarEvent::Exited(SidecarExit { + code: Some(75), + signal: None, + })); + assert_eq!(*seen.lock().unwrap(), vec![(4242, Some(75), Some(75))]); + assert_eq!(watch.lines(), vec!["listening on 10100".to_owned()]); + } - for _ in 0..20 { - tokio::time::sleep(std::time::Duration::from_millis(150)).await; - if proxy.is_alive().await.is_ok() { - return Ok(Some(child)); + #[test] + fn the_exit_code_survives_the_event_stream() { + let watch = SidecarWatch::default(); + watch.record(SidecarEvent::Line("listening on 10100".into())); + watch.record(SidecarEvent::Exited(SidecarExit { + code: Some(1), + signal: None, + })); + assert_eq!(watch.exit().and_then(|exit| exit.code), Some(1)); + assert_eq!(watch.lines(), vec!["listening on 10100".to_owned()]); + } + + #[test] + fn output_is_bounded_and_keeps_the_end() { + let watch = SidecarWatch::default(); + for index in 0..(MAX_LINES + 5) { + watch.record(SidecarEvent::Line(format!("line {index}"))); } + let lines = watch.lines(); + assert_eq!(lines.len(), MAX_LINES); + assert_eq!(lines.first().unwrap(), "line 5"); + assert_eq!(lines.last().unwrap(), &format!("line {}", MAX_LINES + 4)); + } + + #[test] + fn blank_output_is_not_recorded_and_a_reset_forgets_the_attempt() { + let watch = SidecarWatch::default(); + watch.record(SidecarEvent::Line(" \n".into())); + assert!(watch.lines().is_empty()); + watch.record(SidecarEvent::Line("boom".into())); + watch.record(SidecarEvent::Exited(SidecarExit { + code: None, + signal: Some(9), + })); + watch.reset(); + assert!(watch.lines().is_empty()); + assert!(watch.exit().is_none()); + } + + #[test] + fn an_exit_reads_as_a_code_a_signal_or_neither() { + assert_eq!( + SidecarExit { + code: Some(2), + signal: None + } + .describe(), + "exit code 2" + ); + assert_eq!( + SidecarExit { + code: None, + signal: Some(9) + } + .describe(), + "terminated by signal 9" + ); + assert_eq!( + SidecarExit::default().describe(), + "exited without reporting a code" + ); } - let _ = child.kill(); - Err("the OpenCodex sidecar did not become healthy".into()) } diff --git a/desktop/src-tauri/src/startup.rs b/desktop/src-tauri/src/startup.rs new file mode 100644 index 0000000000..85c7145526 --- /dev/null +++ b/desktop/src-tauri/src/startup.rs @@ -0,0 +1,2582 @@ +//! The startup sequence, as named states inside a window the user can already see. +//! +//! Everything below used to run inside `setup()` before any window existed, and the window was +//! then created hidden. That ordering is why a failed start had no surface: the spawn event stream +//! was discarded, so the child's exit code was gone, and a run of probes that time out rather than +//! refuse takes over a minute with nothing on screen to explain it. D7 inverts it. The window is +//! created and shown first, and the sequence runs inside it as named states under one overall +//! deadline, with a retry, the child's exit code and a diagnostic the user can copy. +//! +//! Registration comes first, before the runtime is touched at all. The order looks backwards until +//! you follow the failing case: a login launch starts hidden, and if the tray were installed only +//! after a successful start then a start that failed would leave a running process with no window +//! and no icon — invisible. The app establishes its own surface, then deals with the runtime. +//! +//! A launch that came from login autostart starts hidden, and that is the only difference — except +//! where there is no usable tray to hide into, which is R1 and lives in [`shows_window`]. + +use crate::{ + auth::Auth, + claim, + endpoint::ProxyEndpoint, + first_run::{self, StartAtLogin}, + identity, ownership, + proxy::{ProxyClient, RuntimeIdentity}, + resolve, runtime_stop, + sidecar::{self, SidecarWatch}, + tray_availability::{self, TrayAvailability}, + AppState, +}; +use serde::Serialize; +use std::{ + path::PathBuf, + sync::{ + atomic::{AtomicBool, AtomicU64, Ordering}, + Mutex, MutexGuard, PoisonError, + }, +}; +use tauri::{AppHandle, Emitter, Manager}; +use tokio::sync::oneshot; +use tokio::time::{sleep, sleep_until, Duration, Instant}; + +/// The event the bootstrap page listens on. +pub const PHASE_EVENT: &str = "startup-phase"; + +/// One deadline for the whole sequence. +/// +/// Per-step budgets were what produced the unbounded case: a two-second attach loop whose probes +/// each cost a four-second client timeout, followed by twenty more waits, adds up to something no +/// single number in the code admitted to. One ceiling over the whole run is a promise that can be +/// read — and every probe under it is bounded by the remaining time rather than by its own +/// timeout, because otherwise the last probe overruns the ceiling by the whole client timeout. +pub const DEADLINE: Duration = Duration::from_secs(30); + +const POLL: Duration = Duration::from_millis(250); + +/// How long the deadline guard waits past the ceiling before speaking for a run that has not. +/// +/// The run's own failure names the endpoint, the home and how the child ended; the guard's can +/// only name where it stalled. The grace lets the run lose its own race first, so the better +/// diagnostic is the one on screen. +const SETTLE_GRACE: Duration = Duration::from_secs(2); + +/// How long a child this app started may take to answer before a later run stops waiting on it. +/// +/// The slowest start that still ends well is a hard-pinned port reclaim (60 seconds) plus the +/// pinned prefer-retry (5 seconds). Past this, a child that neither answers nor has reported an +/// exit is wedged, or its exit is held up by a grandchild that kept its output pipes open (the +/// shell plugin reports an exit only once both pipes close), and waiting on it again would keep +/// the port empty for good. +pub const CHILD_START_GRACE: Duration = Duration::from_secs(90); + +/// Whether a run waits on the child this app already started instead of starting another; `age` +/// is how long ago the tracked child was spawned, and nothing when the app tracks none. The caller +/// also requires that the child has not reported an exit. +pub fn waits_on_child(age: Option) -> bool { + age.is_some_and(|age| age < CHILD_START_GRACE) +} + +/// Where the launch came from. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum LaunchOrigin { + /// A person opened the app. + User, + /// The login item started it. + Autostart, +} + +/// The argument the autostart registration passes back to us. Nothing else supplies it, so its +/// presence is the launch origin. +pub const AUTOSTART_FLAG: &str = "--autostart"; + +/// Why a run was started. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Mode { + /// The app's own launch, or a person's retry. It may ask to take a listening runtime over. + Launch, + /// The supervisor bringing a runtime back (`supervisor.rs`). Nobody is looking at this run, so + /// it never shows the window or a prompt: a runtime that answers is attached as a guest. Every + /// other gate is the launch's own — only a proven absence starts a runtime. + Recover, +} + +impl LaunchOrigin { + pub fn from_args(mut args: impl Iterator) -> Self { + if args.any(|argument| argument == AUTOSTART_FLAG) { + Self::Autostart + } else { + Self::User + } + } + + pub fn detect() -> Self { + Self::from_args(std::env::args()) + } +} + +/// Whether this launch shows its window. +/// +/// D7 shows it always and exempts a login launch, which starts hidden. D6 shows it wherever there +/// is no usable tray. A no-tray login launch satisfies both rules and they disagree, so R1 settles +/// it: tray availability wins. Starting hidden is a property of having somewhere to be hidden in, +/// not of how the process was started. +pub fn shows_window(origin: LaunchOrigin, tray: TrayAvailability) -> bool { + !tray.is_available() || origin == LaunchOrigin::User +} + +/// A named state of the startup sequence. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Phase { + /// Nothing has run yet. + /// + /// This is what the sequence's state says before its first report, and it is deliberately not + /// one of the [`PHASES`]: it is the absence of a run, not a step of one. Seeding the state + /// with `Registering` instead made "the sequence has not started" render exactly like "the + /// sequence is registering", so a shell that never began was indistinguishable from one that + /// had — on the one surface whose job is to tell those apart. + NotStarted, + Registering, + Resolving, + Probing, + Attaching, + TakingOver, + Starting, + Waiting, + Ready, + Failed, +} + +/// Every phase, in the order they run. The bootstrap page derives its checklist from this rather +/// than restating it, so a phase cannot exist in one place and be missing from the other. +/// +/// [`Phase::NotStarted`] is absent on purpose. It is the state of not having run, so a checklist +/// row for it would be a step that never completes. +pub const PHASES: [Phase; 9] = [ + Phase::Registering, + Phase::Resolving, + Phase::Probing, + Phase::Attaching, + Phase::TakingOver, + Phase::Starting, + Phase::Waiting, + Phase::Ready, + Phase::Failed, +]; + +impl Phase { + /// The stable identifier the bootstrap page keys on. + pub fn id(self) -> &'static str { + match self { + Self::NotStarted => "not-started", + Self::Registering => "registering", + Self::Resolving => "resolving", + Self::Probing => "probing", + Self::Attaching => "attaching", + Self::TakingOver => "taking-over", + Self::Starting => "starting", + Self::Waiting => "waiting", + Self::Ready => "ready", + Self::Failed => "failed", + } + } + + pub fn label(self) -> &'static str { + match self { + Self::NotStarted => "Waiting for the startup sequence to begin", + Self::Registering => "Registering the tray and the login item", + Self::Resolving => "Resolving the configuration home and port", + Self::Probing => "Looking for a runtime that is already listening", + Self::Attaching => "Attaching to the runtime that answered", + Self::TakingOver => "Taking over the runtime that was already listening", + Self::Starting => "Starting the bundled runtime", + Self::Waiting => "Waiting for the runtime to report healthy", + Self::Ready => "Ready", + Self::Failed => "OpenCodex could not start its runtime", + } + } + + pub fn is_terminal(self) -> bool { + matches!(self, Self::Ready | Self::Failed) + } + + /// The phase a published id came from, for a caller that only has the wire value. + /// + /// Derived from [`PHASES`] rather than restating the mapping, so a phase cannot be resolvable + /// here and missing from the checklist. + pub fn from_id(id: &str) -> Option { + PHASES.into_iter().find(|phase| phase.id() == id) + } +} + +/// One phase, as the bootstrap page sees it. +#[derive(Clone, Debug, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct PhaseInfo { + pub id: &'static str, + pub label: &'static str, + pub terminal: bool, +} + +/// The phase list the page renders. Derived from [`PHASES`] so the two cannot drift. +pub fn phase_list() -> Vec { + PHASES + .iter() + .map(|phase| PhaseInfo { + id: phase.id(), + label: phase.label(), + terminal: phase.is_terminal(), + }) + .collect() +} + +/// What the bootstrap page is told. +/// +/// It carries the phases already finished, not just the current one. An event emitted before the +/// page's listener exists is gone, and the early phases finish in milliseconds, so a page that +/// reconstructed history from events alone would show a run in progress with nothing behind it. +#[derive(Clone, Debug, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct Progress { + pub phase: &'static str, + pub label: &'static str, + pub detail: Option, + pub completed: Vec<&'static str>, + pub failed_phase: Option<&'static str>, + pub elapsed_ms: u64, + pub dashboard: Option, + pub diagnostic: Option, + pub can_retry: bool, + /// Present only while the shell is waiting on the user's takeover decision. + pub consent: Option, +} + +/// What the consent panel renders. `blocked` carries the CLI's refusal reason when a +/// takeover cannot be offered; the panel is shown only for the offerable case today, but +/// the field is part of the wire so a later UI does not need a schema change. +#[derive(Clone, Debug, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct ConsentPrompt { + pub endpoint: String, + pub port: u16, + pub home: String, + pub owner: String, + pub blocked: Option, + /// The version-skew warning when the listening runtime and the bundled CLI differ; + /// the panel renders it so an upgrade, or a downgrade the proxy-newer rule did not + /// already keep unoffered, is a decision made with the versions visible. + pub version_note: Option, +} + +impl Progress { + fn new(phase: Phase, elapsed_ms: u64) -> Self { + Self { + phase: phase.id(), + label: phase.label(), + detail: None, + completed: Vec::new(), + failed_phase: None, + elapsed_ms, + dashboard: None, + diagnostic: None, + can_retry: phase == Phase::Failed, + consent: None, + } + } +} + +/// What the page is told when the sequence's own state is not registered. +/// +/// The command used to answer `None` here, and the page dropped it: `apply` returns early on a +/// falsy progress, so the surface kept its initial markup, no event ever arrived, and nothing on +/// screen distinguished that from a run still in progress. A shell that cannot find its own +/// startup state is a defect, and a defect the user can read and copy beats a window that looks +/// like it is still working. +pub fn unavailable() -> Progress { + let reason = + "the shell's startup state is not registered, so it cannot report on its own startup"; + let mut progress = Progress::new(Phase::Failed, 0); + progress.diagnostic = Some(format!( + "OpenCodex desktop {} on {}\nstate: {}\nreason: {reason}", + env!("CARGO_PKG_VERSION"), + std::env::consts::OS, + Phase::NotStarted.id(), + )); + progress.detail = Some(reason.to_owned()); + progress +} + +/// Where the sequence is pointed, once the CLI has said. +#[derive(Clone)] +struct Target { + endpoint: ProxyEndpoint, + home: PathBuf, +} + +/// What registering established about this installation. +#[derive(Clone, Debug)] +pub struct Registration { + pub login: StartAtLogin, + /// This installation's own id, minted once in the app's config directory. + pub install_id: Option, +} + +struct Live { + latest: Progress, + reported: Vec<&'static str>, + /// The takeover decision state. Answered stays set until the run clears it: the deadline + /// extension lands between the decision and the clear, and the guard must keep waiting + /// through both. + consent: ConsentState, + /// The current run's ceiling. + /// + /// The consent wait moves it by however long the person took, so the deadline guard + /// re-reads it instead of racing a stale copy. It lives under this lock so the expiry + /// decision and the terminal publish are one critical section against consent transitions. + deadline: Instant, +} + +impl Live { + fn is_settled(&self) -> bool { + self.latest.phase == Phase::Ready.id() || self.latest.phase == Phase::Failed.id() + } + + /// Record the latest state. A run that already said how it ended refuses further + /// reports: the terminal state is the page's promise that the screen stopped changing, + /// and a probe resuming after the expiry landed must not move it back — nor reopen the + /// consent gate that reads this state. Returns whether the report was taken. + fn publish(&mut self, progress: &mut Progress, failed_in: Option) -> bool { + if self.is_settled() { + return false; + } + if !self.reported.contains(&progress.phase) + && progress.phase != Phase::Ready.id() + && progress.phase != Phase::Failed.id() + { + self.reported.push(progress.phase); + } + progress.completed = self + .reported + .iter() + .copied() + .filter(|id| *id != progress.phase) + .collect(); + progress.failed_phase = failed_in.map(Phase::id); + self.latest = progress.clone(); + true + } +} + +/// The takeover prompt's decision state. +enum ConsentState { + /// No prompt is up and none was just answered. + Idle, + /// A prompt is up; the sender resolves with the user's decision. + Pending(oneshot::Sender), + /// The user answered and the run has not yet consumed the extension. + Answered, +} + +/// What the deadline guard's expiry step found. +enum Expiry { + /// The run is terminal or superseded; the guard is done. + Dead, + /// A consent prompt is pending or its answer is being consumed; re-check shortly. + Blocked, + /// Not expired yet; the current ceiling plus its grace. + Waiting(Instant), + /// Expired and the failure was published in the same critical section; emit it. + Fired(Box), +} + +/// The sequence's managed state: the latest thing it said, what it has already finished, and +/// whether it is running, so a retry cannot start a second run alongside the first. +pub struct Startup { + live: Mutex, + /// Serialize state publication with its synchronous event dispatch. Always acquired + /// before `live`, and never held across an await. + reporting: Mutex<()>, + running: AtomicBool, + /// Whether the run in flight is a [`Mode::Recover`] run. + recovering: AtomicBool, + /// Set when the run failed because a listener this app cannot use holds the port. The + /// supervisor reads it when the run ends: another attempt would only find the same listener. + held: Mutex>, + /// Whether this window has already left the bundled bootstrap surface. + /// + /// Explicit open actions can arrive repeatedly from the tray, the single-instance hook, and + /// the shell command. Navigating on every action would recreate the React application and + /// discard renderer state, so the transition is owned here and consumed exactly once per run. + dashboard_loaded: AtomicBool, + /// Whether a person asked for the dashboard during this run. + /// + /// An explicit open that arrives while startup is still running only shows the bootstrap page; + /// `finish` reads this after it has recorded Ready, and `open_dashboard` sets it before it + /// reads progress, so whichever of the two runs second sees the other and navigates. + dashboard_requested: AtomicBool, + /// Which run the state belongs to. + /// + /// A run's deadline guard outlives the run it was started for, and a retry that begins before + /// the old guard fires would otherwise be failed by it. + generation: AtomicU64, + /// The outcome of the one-time registration, once it has happened. + registered: Mutex>, +} + +impl Startup { + pub fn new() -> Self { + Self { + live: Mutex::new(Live { + latest: Progress::new(Phase::NotStarted, 0), + reported: Vec::new(), + consent: ConsentState::Idle, + deadline: Instant::now(), + }), + reporting: Mutex::new(()), + running: AtomicBool::new(false), + recovering: AtomicBool::new(false), + held: Mutex::new(None), + dashboard_loaded: AtomicBool::new(false), + dashboard_requested: AtomicBool::new(false), + generation: AtomicU64::new(0), + registered: Mutex::new(None), + } + } + + fn live(&self) -> MutexGuard<'_, Live> { + self.live.lock().unwrap_or_else(PoisonError::into_inner) + } + + fn registration(&self) -> Option { + self.registered + .lock() + .unwrap_or_else(PoisonError::into_inner) + .clone() + } + + fn remember_registration(&self, registration: Registration) { + *self + .registered + .lock() + .unwrap_or_else(PoisonError::into_inner) = Some(registration); + } + + /// The whole state of the run so far, which is what the page asks for when it loads. + pub fn latest(&self) -> Progress { + self.live().latest.clone() + } + + /// Whether a run is in flight. + pub fn is_running(&self) -> bool { + self.running.load(Ordering::Acquire) + } + + /// Whether the last run reported Ready. + pub fn is_ready(&self) -> bool { + self.live().latest.phase == Phase::Ready.id() + } + + fn mode(&self) -> Mode { + if self.recovering.load(Ordering::Acquire) { + Mode::Recover + } else { + Mode::Launch + } + } + + fn held_slot(&self) -> MutexGuard<'_, Option> { + self.held.lock().unwrap_or_else(PoisonError::into_inner) + } + + /// Record that this run found the port held by a listener it cannot use. + fn note_held(&self, pid: Option) { + *self.held_slot() = Some(crate::supervisor::Held { pid }); + } + + fn take_held(&self) -> Option { + self.held_slot().take() + } + + /// The user's answer to a pending takeover prompt. Nothing pending is a no-op: a retry + /// or a late click must never be read as a decision for a prompt that is not up. + pub fn decide_takeover(&self, approved: bool) { + let mut live = self.live(); + match std::mem::replace(&mut live.consent, ConsentState::Idle) { + ConsentState::Pending(sender) => { + live.consent = ConsentState::Answered; + let _ = sender.send(approved); + } + // A late click or a duplicate decision answers nothing: restore what was there. + prior => live.consent = prior, + } + } + + /// Register the pending prompt, unless the run already ended. A guard expiry can win the + /// race against the prompt being posted; posting one anyway would leave a receiver that + /// waits forever on a decision nobody can see. + fn await_consent(&self) -> Option> { + let mut live = self.live(); + if live.is_settled() { + return None; + } + let (sender, receiver) = oneshot::channel(); + live.consent = ConsentState::Pending(sender); + Some(receiver) + } + + /// Consume the decision and publish the extended ceiling in the same critical section, so + /// the guard's next expiry check sees either a pending/answered prompt or the new deadline, + /// never the gap between them. + fn resolve_consent(&self, deadline: Instant) { + let mut live = self.live(); + live.deadline = deadline; + live.consent = ConsentState::Idle; + } + + fn set_deadline(&self, deadline: Instant) { + self.live().deadline = deadline; + } + + fn restart(&self) { + // A retry during a pending consent prompt drops the sender, so the waiting run reads + // the decision as declined rather than pairing an old prompt with a new sequence. + let mut live = self.live(); + live.consent = ConsentState::Idle; + live.reported.clear(); + live.latest = Progress::new(Phase::NotStarted, 0); + *self.held_slot() = None; + self.dashboard_loaded.store(false, Ordering::SeqCst); + self.dashboard_requested.store(false, Ordering::SeqCst); + } + + fn should_navigate_dashboard(&self) -> bool { + !self.dashboard_loaded.swap(true, Ordering::SeqCst) + } + + /// Give the one navigation back when the WebView refused the script, so the next open retries. + fn navigation_failed(&self) { + self.dashboard_loaded.store(false, Ordering::SeqCst); + } + + fn request_dashboard(&self) { + self.dashboard_requested.store(true, Ordering::SeqCst); + } + + fn dashboard_requested(&self) -> bool { + self.dashboard_requested.load(Ordering::SeqCst) + } + + /// The dashboard URL once this run is Ready, otherwise nothing. + fn ready_dashboard(&self) -> Option { + let progress = self.latest(); + (progress.phase == Phase::Ready.id()) + .then_some(progress.dashboard) + .flatten() + } + + /// Whether the run has already said how it ended. + /// + /// A terminal state is the page's only promise that the screen has stopped changing, so it is + /// also what tells a late guard there is nothing left to report. + #[cfg(test)] + fn settled(&self) -> bool { + self.live().is_settled() + } + + fn publish(&self, progress: &mut Progress, failed_in: Option) -> bool { + let mut live = self.live(); + live.publish(progress, failed_in) + } + + fn with_reporting(&self, report: impl FnOnce() -> T) -> T { + let _reporting = self + .reporting + .lock() + .unwrap_or_else(PoisonError::into_inner); + report() + } + + /// Publish a terminal state for a run that did not report one itself. + /// + /// Idempotent and bound to the run it was started for: a run that already said Ready or + /// Failed is left alone, and a caller whose run has been superseded by a retry says + /// nothing. The check and the publish are one critical section, so no other reporter can + /// slip a state between them. + fn settle(&self, started: Instant, generation: u64, reason: String) -> Option { + let mut live = self.live(); + self.settle_locked(&mut live, started, generation, reason) + } + + fn settle_locked( + &self, + live: &mut Live, + started: Instant, + generation: u64, + reason: String, + ) -> Option { + if self.generation.load(Ordering::Acquire) != generation || live.is_settled() { + return None; + } + let stalled_in = live.latest.phase; + let elapsed_ms = elapsed(started); + let mut progress = Progress::new(Phase::Failed, elapsed_ms); + progress.diagnostic = Some( + [ + format!( + "OpenCodex desktop {} on {}", + env!("CARGO_PKG_VERSION"), + std::env::consts::OS + ), + format!("state: {stalled_in}"), + format!("reason: {reason}"), + format!("elapsed: {elapsed_ms}ms"), + ] + .join("\n"), + ); + progress.detail = Some(reason); + live.publish(&mut progress, Phase::from_id(stalled_in)); + Some(progress) + } + + /// The deadline guard's atomic expiry step. The deadline read, the consent state, the + /// terminal check and the failure publish all share one critical section, so a prompt + /// posted or an answer consumed on the other side of the lock can never meet a failure + /// already in flight. + fn expire_run(&self, started: Instant, generation: u64, reason: String) -> Expiry { + let mut live = self.live(); + if self.generation.load(Ordering::Acquire) != generation || live.is_settled() { + return Expiry::Dead; + } + if !matches!(live.consent, ConsentState::Idle) { + // A prompt is up or an answer is being consumed. The budget does not run + // against the person, so there is nothing to expire. + return Expiry::Blocked; + } + let wake = live.deadline + SETTLE_GRACE; + if wake > Instant::now() { + return Expiry::Waiting(wake); + } + match self.settle_locked(&mut live, started, generation, reason) { + Some(progress) => Expiry::Fired(Box::new(progress)), + None => Expiry::Dead, + } + } +} + +impl Default for Startup { + fn default() -> Self { + Self::new() + } +} + +/// Run the sequence, unless it is already running. This is also the retry. +pub fn begin(app: &AppHandle) { + begin_with(app, Mode::Launch); +} + +/// Run the sequence for `mode`, unless one is already running. Returns whether this call started +/// a run. The run reports how it ended to the supervisor, which is what follows up a failed +/// recovery. +pub fn begin_with(app: &AppHandle, mode: Mode) -> bool { + let Some(startup) = app.try_state::() else { + return false; + }; + if startup + .running + .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) + .is_err() + { + return false; + } + startup + .recovering + .store(mode == Mode::Recover, Ordering::Release); + startup.restart(); + let generation = startup.generation.fetch_add(1, Ordering::AcqRel) + 1; + let started = Instant::now(); + startup.set_deadline(started + DEADLINE); + let app = app.clone(); + + // The ceiling is a promise to the page, and something has to keep it when the run does not. + // Every `return` below that reports nothing, and every step that outlives the ceiling, used to + // leave the surface on whatever it was last told — or on its own initial markup when nothing + // had been published at all — for as long as the process lived. That screen is the one a user + // cannot tell from a hung application, which is the whole thing this surface exists to avoid. + let guard = app.clone(); + tauri::async_runtime::spawn(async move { + // The consent wait extends the shared deadline, and while a prompt is up the budget + // does not run at all. The expiry check, the consent state and the terminal publish + // share one critical section, so a prompt posted or an answer consumed can never meet + // a failure already in flight. + loop { + let Some(startup) = guard.try_state::() else { + return; + }; + let expiry = startup.with_reporting(|| { + let expiry = startup.expire_run( + started, + generation, + format!( + "the startup sequence did not finish within {} seconds", + DEADLINE.as_secs() + ), + ); + if let Expiry::Fired(progress) = &expiry { + let _ = guard.emit(PHASE_EVENT, progress); + } + expiry + }); + match expiry { + Expiry::Dead => return, + Expiry::Blocked => { + sleep(POLL).await; + continue; + } + Expiry::Waiting(wake) => { + sleep_until(wake).await; + continue; + } + Expiry::Fired(_) => return, + } + } + }); + + tauri::async_runtime::spawn(async move { + run(&app, started).await; + settle( + &app, + started, + generation, + "the startup sequence ended without reporting a result".to_owned(), + ); + // Read before the flag drops, so a retry that starts at once cannot answer for this run. + let mut outcome = None; + let mut held = None; + if let Some(startup) = app.try_state::() { + outcome = Some(startup.latest()); + held = startup.take_held(); + startup.running.store(false, Ordering::Release); + } + let ready = outcome + .as_ref() + .is_some_and(|progress| progress.phase == Phase::Ready.id()); + let detail = outcome.and_then(|progress| progress.detail); + let ended = match (ready, held) { + (true, _) => crate::supervisor::RunOutcome::Ready, + (false, Some(held)) => crate::supervisor::RunOutcome::Held(held), + (false, None) => crate::supervisor::RunOutcome::Failed, + }; + crate::supervisor::run_finished(&app, mode, ended, detail.as_deref()); + }); + true +} + +/// Report a terminal state for a run that did not report one itself. +/// +/// Idempotent and bound to the run it was started for: a run that already said Ready or Failed is +/// left alone, and a guard whose run has been superseded by a retry says nothing. +fn settle(app: &AppHandle, started: Instant, generation: u64, reason: String) { + let Some(startup) = app.try_state::() else { + return; + }; + startup.with_reporting(|| { + if let Some(progress) = startup.settle(started, generation, reason) { + let _ = app.emit(PHASE_EVENT, progress); + } + }); +} + +async fn run(app: &AppHandle, started: Instant) { + let mut deadline = started + DEADLINE; + // Publishing comes before any lookup that can fail. A sequence that returns before it has + // said anything leaves the page unable to tell "not started" from "still going". + report(app, started, Phase::Registering, None); + let Some(watch) = app.try_state::().map(|state| state.watch.clone()) else { + return; + }; + let registration = register(app, deadline).await; + report( + app, + started, + Phase::Registering, + Some(registration.login.describe().to_owned()), + ); + + report(app, started, Phase::Resolving, None); + // D5: the shell no longer resolves the home, the port or liveness. It asks the bundled CLI, + // which owns the tuned probe budgets that exist because a shell-side reimplementation answered + // "nobody is listening" twice and started duplicate proxies. The call is inside the sequence, so + // a CLI that is missing or slow has a state, a diagnostic and a retry rather than a guess. + let resolution = resolve::run(app, deadline).await; + let Some(answer) = resolution.resolved() else { + // Fail-closed. A resolution that could not be trusted is not an absence, and nothing below + // may read it as one. + fail( + app, + started, + None, + ®istration, + &watch, + Phase::Resolving, + resolution + .reason() + .unwrap_or("the runtime could not be resolved") + .to_owned(), + ); + return; + }; + let endpoint = answer.endpoint(); + let target = Target { + endpoint, + home: answer.home(), + }; + let proxy = match ProxyClient::new(endpoint, Auth::new(answer.home())) { + Ok(proxy) => proxy, + Err(error) => { + fail( + app, + started, + Some(&target), + ®istration, + &watch, + Phase::Resolving, + error.to_string(), + ); + return; + } + }; + if let Some(state) = app.try_state::() { + state.attach(proxy.clone()); + } + report( + app, + started, + Phase::Resolving, + Some(format!( + "{} with a configuration home of {}, resolved by the bundled CLI {}; {}", + target.endpoint.url(""), + target.home.display(), + answer.cli_version, + ownership::describe(&answer.ownership, registration.install_id.as_deref()) + )), + ); + + report( + app, + started, + Phase::Probing, + Some(match answer.liveness.status { + resolve::Status::Live => "a runtime is already listening".to_owned(), + resolve::Status::AbsentProven => { + "no runtime is listening, and that absence was proven".to_owned() + } + }), + ); + let mut took_over = false; + match resolve::live_verdict(&resolution) { + resolve::LiveVerdict::Attach => { + // Without our own id nothing can ever match us, which is the answer Refuse gives. + let consent = match registration.install_id.as_deref() { + Some(install_id) => ownership::consent(&answer.ownership, install_id), + None => ownership::Consent::Refuse, + }; + let mode = app + .try_state::() + .map_or(Mode::Launch, |startup| startup.mode()); + match attach_plan(consent, answer, mode) { + AttachPlan::Guest(detail) => { + attach_as_guest( + app, + started, + &target, + ®istration, + &watch, + &proxy, + endpoint, + deadline, + detail, + ) + .await; + return; + } + AttachPlan::Ask => { + // The prompt has to be visible even when this launch started hidden. + if let Some(window) = app.get_webview_window("main") { + crate::window::show(&window); + } + let Some(startup) = app.try_state::() else { + return; + }; + let Some(receiver) = startup.await_consent() else { + // The run already ended (an expiry won the race to the terminal + // state). Posting the prompt now would wait on a decision nobody + // can see, so the run stops here instead. + return; + }; + let mut progress = Progress::new(Phase::Attaching, elapsed(started)); + progress.detail = Some( + "a runtime was already listening; waiting for a decision on taking it over" + .to_owned(), + ); + progress.consent = Some(ConsentPrompt { + endpoint: target.endpoint.url(""), + port: target.endpoint.port, + home: target.home.display().to_string(), + owner: ownership::owner_label(&answer.ownership), + blocked: None, + version_note: consent_version_note(answer), + }); + emit(app, progress, None); + // The user may take any time; the budget exists to bound the machinery, not + // the person, so the deadline moves by whatever the decision took. + let asked = Instant::now(); + let approved = receiver.await.unwrap_or(false); + deadline += asked.elapsed(); + // The extension and the clear are one critical section: the guard sees + // either a prompt still pending or the moved ceiling, never the gap. + startup.resolve_consent(deadline); + if !approved { + attach_as_guest( + app, + started, + &target, + ®istration, + &watch, + &proxy, + endpoint, + deadline, + format!( + "a runtime was already listening and taking it over was declined, so this app is a guest on it{}", + answer + .skew_warning() + .map(|warning| format!(" ({warning})")) + .unwrap_or_default() + ), + ) + .await; + return; + } + if take_over( + app, + started, + &mut deadline, + &target, + ®istration, + &watch, + &proxy, + answer, + ) + .await + .is_err() + { + return; + } + took_over = true; + } + } + } + // A Child's client runtime. It is never taken over, so there is nothing to ask: attach to + // it, and `bind` makes that ownership when it is the child this app started. + resolve::LiveVerdict::Client => { + attach_as_guest( + app, + started, + &target, + ®istration, + &watch, + &proxy, + endpoint, + deadline, + "a Child's client runtime is listening; it serves Codex through its Home and the Child's dashboard, so this app attached to it and asked nothing" + .to_owned(), + ) + .await; + return; + } + // Something holds the port and this app cannot manage it. That is not an absence, so it + // does not authorise starting a second runtime beside it either. Another attempt would + // find the same listener, which the supervisor is told so it waits for a change instead. + resolve::LiveVerdict::Unusable(reason) => { + if let Some(startup) = app.try_state::() { + startup.note_held(answer.liveness.pid); + } + fail( + app, + started, + Some(&target), + ®istration, + &watch, + Phase::Attaching, + reason, + ); + return; + } + resolve::LiveVerdict::NotLive => {} + } + if !took_over && !resolve::may_start(&resolution) { + // Only a proven absence authorises a start. Nothing else may fall through to one. A + // takeover just proved its own absence by stopping what was there. + fail( + app, + started, + Some(&target), + ®istration, + &watch, + Phase::Probing, + "the runtime's liveness could not be established, so no runtime was started".to_owned(), + ); + return; + } + + // A retry or a recovery must not leave a second proxy behind. A child that has not reported an + // exit is still out there, whatever the last run concluded, so the run waits on that one rather + // than starting another and racing it for the port. This reads the child the app tracks, not + // the ownership confirmation: `attach` above has just reset that, which left the wait dead. + let owns_live_child = app + .try_state::() + .is_some_and(|state| waits_on_child(state.child_age())) + && watch.exit().is_none(); + if owns_live_child { + report( + app, + started, + Phase::Starting, + Some("the runtime this app started has not exited; waiting on it again".to_owned()), + ); + } else { + report(app, started, Phase::Starting, None); + watch.reset(); + match spawn_runtime(app, endpoint, &watch) { + Some(Ok(())) => {} + Some(Err(error)) => { + fail( + app, + started, + Some(&target), + ®istration, + &watch, + Phase::Starting, + error, + ); + return; + } + // An exit is already in flight, so starting a runtime now would orphan it. + None => return, + } + } + + report(app, started, Phase::Waiting, None); + while Instant::now() < deadline { + if matches!(proxy.alive_within(deadline).await, Some(Ok(_))) { + if bind(app, &proxy, deadline).await.is_none() { + // Healthy is not the same as identified: a 200 with a body that does not carry the + // marker is something else holding the port, and the token is never sent to it. + fail( + app, + started, + Some(&target), + ®istration, + &watch, + Phase::Waiting, + "the runtime reported healthy but did not identify itself".to_owned(), + ); + return; + } + finish(app, started, endpoint); + return; + } + // A child that has already exited will never answer, so the deadline is not worth waiting + // out. This is the case the discarded event stream used to hide behind a generic timeout. + if let Some(exit) = watch.exit() { + fail( + app, + started, + Some(&target), + ®istration, + &watch, + Phase::Waiting, + format!("the runtime {}", exit.describe()), + ); + return; + } + sleep(POLL).await; + } + fail( + app, + started, + Some(&target), + ®istration, + &watch, + Phase::Waiting, + format!( + "the runtime did not report healthy within {} seconds", + DEADLINE.as_secs() + ), + ); +} + +/// What an attach turns into once the recorded owner and the CLI's compatibility answer are +/// laid next to each other. The approved resolve answer carries the claim token. +enum AttachPlan { + /// Stay a guest on what answered; the string is the detail the phase reports. + Guest(String), + /// Offer the takeover and wait on the user. + Ask, +} + +fn attach_plan(consent: ownership::Consent, answer: &resolve::Resolved, mode: Mode) -> AttachPlan { + // The wire warning is the same sentence the CLI prints; it is appended verbatim so + // this surface and `ocx status` never describe the same mismatch differently. + let skew_note = || { + answer + .skew_warning() + .map(|warning| format!(" ({warning})")) + .unwrap_or_default() + }; + match consent { + ownership::Consent::Held => AttachPlan::Guest(format!( + "a runtime was already listening and this installation already owns it{}", + skew_note() + )), + ownership::Consent::Refuse => AttachPlan::Guest(format!( + "a runtime was already listening; its recorded owner could not be read, so this app is a guest on it and asked nothing{}", + skew_note() + )), + ownership::Consent::AskFirstTime | ownership::Consent::AskAgain => match &answer.takeover { + resolve::Takeover::Blocked { reason, detail } => AttachPlan::Guest(format!( + "a runtime was already listening, but taking it over is not available ({reason}: {detail}), so this app is a guest on it{}", + skew_note() + )), + // A supported takeover still goes unoffered when the listening runtime is + // NEWER than the bundled one: approving it would stop the newer runtime and + // start the older bundle, a downgrade nobody asked for. Guest keeps serving + // and the note says why no prompt appeared. + resolve::Takeover::Supported { .. } + if answer.runtime_relation() == resolve::VersionRelation::ProxyNewer => + { + AttachPlan::Guest(format!( + "a runtime was already listening and runs a newer OpenCodex than this app bundles ({}, this app {}), so taking it over would downgrade it; this app is a guest on it and asked nothing{}", + answer + .liveness + .version + .as_deref() + .unwrap_or("an unknown version"), + answer.cli_version, + skew_note() + )) + } + // A recovery runs with nobody watching; a prompt would surface a window the person + // never asked for, over a runtime that is serving. It stays a guest instead. + resolve::Takeover::Supported { .. } if mode == Mode::Recover => AttachPlan::Guest( + format!( + "a runtime was already listening when this app came back for its own, so the recovery attached as a guest and asked nothing{}", + skew_note() + ), + ), + resolve::Takeover::Supported { .. } => AttachPlan::Ask, + }, + } +} + +/// What the consent panel shows about versions. The skew warning comes first; when the +/// versions could not be compared at all (fake, empty or unorderable version strings) the +/// panel still gets one honest line instead of silently asking for a takeover. +fn consent_version_note(answer: &resolve::Resolved) -> Option { + answer.skew_warning().map(str::to_owned).or_else(|| { + matches!( + answer.runtime_relation(), + resolve::VersionRelation::Unknown | resolve::VersionRelation::Incomparable, + ) + .then(|| "the listening runtime's version could not be compared".to_owned()) + }) +} + +/// Report, bind and finish as a guest on the runtime that answered. +#[allow(clippy::too_many_arguments)] +async fn attach_as_guest( + app: &AppHandle, + started: Instant, + target: &Target, + registration: &Registration, + watch: &SidecarWatch, + proxy: &ProxyClient, + endpoint: ProxyEndpoint, + deadline: Instant, + detail: String, +) { + report(app, started, Phase::Attaching, Some(detail)); + if bind(app, proxy, deadline).await.is_none() { + fail( + app, + started, + Some(target), + registration, + watch, + Phase::Attaching, + "the runtime answered but did not identify itself, so this app did not attach" + .to_owned(), + ); + return; + } + finish(app, started, endpoint); +} + +fn approval_still_current(approved: &resolve::Resolved, fresh: &resolve::Resolution) -> bool { + let Some(now) = fresh.resolved() else { + return false; + }; + matches!(resolve::live_verdict(fresh), resolve::LiveVerdict::Attach) + && matches!(&now.takeover, resolve::Takeover::Supported { .. }) + && approved.ownership == now.ownership + && approved.takeover == now.takeover + && approved.config_home == now.config_home + && approved.cli_version == now.cli_version + && approved.port == now.port + && approved.liveness == now.liveness +} + +async fn stop_after_approval( + approved: &resolve::Resolved, + fresh: &resolve::Resolution, + stop: F, +) -> Option +where + F: FnOnce() -> Fut, + Fut: std::future::Future, +{ + if !approval_still_current(approved, fresh) { + return None; + } + Some(stop().await) +} + +async fn claim_after_silence( + stopped: &runtime_stop::StopResult, + silent: bool, + claim: F, +) -> Option +where + F: FnOnce() -> Fut, + Fut: std::future::Future, +{ + if !silent || !stopped.may_check_silence() { + return None; + } + Some(claim().await) +} + +/// Stop the runtime that answered, wait for its silence, and record this installation as +/// the owner. An `Err` has already been reported; `Ok` means the Starting branch may run. +#[allow(clippy::too_many_arguments)] +async fn take_over( + app: &AppHandle, + started: Instant, + deadline: &mut Instant, + target: &Target, + registration: &Registration, + watch: &SidecarWatch, + proxy: &ProxyClient, + approved: &resolve::Resolved, +) -> Result<(), ()> { + let fresh = resolve::run(app, *deadline).await; + let stopped = stop_after_approval(approved, &fresh, || async { + report( + app, + started, + Phase::TakingOver, + Some("stopping the runtime that was already listening".to_owned()), + ); + runtime_stop::run_approved(app, *deadline, approved).await + }) + .await; + let Some(stopped) = stopped else { + fail( + app, + started, + Some(target), + registration, + watch, + Phase::TakingOver, + "the runtime or its managing CLI changed after approval; retry to review it".to_owned(), + ); + return Err(()); + }; + if stopped.is_approval_changed() || !stopped.may_check_silence() { + fail( + app, + started, + Some(target), + registration, + watch, + Phase::TakingOver, + format!( + "the guarded stop could not confirm the approved runtime: {}", + stopped.describe() + ), + ); + return Err(()); + } + // A refused connection, not exit 0 and not the probe's deadline, is the receipt: `ocx stop` + // reports exit 79 when the proxy stopped but history cleanup failed after it exited, and a + // `None` from alive_within is only the clock running out — neither is silence. Only a + // validated stop result reaches this loop, and claim still requires active refusal. + let mut silent = false; + let mut still_answering = false; + while Instant::now() < *deadline { + match proxy.alive_within(*deadline).await { + Some(Err(_)) => { + silent = true; + break; + } + Some(Ok(_)) => { + still_answering = true; + sleep(POLL).await; + } + None => { + still_answering = false; + break; + } + } + } + if !silent { + fail( + app, + started, + Some(target), + registration, + watch, + Phase::TakingOver, + format!( + "the runtime that was already listening {} ({})", + if still_answering { + "is still answering after the stop" + } else { + "did not go silent before the deadline" + }, + stopped.describe() + ), + ); + return Err(()); + } + if !stopped.is_stopped() { + report( + app, + started, + Phase::TakingOver, + Some(format!( + "the runtime that was already listening stopped answering (stop reported: {})", + stopped.describe() + )), + ); + } + + report( + app, + started, + Phase::TakingOver, + Some("recording this installation as the runtime owner".to_owned()), + ); + let install_id = registration.install_id.clone().unwrap_or_default(); + // An unknown record reaches here only off the UI path, and the claim has to refuse rather + // than fabricate the subject it is claiming against. + let resolve::Takeover::Supported { token, .. } = &approved.takeover else { + return Err(()); + }; + let Some(argv) = claim::args(&install_id, &approved.ownership, token) else { + fail( + app, + started, + Some(target), + registration, + watch, + Phase::TakingOver, + "the recorded owner could not be read, so no claim was made".to_owned(), + ); + return Err(()); + }; + match claim_after_silence(&stopped, silent, || claim::run(app, argv, *deadline)).await { + Some(claim::ClaimResult::Recorded(ownership)) => { + report( + app, + started, + Phase::TakingOver, + Some(format!( + "this installation now owns the runtime (consent generation {})", + ownership.consent_generation + )), + ); + Ok(()) + } + Some(claim::ClaimResult::Failed(message)) => { + // The runtime is stopped either way. Refusing here leaves the next launch an + // ordinary absence to start into, which is the acceptable end state. + fail( + app, + started, + Some(target), + registration, + watch, + Phase::TakingOver, + format!( + "the runtime was stopped, but this installation could not be recorded as its owner: {message}" + ), + ); + Err(()) + } + None => { + fail( + app, + started, + Some(target), + registration, + watch, + Phase::TakingOver, + "the approved runtime changed, so no ownership claim was attempted".to_owned(), + ); + Err(()) + } + } +} + +/// Establish the app's own surface: the tray verdict, the tray, and the login item. +/// +/// It happens once per process. A retry re-runs the runtime half of the sequence, and running this +/// half again would build a second tray icon with its own refresh loop and its own menu handlers — +/// the failure would look like the app duplicating itself every time the user pressed Retry. +async fn register(app: &AppHandle, deadline: Instant) -> Registration { + if let Some(done) = app + .try_state::() + .and_then(|startup| startup.registration()) + { + return done; + } + + // The probe blocks on a session-bus round trip, so it does not belong on an async worker — and + // it is bounded by the sequence's own deadline, because a bus that never answers would + // otherwise leave the page in this state with a retry that could do nothing about it. + let tray = match tokio::time::timeout_at( + deadline, + tauri::async_runtime::spawn_blocking(tray_availability::detect), + ) + .await + { + Ok(Ok(tray)) => tray, + _ => TrayAvailability::assumed(), + }; + + // Before the tray, so its Start at Login checkbox reads the state this leaves behind rather + // than the state from before first run. + let login = first_run::apply_start_at_login_default(app); + first_run::adopt_launch_origin_argument(app); + + // The verdict is published only once an icon actually exists. Announcing a tray and then + // failing to install it would hide the window into nothing, which is the exact stranding D6 + // exists to prevent. + let verdict = if tray.is_available() && install_tray(app, deadline).await { + TrayAvailability::Available + } else { + TrayAvailability::Unavailable + }; + if let Some(coordinator) = app.try_state::() { + coordinator.set_tray(verdict); + } + + if let Some(window) = app.get_webview_window("main") { + if shows_window(LaunchOrigin::detect(), verdict) { + crate::window::show(&window); + } + } + // This installation's own id, and what the recorded runtime owner says about it. The claim + // lives in the shared service install state and the CLI is what reads it; the comparison + // against our own id is the rule that record publishes. + // This installation's own id; what the recorded runtime owner says about it is part of the + // resolve answer, so the identity line is written where the answer exists. + let registration = Registration { + login, + install_id: identity::install_id(app), + }; + if let Some(startup) = app.try_state::() { + startup.remember_registration(registration.clone()); + } + registration +} + +/// Build the tray on the main thread, which is where GTK requires it on Linux. +async fn install_tray(app: &AppHandle, deadline: Instant) -> bool { + let handle = app.clone(); + let (sender, receiver) = tokio::sync::oneshot::channel(); + if app + .run_on_main_thread(move || { + let _ = sender.send(crate::tray::install(&handle).map_err(|error| error.to_string())); + }) + .is_err() + { + return false; + } + match tokio::time::timeout_at(deadline, receiver).await { + Ok(Ok(Ok(()))) => true, + Ok(Ok(Err(error))) => { + crate::logging::log_once("the tray could not be installed", &error); + false + } + _ => { + crate::logging::log_once( + "the tray could not be installed", + "the main thread did not answer", + ); + false + } + } +} +/// Start the runtime, unless an exit is already in flight. +/// +/// The coordinator reserves the spawn rather than holding its lock across it: holding it would put +/// process creation in front of the main thread's exit handler, so a wedged spawn would be a Quit +/// that never answers. A quit arriving in between is deferred until the child is ours and then +/// drains it, so it cannot observe "we own nothing" and leave a proxy running that nothing stops. +fn spawn_runtime( + app: &AppHandle, + endpoint: ProxyEndpoint, + watch: &SidecarWatch, +) -> Option> { + let coordinator = app.try_state::()?; + if !coordinator.begin_spawn() { + return None; + } + let outcome = match sidecar::start(app, endpoint, watch) { + Ok(child) => { + if let Some(state) = app.try_state::() { + state.adopt(child); + } + Ok(()) + } + Err(error) => Err(error), + }; + if let Some(reason) = coordinator.finish_spawn() { + // A quit landed while the child was being created. It is ours now, so it gets drained. + crate::exit::drain_now(app, reason); + return None; + } + Some(outcome) +} + +/// Establish which instance is answering, and whether it is the child this app started. +/// +/// The health body is unauthenticated and carries the marker, the pid and the port, so identity is +/// settled before any credential is sent. It is also the only thing that grants process ownership: +/// a spawn records a pid, and this is what says that pid is the one holding the port. An answer +/// that cannot be read leaves the app owning nothing, which is the safe way round — an owner's stop +/// sent to a listener that is not ours is a stop sent to somebody else's runtime. +/// +/// The answer is returned rather than swallowed, because a sequence that cannot identify what it is +/// talking to has not finished. Reporting Ready there would navigate the window to a dashboard the +/// shell cannot authenticate against, since the management token is only sent to a bound instance. +async fn bind(app: &AppHandle, proxy: &ProxyClient, deadline: Instant) -> Option { + let identity = match tokio::time::timeout_at(deadline, proxy.identify()).await { + Ok(Ok(identity)) => identity, + _ => return None, + }; + proxy.bind(identity); + if let Some(state) = app.try_state::() { + state.confirm_ownership(identity); + } + Some(identity) +} + +fn finish(app: &AppHandle, started: Instant, endpoint: ProxyEndpoint) { + // Ownership is whatever the confirmation above established, not whatever a spawn assumed. + crate::tray::set_owned( + app, + app.try_state::() + .is_some_and(|state| state.owns_runtime()), + ); + let path = format!( + "/?desktop_session={}#/usage", + app.state::() + .session_id() + ); + let dashboard = endpoint.url(&path); + let mut progress = Progress::new(Phase::Ready, elapsed(started)); + progress.dashboard = Some(dashboard.clone()); + if !emit(app, progress, None) { + // The run already ended — the expiry won while this one was still binding. The + // terminal state stays and the window must not navigate away from it. + return; + } + app.state::().wake(); + if let Some(window) = app.get_webview_window("main") { + let visible = window.is_visible().unwrap_or(true); + let startup = app.try_state::(); + let requested = startup + .as_ref() + .is_some_and(|startup| startup.dashboard_requested()); + let mode = startup + .as_ref() + .map_or(Mode::Launch, |startup| startup.mode()); + if keeps_update_page(mode, crate::window::shows_update_page(&window)) { + return; + } + if loads_dashboard_on_ready(LaunchOrigin::detect(), visible, requested) { + match startup { + Some(startup) => { + navigate_once(&startup, &dashboard, |url| navigate_dashboard(&window, url)); + } + None => { + navigate_dashboard(&window, &dashboard); + } + } + } + } +} + +/// Open the full dashboard only when a person asks for it. +/// +/// A hidden login launch deliberately leaves its WebView on the tiny bundled startup surface after +/// the runtime becomes ready. The tray, a second ordinary application launch, or the bootstrap +/// command reaches this function and pays the dashboard cost at that point. If startup is still in +/// progress the bootstrap is merely shown; `finish` observes the now-visible window and performs +/// the navigation once the endpoint is ready. +pub fn open_dashboard(app: &AppHandle) { + let startup = app.try_state::(); + let Some(window) = app.get_webview_window("main") else { + return; + }; + if let Some(startup) = startup { + // The request is recorded before progress is read; see `dashboard_requested`. + startup.request_dashboard(); + if let Some(dashboard) = startup.ready_dashboard() { + navigate_once(&startup, &dashboard, |url| navigate_dashboard(&window, url)); + } + } + crate::window::show(&window); +} + +pub fn return_to_dashboard(app: &AppHandle) -> Result<(), String> { + let startup = app.try_state::().ok_or("dashboard is not ready")?; + let dashboard = startup.ready_dashboard(); + let window = app + .get_webview_window("main") + .ok_or("dashboard window is unavailable")?; + return_ready_dashboard(dashboard.as_deref(), |url| navigate_dashboard(&window, url))?; + crate::window::show(&window); + Ok(()) +} + +fn return_ready_dashboard( + dashboard: Option<&str>, + navigate: impl FnOnce(&str) -> bool, +) -> Result<(), String> { + let dashboard = dashboard.ok_or("dashboard is not ready")?; + if !navigate(dashboard) { + return Err("dashboard could not be opened".into()); + } + Ok(()) +} + +fn loads_dashboard_on_ready(origin: LaunchOrigin, window_visible: bool, requested: bool) -> bool { + origin == LaunchOrigin::User || window_visible || requested +} + +/// Whether a Ready run leaves the update page on screen instead of loading the dashboard. +/// +/// A recovery nobody asked to watch lands wherever the window is. When that is the update page — +/// a failed install brings the runtime back from there — the page is showing why the install +/// failed, and its own button returns to the dashboard. +fn keeps_update_page(mode: Mode, on_update_page: bool) -> bool { + mode == Mode::Recover && on_update_page +} + +/// Perform this run's single dashboard navigation through `navigate`. +/// +/// `navigate` reports whether the WebView accepted the script. Acceptance is not proof that the +/// page finished loading, but a refusal certainly left the bootstrap page in place, so the claim is +/// returned and the next explicit open tries again instead of being suppressed for the whole run. +fn navigate_once(startup: &Startup, dashboard: &str, navigate: impl FnOnce(&str) -> bool) -> bool { + if !startup.should_navigate_dashboard() { + return false; + } + if navigate(dashboard) { + return true; + } + startup.navigation_failed(); + false +} + +fn navigate_dashboard(window: &tauri::WebviewWindow, dashboard: &str) -> bool { + // justified: replacing the bootstrap page with the dashboard is how this window has always + // navigated, and the string is a URL this process resolved, not anything a page supplied. + window + .eval(format!("window.location.replace({dashboard:?})")) + .is_ok() +} + +#[allow(clippy::too_many_arguments)] +fn fail( + app: &AppHandle, + started: Instant, + target: Option<&Target>, + registration: &Registration, + watch: &SidecarWatch, + phase: Phase, + reason: String, +) { + let elapsed_ms = elapsed(started); + let mut progress = Progress::new(Phase::Failed, elapsed_ms); + progress.diagnostic = Some(diagnostic( + target.map(|target| (target.endpoint, target.home.clone())), + registration, + watch, + phase, + &reason, + elapsed_ms, + )); + progress.detail = Some(reason); + emit(app, progress, Some(phase)); +} + +/// The text the failure surface offers for copying. +/// +/// It names the state it stopped in, the endpoint and home it was using, how the child ended and +/// what the child last said. Those together are what separates "the port was taken" from "the +/// binary will not run on this CPU" from "the home is not the one the accounts are in", and none of +/// them were reachable from the generic health failure this replaces. +pub fn diagnostic( + target: Option<(ProxyEndpoint, PathBuf)>, + registration: &Registration, + watch: &SidecarWatch, + phase: Phase, + reason: &str, + elapsed_ms: u64, +) -> String { + let mut lines = vec![ + format!( + "OpenCodex desktop {} on {}", + env!("CARGO_PKG_VERSION"), + std::env::consts::OS + ), + format!("state: {}", phase.id()), + format!("reason: {reason}"), + format!("elapsed: {elapsed_ms}ms"), + ]; + match target { + Some((endpoint, home)) => { + lines.push(format!("endpoint: {}", endpoint.url(""))); + lines.push(format!("home: {}", home.display())); + } + None => lines.push("endpoint: not resolved".to_owned()), + } + lines.push(format!("start at login: {}", registration.login.describe())); + lines.push(format!( + "installation id: {}", + registration.install_id.as_deref().unwrap_or("not minted") + )); + lines.push(match watch.exit() { + Some(exit) => format!("runtime process: {}", exit.describe()), + None => "runtime process: still running or never started".to_owned(), + }); + let output = watch.lines(); + if output.is_empty() { + lines.push("runtime output: none".to_owned()); + } else { + lines.push("runtime output:".to_owned()); + lines.extend(output.into_iter().map(|line| format!(" {line}"))); + } + lines.join("\n") +} + +fn report(app: &AppHandle, started: Instant, phase: Phase, detail: Option) { + let mut progress = Progress::new(phase, elapsed(started)); + progress.detail = detail; + emit(app, progress, None); +} + +/// Publish and emit one state. A report refused because the run already ended is not +/// emitted either, so a stale event cannot move the page past the terminal state the +/// snapshot keeps. Returns whether the report was published. +fn emit(app: &AppHandle, mut progress: Progress, failed_in: Option) -> bool { + if let Some(startup) = app.try_state::() { + return startup.with_reporting(|| { + if !startup.publish(&mut progress, failed_in) { + return false; + } + let _ = app.emit(PHASE_EVENT, progress); + true + }); + } + let _ = app.emit(PHASE_EVENT, progress); + true +} + +fn elapsed(started: Instant) -> u64 { + started.elapsed().as_millis() as u64 +} + +#[cfg(test)] +mod tests { + use super::{ + approval_still_current, attach_plan, claim_after_silence, consent_version_note, + keeps_update_page, loads_dashboard_on_ready, navigate_once, return_ready_dashboard, + shows_window, stop_after_approval, unavailable, waits_on_child, AttachPlan, ConsentState, + Expiry, LaunchOrigin, Mode, Phase, Progress, Startup, AUTOSTART_FLAG, CHILD_START_GRACE, + DEADLINE, PHASES, POLL, + }; + use crate::claim::ClaimResult; + use crate::ownership::{Claim, Consent, Owner, Recorded}; + use crate::resolve::{ + Liveness, Port, Resolution, Resolved, Status, Takeover, VersionRelation, VersionSkew, + }; + use crate::runtime_stop::{self, StopResult}; + use crate::tray_availability::TrayAvailability; + use std::cell::Cell; + use std::sync::atomic::Ordering; + use tokio::time::{Duration, Instant}; + + fn supported() -> Takeover { + Takeover::Supported { + protocol_version: 1, + minimum_cli_version: "2.61.0".to_owned(), + token: "tok".to_owned(), + } + } + + fn blocked() -> Takeover { + Takeover::Blocked { + reason: "managing-cli-unsupported".to_owned(), + detail: "path uses 2.59.0".to_owned(), + } + } + + fn answer_for(takeover: Takeover) -> Resolved { + Resolved { + schema: "ocx-resolve/1".to_owned(), + cli_version: "2.61.0".to_owned(), + config_home: "/sandbox/a".to_owned(), + port: Port { + effective: 10100, + configured: 10100, + }, + liveness: Liveness { + status: Status::Live, + pid: Some(42), + port: Some(10100), + hostname: None, + version: Some("2.61.0".to_owned()), + role: None, + }, + ownership: Recorded::Owned { + ownership: Claim { + owner: Owner::Cli, + install_id: "cli-a".to_owned(), + consent_generation: 2, + }, + revision: 7, + }, + takeover, + version_skew: None, + } + } + + fn approved_answer() -> Resolved { + answer_for(supported()) + } + + fn skewed_answer(relation: VersionRelation, warning: &str) -> Resolved { + let mut answer = approved_answer(); + answer.liveness.version = Some("2.62.0".to_owned()); + answer.version_skew = Some(VersionSkew { + cli_version: "2.61.0".to_owned(), + proxy_version: Some("2.62.0".to_owned()), + skewed: true, + relation, + warning: Some(warning.to_owned()), + }); + answer + } + + #[test] + fn a_changed_answer_never_invokes_stop() { + tauri::async_runtime::block_on(async { + let approved = approved_answer(); + let mut changed = approved.clone(); + changed.ownership = Recorded::None { revision: 8 }; + let called = Cell::new(false); + let refused = stop_after_approval( + &approved, + &Resolution::Answered(Box::new(changed)), + || async { + called.set(true); + StopResult::Failed("called".to_owned()) + }, + ) + .await; + assert!(refused.is_none()); + assert!(!called.get()); + let accepted = stop_after_approval( + &approved, + &Resolution::Answered(Box::new(approved.clone())), + || async { + called.set(true); + StopResult::Failed("called".to_owned()) + }, + ) + .await; + assert!(accepted.is_some()); + assert!(called.get()); + let mut moved = approved.clone(); + moved.liveness.pid = Some(43); + assert!(!approval_still_current( + &approved, + &Resolution::Answered(Box::new(moved)) + )); + let mut moved = approved.clone(); + moved.port.effective = 10101; + assert!(!approval_still_current( + &approved, + &Resolution::Answered(Box::new(moved)) + )); + let mut moved = approved.clone(); + moved.config_home = "/sandbox/b".to_owned(); + assert!(!approval_still_current( + &approved, + &Resolution::Answered(Box::new(moved)) + )); + let mut moved = approved.clone(); + moved.cli_version = "2.62.0".to_owned(); + assert!(!approval_still_current( + &approved, + &Resolution::Answered(Box::new(moved)) + )); + let mut moved = approved.clone(); + moved.liveness.hostname = Some("localhost".to_owned()); + assert!(!approval_still_current( + &approved, + &Resolution::Answered(Box::new(moved)) + )); + let mut moved = approved.clone(); + if let Takeover::Supported { token, .. } = &mut moved.takeover { + *token = "changed".to_owned(); + } + assert!(!approval_still_current( + &approved, + &Resolution::Answered(Box::new(moved)) + )); + let mut moved = approved.clone(); + moved.takeover = blocked(); + assert!(!approval_still_current( + &approved, + &Resolution::Answered(Box::new(moved)) + )); + assert!(!approval_still_current( + &approved, + &Resolution::Unknown("unreadable".to_owned()) + )); + }); + } + + #[test] + fn terminal_stop_results_never_invoke_claim_after_silence() { + tauri::async_runtime::block_on(async { + let approved = approved_answer(); + let answer = Resolution::Answered(Box::new(approved.clone())); + let called = Cell::new(false); + for result in [ + StopResult::ApprovalChanged("moved".to_owned()), + StopResult::ManagerStillActive("active".to_owned()), + runtime_stop::read(Some(1), b"{", b""), + StopResult::Failed("the bundled CLI timed out".to_owned()), + ] { + let stopped = stop_after_approval(&approved, &answer, || async { result }) + .await + .expect("matching answer"); + assert!(!stopped.may_check_silence()); + let claimed = claim_after_silence(&stopped, true, || async { + called.set(true); + ClaimResult::Failed("called".to_owned()) + }) + .await; + assert!(claimed.is_none()); + assert!(!called.get()); + } + let history = StopResult::HistoryIncomplete("history-incomplete".to_owned()); + let claimed = claim_after_silence(&history, true, || async { + called.set(true); + ClaimResult::Failed("called".to_owned()) + }) + .await; + assert!(matches!(claimed, Some(ClaimResult::Failed(_)))); + assert!(called.get()); + }); + } + + #[test] + fn a_retry_drops_a_prompt_the_run_is_still_waiting_on() { + // The waiting run reads the dropped sender as declined, so a stale prompt can never + // pair a decision meant for it with the retried sequence. + let startup = Startup::new(); + let mut receiver = startup.await_consent().expect("no terminal state yet"); + startup.restart(); + assert!(matches!( + receiver.try_recv(), + Err(tokio::sync::oneshot::error::TryRecvError::Closed) + )); + } + + #[test] + fn an_ask_only_arises_when_the_takeover_can_be_taken() { + // Held and Refuse never ask, whatever the CLI reported about compatibility. + assert!(matches!( + attach_plan(Consent::Held, &answer_for(supported()), Mode::Launch), + AttachPlan::Guest(_) + )); + assert!(matches!( + attach_plan(Consent::Refuse, &answer_for(supported()), Mode::Launch), + AttachPlan::Guest(_) + )); + assert!(matches!( + attach_plan( + Consent::AskFirstTime, + &answer_for(supported()), + Mode::Launch + ), + AttachPlan::Ask + )); + match attach_plan(Consent::AskAgain, &answer_for(blocked()), Mode::Launch) { + AttachPlan::Guest(detail) => { + assert!(detail.contains("managing-cli-unsupported: path uses 2.59.0")) + } + AttachPlan::Ask => panic!("a blocked takeover is not an offer"), + } + } + + #[test] + fn a_newer_runtime_is_never_offered_a_downgrade() { + // A takeover replaces what listens with the bundled runtime; offering it against a + // NEWER listener is an approval prompt for a downgrade. + for consent in [Consent::AskFirstTime, Consent::AskAgain] { + match attach_plan( + consent, + &skewed_answer(VersionRelation::ProxyNewer, "skew"), + Mode::Launch, + ) { + AttachPlan::Guest(detail) => { + assert!(detail.contains("newer OpenCodex")); + assert!(detail.contains("downgrade")); + assert!(detail.contains("2.62.0")); + assert!(detail.contains(" (skew)")); + } + AttachPlan::Ask => panic!("a newer runtime must not be offered a downgrade"), + } + } + // Recover mode already stays a guest; the proxy-newer rule keeps it so. + match attach_plan( + Consent::AskFirstTime, + &skewed_answer(VersionRelation::ProxyNewer, "skew"), + Mode::Recover, + ) { + AttachPlan::Guest(detail) => { + assert!(detail.contains("downgrade")); + assert!(detail.contains(" (skew)")); + } + AttachPlan::Ask => panic!("a recovery must not prompt"), + } + } + + #[test] + fn the_consent_note_reports_an_uncomparable_version() { + // A fake or missing version yields no skew warning, but the consent panel must not + // ask for a takeover with the version row silently empty. + let mut answer = approved_answer(); + answer.version_skew = Some(VersionSkew { + cli_version: "2.61.0".to_owned(), + proxy_version: None, + skewed: false, + relation: VersionRelation::Incomparable, + warning: None, + }); + assert_eq!( + consent_version_note(&answer).as_deref(), + Some("the listening runtime's version could not be compared") + ); + // A comparable version without a warning needs no extra line. + answer.version_skew = Some(VersionSkew { + cli_version: "2.61.0".to_owned(), + proxy_version: Some("2.61.0".to_owned()), + skewed: false, + relation: VersionRelation::Match, + warning: None, + }); + assert_eq!(consent_version_note(&answer), None); + // A real warning always wins over the fallback. + let warned = skewed_answer(VersionRelation::CliNewer, "CLI 2.61.0 does not match"); + assert_eq!( + consent_version_note(&warned).as_deref(), + Some("CLI 2.61.0 does not match") + ); + } + + #[test] + fn an_older_runtime_still_asks_with_the_skew_visible() { + // The same supported takeover stays offerable when it upgrades the listener; the + // warning is what the consent panel shows. + let answer = skewed_answer(VersionRelation::CliNewer, "CLI 2.61.0 does not match"); + assert!(matches!( + attach_plan(Consent::AskFirstTime, &answer, Mode::Launch), + AttachPlan::Ask + )); + assert_eq!(answer.skew_warning(), Some("CLI 2.61.0 does not match")); + } + + #[test] + fn guest_details_carry_the_skew_warning() { + // Held and Refuse stay guests either way, but the phase detail must name the + // mismatch instead of hiding it in the proxy's own log. + let answer = skewed_answer( + VersionRelation::ProxyNewer, + "CLI 2.61.0 does not match the running proxy 2.62.0", + ); + match attach_plan(Consent::Held, &answer, Mode::Launch) { + AttachPlan::Guest(detail) => { + assert!(detail.contains("does not match the running proxy")) + } + AttachPlan::Ask => panic!("held consent never asks"), + } + match attach_plan( + Consent::AskAgain, + &{ + let mut blocked_answer = skewed_answer( + VersionRelation::ProxyNewer, + "CLI 2.61.0 does not match the running proxy 2.62.0", + ); + blocked_answer.takeover = blocked(); + blocked_answer + }, + Mode::Launch, + ) { + AttachPlan::Guest(detail) => { + assert!(detail.contains("managing-cli-unsupported")); + assert!(detail.contains("does not match the running proxy")); + } + AttachPlan::Ask => panic!("a blocked takeover is not an offer"), + } + } + + #[test] + fn a_recovery_never_asks_and_attaches_as_a_guest() { + // Nobody is looking at a recovery: a prompt would show a window nobody asked for. + for consent in [Consent::AskFirstTime, Consent::AskAgain] { + match attach_plan(consent, &answer_for(supported()), Mode::Recover) { + AttachPlan::Guest(detail) => assert!(detail.contains("guest")), + AttachPlan::Ask => panic!("a recovery must not prompt"), + } + // A launch still asks. + assert!(matches!( + attach_plan(consent, &answer_for(supported()), Mode::Launch), + AttachPlan::Ask + )); + } + let startup = Startup::new(); + assert_eq!(startup.mode(), Mode::Launch); + startup.recovering.store(true, Ordering::SeqCst); + assert_eq!(startup.mode(), Mode::Recover); + } + + #[test] + fn not_having_started_is_not_a_step_of_the_run() { + // A checklist row for it would be a step that never completes, and resolving it out of a + // published id would name a phase the page has nowhere to draw. + assert!(!PHASES.contains(&Phase::NotStarted)); + assert_eq!(Phase::from_id(Phase::NotStarted.id()), None); + for phase in PHASES { + assert_eq!(Phase::from_id(phase.id()), Some(phase)); + } + } + + #[test] + fn a_sequence_that_has_not_run_says_so() { + // Seeding the state with Registering made "has not started" render exactly like "started, + // and registering" — on the one surface whose job is to tell those apart. + let startup = Startup::new(); + assert_eq!(startup.latest().phase, Phase::NotStarted.id()); + assert!(!startup.latest().can_retry); + assert!(!startup.settled()); + } + + #[test] + fn the_snapshot_never_answers_with_nothing() { + // The page returns early on a falsy progress, so answering None here was a window frozen + // on its own markup with no diagnostic in it and no event coming. + let progress = unavailable(); + assert_eq!(progress.phase, Phase::Failed.id()); + assert!(progress.can_retry); + assert!(progress.detail.is_some()); + assert!(progress + .diagnostic + .is_some_and(|text| text.contains("reason:"))); + } + + #[test] + fn only_a_terminal_state_settles_a_run() { + // This is what stops the deadline guard from overwriting a run that already reported, and + // what makes it speak for one that never did. + let startup = Startup::new(); + let mut running = Progress::new(Phase::Waiting, 1); + startup.publish(&mut running, None); + assert!(!startup.settled()); + let mut done = Progress::new(Phase::Ready, 2); + startup.publish(&mut done, None); + assert!(startup.settled()); + } + + #[test] + fn only_the_autostart_argument_marks_a_login_launch() { + let user = ["/Applications/OpenCodex.app".to_owned()]; + assert_eq!( + LaunchOrigin::from_args(user.into_iter()), + LaunchOrigin::User + ); + let login = [ + "/Applications/OpenCodex.app".to_owned(), + AUTOSTART_FLAG.to_owned(), + ]; + assert_eq!( + LaunchOrigin::from_args(login.into_iter()), + LaunchOrigin::Autostart + ); + } + + #[test] + fn a_manual_launch_always_shows_the_window() { + assert!(shows_window( + LaunchOrigin::User, + TrayAvailability::Available + )); + assert!(shows_window( + LaunchOrigin::User, + TrayAvailability::Unavailable + )); + } + + #[test] + fn only_a_hidden_login_launch_defers_the_full_dashboard() { + assert!(loads_dashboard_on_ready(LaunchOrigin::User, false, false)); + assert!(loads_dashboard_on_ready(LaunchOrigin::User, true, false)); + assert!(loads_dashboard_on_ready( + LaunchOrigin::Autostart, + true, + false + )); + assert!(!loads_dashboard_on_ready( + LaunchOrigin::Autostart, + false, + false + )); + // An open that arrived during startup counts even if the queued show has not landed yet. + assert!(loads_dashboard_on_ready( + LaunchOrigin::Autostart, + false, + true + )); + } + + #[test] + fn a_recovery_leaves_the_update_page_where_it_is() { + // A failed install brings the runtime back while the page shows why the install failed. + assert!(keeps_update_page(Mode::Recover, true)); + // Anywhere else a recovery reloads the dashboard, and a launch always moves on. + assert!(!keeps_update_page(Mode::Recover, false)); + assert!(!keeps_update_page(Mode::Launch, true)); + assert!(!keeps_update_page(Mode::Launch, false)); + } + + #[test] + fn a_run_waits_on_a_child_still_starting_and_not_on_one_that_never_will() { + // The app tracks no child: nothing to wait on. + assert!(!waits_on_child(None)); + // A child spawned moments ago, or one still inside the slowest good start, is waited on. + assert!(waits_on_child(Some(Duration::ZERO))); + assert!(waits_on_child(Some(Duration::from_secs(65)))); + // Past the grace it is wedged, or its exit event is held up: a start goes ahead. + assert!(!waits_on_child(Some(CHILD_START_GRACE))); + } + + #[test] + fn explicit_dashboard_navigation_is_consumed_once_per_run() { + let startup = Startup::new(); + let mut navigations = Vec::new(); + assert!(navigate_once( + &startup, + "http://127.0.0.1:10100/#/usage", + |url| { + navigations.push(url.to_string()); + true + } + )); + assert!(!navigate_once( + &startup, + "http://127.0.0.1:10100/#/usage", + |url| { + navigations.push(url.to_string()); + true + } + )); + assert_eq!( + navigations, + vec!["http://127.0.0.1:10100/#/usage".to_string()] + ); + + startup.restart(); + assert!(navigate_once( + &startup, + "http://127.0.0.1:10101/#/usage", + |_| true + )); + assert!(!navigate_once( + &startup, + "http://127.0.0.1:10101/#/usage", + |_| true + )); + } + + #[test] + fn a_refused_dashboard_navigation_is_retried_on_the_next_open() { + let startup = Startup::new(); + assert!(!navigate_once( + &startup, + "http://127.0.0.1:10100/#/usage", + |_| false + )); + let mut attempts = 0; + assert!(navigate_once( + &startup, + "http://127.0.0.1:10100/#/usage", + |_| { + attempts += 1; + true + } + )); + assert_eq!(attempts, 1); + assert!(!navigate_once( + &startup, + "http://127.0.0.1:10100/#/usage", + |_| true + )); + } + + #[test] + fn an_open_during_startup_is_remembered_until_the_run_restarts() { + let startup = Startup::new(); + assert!(!startup.dashboard_requested()); + assert_eq!(startup.ready_dashboard(), None); + startup.request_dashboard(); + assert!(startup.dashboard_requested()); + startup.restart(); + assert!(!startup.dashboard_requested()); + } + + #[test] + fn update_page_return_requires_a_ready_dashboard_and_retries_refused_navigation() { + assert_eq!( + return_ready_dashboard(None, |_| true).unwrap_err(), + "dashboard is not ready" + ); + assert_eq!( + return_ready_dashboard(Some("http://127.0.0.1:10100/#/usage"), |_| false).unwrap_err(), + "dashboard could not be opened" + ); + let mut visited = None; + assert!( + return_ready_dashboard(Some("http://127.0.0.1:10100/#/usage"), |url| { + visited = Some(url.to_owned()); + true + }) + .is_ok() + ); + assert_eq!(visited.as_deref(), Some("http://127.0.0.1:10100/#/usage")); + } + + #[test] + fn a_login_launch_hides_only_where_there_is_a_tray_to_hide_in() { + assert!(!shows_window( + LaunchOrigin::Autostart, + TrayAvailability::Available + )); + assert!(shows_window( + LaunchOrigin::Autostart, + TrayAvailability::Unavailable + )); + } + + #[test] + fn registration_runs_before_the_runtime_is_touched() { + let order: Vec<&str> = PHASES.iter().map(|phase| phase.id()).collect(); + let registering = order.iter().position(|id| *id == "registering").unwrap(); + for later in ["resolving", "probing", "starting", "waiting"] { + assert!(registering < order.iter().position(|id| *id == later).unwrap()); + } + } + + #[test] + fn every_phase_has_a_distinct_identifier_and_a_label() { + let mut ids: Vec<&str> = PHASES.iter().map(|phase| phase.id()).collect(); + ids.sort_unstable(); + ids.dedup(); + assert_eq!(ids.len(), PHASES.len()); + assert!(PHASES.iter().all(|phase| !phase.label().is_empty())); + assert_eq!(PHASES.iter().filter(|phase| phase.is_terminal()).count(), 2); + assert!(PHASES.contains(&Phase::Ready)); + } + + #[test] + fn the_whole_sequence_is_bounded_well_under_the_minute_it_used_to_take() { + let budgets = [DEADLINE, POLL]; + assert!(budgets + .iter() + .all(|budget| *budget <= Duration::from_secs(45))); + assert!(POLL < DEADLINE); + } + + #[test] + fn an_expired_run_publishes_failed_in_the_same_step() { + // The expiry decision and the terminal publish share one critical section: an expired + // deadline with no prompt up fails the run, and the failure is already there when the + // call returns. + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + startup.set_deadline(tokio::time::Instant::now() - Duration::from_secs(60)); + match startup.expire_run( + tokio::time::Instant::now() - Duration::from_secs(60), + 1, + "expired".to_owned(), + ) { + Expiry::Fired(progress) => { + assert_eq!(progress.phase, Phase::Failed.id()); + assert!(progress.can_retry); + } + _ => panic!("an expired deadline with no consent must fire"), + } + assert!(startup.settled()); + // A second expiry for the same run says nothing. + assert!(matches!( + startup.expire_run(tokio::time::Instant::now(), 1, "again".to_owned()), + Expiry::Dead + )); + } + + #[test] + fn a_pending_prompt_blocks_expiry_and_the_prompt_still_resolves() { + // The losing side of the race the guard used to win: the prompt is up while the + // deadline sits in the past. Expiry must yield, and the user's answer must still + // reach the waiting run. + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + startup.set_deadline(tokio::time::Instant::now() - Duration::from_secs(60)); + let mut receiver = startup.await_consent().expect("no terminal state yet"); + assert!(matches!( + startup.expire_run(tokio::time::Instant::now(), 1, "expired".to_owned()), + Expiry::Blocked + )); + startup.decide_takeover(true); + assert_eq!(receiver.try_recv(), Ok(true)); + // The answer was consumed but the extension has not landed yet: still not expirable. + assert!(matches!( + startup.expire_run(tokio::time::Instant::now(), 1, "expired".to_owned()), + Expiry::Blocked + )); + // Once the run publishes the moved ceiling the guard waits on it instead of firing. + startup.resolve_consent(tokio::time::Instant::now() + Duration::from_secs(60)); + assert!(matches!( + startup.expire_run(tokio::time::Instant::now(), 1, "expired".to_owned()), + Expiry::Waiting(_) + )); + } + + #[test] + fn a_terminal_run_posts_no_prompt() { + // The other half of the race: the failure already landed, so the ask path must not + // register a prompt that would wait on a decision nobody can see. + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + let mut terminal = Progress::new(Phase::Failed, 1); + startup.publish(&mut terminal, None); + assert!(startup.await_consent().is_none()); + // And the consent state stays idle, so a later run is not shadowed by a stale prompt. + assert!(matches!(startup.live().consent, ConsentState::Idle)); + } + + #[test] + fn a_terminal_state_is_not_moved_by_a_late_report() { + // The expiry lands while the run is still inside a probe; the probe then resumes and + // reports. Neither the snapshot nor the consent gate may move: the terminal state is + // the page's promise that it stopped changing, and a report that could undo it would + // also reopen the prompt the terminal state just ruled out. + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + startup.set_deadline(tokio::time::Instant::now() - Duration::from_secs(60)); + match startup.expire_run( + tokio::time::Instant::now() - Duration::from_secs(60), + 1, + "expired".to_owned(), + ) { + Expiry::Fired(progress) => assert_eq!(progress.phase, Phase::Failed.id()), + _ => panic!("an expired deadline with no consent must fire"), + } + let mut late = Progress::new(Phase::Probing, 2); + assert!(!startup.publish(&mut late, None)); + assert_eq!(startup.latest().phase, Phase::Failed.id()); + assert!(startup.await_consent().is_none()); + // A second terminal report is refused as well: the first ending stands. + let mut ready = Progress::new(Phase::Ready, 3); + assert!(!startup.publish(&mut ready, None)); + assert_eq!(startup.latest().phase, Phase::Failed.id()); + } + + #[test] + fn expiry_waits_for_an_accepted_report_to_be_dispatched() { + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + startup.set_deadline(Instant::now() - Duration::from_secs(60)); + let events = std::sync::Mutex::new(Vec::new()); + let (checked_tx, checked_rx) = std::sync::mpsc::channel(); + std::thread::scope(|scope| { + startup.with_reporting(|| { + let mut progress = Progress::new(Phase::Probing, 1); + assert!(startup.publish(&mut progress, None)); + let startup = &startup; + let events = &events; + scope.spawn(move || { + // The report was accepted but has not dispatched yet. Expiry cannot + // overtake it, even though the state mutex itself is no longer held. + assert!(matches!( + startup.reporting.try_lock(), + Err(std::sync::TryLockError::WouldBlock) + )); + checked_tx.send(()).unwrap(); + startup.with_reporting(|| { + match startup.expire_run(Instant::now(), 1, "expired".to_owned()) { + Expiry::Fired(progress) => events.lock().unwrap().push(progress.phase), + _ => panic!("the unblocked expiry must publish failure"), + } + }); + }); + checked_rx.recv().unwrap(); + events.lock().unwrap().push(progress.phase); + }); + }); + assert_eq!( + *events.lock().unwrap(), + vec![Phase::Probing.id(), Phase::Failed.id()] + ); + } + + #[test] + fn a_superseded_guard_reports_nothing() { + // A retry bumped the generation: the old guard's expiry is dead even with the + // deadline in the past. + let startup = Startup::new(); + startup.generation.store(2, Ordering::SeqCst); + startup.set_deadline(tokio::time::Instant::now() - Duration::from_secs(60)); + assert!(matches!( + startup.expire_run(tokio::time::Instant::now(), 1, "expired".to_owned()), + Expiry::Dead + )); + } +} diff --git a/desktop/src-tauri/src/supervisor.rs b/desktop/src-tauri/src/supervisor.rs new file mode 100644 index 0000000000..ffccc60b52 --- /dev/null +++ b/desktop/src-tauri/src/supervisor.rs @@ -0,0 +1,896 @@ +//! Keeping the runtime alive: what the app does when the runtime it runs goes away unasked. +//! +//! The startup sequence used to be the only thing that ever started a runtime, and it ran at +//! launch and on the failure page's retry. A runtime that exited afterwards — a crash, a terminal +//! `ocx stop`, or a restart the runtime performed itself by handing the port to a detached +//! grandchild this app could not see — was recorded in memory and nothing more. The port kept +//! refusing connections until somebody quit and reopened the app. +//! +//! Two signals start a recovery here, and both only reach the same startup sequence in +//! [`Mode::Recover`], so every gate it has still holds: only a proven absence starts a runtime, a +//! runtime that answers is attached as a guest, and nothing is ever killed or signalled. +//! +//! - The exit of the child this app started ([`on_exit`]). Exit code +//! [`REQUESTED_RESTART_EXIT_CODE`] is the runtime asking for exactly this: under the marker +//! `sidecar.rs` sets, a restart (a join into a Child, a memory restart, a disconnect) exits with +//! it instead of spawning past the app, and the replacement starts almost at once. Any other exit +//! waits a capped backoff first, so a replacement or a service wrapper that already owns the +//! port binds before the app looks. +//! - The watchdog ([`watchdog_tick`]), every few seconds while a run is Ready: a different process +//! answering the endpoint, or several refused connections in a row. It covers a runtime this app +//! is only a guest on and an exit event that never arrived. Timeouts and unauthorized or +//! unreadable answers never count; they say something is there. +//! +//! A recovery that finds the port held by a listener this app cannot use (one bound off loopback) +//! is not retried: another attempt would find the same listener, and each one costs a resolve. +//! The supervisor parks instead, and the same watchdog tick asks the endpoint only whether +//! something changed there ([`Parked`]). +//! +//! The tray's Stop, Quit and an update's drain clear the app's wish for a runtime before the +//! runtime's exit can arrive (`exit.rs`), so none of them is ever undone. Every decision is +//! appended to `runtime-supervisor.log` in the app's log directory, bounded, because the exit +//! code otherwise dies with the app. + +use crate::{ + exit::{ExitCoordinator, ExitPhase}, + sidecar::SidecarExit, + startup::{self, Mode, Startup}, + AppState, +}; +use std::{ + fs::{self, OpenOptions}, + io::Write, + path::{Path, PathBuf}, + sync::{Mutex, MutexGuard, PoisonError}, + time::{SystemTime, UNIX_EPOCH}, +}; +use tauri::{AppHandle, Manager}; +use tokio::time::{sleep, Duration, Instant}; + +/// The exit code a runtime ends with to hand its restart to this app +/// (`DESKTOP_RESTART_EXIT_CODE` in `src/lib/system-restart-contract.ts`; EX_TEMPFAIL). +pub const REQUESTED_RESTART_EXIT_CODE: i32 = 75; +/// A requested restart has already released the port, so its replacement starts almost at once. +pub const REQUESTED_RESTART_DELAY: Duration = Duration::from_millis(500); +/// The wait before bringing back a runtime that ended unasked, by how many recoveries ran recently. +/// The first step leaves a replacement or a service wrapper that owns the port time to bind first. +pub const BACKOFF: [Duration; 5] = [ + Duration::from_secs(3), + Duration::from_secs(6), + Duration::from_secs(12), + Duration::from_secs(24), + Duration::from_secs(30), +]; +/// How long a runtime has to stay healthy before the backoff starts over. +pub const HEALTHY_RESET: Duration = Duration::from_secs(120); +/// How often the watchdog asks the endpoint who it is, and only while a run is Ready. +pub const WATCHDOG_INTERVAL: Duration = Duration::from_secs(5); +/// Refused connections in a row before the watchdog treats the runtime this app started as gone. +pub const UNREACHABLE_LIMIT: u32 = 3; +/// The same for a runtime this app is only a guest on. Its own manager (a service, a terminal, an +/// update in progress) gets about a minute to bring it back before the app starts one of its own. +pub const GUEST_UNREACHABLE_LIMIT: u32 = 12; +// A runtime somebody else manages gets its manager's grace; one this app started does not. +const _: () = assert!(UNREACHABLE_LIMIT < GUEST_UNREACHABLE_LIMIT); +/// The supervisor log's cap: at this size it is emptied before the next line. +pub const LOG_CAP_BYTES: u64 = 256 * 1024; +pub const LOG_FILE: &str = "runtime-supervisor.log"; + +/// What an exit is judged on. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Input { + pub phase: ExitPhase, + pub wanted: bool, + pub reason_set: bool, + pub startup_running: bool, + /// The exit belongs to a child this app no longer tracks. + pub stale_pid: bool, + pub requested_restart: bool, + /// Recoveries scheduled since the runtime was last healthy for [`HEALTHY_RESET`]. + pub attempts: u32, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Verdict { + /// Leave it: the reason says why. + Ignore(&'static str), + /// A startup run is in flight. It reports how it ended, and a failure is followed up then. + Defer, + /// Bring a runtime back after this long. + RespawnAfter(Duration), +} + +/// Decide what an exit means. Only an exit of the tracked child, with nothing in flight, nobody +/// having asked for the runtime to stop and the app not ending, brings a runtime back. +pub fn decide(input: Input) -> Verdict { + if input.stale_pid { + return Verdict::Ignore("the exit belongs to a runtime this app no longer tracks"); + } + if input.phase != ExitPhase::Idle { + return Verdict::Ignore("the app is starting, stopping, draining or updating its runtime"); + } + if input.reason_set { + return Verdict::Ignore("the app is quitting or restarting"); + } + if !input.wanted { + return Verdict::Ignore("the runtime was stopped from the tray, by Quit or for an update"); + } + if input.startup_running { + return Verdict::Defer; + } + Verdict::RespawnAfter(respawn_delay(input.requested_restart, input.attempts)) +} + +/// A requested restart right after a healthy stretch goes almost at once; everything else, and a +/// requested restart that keeps recurring, waits the capped backoff. +pub fn respawn_delay(requested_restart: bool, attempts: u32) -> Duration { + if requested_restart && attempts == 0 { + return REQUESTED_RESTART_DELAY; + } + let step = usize::try_from(attempts).unwrap_or(usize::MAX); + BACKOFF[step.min(BACKOFF.len() - 1)] +} + +/// The backoff starts over once the runtime has been healthy for [`HEALTHY_RESET`]. +pub fn attempts_after(attempts: u32, healthy_for: Duration) -> u32 { + if healthy_for >= HEALTHY_RESET { + 0 + } else { + attempts + } +} + +/// What one watchdog question established. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Probe { + /// The endpoint identified itself with this pid. + Identified(u32), + /// Nothing is listening: the connection was refused. + Unreachable, + /// A timeout, an unauthorized or unreadable answer, or something that is not this proxy. + /// Something may be there, so it proves nothing. + Inconclusive, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Watch { + /// The bound runtime answered. + Healthy, + /// Not proven gone; the refused connections in a row so far. + Counting(u32), + /// Bring the runtime back now. + Recover(&'static str), +} + +/// Judge one watchdog answer against the runtime the app is bound to; `owned` says whether this +/// app started it. +pub fn classify(probe: Probe, bound_pid: u32, streak: u32, owned: bool) -> Watch { + let limit = if owned { + UNREACHABLE_LIMIT + } else { + GUEST_UNREACHABLE_LIMIT + }; + match probe { + Probe::Identified(pid) if pid == bound_pid => Watch::Healthy, + Probe::Identified(_) => Watch::Recover("a different process answers the endpoint"), + Probe::Unreachable => { + let streak = streak.saturating_add(1); + if streak >= limit { + Watch::Recover("the endpoint refused every connection the watchdog made") + } else { + Watch::Counting(streak) + } + } + Probe::Inconclusive => Watch::Counting(0), + } +} + +/// A listener this app cannot use held the port when a run looked; its pid, when the resolve named +/// one. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Held { + pub pid: Option, +} + +/// How a startup run ended, as far as supervision is concerned. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum RunOutcome { + Ready, + /// Anything another attempt may change: a resolve that failed, a start that did not answer. + Failed, + /// The port is held by a listener this app cannot use. Another attempt finds the same one. + Held(Held), +} + +/// What follows a run that did not reach Ready. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum FollowUp { + /// A launch somebody is looking at: their retry decides, as it always has. + Wait, + /// Another attempt after the backoff, if supervision still allows one. + Retry, + /// Watch the endpoint for a change instead of retrying ([`Parked`]). + Park, +} + +/// A failed recovery, or a failed run that swallowed an exit of this app's child, is followed up; +/// any other failed launch waits for the person. A port held by a listener this app cannot use is +/// never retried, because the retry would find the same listener again every time. +pub fn follow_up(mode: Mode, owed: bool, held: bool) -> FollowUp { + if mode != Mode::Recover && !owed { + FollowUp::Wait + } else if held { + FollowUp::Park + } else { + FollowUp::Retry + } +} + +/// A parked supervisor's view of the endpoint after a run found the port held by a listener this +/// app cannot use. Only a change there is worth another recovery: a different process answering, +/// or the listener that answered going silent. A listener that refused loopback from the start is +/// bound where loopback cannot see it, so its refusals are the steady state, not a change. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct Parked { + holder: Option, + /// The holder has answered on loopback at least once. + answered: bool, + streak: u32, +} + +impl Parked { + pub fn new(holder: Option) -> Self { + Self { + holder, + ..Self::default() + } + } + + /// Judge one watchdog answer. `Some(reason)` is a change worth another recovery. + pub fn observe(&mut self, probe: Probe) -> Option<&'static str> { + match probe { + Probe::Identified(pid) => { + self.streak = 0; + if self.holder.is_some_and(|holder| holder != pid) { + return Some("a different process answers the endpoint"); + } + self.holder = Some(pid); + self.answered = true; + None + } + Probe::Unreachable if self.answered => { + self.streak = self.streak.saturating_add(1); + // Not ours: its own manager gets the guest's grace to bring it back first. + (self.streak >= GUEST_UNREACHABLE_LIMIT) + .then_some("the listener that held the port stopped answering") + } + Probe::Unreachable => None, + Probe::Inconclusive => { + self.streak = 0; + None + } + } + } +} + +#[derive(Default)] +struct State { + attempts: u32, + /// A recovery is scheduled and has not started yet. One at a time. + pending: bool, + /// An exit arrived while a run was in flight; a failure of that run is followed up. + owed: bool, + streak: u32, + healthy_since: Option, + /// The last run found the port held by a listener this app cannot use. + parked: Option, +} + +/// The supervisor's managed state. +pub struct Supervisor { + state: Mutex, + log: Option, +} + +impl Supervisor { + fn new(log: Option) -> Self { + Self { + state: Mutex::new(State::default()), + log, + } + } + + fn state(&self) -> MutexGuard<'_, State> { + self.state.lock().unwrap_or_else(PoisonError::into_inner) + } + + fn note(&self, message: &str) { + let Some(path) = &self.log else { + return; + }; + if let Err(error) = append_bounded(path, &log_line(SystemTime::now(), message)) { + crate::logging::log_once( + "the runtime supervisor log could not be written", + &error.to_string(), + ); + } + } +} + +/// Register the exit hook and start the watchdog. Called once from setup; it starts no runtime. +pub fn watch_runtime(app: &AppHandle) { + let log = app.path().app_log_dir().ok().map(|dir| dir.join(LOG_FILE)); + app.manage(Supervisor::new(log)); + if let Some(state) = app.try_state::() { + let handle = app.clone(); + state + .watch + .on_exit(move |pid, exit| on_exit(&handle, pid, exit)); + } + let handle = app.clone(); + tauri::async_runtime::spawn(async move { + loop { + sleep(WATCHDOG_INTERVAL).await; + watchdog_tick(&handle).await; + } + }); +} + +/// A person asked for a runtime again: supervision resumes and the backoff starts over. +pub fn resume(app: &AppHandle) { + if let Some(coordinator) = app.try_state::() { + coordinator.resume(); + } + if let Some(supervisor) = app.try_state::() { + let mut state = supervisor.state(); + state.attempts = 0; + state.parked = None; + } +} + +fn input(app: &AppHandle, stale_pid: bool, requested_restart: bool, attempts: u32) -> Input { + let supervision = app + .try_state::() + .map(|coordinator| coordinator.supervision()); + Input { + phase: supervision.map_or(ExitPhase::Draining, |s| s.phase), + wanted: supervision.is_some_and(|s| s.wanted), + reason_set: supervision.is_none_or(|s| s.reason_set), + startup_running: app + .try_state::() + .is_some_and(|startup| startup.is_running()), + stale_pid, + requested_restart, + attempts, + } +} + +/// The child this app started has exited. +fn on_exit(app: &AppHandle, pid: u32, exit: SidecarExit) { + let Some(supervisor) = app.try_state::() else { + return; + }; + let stale = app + .try_state::() + .is_none_or(|state| state.child_pid() != Some(pid)); + let requested = exit.code == Some(REQUESTED_RESTART_EXIT_CODE); + let attempts = supervisor.state().attempts; + let verdict = decide(input(app, stale, requested, attempts)); + supervisor.note(&format!( + "runtime pid {pid} ended ({}); {}", + exit.describe(), + describe(verdict, attempts) + )); + match verdict { + Verdict::Ignore(_) => {} + Verdict::Defer => supervisor.state().owed = true, + Verdict::RespawnAfter(delay) => { + // The child is gone: let go of its handle without signalling anything, so nothing + // later reads it as a runtime this app still owns. + if let Some(state) = app.try_state::() { + state.release(); + } + crate::tray::set_owned(app, false); + schedule(app, delay); + } + } +} + +fn describe(verdict: Verdict, attempts: u32) -> String { + match verdict { + Verdict::Ignore(reason) => format!("left alone: {reason}"), + Verdict::Defer => "a startup run is in flight; its outcome decides".to_owned(), + Verdict::RespawnAfter(delay) => format!( + "bringing a runtime back in {} ms (recovery {})", + delay.as_millis(), + attempts.saturating_add(1) + ), + } +} + +/// Start one recovery after `delay`, unless one is already scheduled. +fn schedule(app: &AppHandle, delay: Duration) { + let Some(supervisor) = app.try_state::() else { + return; + }; + { + let mut state = supervisor.state(); + if state.pending { + return; + } + state.pending = true; + state.attempts = state.attempts.saturating_add(1); + state.streak = 0; + state.healthy_since = None; + state.parked = None; + } + let app = app.clone(); + tauri::async_runtime::spawn(async move { + sleep(delay).await; + let Some(supervisor) = app.try_state::() else { + return; + }; + let attempts = { + let mut state = supervisor.state(); + state.pending = false; + state.attempts + }; + // The world may have moved during the wait: a Stop, a Quit or an update since then wins. + match decide(input(&app, false, false, attempts)) { + Verdict::RespawnAfter(_) => { + if !startup::begin_with(&app, Mode::Recover) { + supervisor.state().owed = true; + } + } + Verdict::Defer => supervisor.state().owed = true, + Verdict::Ignore(reason) => supervisor.note(&format!("recovery skipped: {reason}")), + } + }); +} + +/// A startup run ended. A Ready run starts the healthy clock. A failed run is followed up as +/// [`follow_up`] says: another attempt after the backoff, a parked watch when a listener this app +/// cannot use holds the port, or nothing until the person retries. `detail` is the run's own last +/// word, for the log. +pub fn run_finished(app: &AppHandle, mode: Mode, outcome: RunOutcome, detail: Option<&str>) { + let Some(supervisor) = app.try_state::() else { + return; + }; + let ready = outcome == RunOutcome::Ready; + let (owed, attempts) = { + let mut state = supervisor.state(); + let owed = std::mem::take(&mut state.owed); + state.parked = None; + if ready { + state.streak = 0; + state.healthy_since = Some(Instant::now()); + } + (owed, state.attempts) + }; + if ready { + if mode == Mode::Recover { + supervisor.note("recovery run ready"); + } + return; + } + let held = match outcome { + RunOutcome::Held(held) => Some(held), + RunOutcome::Ready | RunOutcome::Failed => None, + }; + let reason = detail.unwrap_or("no reason reported"); + match follow_up(mode, owed, held.is_some()) { + FollowUp::Wait => {} + FollowUp::Park => { + let holder = held.and_then(|held| held.pid); + supervisor.state().parked = Some(Parked::new(holder)); + supervisor.note(&format!( + "startup run failed ({reason}); the port is held by a listener this app cannot use, so no attempt is scheduled until the endpoint changes" + )); + } + FollowUp::Retry => { + let verdict = decide(input(app, false, false, attempts)); + supervisor.note(&format!( + "startup run failed ({reason}); {}", + describe(verdict, attempts) + )); + if let Verdict::RespawnAfter(delay) = verdict { + schedule(app, delay); + } + } + } +} + +/// The watchdog's one question: who answers the endpoint, if anything. +async fn probe(proxy: &crate::proxy::ProxyClient) -> Probe { + match proxy.identify().await { + Ok(identity) => Probe::Identified(identity.pid), + Err(error) if error.is_unreachable() => Probe::Unreachable, + Err(_) => Probe::Inconclusive, + } +} + +/// One parked watchdog step: a recovery only once the endpoint has changed. +async fn parked_tick(app: &AppHandle, supervisor: &Supervisor) { + let Some(proxy) = app.try_state::().and_then(|state| state.proxy()) else { + return; + }; + let answer = probe(&proxy).await; + let changed = { + let mut state = supervisor.state(); + let Some(parked) = state.parked.as_mut() else { + return; + }; + parked.observe(answer) + }; + let Some(reason) = changed else { + return; + }; + // Re-read after the await: a Stop, a Quit or a run that began meanwhile wins. + let allowed = app + .try_state::() + .is_some_and(|coordinator| coordinator.supervision_allowed()); + let idle = app + .try_state::() + .is_some_and(|startup| !startup.is_running()); + if !allowed || !idle { + return; + } + supervisor.note(&format!( + "watchdog: {reason} on the port that was held; bringing a runtime back" + )); + schedule(app, Duration::ZERO); +} + +/// One watchdog step. It asks only while supervision is allowed and a run is Ready or the +/// supervisor is parked, and it asks the unauthenticated health endpoint, so nothing secret is sent. +async fn watchdog_tick(app: &AppHandle) { + let Some(supervisor) = app.try_state::() else { + return; + }; + let allowed = app + .try_state::() + .is_some_and(|coordinator| coordinator.supervision_allowed()); + let (idle, ready) = app + .try_state::() + .map_or((false, false), |startup| { + (!startup.is_running(), startup.is_ready()) + }); + let parked = { + let mut state = supervisor.state(); + let parked = state.parked.is_some(); + if !allowed || !idle || state.pending || !(ready || parked) { + state.streak = 0; + return; + } + parked && !ready + }; + if parked { + parked_tick(app, &supervisor).await; + return; + } + let Some((proxy, owned)) = app + .try_state::() + .and_then(|state| state.proxy().map(|proxy| (proxy, state.owns_runtime()))) + else { + return; + }; + let Some(binding) = proxy.binding() else { + return; + }; + let answer = probe(&proxy).await; + let verdict = { + let mut state = supervisor.state(); + let verdict = classify(answer, binding.identity.pid, state.streak, owned); + match verdict { + Watch::Healthy => { + state.streak = 0; + let since = *state.healthy_since.get_or_insert_with(Instant::now); + state.attempts = attempts_after(state.attempts, since.elapsed()); + } + Watch::Counting(streak) => state.streak = streak, + Watch::Recover(_) => state.streak = 0, + } + verdict + }; + let Watch::Recover(reason) = verdict else { + return; + }; + // Re-read after the await: a Stop, a Quit or a run that began meanwhile wins. + if !app + .try_state::() + .is_some_and(|coordinator| coordinator.supervision_allowed()) + { + return; + } + supervisor.note(&format!( + "watchdog: {reason} (bound pid {}); bringing a runtime back", + binding.identity.pid + )); + crate::tray::set_owned(app, false); + schedule(app, Duration::ZERO); +} + +/// `2026-09-26T14:33:10Z message\n`, in UTC without a date library. +pub fn log_line(at: SystemTime, message: &str) -> String { + let seconds = at + .duration_since(UNIX_EPOCH) + .map_or(0, |elapsed| elapsed.as_secs()); + let days = i64::try_from(seconds / 86_400).unwrap_or(0); + let rest = seconds % 86_400; + let (year, month, day) = civil_from_days(days); + format!( + "{year:04}-{month:02}-{day:02}T{:02}:{:02}:{:02}Z {message}\n", + rest / 3_600, + rest % 3_600 / 60, + rest % 60 + ) +} + +/// Days since 1970-01-01 to a proleptic Gregorian date (Howard Hinnant's `civil_from_days`). +fn civil_from_days(days: i64) -> (i64, u32, u32) { + let z = days + 719_468; + let era = z.div_euclid(146_097); + let doe = z.rem_euclid(146_097); + let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365; + let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); + let mp = (5 * doy + 2) / 153; + let day = u32::try_from(doy - (153 * mp + 2) / 5 + 1).unwrap_or(1); + let month = u32::try_from(if mp < 10 { mp + 3 } else { mp - 9 }).unwrap_or(1); + let year = yoe + era * 400 + i64::from(month <= 2); + (year, month, day) +} + +/// Append one line, emptying the file first once it has reached [`LOG_CAP_BYTES`]. A symlink in +/// its place is refused rather than followed. +pub fn append_bounded(path: &Path, line: &str) -> std::io::Result<()> { + if let Some(parent) = path.parent() { + fs::create_dir_all(parent)?; + } + let existing = match fs::symlink_metadata(path) { + Ok(metadata) if metadata.file_type().is_symlink() => { + return Err(std::io::Error::other("the log path is a symlink")); + } + Ok(metadata) => metadata.len(), + Err(error) if error.kind() == std::io::ErrorKind::NotFound => 0, + Err(error) => return Err(error), + }; + let mut options = OpenOptions::new(); + if existing >= LOG_CAP_BYTES { + options.write(true).truncate(true); + } else { + options.append(true).create(true); + } + let mut file = options.open(path)?; + if existing >= LOG_CAP_BYTES { + file.write_all( + log_line( + SystemTime::now(), + &format!("log emptied at the {} KiB cap", LOG_CAP_BYTES / 1024), + ) + .as_bytes(), + )?; + } + file.write_all(line.as_bytes()) +} + +#[cfg(test)] +mod tests { + use super::{ + append_bounded, attempts_after, classify, decide, follow_up, log_line, respawn_delay, + FollowUp, Input, Parked, Probe, Verdict, Watch, BACKOFF, GUEST_UNREACHABLE_LIMIT, + HEALTHY_RESET, LOG_CAP_BYTES, REQUESTED_RESTART_DELAY, REQUESTED_RESTART_EXIT_CODE, + UNREACHABLE_LIMIT, + }; + use crate::exit::ExitPhase; + use crate::startup::Mode; + use std::time::{Duration, UNIX_EPOCH}; + + const PHASES: [ExitPhase; 7] = [ + ExitPhase::Idle, + ExitPhase::Spawning, + ExitPhase::Stopping, + ExitPhase::Draining, + ExitPhase::Drained, + ExitPhase::DrainFailed, + ExitPhase::OwnershipUnknown, + ]; + + fn open() -> Input { + Input { + phase: ExitPhase::Idle, + wanted: true, + reason_set: false, + startup_running: false, + stale_pid: false, + requested_restart: false, + attempts: 0, + } + } + + #[test] + fn only_an_idle_wanted_unclaimed_settled_exit_of_the_tracked_child_brings_a_runtime_back() { + for phase in PHASES { + for wanted in [false, true] { + for reason_set in [false, true] { + for startup_running in [false, true] { + for stale_pid in [false, true] { + let input = Input { + phase, + wanted, + reason_set, + startup_running, + stale_pid, + ..open() + }; + let gates_open = + phase == ExitPhase::Idle && wanted && !reason_set && !stale_pid; + let expected = match (gates_open, startup_running) { + (true, false) => "respawn", + (true, true) => "defer", + (false, _) => "ignore", + }; + let verdict = match decide(input) { + Verdict::RespawnAfter(_) => "respawn", + Verdict::Defer => "defer", + Verdict::Ignore(_) => "ignore", + }; + assert_eq!(verdict, expected, "{input:?}"); + } + } + } + } + } + } + + #[test] + fn a_requested_restart_goes_at_once_and_an_unasked_exit_backs_off() { + assert_eq!(REQUESTED_RESTART_EXIT_CODE, 75); + assert_eq!(respawn_delay(true, 0), REQUESTED_RESTART_DELAY); + assert!(REQUESTED_RESTART_DELAY < Duration::from_secs(1)); + // The first unasked step leaves a replacement or a service wrapper time to bind first. + assert_eq!(respawn_delay(false, 0), Duration::from_secs(3)); + // A requested restart that keeps recurring is not exempt from the backoff. + assert_eq!(respawn_delay(true, 1), BACKOFF[1]); + let mut previous = Duration::ZERO; + for attempts in 0..20 { + let delay = respawn_delay(false, attempts); + assert!(delay >= previous, "the backoff never shrinks"); + assert!(delay <= Duration::from_secs(30), "the backoff is capped"); + previous = delay; + } + assert_eq!(respawn_delay(false, u32::MAX), Duration::from_secs(30)); + assert_eq!( + decide(Input { + requested_restart: true, + ..open() + }), + Verdict::RespawnAfter(REQUESTED_RESTART_DELAY) + ); + } + + #[test] + fn the_backoff_starts_over_only_after_a_healthy_stretch() { + assert_eq!(attempts_after(4, HEALTHY_RESET - Duration::from_secs(1)), 4); + assert_eq!(attempts_after(4, HEALTHY_RESET), 0); + } + + #[test] + fn only_refused_connections_count_and_a_different_pid_recovers_at_once() { + for (owned, limit) in [(true, UNREACHABLE_LIMIT), (false, GUEST_UNREACHABLE_LIMIT)] { + assert_eq!( + classify(Probe::Identified(42), 42, 2, owned), + Watch::Healthy + ); + assert!(matches!( + classify(Probe::Identified(43), 42, 0, owned), + Watch::Recover(_) + )); + let mut streak = 0; + for _ in 1..limit { + match classify(Probe::Unreachable, 42, streak, owned) { + Watch::Counting(next) => streak = next, + other => panic!("recovered too early: {other:?}"), + } + } + // A timeout or an unreadable answer says something may be there: it breaks the run. + assert_eq!( + classify(Probe::Inconclusive, 42, streak, owned), + Watch::Counting(0) + ); + assert!(matches!( + classify(Probe::Unreachable, 42, streak, owned), + Watch::Recover(_) + )); + } + } + + #[test] + fn a_failed_recovery_on_a_held_port_parks_instead_of_scheduling_another() { + // A recovery that found a listener this app cannot use would find it again every time, + // and each attempt costs a resolve: it is never rescheduled. + assert_eq!(follow_up(Mode::Recover, false, true), FollowUp::Park); + assert_eq!(follow_up(Mode::Launch, true, true), FollowUp::Park); + // Anything else another attempt may change keeps its backoff. + assert_eq!(follow_up(Mode::Recover, false, false), FollowUp::Retry); + assert_eq!(follow_up(Mode::Recover, true, false), FollowUp::Retry); + assert_eq!(follow_up(Mode::Launch, true, false), FollowUp::Retry); + // A launch somebody is looking at still waits for their retry. + assert_eq!(follow_up(Mode::Launch, false, false), FollowUp::Wait); + assert_eq!(follow_up(Mode::Launch, false, true), FollowUp::Wait); + } + + #[test] + fn a_parked_supervisor_recovers_only_on_a_change_at_the_endpoint() { + // Bound off loopback: refused from the start, and forever. That is the steady state. + let mut hidden = Parked::new(Some(42)); + for _ in 0..(GUEST_UNREACHABLE_LIMIT * 4) { + assert_eq!(hidden.observe(Probe::Unreachable), None); + } + assert_eq!(hidden.observe(Probe::Inconclusive), None); + // Something this app can use now answers on loopback. + assert!(hidden.observe(Probe::Identified(43)).is_some()); + + // A holder loopback can see: the same pid is no change, a different one is. + let mut seen = Parked::new(Some(42)); + assert_eq!(seen.observe(Probe::Identified(42)), None); + assert!(seen.observe(Probe::Identified(7)).is_some()); + + // It going silent is a change too, after the grace its own manager gets. + let mut silent = Parked::new(None); + assert_eq!(silent.observe(Probe::Identified(42)), None); + for _ in 1..GUEST_UNREACHABLE_LIMIT { + assert_eq!(silent.observe(Probe::Unreachable), None); + } + // A timeout breaks the run of refusals: something may be there. + assert_eq!(silent.observe(Probe::Inconclusive), None); + for _ in 1..GUEST_UNREACHABLE_LIMIT { + assert_eq!(silent.observe(Probe::Unreachable), None); + } + assert!(silent.observe(Probe::Unreachable).is_some()); + } + + #[test] + fn a_log_line_is_utc_iso_and_carries_the_message() { + assert_eq!( + log_line(UNIX_EPOCH, "start"), + "1970-01-01T00:00:00Z start\n" + ); + assert_eq!( + log_line(UNIX_EPOCH + Duration::from_secs(951_782_400), "leap"), + "2000-02-29T00:00:00Z leap\n" + ); + assert_eq!( + log_line(UNIX_EPOCH + Duration::from_secs(1_700_000_000), "x"), + "2023-11-14T22:13:20Z x\n" + ); + } + + #[test] + fn the_log_is_bounded_and_a_symlink_is_not_followed() { + let dir = std::env::temp_dir().join(format!( + "ocx-supervisor-log-{}-{}", + std::process::id(), + uuid::Uuid::new_v4() + )); + let path = dir.join("nested").join("runtime-supervisor.log"); + append_bounded(&path, "first\n").unwrap(); + append_bounded(&path, "second\n").unwrap(); + assert_eq!(std::fs::read_to_string(&path).unwrap(), "first\nsecond\n"); + let cap = usize::try_from(LOG_CAP_BYTES).unwrap(); + std::fs::write(&path, vec![b'x'; cap]).unwrap(); + append_bounded(&path, "after\n").unwrap(); + let text = std::fs::read_to_string(&path).unwrap(); + assert!(text.len() < 1024, "the full log was emptied first"); + assert!(text.contains("log emptied at the 256 KiB cap")); + assert!(text.ends_with("after\n")); + #[cfg(unix)] + { + let target = dir.join("elsewhere"); + let link = dir.join("link.log"); + std::os::unix::fs::symlink(&target, &link).unwrap(); + assert!(append_bounded(&link, "nope\n").is_err()); + assert!(!target.exists()); + } + let _ = std::fs::remove_dir_all(&dir); + } +} diff --git a/desktop/src-tauri/src/tray.rs b/desktop/src-tauri/src/tray.rs index 56f2550645..479a5a9090 100644 --- a/desktop/src-tauri/src/tray.rs +++ b/desktop/src-tauri/src/tray.rs @@ -1,4 +1,9 @@ -use crate::{formatting, proxy::ProxyClient, updater, widget, window}; +use crate::{ + exit::{self, ExitReason}, + formatting, popup, + proxy::ProxyClient, + updater, widget, window, +}; use serde_json::Value; use std::sync::{ atomic::{AtomicBool, Ordering}, @@ -13,13 +18,16 @@ use tauri_plugin_autostart::ManagerExt; use tauri_plugin_opener::OpenerExt; pub struct TrayState { - pub menu: Mutex>, + pub menu: Mutex>, pub installing: AtomicBool, + pub update_pending: AtomicBool, } -pub struct UpdateMenu { +#[derive(Clone)] +pub struct TrayMenu { check_updates: MenuItem, install_update: MenuItem, + stop: MenuItem, } impl Default for TrayState { @@ -27,11 +35,41 @@ impl Default for TrayState { Self { menu: Mutex::new(None), installing: AtomicBool::new(false), + update_pending: AtomicBool::new(false), } } } -pub fn install(app: &AppHandle, proxy: ProxyClient) -> tauri::Result<()> { +#[cfg(any(not(target_os = "macos"), test))] +fn tray_icon_bytes(pending: bool) -> &'static [u8] { + if pending { + include_bytes!("../icons/tray/icon-update.png") + } else { + include_bytes!("../icons/tray/icon.png") + } +} + +fn apply_update_indicator(app: &AppHandle, pending: bool) { + #[cfg(target_os = "macos")] + popup::set_update_dot(app, pending); + #[cfg(not(target_os = "macos"))] + if let Some(tray) = app.tray_by_id("main") { + let image = + tauri::image::Image::from_bytes(tray_icon_bytes(pending)).expect("generated tray icon"); + let _ = tray.set_icon(Some(image)); + } +} + +fn update_pending(app: &AppHandle) -> bool { + app.try_state::() + .is_some_and(|state| state.update_pending.load(Ordering::Acquire)) +} + +/// Build the tray. +/// +/// The proxy is not passed in. The tray is installed before a runtime has been resolved, so every +/// use reads the current client from the app instead of holding one that might not exist yet. +pub fn install(app: &AppHandle) -> tauri::Result<()> { let open = MenuItem::with_id(app, "open-dashboard", "Open Dashboard", true, None::<&str>)?; let browser = MenuItem::with_id(app, "open-browser", "Open in Browser", true, None::<&str>)?; let login = CheckMenuItem::with_id( @@ -42,12 +80,12 @@ pub fn install(app: &AppHandle, proxy: ProxyClient) -> tauri::Result<()> { app.autolaunch().is_enabled().unwrap_or(false), None::<&str>, )?; - let spawned_by_us = app - .state::() - .spawned_by_us - .load(Ordering::Relaxed); - let stop = MenuItem::with_id(app, "stop-proxy", "Stop proxy", spawned_by_us, None::<&str>)?; - let stop_item = stop.clone(); + // The tray is built before the startup sequence has decided anything, so nothing owns a + // runtime yet. Ownership arrives later and reaches this item through [`set_owned`]. + let owned = app + .try_state::() + .is_some_and(|state| state.owns_runtime()); + let stop = MenuItem::with_id(app, "stop-proxy", "Stop proxy", owned, None::<&str>)?; let check_updates = MenuItem::with_id( app, "check-updates", @@ -57,10 +95,19 @@ pub fn install(app: &AppHandle, proxy: ProxyClient) -> tauri::Result<()> { )?; let install_update = MenuItem::with_id(app, "install-update", "Install update", false, None::<&str>)?; + // Every platform needs a menu path to the popup, not only Linux. + // + // On macOS the icon click cannot be the only way in: `tray-icon` assigns the menu to the + // NSStatusItem itself, so AppKit pops that menu on mouse-down before the crate's own click + // handler runs, and `show_menu_on_left_click(false)` cannot take it back. Linux tray hosts + // differ in whether a click reaches the application at all. That leaves Windows as the only + // platform where the icon alone would have worked. + let show_usage = MenuItem::with_id(app, "show-usage", "Show Usage", true, None::<&str>)?; let quit = MenuItem::with_id(app, "quit", "Quit", true, None::<&str>)?; let menu = Menu::with_items( app, &[ + &show_usage, &open, &browser, &PredefinedMenuItem::separator(app)?, @@ -74,36 +121,84 @@ pub fn install(app: &AppHandle, proxy: ProxyClient) -> tauri::Result<()> { ], )?; if let Ok(mut state) = app.state::().menu.lock() { - *state = Some(UpdateMenu { + *state = Some(TrayMenu { check_updates: check_updates.clone(), install_update: install_update.clone(), + stop: stop.clone(), }); } - let tray = TrayIconBuilder::with_id("main") + let builder = TrayIconBuilder::with_id("main") .icon(icon()) .icon_as_template(true) - .menu(&menu) + .menu(&menu); + // Attaching a menu makes the left click open that menu by default, which swallows the click + // before `on_tray_icon_event` can do anything visible. On macOS and Windows that left the + // usage popup with no way to open at all: the icon showed the menu, and the menu item that + // opens the popup is Linux-only. Left click is the popup, right click is the menu. + // + // Linux keeps the default. Its StatusNotifier hosts deliver no usable click event, so the + // menu is the entire interaction there and turning it off would remove the only way in. + #[cfg(not(target_os = "linux"))] + let builder = builder.show_menu_on_left_click(false); + let tray = builder .on_tray_icon_event(|tray, event| { if let TrayIconEvent::Click { button: MouseButton::Left, button_state: MouseButtonState::Up, + position, .. } = event { - if let Some(window) = tray.app_handle().get_webview_window("main") { - window::show(&window); + // The icon opens the usage popup rather than the dashboard. Reading the + // current numbers is the reason to look at a tray icon at all, and the + // dashboard remains one menu item away. With no runtime resolved there is + // nothing to report, so the window stays the answer. + let app = tray.app_handle(); + match app + .state::() + .proxy() + .map(|proxy| proxy.endpoint()) + { + Some(endpoint) => { + let _ = popup::toggle(app, endpoint, position); + } + None => { + if let Some(window) = app.get_webview_window("main") { + window::show(&window); + } + } } } }) .on_menu_event(move |app, event| match event.id().as_ref() { + "show-usage" => { + let Some(endpoint) = app + .state::() + .proxy() + .map(|proxy| proxy.endpoint()) + else { + return; + }; + // Anchor on the icon the user just clicked. A zero anchor clamps the popup into + // the top-left corner of the work area, which reads as a misplaced window rather + // than a menu, and on macOS the menu is now the ordinary way in rather than a + // fallback. Hosts that cannot report a rect still get the clamped corner, which + // is the best available answer there. + let anchor = tray_anchor(app); + let _ = popup::show(app, endpoint, anchor); + } "open-dashboard" => { - if let Some(window) = app.get_webview_window("main") { - window::show(&window); - } + crate::startup::open_dashboard(app); } "open-browser" => { - let endpoint = app.state::().proxy.endpoint(); + let Some(endpoint) = app + .state::() + .proxy() + .map(|proxy| proxy.endpoint()) + else { + return; + }; let _ = app .opener() .open_url(format!("{}#/usage", endpoint.url("/")), None::); @@ -117,68 +212,46 @@ pub fn install(app: &AppHandle, proxy: ProxyClient) -> tauri::Result<()> { } } "stop-proxy" => { - if app - .state::() - .spawned_by_us - .load(Ordering::Relaxed) - { - let proxy = app.state::().proxy.clone(); - let app = app.clone(); - let stop_item = stop_item.clone(); - tauri::async_runtime::spawn(async move { - let stopped = proxy.stop().await.is_ok() || proxy.is_alive().await.is_err(); - if stopped { - app.state::().shutdown_child(); - let _ = stop_item.set_enabled(false); - } - }); - } + // Through the coordinator, not beside it: Stop pressed twice, Stop then Quit, and + // Stop during an update all have to be one execution over one child. + exit::request_stop(app); } "check-updates" => { let app = app.clone(); tauri::async_runtime::spawn(async move { - updater::check_and_show(&app).await; + let _ = updater::check_and_show(&app).await; }); } "install-update" => { let app = app.clone(); tauri::async_runtime::spawn(async move { - let update = app - .state::() - .0 - .lock() - .ok() - .and_then(|mut pending| pending.take()); - let Some(update) = update else { - return; - }; - let version = update.version.clone(); - let retry_update = update.clone(); - set_installing(&app, &version); - if let Err(error) = updater::install(&app, update).await { - if let Ok(mut pending) = - app.state::().0.lock() - { - *pending = Some(retry_update); - } - set_install_failed(&app, &version); + if let Err(error) = updater::install_pending(&app).await { crate::logging::log_once("updater install failed", &error); } }); } - "quit" => app.exit(0), + // The only gesture that ends the app. It does not call `exit` itself: the coordinator + // holds the exit, drains an app-owned runtime and only then lets the process end. + "quit" => exit::request(app, ExitReason::UserQuit), _ => {} }) .build(app)?; - refresh_title(&tray, &proxy); - widget::refresh(&proxy); + apply_update_indicator(app, update_pending(app)); + refresh(app, &tray); let tray = tray.clone(); + let app = app.clone(); tauri::async_runtime::spawn(async move { let mut tick = 0; loop { tokio::time::sleep(std::time::Duration::from_secs(60)).await; - refresh_title(&tray, &proxy); + let Some(proxy) = app + .try_state::() + .and_then(|state| state.proxy()) + else { + continue; + }; + refresh_title(&app, &tray, &proxy); tick += 1; if tick % 5 == 0 { widget::refresh(&proxy); @@ -188,31 +261,55 @@ pub fn install(app: &AppHandle, proxy: ProxyClient) -> tauri::Result<()> { Ok(()) } +fn refresh(app: &AppHandle, tray: &tauri::tray::TrayIcon) { + let Some(proxy) = app + .try_state::() + .and_then(|state| state.proxy()) + else { + return; + }; + refresh_title(app, tray, &proxy); + widget::refresh(&proxy); +} + +/// Take a copy of the menu handles, holding the lock only for the copy. +/// +/// Every Tauri menu setter dispatches to the main thread and waits for it. The tray is built *on* +/// the main thread and takes this same mutex while doing so, so calling a setter with the lock held +/// is a cycle: a background update owns the mutex and waits for the main thread, and the main +/// thread waits for the mutex. The app would stop answering Quit. +fn menu_handles(app: &AppHandle) -> Option { + let state = app.try_state::()?; + let handles = state.menu.lock().ok()?; + handles.as_ref().cloned() +} + +/// Reflect who owns the runtime in the tray's Stop item. +pub fn set_owned(app: &AppHandle, owned: bool) { + if let Some(menu) = menu_handles(app) { + let _ = menu.stop.set_enabled(owned); + } +} + pub fn show_update_available(app: &AppHandle, version: &str) { - if let Some(state) = app.try_state::() { - if let Ok(menu) = state.menu.lock() { - if let Some(menu) = menu.as_ref() { - let _ = menu.install_update.set_text(updater::update_label(version)); - let _ = menu.install_update.set_enabled(true); - let _ = menu.check_updates.set_enabled(true); - let _ = menu.check_updates.set_text("Check for Updates…"); - } - } + if let Some(menu) = menu_handles(app) { + let _ = menu.install_update.set_text(updater::update_label(version)); + let _ = menu.install_update.set_enabled(true); + let _ = menu.check_updates.set_enabled(true); + let _ = menu.check_updates.set_text("Check for Updates…"); } + apply_update_indicator(app, true); } pub fn show_up_to_date(app: &AppHandle) { - if let Some(state) = app.try_state::() { - if let Ok(menu) = state.menu.lock() { - if let Some(menu) = menu.as_ref() { - let _ = menu - .check_updates - .set_text(format!("Up to date (v{})", env!("CARGO_PKG_VERSION"))); - let _ = menu.check_updates.set_enabled(true); - let _ = menu.install_update.set_enabled(false); - } - } + if let Some(menu) = menu_handles(app) { + let _ = menu + .check_updates + .set_text(format!("Up to date (v{})", env!("CARGO_PKG_VERSION"))); + let _ = menu.check_updates.set_enabled(true); + let _ = menu.install_update.set_enabled(false); } + apply_update_indicator(app, false); } pub fn is_installing(app: &AppHandle) -> bool { @@ -220,73 +317,85 @@ pub fn is_installing(app: &AppHandle) -> bool { .is_some_and(|state| state.installing.load(Ordering::Acquire)) } -fn set_installing(app: &AppHandle, version: &str) { - if let Some(state) = app.try_state::() { - state.installing.store(true, Ordering::Release); - if let Ok(menu) = state.menu.lock() { - if let Some(menu) = menu.as_ref() { - let _ = menu - .install_update - .set_text(format!("Installing update v{version}…")); - let _ = menu.install_update.set_enabled(false); - let _ = menu.check_updates.set_enabled(false); - } - } - } -} - -fn set_install_failed(app: &AppHandle, version: &str) { - if let Some(state) = app.try_state::() { - state.installing.store(false, Ordering::Release); +pub fn show_installing(app: &AppHandle, version: &str) { + if let Some(menu) = menu_handles(app) { + let _ = menu + .install_update + .set_text(format!("Installing update v{version}…")); + let _ = menu.install_update.set_enabled(false); + let _ = menu.check_updates.set_enabled(false); } - show_update_available(app, version); } -fn refresh_title(tray: &tauri::tray::TrayIcon, proxy: &ProxyClient) { +fn refresh_title(app: &AppHandle, tray: &tauri::tray::TrayIcon, proxy: &ProxyClient) { + #[cfg(target_os = "macos")] + let app = app.clone(); + #[cfg(not(target_os = "macos"))] + let _ = app; let proxy = proxy.clone(); let tray = tray.clone(); tauri::async_runtime::spawn(async move { let Ok(settings) = proxy.companion_settings().await else { return; }; - let Ok(usage) = proxy.usage_summary().await else { + let Ok(usage) = proxy.usage_today().await else { return; }; let quotas = proxy.quotas().await.unwrap_or(Value::Null); - if let Some(title) = render_title(&settings, &usage, "as) { - let _ = tray.set_title(Some(&title)); - } + let title = render_title(&settings, &usage, "as); + let _ = tray.set_title(title.as_deref()); + #[cfg(target_os = "macos")] + apply_update_indicator(&app, update_pending(&app)); }); } pub(crate) fn render_title(settings: &Value, usage: &Value, quotas: &Value) -> Option { + let settings = settings.get("settings").unwrap_or(settings); let metric = settings - .pointer("/settings/menuBarMetric") + .get("menuBarMetric") .and_then(Value::as_str) .unwrap_or("tokens"); - let summary = usage.get("summary").unwrap_or(usage); - let quota = quota_percent(quotas); + let visible_summary = crate::companion_usage::filtered_summary(usage, settings); + let summary = visible_summary.as_ref().unwrap_or(&Value::Null); + let quota = quota_percent(quotas, settings); + let template = settings + .get("menuBarTemplate") + .and_then(Value::as_str) + .filter(|value| !value.trim().is_empty()); let value = match metric { - "requests" => formatting::count(summary.get("requests").and_then(Value::as_i64)), + "requests" => formatting::count( + summary + .get("requests") + .and_then(crate::companion_usage::integer), + ), "cost" => formatting::cost(summary.get("estimatedCostUsd").and_then(Value::as_f64)), "quota" => format_percent(quota), - "none" => return None, - _ => formatting::tokens(summary.get("totalTokens").and_then(Value::as_i64)), + "none" if template.is_none() => return None, + "none" => String::new(), + _ => formatting::tokens( + summary + .get("totalTokens") + .and_then(crate::companion_usage::integer), + ), }; - let template = settings - .pointer("/settings/menuBarTemplate") - .and_then(Value::as_str) - .filter(|value| !value.trim().is_empty()); let rendered = template .map(|value| { value .replace( "{requests}", - &formatting::count(summary.get("requests").and_then(Value::as_i64)), + &formatting::count( + summary + .get("requests") + .and_then(crate::companion_usage::integer), + ), ) .replace( "{totalTokens}", - &formatting::tokens(summary.get("totalTokens").and_then(Value::as_i64)), + &formatting::tokens( + summary + .get("totalTokens") + .and_then(crate::companion_usage::integer), + ), ) .replace( "{costUsd}", @@ -294,11 +403,19 @@ pub(crate) fn render_title(settings: &Value, usage: &Value, quotas: &Value) -> O ) .replace( "{inputTokens}", - &formatting::tokens(summary.get("inputTokens").and_then(Value::as_i64)), + &formatting::tokens( + summary + .get("inputTokens") + .and_then(crate::companion_usage::integer), + ), ) .replace( "{outputTokens}", - &formatting::tokens(summary.get("outputTokens").and_then(Value::as_i64)), + &formatting::tokens( + summary + .get("outputTokens") + .and_then(crate::companion_usage::integer), + ), ) .replace("{quotaPercent}", &format_percent(quota)) }) @@ -316,10 +433,16 @@ pub(crate) fn render_title(settings: &Value, usage: &Value, quotas: &Value) -> O } } -fn quota_percent(value: &Value) -> Option { +fn quota_percent(value: &Value, settings: &Value) -> Option { let reports = value.get("reports")?.as_array()?; let mut values = Vec::new(); for report in reports { + if crate::companion_usage::hidden( + settings, + crate::companion_usage::text(report, "provider"), + ) { + continue; + } let Some(quota) = report.get("quota") else { continue; }; @@ -349,3 +472,166 @@ fn icon() -> tauri::image::Image<'static> { tauri::image::Image::from_bytes(include_bytes!("../icons/tray/icon.png")) .expect("valid tray icon") } + +/// Centre of the tray icon in physical pixels, for anchoring the popup. +/// +/// Returns the origin when the platform cannot report a rect. `popup::geometry` clamps that into +/// the work area, so the window still appears; it simply cannot point at anything. +fn tray_anchor(app: &AppHandle) -> tauri::PhysicalPosition { + app.tray_by_id("main") + .and_then(|tray| tray.rect().ok().flatten()) + .map(|rect| { + let position: tauri::PhysicalPosition = match rect.position { + tauri::Position::Physical(value) => { + tauri::PhysicalPosition::new(value.x as f64, value.y as f64) + } + tauri::Position::Logical(value) => tauri::PhysicalPosition::new(value.x, value.y), + }; + let size: tauri::PhysicalSize = match rect.size { + tauri::Size::Physical(value) => { + tauri::PhysicalSize::new(value.width as f64, value.height as f64) + } + tauri::Size::Logical(value) => tauri::PhysicalSize::new(value.width, value.height), + }; + tauri::PhysicalPosition::new( + position.x + size.width / 2.0, + position.y + size.height / 2.0, + ) + }) + .unwrap_or_else(|| tauri::PhysicalPosition::new(0.0, 0.0)) +} + +#[cfg(test)] +mod tests { + use super::{render_title, tray_icon_bytes}; + use serde_json::json; + + #[test] + fn dotted_tray_variant_is_distinct_and_both_variants_are_png() { + let normal = tray_icon_bytes(false); + let dotted = tray_icon_bytes(true); + assert_eq!(&normal[..8], b"\x89PNG\r\n\x1a\n"); + assert_eq!(&dotted[..8], b"\x89PNG\r\n\x1a\n"); + assert_ne!(normal, dotted); + } + + #[test] + fn icon_only_clears_the_title_but_a_template_and_unavailable_data_keep_their_meaning() { + let usage = json!({"summary":{"requests":7,"totalTokens":12}}); + assert_eq!( + render_title( + &json!({"settings":{"menuBarMetric":"none"}}), + &usage, + &json!({}) + ), + None + ); + assert_eq!( + render_title( + &json!({"settings":{"menuBarMetric":"none","menuBarTemplate":"{requests}"}}), + &usage, + &json!({}) + ), + Some("7".into()) + ); + assert_eq!( + render_title( + &json!({"settings":{"menuBarMetric":"tokens","hiddenProviders":["hidden"]}}), + &usage, + &json!({}) + ), + Some("—".into()) + ); + } + + #[test] + fn title_uses_filtered_whole_counts_and_ignores_hidden_quota_reports() { + let settings = + json!({"settings":{"menuBarMetric":"requests","hiddenProviders":["hidden"]}}); + let usage = json!({"summary":{"requests":99},"models":[ + {"provider":"hidden","model":"m","requests":97}, + {"provider":"visible","model":"m","requests":2} + ]}); + assert_eq!( + render_title(&settings, &usage, &json!({})), + Some("2".into()) + ); + let settings = json!({"settings":{"menuBarMetric":"quota","hiddenProviders":["hidden"]}}); + let quotas = json!({"reports":[{"provider":"hidden","quota":{"weeklyPercent":1}}, {"provider":"visible","quota":{"weeklyPercent":75}}]}); + assert_eq!(render_title(&settings, &usage, "as), Some("75%".into())); + } + + /// This file's own source, read at compile time, with the test module cut off. + /// + /// Slicing at the test attribute matters: the assertions below quote the very call names they + /// look for, so scanning the whole file would find the test's own string literals and pass + /// after the real calls were deleted. + fn production_source() -> &'static str { + include_str!("tray.rs") + .split("#[cfg(te") + .next() + .expect("source has a production half") + } + + /// A tray with a menu opens that menu on left click unless the builder says otherwise, and + /// nothing in the type system connects the two calls. The usage popup was unreachable on + /// macOS and Windows for exactly that reason, and the failure is quiet: the icon still + /// responds to the click, just with the wrong surface. The menu item that opens the popup is + /// Linux-only, so there was no second way in. + #[test] + fn attaching_a_menu_leaves_the_left_click_for_the_popup() { + let source = production_source(); + assert!( + source.contains(".menu(&menu)"), + "tray.rs no longer attaches a menu; this pairing may no longer apply" + ); + // The call site, not the name: the comments above explain why the flag is inert on + // macOS, and a bare substring matched that prose instead of the builder. + assert!( + source.contains("builder.show_menu_on_left_click(false)"), + "a tray with a menu must release the left click, or the popup cannot open" + ); + assert!( + source.contains("#[cfg(not(target_os = \"linux\"))]"), + "Linux delivers no usable click event, so it must keep the menu on left click" + ); + } + + /// macOS pops the attached menu from AppKit before the crate's click handler runs, so the + /// icon click cannot be the only way to the popup. The menu item is the path that works + /// everywhere, and platform-gating it once already left two platforms with no way in. + #[test] + fn the_usage_menu_item_is_not_platform_gated() { + let source = production_source(); + let declaration = source + .lines() + .position(|line| line.contains("let show_usage =")) + .expect("the menu no longer declares the usage item"); + let lines: Vec<&str> = source.lines().collect(); + // Every line that mentions the item: its declaration, its place in the menu, and the + // event arm. None of them may sit under a platform attribute. + let mentions = lines + .iter() + .enumerate() + .filter(|(_, line)| line.contains("show_usage") || line.contains("\"show-usage\"")) + .map(|(index, _)| index); + for index in mentions { + let previous = lines[..index] + .iter() + .rev() + .find(|line| !line.trim().is_empty()) + .copied() + .unwrap_or_default(); + assert!( + !previous.trim_start().starts_with("#[cfg("), + "the usage item is platform-gated at line {}; every platform needs a menu path \ + to the popup", + index + 1 + ); + } + assert!( + declaration > 0, + "the declaration is the first line of the file" + ); + } +} diff --git a/desktop/src-tauri/src/tray_availability.rs b/desktop/src-tauri/src/tray_availability.rs new file mode 100644 index 0000000000..404c5b75e3 --- /dev/null +++ b/desktop/src-tauri/src/tray_availability.rs @@ -0,0 +1,131 @@ +//! Whether this session actually has a tray, as opposed to a tray backend that accepts an icon. +//! +//! `TrayIconBuilder::build` returning `Ok` proves nothing on Linux. The pinned backend creates an +//! AppIndicator and reports success without checking that anything will display it, so on stock +//! GNOME — which ships no AppIndicator extension — construction succeeds and no icon ever appears. +//! The shell's macOS-shaped assumptions then compound it: the window was created hidden and close +//! always hid, which leaves a running process with no way back in. +//! +//! So the question is asked of the session bus. Not whether the watcher exists — a watcher with no +//! host attached still accepts registrations and still draws nothing — but whether it reports a +//! host registered, which is the StatusNotifier specification's own answer to "is there somewhere +//! for an icon to appear". macOS and Windows have a status area that is always present and answer +//! without a probe. + +/// The result of asking whether this session can display a tray icon. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum TrayAvailability { + /// There is somewhere for the icon to appear, so hiding to the tray is a real place to hide. + Available, + /// There is not. The window is shown on launch and closing it ends the app through the drain. + Unavailable, +} + +impl TrayAvailability { + pub fn is_available(self) -> bool { + matches!(self, Self::Available) + } + + /// Whether a window close or a platform quit gesture may hide the app instead of ending it. + pub fn hides_to_tray(self) -> bool { + self.is_available() + } + + /// What to assume before the probe has answered. + /// + /// The probe is asynchronous, and a window close can land before it returns. Assuming a tray + /// that turns out not to exist is the failure this whole module is about, so the platforms that + /// need a probe assume nothing until they have one. + pub fn assumed() -> Self { + if cfg!(target_os = "linux") { + Self::Unavailable + } else { + Self::Available + } + } +} + +/// The StatusNotifier watcher, and the property that says a host is attached to it. +#[cfg(target_os = "linux")] +pub const WATCHER_NAME: &str = "org.kde.StatusNotifierWatcher"; +#[cfg(target_os = "linux")] +pub const WATCHER_PATH: &str = "/StatusNotifierWatcher"; +#[cfg(target_os = "linux")] +pub const HOST_REGISTERED: &str = "IsStatusNotifierHostRegistered"; + +/// How long the session-bus probe may take. +#[cfg(target_os = "linux")] +const PROBE_TIMEOUT_MS: u64 = 750; + +/// Read a host-registered answer as an availability verdict. +/// +/// `None` means the question could not be asked at all — no session bus, no watcher on it, no +/// reply, a malformed one. That is deliberately folded into the same answer as a watcher with no +/// host, because the two are indistinguishable from here and the safe response to both is +/// identical: show the window and let close mean close. Guessing the other way strands the user. +pub fn from_host_registered(registered: Option) -> TrayAvailability { + match registered { + Some(true) => TrayAvailability::Available, + Some(false) | None => TrayAvailability::Unavailable, + } +} + +#[cfg(not(target_os = "linux"))] +pub fn detect() -> TrayAvailability { + // macOS and Windows both have a status area that is always there, so the answer is known + // without asking anything. It still goes through the same reading so there is one place where + // an availability verdict is produced. + from_host_registered(Some(true)) +} + +#[cfg(target_os = "linux")] +pub fn detect() -> TrayAvailability { + from_host_registered(host_registered()) +} + +#[cfg(target_os = "linux")] +fn host_registered() -> Option { + use dbus::blocking::{stdintf::org_freedesktop_dbus::Properties, Connection}; + use std::time::Duration; + + let connection = Connection::new_session().ok()?; + let watcher = connection.with_proxy( + WATCHER_NAME, + WATCHER_PATH, + Duration::from_millis(PROBE_TIMEOUT_MS), + ); + // A watcher nobody owns makes this call fail rather than answer, which is the same verdict. + watcher.get(WATCHER_NAME, HOST_REGISTERED).ok() +} + +#[cfg(test)] +mod tests { + use super::{from_host_registered, TrayAvailability}; + + #[test] + fn only_a_registered_host_is_an_available_tray() { + assert_eq!( + from_host_registered(Some(true)), + TrayAvailability::Available + ); + assert_eq!( + from_host_registered(Some(false)), + TrayAvailability::Unavailable + ); + assert_eq!(from_host_registered(None), TrayAvailability::Unavailable); + } + + #[test] + fn hiding_is_only_offered_where_the_icon_would_be_drawn() { + assert!(TrayAvailability::Available.hides_to_tray()); + assert!(!TrayAvailability::Unavailable.hides_to_tray()); + } + + #[test] + fn nothing_is_assumed_on_the_platform_that_needs_a_probe() { + assert_eq!( + TrayAvailability::assumed().is_available(), + !cfg!(target_os = "linux") + ); + } +} diff --git a/desktop/src-tauri/src/updater.rs b/desktop/src-tauri/src/updater.rs index 4638c1565d..bcaf334953 100644 --- a/desktop/src-tauri/src/updater.rs +++ b/desktop/src-tauri/src/updater.rs @@ -1,12 +1,384 @@ -use crate::{logging, tray}; +use crate::{ + exit::{AbortedRestart, ExitCoordinator, ExitPhase, RestartReadiness}, + logging, tray, +}; +use serde::Serialize; +use serde_json::to_value; +use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::Mutex; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; use tauri::{AppHandle, Manager}; use tauri_plugin_updater::{Update, UpdaterExt}; +use tokio::sync::watch; +use uuid::Uuid; + +#[derive(Clone)] +pub enum UiProjection { + Available(String), + Current, + Installing(String), +} + +#[derive(Clone)] +struct UiUpdate { + revision: u64, + projection: UiProjection, +} + +pub struct CheckGeneration { + latest_started: AtomicU64, + install_epoch: AtomicU64, + application: Mutex<()>, + latest_ui_revision: AtomicU64, + ui: watch::Sender>, +} + +impl Default for CheckGeneration { + fn default() -> Self { + let (ui, _) = watch::channel(None); + Self { + latest_started: AtomicU64::new(0), + install_epoch: AtomicU64::new(0), + application: Mutex::new(()), + latest_ui_revision: AtomicU64::new(0), + ui, + } + } +} + +impl CheckGeneration { + pub fn begin_if_not_installing( + &self, + installing: &std::sync::atomic::AtomicBool, + publish_checking: impl FnOnce(), + ) -> Option<(u64, u64)> { + let _guard = self + .application + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + if installing.load(Ordering::Acquire) { + return None; + } + let generation = self.latest_started.fetch_add(1, Ordering::AcqRel) + 1; + let epoch = self.install_epoch.load(Ordering::Acquire); + publish_checking(); + Some((generation, epoch)) + } + + pub fn claim_install( + &self, + installing: &std::sync::atomic::AtomicBool, + pending_version: impl FnOnce() -> Option, + on_claim: impl FnOnce(), + ) -> InstallClaim { + let _guard = self + .application + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + if installing.load(Ordering::Acquire) { + return InstallClaim::Busy; + } + let Some(version) = pending_version() else { + return InstallClaim::NoPending; + }; + if installing + .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) + .is_err() + { + return InstallClaim::Busy; + } + self.install_epoch.fetch_add(1, Ordering::AcqRel); + self.latest_ui_revision.fetch_add(1, Ordering::AcqRel); + on_claim(); + self.queue_ui(UiProjection::Installing(version)); + InstallClaim::Claimed + } + + pub fn epoch_is_current(&self, epoch: u64) -> bool { + self.install_epoch.load(Ordering::Acquire) == epoch + } + + pub fn apply_if_current(&self, generation: u64, apply: impl FnOnce() -> T) -> Option { + let _guard = self + .application + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + if self.latest_started.load(Ordering::Acquire) != generation { + return None; + } + Some(apply()) + } + + pub fn inspect(&self, read: impl FnOnce() -> T) -> T { + let _guard = self + .application + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + read() + } + + // Call only inside application/inspect. This is an in-memory send, never a Tauri setter. + fn queue_ui(&self, projection: UiProjection) { + let revision = self.latest_ui_revision.fetch_add(1, Ordering::AcqRel) + 1; + self.ui.send_replace(Some(UiUpdate { + revision, + projection, + })); + } + + fn apply_ui_projection_if_current( + &self, + update: UiUpdate, + apply: impl FnOnce(UiProjection), + ) -> bool { + if update.revision != self.latest_ui_revision.load(Ordering::Acquire) { + return false; + } + apply(update.projection); + true + } +} + +pub fn start_ui_projection_worker(app: AppHandle) { + let mut receiver = app.state::().ui.subscribe(); + tauri::async_runtime::spawn(async move { + while receiver.changed().await.is_ok() { + let Some(update) = receiver.borrow_and_update().clone() else { + continue; + }; + app.state::() + .apply_ui_projection_if_current(update, |projection| match projection { + UiProjection::Available(version) => tray::show_update_available(&app, &version), + UiProjection::Current => tray::show_up_to_date(&app), + UiProjection::Installing(version) => tray::show_installing(&app, &version), + }); + } + }); +} + +#[derive(Clone, Debug, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct DesktopSnapshot { + session_id: String, + current_version: String, + latest_version: Option, + available: bool, + checked_at_ms: Option, + phase: &'static str, +} + +pub struct DesktopUpdateState { + session_id: String, + tx: watch::Sender, +} + +impl DesktopUpdateState { + pub fn new(current_version: String) -> Self { + let session_id = Uuid::new_v4().to_string(); + let (tx, _) = watch::channel(DesktopSnapshot { + session_id: session_id.clone(), + current_version, + latest_version: None, + available: false, + checked_at_ms: None, + phase: "idle", + }); + Self { session_id, tx } + } + + pub fn session_id(&self) -> &str { + &self.session_id + } + + pub fn publish(&self, phase: &'static str, latest: Option, checked: Option) { + let previous = self.tx.borrow().clone(); + let next = DesktopSnapshot { + session_id: self.session_id.clone(), + current_version: previous.current_version, + available: latest.is_some(), + latest_version: latest, + checked_at_ms: checked, + phase, + }; + self.tx.send_replace(next); + } + + pub fn retain_phase(&self, phase: &'static str) { + let previous = self.tx.borrow().clone(); + self.publish(phase, previous.latest_version, previous.checked_at_ms); + } + + pub fn wake(&self) { + self.wake_with_before_notify(|| {}); + } + + fn wake_with_before_notify(&self, before_notify: impl FnOnce()) { + before_notify(); + self.tx.send_modify(|_| {}); + } +} + +fn now_ms() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_millis() + .min(u128::from(u64::MAX)) as u64 +} + +pub fn start_snapshot_publisher(app: AppHandle) { + let mut receiver = app.state::().tx.subscribe(); + tauri::async_runtime::spawn(async move { + loop { + let snapshot = receiver.borrow_and_update().clone(); + if let Some(proxy) = app + .try_state::() + .and_then(|state| state.proxy()) + { + if let Ok(body) = to_value(&snapshot) { + let _ = proxy.post_desktop_snapshot(&body).await; + } + } + if matches!( + tokio::time::timeout(Duration::from_secs(60), receiver.changed()).await, + Ok(Err(_)) + ) { + break; + } + } + }); +} pub struct PendingUpdate(pub Mutex>); +#[derive(Debug, PartialEq, Eq)] +pub enum InstallClaim { + Claimed, + Busy, + NoPending, +} + +#[derive(Serialize)] +#[serde(rename_all = "camelCase")] +pub struct PageUpdateStatus { + pub current_version: String, + pub latest_version: Option, + pub available: bool, + pub installing: bool, + pub checking: bool, +} + +pub fn page_status(app: &AppHandle) -> PageUpdateStatus { + app.state::().inspect(|| { + let pending = app.state::(); + let pending = pending + .0 + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + let latest_version = pending.as_ref().map(|update| update.version.clone()); + let installing = tray::is_installing(app); + let checking = app.state::().tx.borrow().phase == "checking"; + PageUpdateStatus { + current_version: env!("CARGO_PKG_VERSION").to_owned(), + available: latest_version.is_some(), + latest_version, + installing, + checking, + } + }) +} + +pub async fn install_pending(app: &AppHandle) -> Result { + let state = app.state::(); + let gate = app.state::(); + match gate.claim_install( + &state.installing, + || { + app.state::() + .0 + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .as_ref() + .map(|update| update.version.clone()) + }, + || app.state::().retain_phase("installing"), + ) { + InstallClaim::Claimed => {} + InstallClaim::Busy => return Err("an update is already installing".into()), + InstallClaim::NoPending => return Err("no update is ready to install".into()), + } + let pending = app.state::(); + let update = pending + .0 + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .take(); + let Some(update) = update else { + gate.inspect(|| { + state.installing.store(false, Ordering::Release); + app.state::().retain_phase("current"); + gate.queue_ui(UiProjection::Current); + }); + return Err("no update is ready to install".into()); + }; + let version = update.version.clone(); + let retry_update = update.clone(); + let result = install(app, update).await; + if let Err(error) = result { + gate.inspect(|| { + let pending = app.state::(); + *pending + .0 + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) = Some(retry_update); + state.installing.store(false, Ordering::Release); + state.update_pending.store(true, Ordering::Release); + app.state::() + .retain_phase("install-failed"); + gate.queue_ui(UiProjection::Available(version)); + }); + return Err(error); + } + state.installing.store(false, Ordering::Release); + Ok(page_status(app)) +} + +/// The manifest key a Linux install must resolve, or None to keep the updater's default +/// os-arch key (linux-x86_64, windows-x86_64, darwin-*). +/// +/// A deb install cannot apply the AppImage payload: the updater validates the downloaded +/// bytes as a real .deb before installing through package-manager elevation, so it must +/// resolve the deb's own manifest key. The bundle type is patched into the binary at +/// packaging time, so the answer is embedded per artifact, not detected at runtime. The +/// AppImage keeps the default key, which is also what installs from releases before the +/// deb target existed already resolve. +#[cfg(any(target_os = "linux", test))] +pub fn linux_updater_target( + bundle: Option, +) -> Option<&'static str> { + match bundle { + Some(tauri_utils::config::BundleType::Deb) => Some("linux-x86_64-deb"), + _ => None, + } +} + +#[cfg(target_os = "linux")] +fn configured_updater_target() -> Option<&'static str> { + linux_updater_target(tauri_utils::platform::bundle_type()) +} + +#[cfg(not(target_os = "linux"))] +fn configured_updater_target() -> Option<&'static str> { + None +} + pub async fn check(app: &AppHandle) -> Result, String> { - app.updater() + let mut builder = app.updater_builder(); + if let Some(target) = configured_updater_target() { + builder = builder.target(target); + } + builder + .build() .map_err(|error| error.to_string())? .check() .await @@ -14,11 +386,59 @@ pub async fn check(app: &AppHandle) -> Result, String> { } pub async fn install(app: &AppHandle, update: Update) -> Result<(), String> { - update - .download_and_install(|_, _| {}, || {}) + // Download and verify first, and separately from installing. The pinned updater checks the + // release signature inside `download`, so these bytes are the ones the key signed; nothing has + // been replaced yet, and a failure here costs only the download. + let package = update + .download(|_, _| {}, || {}) .await .map_err(|error| error.to_string())?; - app.restart(); + + // Then stop the runtime, and confirm it stopped, *before* anything is replaced. Asking for the + // restart after `install` is the shape that does not work: the pinned Windows installer hands off + // to the installer process and ends this one, so the call after it is never reached and the + // update would replace files under a runtime that is still serving. R2 still holds — this is a + // coordinated restart and not a quit — but the coordination has to finish first. + let readiness = crate::exit::prepare_restart(app).await; + if readiness != RestartReadiness::Ready { + // Not an ending after all: the app goes back to running, so a close hides again, Quit + // works and Install can be retried. A drain somebody else owns is left alone. + recover_after_failed_install(app); + return Err(format!( + "the update was downloaded but not installed: {}", + readiness.describe() + )); + } + + if let Err(error) = update.install(package) { + // The installer returned a failure. The runtime was stopped for an install that did not + // happen, so the app brings one back instead of sitting drained. + recover_after_failed_install(app); + return Err(error.to_string()); + } + // Only reached where the installer returns. On Windows it does not. + crate::exit::complete_restart(app) +} + +/// Hand a failed install back to a running app. True when the drain had already stopped the +/// runtime, so the startup sequence has to bring one back; a drain that failed left it running. +/// Intent captured before the drain and a newer startup retry are both authoritative. +fn after_install_failure(coordinator: &ExitCoordinator) -> bool { + coordinator.abort_restart() + == Some(AbortedRestart { + phase: ExitPhase::Drained, + runtime_was_wanted: true, + }) +} + +fn recover_after_failed_install(app: &AppHandle) { + let restart = app + .try_state::() + .is_some_and(|coordinator| after_install_failure(&coordinator)); + if restart { + // Recover, not Launch: nobody is waiting on a prompt, and only a proven absence starts one. + crate::startup::begin_with(app, crate::startup::Mode::Recover); + } } pub fn update_label(version: &str) -> String { @@ -29,43 +449,458 @@ pub fn start_background_checks(app: AppHandle) { tauri::async_runtime::spawn(async move { tokio::time::sleep(std::time::Duration::from_secs(30)).await; loop { - check_and_show(&app).await; + let _ = check_and_show(&app).await; tokio::time::sleep(std::time::Duration::from_secs(6 * 60 * 60)).await; } }); } -pub async fn check_and_show(app: &AppHandle) { - if tray::is_installing(app) { - return; - } - match check(app).await { - Ok(Some(update)) => { - if tray::is_installing(app) { - return; +pub async fn check_and_show(app: &AppHandle) -> Result<(), String> { + let gate = app.state::(); + let state = app.state::(); + let Some((generation, epoch)) = gate.begin_if_not_installing(&state.installing, || { + app.state::().retain_phase("checking"); + }) else { + return Ok(()); + }; + let answer = check(app).await; + let applied = gate.apply_if_current(generation, || { + if tray::is_installing(app) || !gate.epoch_is_current(epoch) { + return Ok(()); + } + match answer { + Ok(Some(update)) => { + let version = update.version.clone(); + if let Ok(mut pending) = app.state::().0.lock() { + *pending = Some(update); + } + app.state::().publish( + "available", + Some(version.clone()), + Some(now_ms()), + ); + state.update_pending.store(true, Ordering::Release); + gate.queue_ui(UiProjection::Available(version)); + Ok(()) } - let version = update.version.clone(); - if let Ok(mut pending) = app.state::().0.lock() { - *pending = Some(update); + Ok(None) => { + if let Ok(mut pending) = app.state::().0.lock() { + *pending = None; + } + app.state::() + .publish("current", None, Some(now_ms())); + state.update_pending.store(false, Ordering::Release); + gate.queue_ui(UiProjection::Current); + Ok(()) } - tray::show_update_available(app, &version); - } - Ok(None) => { - if let Ok(mut pending) = app.state::().0.lock() { - *pending = None; + Err(error) => { + app.state::().retain_phase("error"); + Err(error) } - tray::show_up_to_date(app); } - Err(error) => logging::log_once("updater check failed", &error), + }); + if let Some(Err(error)) = &applied { + logging::log_once("updater check failed", error); } + applied.unwrap_or(Ok(())) } #[cfg(test)] mod tests { - use super::update_label; + use super::{ + after_install_failure, linux_updater_target, update_label, CheckGeneration, + DesktopUpdateState, InstallClaim, UiProjection, + }; + use crate::exit::{DrainVerdict, ExitCoordinator, ExitDecision, ExitReason}; + use crate::tray_availability::TrayAvailability; + use std::sync::atomic::{AtomicBool, Ordering}; + use std::sync::{mpsc, Arc}; + use tauri_utils::config::BundleType; + + #[test] + fn a_failed_install_restarts_the_runtime_only_when_the_drain_had_stopped_it() { + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.finish_drain(DrainVerdict::Drained); + assert!(after_install_failure(&coordinator)); + assert!(coordinator.supervision_allowed()); + assert!(coordinator.begin_spawn()); + + // A drain that failed left the runtime serving: back to running, nothing to start. + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.finish_drain(DrainVerdict::Failed); + assert!(!after_install_failure(&coordinator)); + assert!(coordinator.supervision_allowed()); + + // A quit that took the drain is never turned back into a running app. + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::UserQuit); + coordinator.finish_drain(DrainVerdict::Drained); + assert!(!after_install_failure(&coordinator)); + assert_eq!(coordinator.decision(), ExitDecision::Proceed); + + // A completed tray Stop remains the person's intent across repeated failed updates. + let coordinator = ExitCoordinator::new(); + coordinator.set_tray(TrayAvailability::Available); + assert!(coordinator.begin_stop()); + assert_eq!(coordinator.finish_stop(), None); + for _ in 0..2 { + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.finish_drain(DrainVerdict::Drained); + assert!(!after_install_failure(&coordinator)); + assert!(!coordinator.supervision_allowed()); + assert_eq!(coordinator.decision(), ExitDecision::Hide); + } + + // A newer retry wins over the stopped intent that the update captured at claim time. + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.resume(); + coordinator.finish_drain(DrainVerdict::Drained); + assert!(after_install_failure(&coordinator)); + assert!(coordinator.supervision_allowed()); + } + + #[test] + fn desktop_snapshot_serializes_the_bounded_wire_fields() { + let state = DesktopUpdateState::new("2.61.0".into()); + state.publish("available", Some("2.62.0".into()), Some(1_790_000_000_000)); + let value = serde_json::to_value(state.tx.borrow().clone()).unwrap(); + assert!(uuid::Uuid::parse_str(state.session_id()).is_ok()); + assert_eq!(value["sessionId"], state.session_id()); + assert_eq!(value["currentVersion"], "2.61.0"); + assert_eq!(value["latestVersion"], "2.62.0"); + assert_eq!(value["available"], true); + assert_eq!(value["checkedAtMs"], 1_790_000_000_000u64); + assert!(value["checkedAtMs"].as_u64().unwrap() >= 946_684_800_000); + assert_eq!(value["phase"], "available"); + assert_eq!(value.as_object().unwrap().len(), 6); + } + + #[test] + fn wake_preserves_a_snapshot_published_during_notification() { + let state = DesktopUpdateState::new("2.61.0".into()); + state.publish("checking", None, None); + state.wake_with_before_notify(|| { + state.publish("available", Some("2.62.0".into()), Some(123)); + }); + let snapshot = state.tx.borrow(); + assert_eq!(snapshot.phase, "available"); + assert_eq!(snapshot.latest_version.as_deref(), Some("2.62.0")); + assert_eq!(snapshot.checked_at_ms, Some(123)); + } + + #[test] + fn wake_notifies_without_changing_the_snapshot() { + let state = DesktopUpdateState::new("2.61.0".into()); + let mut receiver = state.tx.subscribe(); + let before = serde_json::to_value(receiver.borrow_and_update().clone()).unwrap(); + state.wake(); + assert!(receiver.has_changed().unwrap()); + let after = serde_json::to_value(receiver.borrow_and_update().clone()).unwrap(); + assert_eq!(after, before); + } + + #[test] + fn a_delayed_older_none_cannot_clear_a_newer_pending_update() { + let checks = CheckGeneration::default(); + let installing = AtomicBool::new(false); + let (older, _) = checks.begin_if_not_installing(&installing, || {}).unwrap(); + let (newer, _) = checks.begin_if_not_installing(&installing, || {}).unwrap(); + let mut pending: Option<&str> = None; + let mut phase = "checking"; + assert_eq!( + checks.apply_if_current(newer, || { + pending = Some("2.62.0"); + phase = "available"; + }), + Some(()) + ); + assert_eq!( + checks.apply_if_current(older, || { + pending = None; + phase = "current"; + }), + None + ); + assert_eq!(pending, Some("2.62.0")); + assert_eq!(phase, "available"); + let (third, _) = checks.begin_if_not_installing(&installing, || {}).unwrap(); + assert_eq!( + checks.apply_if_current(newer, || { + pending = None; + }), + None + ); + assert_eq!(pending, Some("2.62.0")); + assert_eq!( + checks.apply_if_current(third, || { + pending = None; + }), + Some(()) + ); + assert_eq!(pending, None); + } + + #[test] + fn checking_publication_rechecks_install_claim_inside_the_gate() { + let checks = CheckGeneration::default(); + let installing = AtomicBool::new(false); + assert_eq!( + checks.claim_install(&installing, || Some("2.66.0".into()), || {}), + InstallClaim::Claimed + ); + let mut published = false; + assert_eq!( + checks.begin_if_not_installing(&installing, || { + published = true; + }), + None + ); + assert!(!published); + } + + #[test] + fn install_claim_has_one_winner_and_can_retry_after_failure() { + let gate = CheckGeneration::default(); + let installing = AtomicBool::new(false); + assert_eq!( + gate.claim_install(&installing, || Some("2.66.0".into()), || {}), + InstallClaim::Claimed + ); + assert_eq!(gate.install_epoch.load(Ordering::Acquire), 1); + assert_eq!( + gate.claim_install(&installing, || Some("2.66.0".into()), || {}), + InstallClaim::Busy + ); + assert_eq!(gate.install_epoch.load(Ordering::Acquire), 1); + installing.store(false, Ordering::Release); + assert_eq!( + gate.claim_install(&installing, || Some("2.66.0".into()), || {}), + InstallClaim::Claimed + ); + assert_eq!(gate.install_epoch.load(Ordering::Acquire), 2); + } + + #[test] + fn install_click_without_pending_leaves_in_flight_check_valid() { + let gate = CheckGeneration::default(); + let installing = AtomicBool::new(false); + let (generation, epoch) = gate.begin_if_not_installing(&installing, || {}).unwrap(); + let revision = gate.latest_ui_revision.load(Ordering::Acquire); + let mut claimed_hook = false; + assert_eq!( + gate.claim_install( + &installing, + || None, + || { + claimed_hook = true; + } + ), + InstallClaim::NoPending + ); + assert!(!claimed_hook); + assert!(!installing.load(Ordering::Acquire)); + assert_eq!(gate.install_epoch.load(Ordering::Acquire), 0); + assert_eq!(gate.latest_ui_revision.load(Ordering::Acquire), revision); + assert!(gate.epoch_is_current(epoch)); + assert_eq!( + gate.apply_if_current(generation, || "current"), + Some("current") + ); + } + + #[test] + fn page_check_started_before_tray_check_cannot_override_it_in_either_completion_order() { + let gate = CheckGeneration::default(); + let installing = AtomicBool::new(false); + let mut pending = Some("previous"); + let mut phase = "available"; + + let (page, _) = gate + .begin_if_not_installing(&installing, || { + phase = "checking"; + }) + .unwrap(); + let (tray, _) = gate + .begin_if_not_installing(&installing, || { + phase = "checking"; + }) + .unwrap(); + assert_eq!( + gate.apply_if_current(page, || { + pending = None; + phase = "current"; + }), + None + ); + assert_eq!((pending, phase), (Some("previous"), "checking")); + assert_eq!( + gate.apply_if_current(tray, || { + pending = Some("tray"); + phase = "available"; + }), + Some(()) + ); + assert_eq!((pending, phase), (Some("tray"), "available")); + + let (page, _) = gate + .begin_if_not_installing(&installing, || { + phase = "checking"; + }) + .unwrap(); + let (tray, _) = gate + .begin_if_not_installing(&installing, || { + phase = "checking"; + }) + .unwrap(); + assert_eq!( + gate.apply_if_current(tray, || { + pending = Some("new tray"); + phase = "available"; + }), + Some(()) + ); + assert_eq!( + gate.apply_if_current(page, || { + pending = None; + phase = "current"; + }), + None + ); + assert_eq!((pending, phase), (Some("new tray"), "available")); + } + + #[test] + fn install_claim_cannot_land_between_check_guard_and_pending_tray_write() { + let gate = Arc::new(CheckGeneration::default()); + let installing = Arc::new(AtomicBool::new(false)); + let (generation, epoch) = gate.begin_if_not_installing(&installing, || {}).unwrap(); + let (attempt_tx, attempt_rx) = mpsc::channel(); + let (claimed_tx, claimed_rx) = mpsc::channel(); + let mut pending = None; + let mut tray_visible = false; + + let claim_thread = gate + .apply_if_current(generation, || { + assert!(!installing.load(Ordering::Acquire)); + assert!(gate.epoch_is_current(epoch)); + let claim_gate = Arc::clone(&gate); + let claim_flag = Arc::clone(&installing); + let thread = std::thread::spawn(move || { + attempt_tx.send(()).unwrap(); + claimed_tx + .send(claim_gate.claim_install( + &claim_flag, + || Some("signed update".into()), + || {}, + )) + .unwrap(); + }); + attempt_rx + .recv_timeout(std::time::Duration::from_secs(1)) + .unwrap(); + assert_eq!( + claimed_rx.recv_timeout(std::time::Duration::from_millis(25)), + Err(mpsc::RecvTimeoutError::Timeout) + ); + pending = Some("signed update"); + tray_visible = true; + assert!(!installing.load(Ordering::Acquire)); + thread + }) + .unwrap(); + assert_eq!((pending, tray_visible), (Some("signed update"), true)); + assert_eq!( + claimed_rx + .recv_timeout(std::time::Duration::from_secs(1)) + .unwrap(), + InstallClaim::Claimed + ); + claim_thread.join().unwrap(); + assert!(installing.load(Ordering::Acquire)); + assert!(!gate.epoch_is_current(epoch)); + } + + #[test] + fn status_read_completes_while_check_ui_setter_is_blocked() { + let gate = Arc::new(CheckGeneration::default()); + let installing = AtomicBool::new(false); + let (generation, _) = gate.begin_if_not_installing(&installing, || {}).unwrap(); + let mut pending = None; + assert_eq!( + gate.apply_if_current(generation, || { + pending = Some("signed update"); + gate.queue_ui(UiProjection::Available("2.66.0".into())); + }), + Some(()) + ); + let projected = gate.ui.borrow().clone().unwrap(); + let (setter_entered_tx, setter_entered_rx) = mpsc::channel(); + let (status_returned_tx, status_returned_rx) = mpsc::channel(); + let setter_gate = Arc::clone(&gate); + let setter = std::thread::spawn(move || { + setter_gate.apply_ui_projection_if_current(projected, |_| { + setter_entered_tx.send(()).unwrap(); + status_returned_rx + .recv_timeout(std::time::Duration::from_secs(1)) + .unwrap(); + }) + }); + setter_entered_rx + .recv_timeout(std::time::Duration::from_secs(1)) + .unwrap(); + let read_gate = Arc::clone(&gate); + let (read_tx, read_rx) = mpsc::channel(); + let reader = std::thread::spawn(move || { + read_tx.send(read_gate.inspect(|| "available")).unwrap(); + }); + assert_eq!( + read_rx + .recv_timeout(std::time::Duration::from_secs(1)) + .unwrap(), + "available" + ); + status_returned_tx.send(()).unwrap(); + reader.join().unwrap(); + assert!(setter.join().unwrap()); + assert_eq!(pending, Some("signed update")); + } + + #[test] + fn superseded_ui_projection_never_enters_its_setter() { + let gate = CheckGeneration::default(); + gate.inspect(|| gate.queue_ui(UiProjection::Current)); + let old = gate.ui.borrow().clone().unwrap(); + gate.inspect(|| gate.queue_ui(UiProjection::Available("2.66.0".into()))); + let newest = gate.ui.borrow().clone().unwrap(); + assert!(!gate.apply_ui_projection_if_current(old, |_| panic!("stale setter ran"))); + let mut applied = false; + assert!(gate.apply_ui_projection_if_current(newest, |_| applied = true)); + assert!(applied); + } #[test] fn formats_update_menu_label() { assert_eq!(update_label("2.62.0"), "Install update v2.62.0"); } + + #[test] + fn deb_installs_resolve_their_own_updater_key() { + assert_eq!( + linux_updater_target(Some(BundleType::Deb)), + Some("linux-x86_64-deb") + ); + } + + #[test] + fn appimage_installs_keep_the_default_updater_key() { + assert_eq!(linux_updater_target(Some(BundleType::AppImage)), None); + } + + #[test] + fn unbundled_builds_keep_the_default_updater_key() { + // Dev builds and any format without a patcher entry resolve the default key. + assert_eq!(linux_updater_target(None), None); + } } diff --git a/desktop/src-tauri/src/widget.rs b/desktop/src-tauri/src/widget.rs index 4db3005f03..40c4e2c44c 100644 --- a/desktop/src-tauri/src/widget.rs +++ b/desktop/src-tauri/src/widget.rs @@ -1,5 +1,6 @@ #[cfg(target_os = "macos")] mod macos { + use crate::companion_query::{timeline_query, timeline_rows}; use crate::{ proxy::{ProxyClient, ProxyError}, tray, @@ -44,6 +45,12 @@ mod macos { bucket_seconds: i64, style: String, series: Vec, + #[serde(default, skip_serializing_if = "is_false")] + incomplete: bool, + } + + fn is_false(value: &bool) -> bool { + !value } #[derive(Debug, Serialize, serde::Deserialize, Clone, PartialEq)] @@ -68,6 +75,8 @@ mod macos { Unauthorized, Http, Decode, + /// The port answered, but as something other than the runtime this shell is bound to. + Foreign, } fn state_for_error( @@ -85,6 +94,18 @@ mod macos { "Needs API key", Some("This proxy requires an API key.".into()), ), + // A runtime this app did not start is a different event from a fault, so it does not + // borrow the vocabulary of one. "degraded" would claim the proxy is misbehaving and + // "unreachable" would claim nothing is there; a user who started the runtime from npm + // or the CLI themselves would read either as a defect in a setup that is working. + // The widget has no red for this: `tone` in `app/Sources/OpenCodexWidget/Views.swift` + // maps a state it does not know to the neutral secondary colour, which is the right + // signal for "serving, just not ours". + ErrorKind::Foreign => ( + "foreign", + "External runtime", + Some("This port is served by a runtime this app did not start.".into()), + ), ErrorKind::Http | ErrorKind::Decode => ("degraded", "Degraded", detail), } } @@ -95,6 +116,7 @@ mod macos { ProxyError::Unauthorized => (ErrorKind::Unauthorized, None), ProxyError::Http(status) => (ErrorKind::Http, Some(format!("HTTP {status}"))), ProxyError::Decode(error) => (ErrorKind::Decode, Some(error.to_string())), + ProxyError::Foreign => (ErrorKind::Foreign, None), } } @@ -103,7 +125,7 @@ mod macos { } fn integer(value: Option<&Value>) -> Option { - value.and_then(Value::as_i64) + value.and_then(crate::companion_usage::integer) } fn reset_at(value: Option<&Value>) -> Option { @@ -115,12 +137,18 @@ mod macos { }) } - fn quotas(value: &Value) -> Vec { + fn quotas(value: &Value, settings: &Value) -> Vec { let Some(reports) = value.get("reports").and_then(Value::as_array) else { return Vec::new(); }; let mut rows = Vec::new(); for report in reports { + if crate::companion_usage::hidden( + settings.get("settings").unwrap_or(settings), + crate::companion_usage::text(report, "provider"), + ) { + continue; + } let provider_label = report .get("label") .or_else(|| report.get("provider")) @@ -131,12 +159,19 @@ mod macos { continue; }; let mut push = |percent: Option<&Value>, window_label: &str, reset: Option<&Value>| { + // JSON null (or any non-number) is an absent window, not a row of dashes: a + // weekly-only plan reports `fiveHourPercent: null` and must show weekly only. + let percent = number(percent).filter(|value| value.is_finite() && *value >= 0.0); + // Same bounds as the native panel's `reset`: after the millisecond conversion, + // a time past year 9999 is not a reset the widget can show. + let reset = + reset_at(reset).filter(|value| *value > 0.0 && *value < 253_402_300_800.0); if percent.is_some() || reset.is_some() { rows.push(Quota { provider_label: provider_label.clone(), window_label: window_label.to_owned(), - percent: number(percent), - reset_at: reset_at(reset), + percent, + reset_at: reset, }); } }; @@ -177,10 +212,9 @@ mod macos { .and_then(Value::as_str) .unwrap_or("line") .to_owned(); - let series = value - .get("series") - .and_then(Value::as_array)? - .iter() + let (rows, incomplete) = timeline_rows(value, settings)?; + let series = rows + .into_iter() .take(6) .filter_map(|item| { Some(Series { @@ -199,46 +233,10 @@ mod macos { bucket_seconds, style, series, + incomplete, }) } - fn timeline_query(settings: &Value) -> String { - let settings = settings.get("settings").unwrap_or(settings); - let get = |key: &str, fallback: &str| { - settings - .get(key) - .and_then(Value::as_str) - .unwrap_or(fallback) - .to_owned() - }; - let hours = settings - .get("chartHours") - .and_then(Value::as_i64) - .unwrap_or(24); - let bucket_minutes = settings - .get("bucketMinutes") - .and_then(Value::as_i64) - .unwrap_or(60); - let metric = get("tokenMetric", "total"); - let aggregation = get("aggregation", "sum"); - let grouping = get("chartGrouping", "model"); - let mut query = format!( - "hours={hours}&bucketMinutes={bucket_minutes}&metric={metric}&aggregation={aggregation}&grouping={grouping}" - ); - if let Some(models) = settings.get("models").and_then(Value::as_array) { - let models = models - .iter() - .filter_map(Value::as_str) - .collect::>() - .join(","); - if !models.is_empty() { - query.push_str("&models="); - query.push_str(&models); - } - } - query - } - fn snapshot_path() -> PathBuf { let home = std::env::var_os("HOME") .map(PathBuf::from) @@ -259,12 +257,57 @@ mod macos { snapshot } + /// What the widget displays apart from its age caption. `last_updated` moves on every + /// successful poll; reloading for it alone would spend WidgetKit's budget every five minutes + /// while the widget already renders that age as a self-updating relative date. + fn displayed(snapshot: &Snapshot) -> Snapshot { + let mut snapshot = without_generated_at(snapshot); + snapshot.last_updated = None; + snapshot + } + + /// Rewrite an unchanged snapshot after this long, so the widget can still tell a live app from + /// one that stopped writing. The widget marks a snapshot stale after two heartbeats + /// (`WidgetSnapshot.staleAfter` in app/Sources/MenuBarCore/WidgetSnapshot.swift). + const HEARTBEAT_SECONDS: f64 = 15.0 * 60.0; + + /// Whether `snapshot` should replace `previous` on disk. The heartbeat is measured from the + /// file's own `generated_at`, so a restarted app decides the same way as a running one. + /// Writing spends no WidgetKit budget, so the file always carries the latest poll time. + fn should_write(previous: Option<&Snapshot>, snapshot: &Snapshot) -> bool { + let Some(previous) = previous else { + return true; + }; + without_generated_at(previous) != without_generated_at(snapshot) + || snapshot.generated_at - previous.generated_at >= HEARTBEAT_SECONDS + } + + /// Minimum spacing between reload requests: at most 72 a day, inside the 40-70 Apple quotes + /// as a typical budget once the widget's own 30-minute fallback timeline is counted separately. + /// A change that lands inside the window is already on disk, and that fallback rereads it. + const RELOAD_INTERVAL_SECONDS: f64 = 20.0 * 60.0; + + /// Whether a snapshot that was just written should ask WidgetKit for a reload: only when what + /// the widget displays changed, and not sooner than `RELOAD_INTERVAL_SECONDS` after the last + /// request. Timestamp-only writes and heartbeats never reload. + fn should_reload( + previous: Option<&Snapshot>, + snapshot: &Snapshot, + last_reload: Option, + now: f64, + ) -> bool { + previous.is_none_or(|previous| displayed(previous) != displayed(snapshot)) + && last_reload.is_none_or(|last| now - last >= RELOAD_INTERVAL_SECONDS) + } + + static LAST_RELOAD: std::sync::Mutex> = std::sync::Mutex::new(None); + fn write_if_changed( path: &std::path::Path, previous: Option<&Snapshot>, snapshot: &Snapshot, ) -> std::io::Result { - if previous.map(without_generated_at).as_ref() == Some(&without_generated_at(snapshot)) { + if !should_write(previous, snapshot) { return Ok(false); } let Some(directory) = path.parent() else { @@ -286,6 +329,35 @@ mod macos { Ok(true) } + extern "C" { + fn ocx_widget_reload_timelines(); + } + + /// Persist the snapshot and, when it was actually written, ask WidgetKit to reload the widget. + fn publish(snapshot: &Snapshot) { + let path = snapshot_path(); + let previous = fs::read(&path) + .ok() + .and_then(|bytes| serde_json::from_slice::(&bytes).ok()); + match write_if_changed(&path, previous.as_ref(), snapshot) { + Ok(true) => { + let now = now_seconds(); + let mut last = LAST_RELOAD + .lock() + .unwrap_or_else(|poison| poison.into_inner()); + if should_reload(previous.as_ref(), snapshot, *last, now) { + *last = Some(now); + // SAFETY: a no-argument Swift export that only enqueues work on the main queue. + unsafe { ocx_widget_reload_timelines() }; + } + } + Ok(false) => {} + Err(error) => { + crate::logging::log_once("widget snapshot write failed", &error.to_string()) + } + } + } + fn make_snapshot( proxy: &ProxyClient, settings: &Value, @@ -304,7 +376,7 @@ mod macos { (!parts.is_empty()).then(|| parts.join(" · ")) }; let today_snapshot = today - .and_then(|value| value.get("summary").or(Some(value))) + .and_then(|value| crate::companion_usage::filtered_summary(value, settings)) .map(|summary| Today { requests: integer(summary.get("requests")), total_tokens: integer(summary.get("totalTokens")), @@ -321,7 +393,7 @@ mod macos { endpoint_display: format!("{}:{}", endpoint.host, endpoint.port), menu_title, today: today_snapshot, - quotas: quotas(quotas_value), + quotas: quotas(quotas_value, settings), chart, last_updated: timeline_value.map(|_| now_seconds()), generated_at: now_seconds(), @@ -351,13 +423,7 @@ mod macos { last_updated: None, generated_at: now_seconds(), }; - let path = snapshot_path(); - let previous = fs::read(&path) - .ok() - .and_then(|bytes| serde_json::from_slice::(&bytes).ok()); - if let Err(error) = write_if_changed(&path, previous.as_ref(), &snapshot) { - crate::logging::log_once("widget snapshot write failed", &error.to_string()); - } + publish(&snapshot); crate::logging::log_once("widget snapshot health failed", state); return; } @@ -377,13 +443,7 @@ mod macos { quota_value.as_ref(), timeline_value.as_ref(), ); - let path = snapshot_path(); - let previous = fs::read(&path) - .ok() - .and_then(|bytes| serde_json::from_slice::(&bytes).ok()); - if let Err(error) = write_if_changed(&path, previous.as_ref(), &snapshot) { - crate::logging::log_once("widget snapshot write failed", &error.to_string()); - } + publish(&snapshot); } pub fn refresh(proxy: &ProxyClient) { @@ -423,6 +483,7 @@ mod macos { id: "openai/gpt".into(), points: vec![1.0, 2.0], }], + incomplete: false, }), last_updated: Some(2.0), generated_at: 3.0, @@ -434,7 +495,7 @@ mod macos { } #[test] - fn error_state_mapping_covers_four_kinds() { + fn error_state_mapping_covers_every_kind() { assert_eq!( state_for_error(ErrorKind::Unreachable, None).0, "unreachable" @@ -451,6 +512,19 @@ mod macos { state_for_error(ErrorKind::Decode, Some("bad".into())).0, "degraded" ); + assert_eq!(state_for_error(ErrorKind::Foreign, None).0, "foreign"); + } + + #[test] + fn a_foreign_runtime_is_not_reported_as_a_failure() { + // The mapping is the whole point of the variant. Folding it into either neighbour + // tells a user whose own CLI or npm runtime holds the port that something is broken, + // and the widget is the one surface where that claim is read without any context. + assert_eq!(proxy_error(&ProxyError::Foreign).0, ErrorKind::Foreign); + let (state, title, detail) = state_for_error(ErrorKind::Foreign, None); + assert_eq!(state, "foreign"); + assert_eq!(title, "External runtime"); + assert!(detail.unwrap().contains("did not start")); } #[test] @@ -476,6 +550,107 @@ mod macos { let _ = fs::remove_file(path); } + #[test] + fn polls_are_written_but_only_visible_changes_reload_and_not_too_often() { + let previous = Snapshot { + schema_version: 1, + state: "running".into(), + state_title: "Running".into(), + detail: None, + endpoint_display: "127.0.0.1:10100".into(), + menu_title: Some("12".into()), + today: None, + quotas: Vec::new(), + chart: None, + last_updated: Some(1_000.0), + generated_at: 1_000.0, + }; + assert!(should_write(None, &previous), "first write"); + assert!( + should_reload(None, &previous, None, 1_000.0), + "first write reloads" + ); + // Same content and poll time: nothing to write before the heartbeat. + let mut idle = previous.clone(); + idle.generated_at = 1_300.0; + assert!(!should_write(Some(&previous), &idle)); + idle.generated_at = previous.generated_at + HEARTBEAT_SECONDS; + assert!(should_write(Some(&previous), &idle), "heartbeat rewrites"); + assert!( + !should_reload(Some(&previous), &idle, None, idle.generated_at), + "heartbeat never reloads" + ); + // A new poll time is written so the file's "Updated" is current, but costs no reload. + let mut polled = previous.clone(); + polled.generated_at = 1_300.0; + polled.last_updated = Some(1_300.0); + assert!(should_write(Some(&previous), &polled)); + assert!(!should_reload(Some(&previous), &polled, None, 1_300.0)); + // A visible change reloads, but not within the interval of the previous request. + let mut counted = polled.clone(); + counted.menu_title = Some("13".into()); + assert!(should_write(Some(&previous), &counted)); + assert!(should_reload(Some(&previous), &counted, None, 1_300.0)); + assert!(!should_reload( + Some(&previous), + &counted, + Some(1_300.0 - 60.0), + 1_300.0 + )); + assert!(should_reload( + Some(&previous), + &counted, + Some(1_300.0 - RELOAD_INTERVAL_SECONDS), + 1_300.0 + )); + } + + #[test] + fn a_failed_write_reports_an_error_instead_of_a_write() { + // The parent is a file, so the directory cannot be created: `publish` must see an + // error here and never reach the reload call. + let blocker = + std::env::temp_dir().join(format!("ocx-widget-blocker-{}", std::process::id())); + fs::write(&blocker, b"x").unwrap(); + let snapshot = Snapshot { + schema_version: 1, + state: "running".into(), + state_title: "Running".into(), + detail: None, + endpoint_display: "127.0.0.1:10100".into(), + menu_title: None, + today: None, + quotas: Vec::new(), + chart: None, + last_updated: None, + generated_at: 1.0, + }; + assert!(write_if_changed(&blocker.join("snapshot.json"), None, &snapshot).is_err()); + let _ = fs::remove_file(blocker); + } + + #[test] + fn quota_rows_skip_windows_the_plan_does_not_report() { + let reports = json!({ "reports": [ + { "provider": "openai", "label": "OpenAI", "quota": { + "fiveHourPercent": null, "fiveHourResetAt": null, + "weeklyPercent": 49.0, "weeklyResetAt": 1_900_000_000 } }, + { "provider": "kimi", "label": "Kimi", "quota": { + "fiveHourPercent": 0, "weeklyPercent": 35 } }, + // A reset-only window whose time is out of range is not a window either. + { "provider": "far", "label": "Far", "quota": { "weeklyResetAt": 1e20 } } + ] }); + let rows = quotas(&reports, &json!({ "settings": {} })); + let windows: Vec<_> = rows + .iter() + .map(|row| (row.provider_label.as_str(), row.window_label.as_str())) + .collect(); + assert_eq!( + windows, + [("OpenAI", "week"), ("Kimi", "5h"), ("Kimi", "week")] + ); + } + #[test] fn chart_series_are_truncated_to_six() { let series = (0..8) diff --git a/desktop/src-tauri/src/window.rs b/desktop/src-tauri/src/window.rs index f9550247ae..b5a90c8cbf 100644 --- a/desktop/src-tauri/src/window.rs +++ b/desktop/src-tauri/src/window.rs @@ -1,5 +1,6 @@ -use crate::{auth::Auth, discovery::ProxyEndpoint}; -use tauri::{AppHandle, Manager, Url, WebviewWindow, WindowEvent}; +use crate::{auth::Auth, exit, AppState}; +use tauri::webview::{NewWindowFeatures, NewWindowResponse}; +use tauri::{AppHandle, Manager, Runtime, Url, WebviewWindow, WindowEvent}; pub fn webview_user_agent() -> String { let platform = if cfg!(target_os = "macos") { @@ -12,44 +13,157 @@ pub fn webview_user_agent() -> String { format!("{platform} {}", Auth::user_agent()) } +/// Decide what closing this window means, at the moment it is closed. +/// +/// The answer is not known when the window is built: on Linux it depends on a session-bus probe +/// that the startup sequence runs afterwards. So it is read here rather than captured. With a tray +/// a close hides and the runtime keeps serving; without one there is nowhere to hide, so D6 makes +/// the close a quit — and it takes the same graceful drain the tray's Quit does. pub fn configure(window: &WebviewWindow) { let window_for_close = window.clone(); window.on_window_event(move |event| { if let WindowEvent::CloseRequested { api, .. } = event { api.prevent_close(); - let _ = window_for_close.hide(); - apply_tray_policy(window_for_close.app_handle(), false); + exit::gesture(window_for_close.app_handle()); } }); } -pub fn navigation_allowed(endpoint: ProxyEndpoint) -> impl Fn(&Url) -> bool { +/// Where this window may navigate. +/// +/// The loopback endpoint is read from the app rather than captured, because the window now exists +/// before anything has been resolved. Until it has, an http target is refused outright instead of +/// being handed to the browser: nothing should be navigating anywhere yet, and opening an +/// unresolved address in the user's browser is a worse answer than doing nothing. +pub fn navigation_allowed(app: AppHandle) -> impl Fn(&Url) -> bool { move |url| { - if url.scheme() == "tauri" { + if is_app_origin(url) { return true; } - if url.scheme() == "http" && url.host_str() == Some(endpoint.host) { - return url.port_or_known_default() == Some(endpoint.port); + if url.scheme() == "about" && url.as_str() == "about:blank" { + return true; } - if matches!(url.scheme(), "http" | "https") { - let _ = tauri_plugin_opener::open_url(url.as_str(), None::<&str>); - return false; + let endpoint = app + .try_state::() + .and_then(|state| state.proxy()) + .map(|proxy| proxy.endpoint()); + if let Some(endpoint) = endpoint { + if url.scheme() == "http" && url.host_str() == Some(endpoint.host) { + return url.port_or_known_default() == Some(endpoint.port); + } + open_in_default_browser(url); } - url.scheme() == "about" && url.as_str() == "about:blank" + false + } +} + +/// Whether a URL the dashboard asked for belongs in the user's default browser. +/// +/// Only web addresses leave the app. Anything else a page could name (`file:`, `javascript:`, +/// a custom scheme) has no business being handed to the OS launcher from a webview. +fn opens_in_default_browser(url: &Url) -> bool { + matches!(url.scheme(), "http" | "https") +} + +pub fn open_in_default_browser(url: &Url) { + if opens_in_default_browser(url) { + let _ = tauri_plugin_opener::open_url(url.as_str(), None::<&str>); + } +} + +/// What a `window.open` or `target="_blank"` link from a shell webview does. +/// +/// The shell never grows a second webview: every such request is answered in the default +/// browser and the in-app window is denied. Without this handler the pinned wry answers the +/// request itself, and on no platform does that reach a browser: WebView2 marks it handled and +/// drops it, WebKitGTK creates nothing, and WKWebView only gets there when its navigation policy +/// happens to see the URL first. That is how the dashboard's "didn't open? open the login page" +/// link, and the device-code logins that rely on it, did nothing in the app. +/// +/// It is also why the opener plugin's click interceptor is switched off in `lib.rs`: that script +/// cancels a `_blank` click and asks for `plugin:opener|open_url` over IPC, which the loopback +/// dashboard is not granted, so the click was consumed and nothing opened. Routing links here +/// instead keeps the decision in one Rust function and adds no IPC grant to a remote origin. +pub fn open_new_windows_in_default_browser( +) -> impl Fn(Url, NewWindowFeatures) -> NewWindowResponse + Send + 'static { + |url, _features| { + open_in_default_browser(&url); + NewWindowResponse::Deny } } +/// The bundled `frontendDist` origin. +/// +/// Tauri serves it as `tauri://localhost` on macOS and Linux, and as `http://tauri.localhost` on +/// Windows, where WebView2 has no custom-scheme support. Without that second spelling the window's +/// first navigation to its own page on Windows falls through to the branch that hands a URL to the +/// external browser. +/// +/// It is that one host and nothing near it. `https` is not the scheme the pinned Tauri serves the +/// app over, and a port means something else is answering rather than the app — neither localhost +/// generally, nor a name that merely ends in it, is this origin. +fn is_app_origin(url: &Url) -> bool { + match url.scheme() { + "tauri" => url.host_str() == Some("localhost") && url.port().is_none(), + "http" => url.host_str() == Some("tauri.localhost") && url.port().is_none(), + _ => false, + } +} + +pub fn require_update_page(window: &WebviewWindow) -> Result<(), String> { + if window.label() != "main" { + return Err("update page unavailable".into()); + } + let url = window.url().map_err(|_| "update page unavailable")?; + if !is_update_page_url(&url) { + return Err("update page unavailable".into()); + } + Ok(()) +} + +fn is_update_page_url(url: &Url) -> bool { + is_app_origin(url) && url.path() == "/update.html" +} + +/// Whether the window currently shows the bundled update page. An unreadable URL reads as not. +pub fn shows_update_page(window: &WebviewWindow) -> bool { + window.url().is_ok_and(|url| is_update_page_url(&url)) +} + pub fn show(window: &WebviewWindow) { let _ = window.show(); let _ = window.set_focus(); + report_visibility(window, true); apply_tray_policy(window.app_handle(), true); } pub fn hide(window: &WebviewWindow) { let _ = window.hide(); + report_visibility(window, false); apply_tray_policy(window.app_handle(), false); } +/// Tell the main window's page whether its host window is visible. +/// +/// Windows WebView2 does not flip `document.visibilityState` when the host window is hidden +/// (tauri issues #10592 and #6864), so the dashboard's pollers keep running while the app sits in +/// the tray; macOS WKWebView does flip it. Publishing the host's own answer gives the GUI one +/// signal on every platform instead of one that is correct on only some of them. +/// +/// Only the `main` window publishes: `exit::hide_windows` hides every window through `hide`, +/// and the tray popup carries its own equivalent bridge, so an unguarded report would claim the +/// dashboard was hidden because a popup was. A page that has not loaded yet simply misses the eval; +/// the page-load hook re-sends the current state. +pub fn report_visibility(window: &WebviewWindow, visible: bool) { + if window.label() != "main" { + return; + } + let script = format!( + "window.__OPENCODEX_HOST_VISIBLE__ = {visible}; window.dispatchEvent(new CustomEvent('opencodex:host-visibility', {{detail: {visible}}}));" + ); + let _ = window.eval(script); +} + #[cfg(target_os = "macos")] fn apply_tray_policy(app: &AppHandle, visible: bool) { let policy = if visible { @@ -70,7 +184,80 @@ pub fn set_tray_policy(app: &AppHandle, visible: bool) { #[cfg(test)] mod tests { - use super::webview_user_agent; + use super::{is_app_origin, is_update_page_url, opens_in_default_browser, webview_user_agent}; + use tauri::Url; + + fn url(value: &str) -> Url { + Url::parse(value).expect("a url") + } + + #[test] + fn only_web_addresses_are_handed_to_the_default_browser() { + for value in [ + "https://auth.openai.com/oauth/authorize?client_id=x", + "https://github.com/login/device", + "http://127.0.0.1:1455/auth/callback", + ] { + assert!(opens_in_default_browser(&url(value)), "{value}"); + } + for value in [ + "about:blank", + "file:///etc/passwd", + "javascript:alert(1)", + "tauri://localhost/index.html", + "mailto:someone@example.com", + ] { + assert!(!opens_in_default_browser(&url(value)), "{value}"); + } + } + + #[test] + fn the_app_origin_is_allowed_by_both_spellings_on_every_platform() { + // The custom scheme everywhere, and the http spelling WebView2 needs on Windows. The + // second is not gated on the platform: the origin is the app's wherever it is served. + assert!(is_app_origin(&url( + "tauri://localhost/index.html?port=10100" + ))); + assert!(is_app_origin(&url( + "http://tauri.localhost/index.html?port=10100" + ))); + } + + #[test] + fn nothing_near_that_origin_is_that_origin() { + for value in [ + // Not the scheme the pinned Tauri serves the app over. + "https://tauri.localhost/index.html", + // A port means something else is answering. + "http://tauri.localhost:8080/", + // Neither localhost generally nor a name that merely contains it. + "http://localhost/", + "http://127.0.0.1/", + "http://evil.tauri.localhost/", + "http://tauri.localhost.example.com/", + "file:///C:/index.html", + ] { + assert!(!is_app_origin(&url(value)), "{value}"); + } + } + + #[test] + fn only_the_bundled_update_page_has_update_commands() { + for value in [ + "tauri://localhost/update.html", + "http://tauri.localhost/update.html", + ] { + assert!(is_update_page_url(&url(value)), "{value}"); + } + for value in [ + "http://127.0.0.1:10100/update.html", + "tauri://evil/update.html", + "tauri://localhost/index.html", + "http://tauri.localhost/update.html.evil", + ] { + assert!(!is_update_page_url(&url(value)), "{value}"); + } + } #[test] fn webview_user_agent_marks_the_desktop_shell() { @@ -85,4 +272,86 @@ mod tests { assert!(user_agent.contains("(X11; Linux x86_64)")); } } + + /// The zoom polyfill runs inside the loopback dashboard, which is a remote origin to Tauri. The + /// capability that lets it call `set_webview_zoom` is the only one reaching that origin, so it + /// stays pinned to this window, this origin and this one command. + #[test] + fn the_dashboard_reaches_only_the_zoom_command() { + let zoom: serde_json::Value = + serde_json::from_str(include_str!("../capabilities/dashboard-zoom.json")) + .expect("dashboard-zoom capability is JSON"); + assert_eq!(zoom["windows"], serde_json::json!(["main"])); + assert_eq!( + zoom["remote"]["urls"], + serde_json::json!(["http://127.0.0.1:*"]) + ); + assert_eq!( + zoom["permissions"], + serde_json::json!(["core:webview:allow-set-webview-zoom"]) + ); + + // Tauri matches the origin with URLPattern; the dashboard is the loopback endpoint on + // whatever port it resolved to, and nothing beside it. + let pattern: tauri_utils::acl::RemoteUrlPattern = + "http://127.0.0.1:*".parse().expect("a URL pattern"); + let dashboard = crate::endpoint::ProxyEndpoint { + host: "127.0.0.1", + port: 10100, + } + .url("/#/usage"); + assert!(pattern.test(&url(&dashboard)), "{dashboard}"); + for value in [ + "http://localhost:10100/", + "https://127.0.0.1:10100/", + "http://127.0.0.2:10100/", + "http://example.com/", + ] { + assert!(!pattern.test(&url(value)), "{value}"); + } + + let default: serde_json::Value = + serde_json::from_str(include_str!("../capabilities/default.json")) + .expect("default capability is JSON"); + assert!(default.get("remote").is_none()); + } + + /// The overlay title bar is moved and zoomed from the page: the dashboard's top strips call + /// `plugin:window|start_dragging` and `plugin:window|toggle_maximize` from the loopback + /// origin, and the bundled pages do the same from the app origin. The dashboard also reads + /// the window's native scale so its traffic-light clearance survives page zoom. Grants are + /// pinned to `main`; the app origin needs only the two window commands. + #[test] + fn the_titlebar_commands_are_granted_on_each_origin() { + let titlebar: serde_json::Value = + serde_json::from_str(include_str!("../capabilities/dashboard-titlebar.json")) + .expect("dashboard-titlebar capability is JSON"); + assert_eq!(titlebar["windows"], serde_json::json!(["main"])); + assert_eq!( + titlebar["remote"]["urls"], + serde_json::json!(["http://127.0.0.1:*"]) + ); + assert_eq!( + titlebar["permissions"], + serde_json::json!([ + "core:window:allow-start-dragging", + "core:window:allow-toggle-maximize", + "core:window:allow-scale-factor" + ]) + ); + + let default: serde_json::Value = + serde_json::from_str(include_str!("../capabilities/default.json")) + .expect("default capability is JSON"); + let permissions = default["permissions"].as_array().expect("permissions list"); + for permission in [ + "core:window:allow-start-dragging", + "core:window:allow-toggle-maximize", + ] { + assert!( + permissions.contains(&serde_json::json!(permission)), + "default capability grants {permission}" + ); + } + } } diff --git a/desktop/src-tauri/tauri.conf.json b/desktop/src-tauri/tauri.conf.json index 6af54b5e1c..688678b4f6 100644 --- a/desktop/src-tauri/tauri.conf.json +++ b/desktop/src-tauri/tauri.conf.json @@ -1,13 +1,15 @@ { "$schema": "https://schema.tauri.app/config/2", "productName": "OpenCodex", - "version": "2.61.0", + "version": "2.76.0", "identifier": "com.opencodex.desktop", "build": { "frontendDist": "../ui", "devUrl": "http://localhost:1420" }, "app": { + "withGlobalTauri": true, + "macOSPrivateApi": true, "security": { "csp": "default-src 'self'; connect-src 'self' http://127.0.0.1:*; style-src 'self' 'unsafe-inline'; script-src 'self'" } @@ -20,7 +22,8 @@ "binaries/ocx" ], "resources": { - "resources/gui/dist": "gui/dist" + "resources/gui/dist": "gui/dist", + "resources/keyring": "keyring" }, "icon": [ "icons/icon.icns", @@ -29,6 +32,7 @@ ], "macOS": { "minimumSystemVersion": "13.0", + "entitlements": "Entitlements.plist", "files": { "PlugIns/OpenCodexWidget.appex": "widget/OpenCodexWidget.appex" }, diff --git a/desktop/ui/index.html b/desktop/ui/index.html index e4d5c9a7de..60eed6cf2d 100644 --- a/desktop/ui/index.html +++ b/desktop/ui/index.html @@ -5,25 +5,378 @@ OpenCodex -
-

OpenCodex

-

Connecting to the OpenCodex proxy…

- + +
+
+ +

OpenCodex

+
+ +

Starting OpenCodex…

+

+ +

+
+ Details +
    +
    + +
    - + + diff --git a/desktop/ui/logo.png b/desktop/ui/logo.png new file mode 100644 index 0000000000..894ad8ca71 Binary files /dev/null and b/desktop/ui/logo.png differ diff --git a/desktop/ui/main.js b/desktop/ui/main.js deleted file mode 100644 index 85c4ea8732..0000000000 --- a/desktop/ui/main.js +++ /dev/null @@ -1,34 +0,0 @@ -const params = new URLSearchParams(window.location.search); -const port = Number(params.get("port") || "10100"); -const origin = `http://127.0.0.1:${port}`; -const dashboardUrl = `${origin}/#/usage`; -const status = document.querySelector("#status"); -const retry = document.querySelector("#retry"); -let checking = false; - -async function check() { - if (checking) return; - checking = true; - status.textContent = `Connecting to OpenCodex proxy at 127.0.0.1:${port}…`; - retry.disabled = true; - try { - const response = await fetch(`${origin}/healthz`, { - cache: "no-store", - }); - if (response.ok) { - status.textContent = "Proxy is ready. Loading dashboard…"; - window.location.replace(dashboardUrl); - return; - } - throw new Error(`HTTP ${response.status}`); - } catch { - status.textContent = "The proxy is not reachable yet."; - } finally { - checking = false; - retry.disabled = false; - } -} - -retry.addEventListener("click", check); -check(); -setInterval(check, 1500); diff --git a/desktop/ui/update.html b/desktop/ui/update.html new file mode 100644 index 0000000000..22ca486c13 --- /dev/null +++ b/desktop/ui/update.html @@ -0,0 +1,144 @@ + + + + + + OpenCodex update + + + + +
    +

    OpenCodex update

    +

    Reading update status…

    + +
    + + + +
    +
    + + + diff --git a/devlog/_fin/260904_repo_hygiene_campaign/000_plan.md b/devlog/_fin/260904_repo_hygiene_campaign/000_plan.md index f7f030084a..a27c412b9e 100644 --- a/devlog/_fin/260904_repo_hygiene_campaign/000_plan.md +++ b/devlog/_fin/260904_repo_hygiene_campaign/000_plan.md @@ -17,15 +17,17 @@ every contributor whose work is carried. ## Classification of local branches -Every branch was scored on four independent axes rather than by name: +Every branch was scored on four independent axes rather than by name. Axis 3 is +shown in its corrected form; the campaign itself ran it without `--no-renames` +(see the 2026-09-21 correction in 010_method.md): 1. `git merge-base --is-ancestor
    origin/dev` — plain ancestry. 2. `git cherry origin/dev
    ` — patch-equivalence, which catches rebases. 3. Content landing — the files the branch touches - (`git diff --name-only origin/dev...
    `) are compared two-dot against - `origin/dev` restricted to exactly those paths. Zero remaining difference - means the branch's content is already on `dev` even though a squash merge - destroyed its commit identity. + (`git diff --no-renames --name-only origin/dev...
    `) are compared two-dot + against `origin/dev` restricted to exactly those paths. Zero remaining + difference means the branch's content is already on `dev` even though a + squash merge destroyed its commit identity. 4. Exact reference matching against live GitHub state: open-PR head refs, worktree-backing refs, and the PR number a scratch branch was cut for. diff --git a/devlog/_fin/260904_repo_hygiene_campaign/010_method.md b/devlog/_fin/260904_repo_hygiene_campaign/010_method.md index 773b6db649..c8e98858bc 100644 --- a/devlog/_fin/260904_repo_hygiene_campaign/010_method.md +++ b/devlog/_fin/260904_repo_hygiene_campaign/010_method.md @@ -7,8 +7,8 @@ A local branch is deletable when at least one holds, and no guard fires. ``` T1 ancestry git merge-base --is-ancestor
    origin/dev T2 patch-equiv git cherry origin/dev
    -> no '+' lines -T3 content paths = git diff --name-only origin/dev...
    - git diff --name-only origin/dev
    -- -> empty +T3 content paths = git diff --no-renames --name-only origin/dev...
    + git diff --no-renames --name-only origin/dev
    -- -> empty T4 scratch branch name encodes a PR number whose state is MERGED or CLOSED AND the name matches the scratch prefix set AND the number is a WHOLE numeric token of the branch name @@ -22,6 +22,17 @@ report "unmerged" for work that is fully shipped. T3 asks the only question that is actually load-bearing — is there any difference left in the files this branch claims to change. +Correction, 2026-09-21: the 71 deletions recorded below ran the listing command +without `--no-renames`. Rename detection must be disabled while collecting that +path set, and any rerun after this date should use the form shown above. +Otherwise a rename contributes only its destination: if `dev` independently +contains the same destination but retains the source, the restricted second +diff is empty even though the complete tip trees differ. `--no-renames` emits +both the deleted source and added destination, so the source-side difference +prevents a false LANDED verdict. No wrongly-LANDED branch has been identified +from the earlier run; this is a preventive correction for the next sweep, not a +measured incident. + T4 is deliberately narrow. It fires only for throwaway prefixes (`pr*`, `rb-`, `jrb-`, `mtp/`, `big-`, `cf-`, `ocx-`, `wip/`, `backup/`, `candidate`, `cursor-`, `midstream`) created by earlier review and rebase runs, diff --git a/devlog/_fin/260904_repo_hygiene_campaign/100_pr_verdicts.md b/devlog/_fin/260904_repo_hygiene_campaign/100_pr_verdicts.md index bf8e5d701c..3ff033a1c9 100644 --- a/devlog/_fin/260904_repo_hygiene_campaign/100_pr_verdicts.md +++ b/devlog/_fin/260904_repo_hygiene_campaign/100_pr_verdicts.md @@ -1,10 +1,12 @@ # 100 — Per-PR verdicts Full classification of the 53 pull requests open when the campaign started. -Method: fetch each PR head, take the files it touches -(`git diff --name-only origin/dev...`), then compare those exact paths -two-dot against `origin/dev`. Remaining differences mean the work has not -landed. +Method: fetch each PR head, take the files it touches, then compare those exact +paths two-dot against `origin/dev`. Remaining differences mean the work has not +landed. The verdicts below were produced with +`git diff --name-only origin/dev...`; any rerun must use +`git diff --no-renames --name-only origin/dev...` so a rename cannot hide +the deleted source side from the path set (2026-09-21; see 010_method.md). ## Closed diff --git a/devlog/_fin/260921_brand_icon_and_menu_bar_mark/000_plan.md b/devlog/_fin/260921_brand_icon_and_menu_bar_mark/000_plan.md new file mode 100644 index 0000000000..35d6b14a3b --- /dev/null +++ b/devlog/_fin/260921_brand_icon_and_menu_bar_mark/000_plan.md @@ -0,0 +1,95 @@ +# Brand icon and menu bar mark + +The shipped app icon is not the product's mark. `desktop/src-tauri/icons/*` derives from an empty +rounded-square ring that came in with the Tauri template, and the vector source written for it in +#5329 reproduced that ring faithfully — the measurement was right and the subject was wrong. + +Two consequences, both visible on a Mac today. The artwork covers 21.4% of the 1024px canvas and +has transparent corners, so macOS 26/27 classifies it as a uniquely shaped icon, strips it onto a +default grey tile and scales it down; Finder shows a grey square with a small black ring in it. +And the menu bar carries the same ring, so nothing on screen says which product this is. + +The real mark already exists in the repository. `assets/logo-light.png` is the mark on +transparency at 512px and `gui/public/favicon.png` is its app-icon composition at 128px: a light +squircle behind a dark six-lobed cloud that holds a `>` and a `_`, flanked by `{` and `}`, inside a +dashed orbit with a dot at top and bottom. Neither has a vector source, and `gui/src/icons.tsx` +holds only 24x24 line icons, so there is nothing to reuse — the vector has to be produced. + +## Where the geometry comes from + +The silhouette is measured, not redrawn. `assets/logo-light.png` is pure black with a shaped alpha +channel, so the outline is the alpha channel: upsample it 4x to 2048px, threshold at alpha 110, +trace with potrace, and map the result back into the 512-unit source space. That yields exactly +nine subpaths — cloud, two braces, four orbit arcs, two dots — and re-rendering them at 512px +disagrees with the thresholded source in **188 of 262144 pixels (0.072%)**, which is antialiasing +rather than a different shape. + +The prompt glyphs cannot be traced. In the source they are engraved: alpha 217+ against a 206 body, +with a lit rim along one edge. Composited at 512px that reads as depth; at 128px and below it reads +as nothing, and an app icon spends most of its life at 32px. Thresholding the emboss produces a +ragged chevron because the lit edge falls below the threshold asymmetrically. + +They are redrawn as flat geometry on the measured centreline instead: + +| glyph | measurement (512 source space) | drawn as | +| --- | --- | --- | +| `>` | rows 214-238 give the upper arm centreline slope 0.5625; rows 254-278 give the lower arm slope -0.5833; the two meet at (214.6, 244); tips at y 196.5 and 292.5 | polyline `193.4 206.5 -> 214.6 244 -> 193.4 281.5`, stroke 22, round cap and join | +| `_` | x 253-324.5, y 271-293.5, ends semicircular | rect 71 x 22.5, rx 11.25 at (253.5, 271) | + +Checked against the source: the chevron's predicted horizontal cross-section is 25.3px against 25px +measured, and the underscore's cap curvature lands within one pixel at both ends. + +## Shape of the change + +The glyphs are a **mask** rather than a lighter fill. Cutting them out of the mark makes the +backdrop show through, which is the flat reading of an engraved groove, and it is also what gives +the menu bar template real holes instead of a black blob. + +The backdrop is a **full-bleed opaque square**, not a pre-rounded tile. Apple's current app icon +guidance asks for a square, unmasked, full-bleed 1024px source and applies the rounded-rectangle +mask and material itself; a baked corner fights that and shows as jagged edges. The 824px inner +tile with a transparent margin is the pre-Tahoe recipe, and the transparent margin is precisely +what triggers today's grey fallback. + +Files: + +- `desktop/src-tauri/icons/icon.svg` — replaced. Full-bleed `#fcfcfc` backdrop, mark in `#2c2c2c`, + glyphs cut by `mask#prompt`, mark placed by `translate(2 26) scale(2)` so the orbit centre sits on + the canvas centre and the ink keeps the 77% coverage the favicon composition uses. +- `desktop/src-tauri/icons/tray/icon.svg` — new. Same curves, no backdrop, black fill, orbit and + dots dropped because at 22pt a dashed circle resolves into grey specks. viewBox is the ink bounds + of what is left plus 6%, so the glyph fills the menu bar height rather than the source margin. +- `desktop/scripts/generate-icons.ts` — `render()` takes a source, and the run emits + `tray/icon.png` at 44px (22pt at @2x) alongside the existing seventeen. Both `icons` and + `icons:check` cover it. +- `tests/ci-workflows/build-desktop-icon-set.test.ts` — two additions. The tray raster has to be the + size the generator declares and the generator has to actually render and report it, and the tray + source has to carry the app icon's mask verbatim, wire it onto the mark, and draw distinct curves + that all appear in `icon.svg`. + + Subset alone was too weak, and a review caught it: every interesting way of breaking the tray + removes something, so a strict subset stays a subset. Dropping the mask, deleting the underscore + or repeating a brace in place of the cloud each ship a black blob with green CI. Each of those, + plus removing the generator's tray render and removing its `produced.push`, was applied and run: + all five turn the suite red at 6 pass / 1 fail, and the restored tree is 7 pass / 0 fail. + +Nothing in `desktop/src-tauri/src/tray.rs` changes: it already builds the tray with +`.icon_as_template(true)`, and the asset it includes is the file being replaced. + +## Acceptance + +1. `cd desktop && bun run icons:check` reports every generated artifact matching the source, tray included. +2. `tests/ci-workflows/build-desktop-icon-set.test.ts` passes, and its drift guard fails when the + tray source is perturbed. +3. `icon.png` is fully opaque, and `tray/icon.png` is 44x44 with no non-black opaque pixel. +4. The change lands on `dev`, and a locally built and installed app shows the mark in Finder, the + Dock and the menu bar. + +## Recorded results + +- silhouette trace vs source alpha: 188 / 262144 px (0.072%). +- `icon.png` opaque coverage: 21.4% before, 100.0% after. +- `tray/icon.png`: 44x44, 759 pixels with alpha above zero — 498 fully opaque and 261 antialiased + — and no pixel with alpha whose colour is anything but black, which is what a template image has + to be. Both prompt glyphs are transparent holes rather than white fill. +- `cd desktop && bun run icons` regenerated 18 artifacts; `bun run icons:check` reported 18 matching. Both are desktop package scripts and fail from the repository root. diff --git a/devlog/_fin/260921_brand_icon_and_menu_bar_mark/010_favicons.md b/devlog/_fin/260921_brand_icon_and_menu_bar_mark/010_favicons.md new file mode 100644 index 0000000000..6e09e5c597 --- /dev/null +++ b/devlog/_fin/260921_brand_icon_and_menu_bar_mark/010_favicons.md @@ -0,0 +1,54 @@ +# The dashboard and documentation favicons + +The desktop app now renders its icon and its menu bar image from one traced vector. The two +favicons the product serves are still hand-made rasters with no source, and one of them is broken +in a way the file itself does not show. + +`docs-site/public/favicon.png` is the dark variant of the mark: white on transparency, 192px, +36225 of 36864 pixels carrying some alpha but every one of them RGB `(255,255,255)`, exactly one +fully opaque pixel, corner `(255,255,255,3)`. Composited on white it is a white square. A browser +tab strip is light by default and Starlight names the favicon unconditionally, so the documentation +site effectively has no favicon in light mode. `favicon.ico` beside it carries the same artwork at +16, 32 and 48, also with no fully opaque pixel. + +`gui/public/favicon.png` is the light composition and looks right at 128px, but it is a bitmap no +source can regenerate, and it is the shaded artwork rather than the flat mark: its engraved prompt +all but disappears at 16 and 32. Rendering the vector at 128 differs from it in 68% of pixels. That +is a visible simplification, not only a deduplication, and it is the same trade the app icon made. + +This unit covers the favicons and nothing else. Starlight's `logo-light.png` and `logo-dark.png` +stay independent 512px rasters — they are the brand artwork the vector was traced from, not +derived assets. `og.png` also stays, and carries a separate pre-existing defect worth its own +scope: `docs-site/astro.config.mjs` declares it 1200x630 while the committed file is 1536x1024. + +## Shape of the change + +- `scripts/lib/icon-render.ts` — new. The renderer, the RGBA re-encode and the ICO packer move here + out of `desktop/scripts/generate-icons.ts`, which keeps its size tables and imports them. Two + generators sharing one renderer is the point; a second copy of the PNG re-encode would be a + second place for the alpha bug to come back. +- `scripts/brand-favicons.ts` — new. Renders `desktop/src-tauri/icons/icon.svg` into + `gui/public/favicon.png` (128), `docs-site/public/favicon.png` (192) and + `docs-site/public/favicon.ico` (16, 32, 48) — the names, sizes and formats the two sites already + reference, so no page or config changes. `--check` regenerates into scratch and compares bytes, + the same contract the desktop set has. +- `package.json` — `favicons` and `favicons:check`. +- `tests/ci-workflows/brand-favicons.test.ts` — new, registered in `scripts/test-layout/layout.json` + and `tests/fixtures/test-layout-expected.json`. Asserts the declared sizes match what the two + sites ask for, that each committed favicon is that size with its alpha channel intact, that the + ICO carries exactly the declared sizes as embedded PNGs, and that both favicons read on a light + tab. + + That last one is why the test decodes pixels. An opaque corner alone is not enough: a plain white + square has an opaque corner and is still invisible. So it requires an opaque light corner **and** + at least 10% of the image to be opaque pixels whose luminance differs from that corner by more + than 64 — the mark actually being there. + +## Acceptance + +1. `bun run favicons:check` reports every favicon matching the source. +2. The new test fails on both ways of being invisible, checked by applying each: the + white-on-transparent artwork this replaces gives 3 pass / 1 fail, and a solid `#fcfcfc` square + gives 3 pass / 1 fail. The generated favicons give 4 pass / 0 fail. +3. `bun run privacy:scan`, `bun run structure:check` and the two test-layout guards stay green. + diff --git a/devlog/_fin/260921_brand_icon_and_menu_bar_mark/020_closure.md b/devlog/_fin/260921_brand_icon_and_menu_bar_mark/020_closure.md new file mode 100644 index 0000000000..858cadede9 --- /dev/null +++ b/devlog/_fin/260921_brand_icon_and_menu_bar_mark/020_closure.md @@ -0,0 +1,66 @@ +# Outcome + +Three changes landed on `dev`, in this order: + +| commit | pull request | change | +| --- | --- | --- | +| `a2d35a609e` | #5355 | the app icon and the menu bar mark, traced from the brand artwork | +| `5794348b0d` | #5356 | the alpha channel every generated icon needs | +| `917d690ecb` | #5361 | the dashboard and documentation favicons, from the same vector | + +## Verified on screen + +- **Finder and Dock.** Built locally with `bun run build:local`, signed with the Developer ID + identity the installed app already carried, installed to `/Applications` and relaunched. Finder + shows the mark on the system rounded rectangle: macOS masks the full-bleed square itself, which + is what the square, unmasked source is for. The widget extension still registers with + `pluginkit` under `com.opencodex.desktop.widget`, so replacing the bundle did not cost it. +- **Menu bar.** The status item renders the template mark with both prompt glyphs as holes, + tinted by macOS, next to the usage label. Both that and the Finder icon are recorded in + `assets/pr-screenshots/app-icon-finder-menubar.png`. +- **Browser tabs.** The two favicons were served over loopback HTTP and opened in a browser. Both + read as the mark on a light tile at tab size. This is the claim `010_favicons.md` makes, checked + in a tab rather than in a composite. `assets/pr-screenshots/favicon-browser-tab.png` is the tab + strip itself, not a rendering of one. + +## The defect the build caught + +`bun run build:local` failed on the first head with +`error: proc macro panicked ... icon .../icons/icon.png is not RGBA`. The new backdrop is opaque, +and librsvg drops the alpha channel when nothing in a render is transparent; `generate_context!` +rejects a window icon that is not RGBA. Fifteen of the sixteen rasters were affected — only the +tray image, which has real transparency, kept its alpha. + +Nothing in the repository could have seen it. The icon tests read dimensions and container +structure, and no test or hosted job builds the Tauri bundle. The generator now re-encodes, and +the colour type of every committed raster is asserted rather than trusted. + +## CI at the head + +`917d690ecb`, read at the exact SHA. Green: all four test shards, `gates`, `desktop shell`, +`macos 2/2`, `macos widget + bundle`, `docs site build`, `docker smoke`, `storage policy`, +`api usage`, all three `npm-global` legs, all three keyring legs, the three service legs. +Skipped, and named rather than counted: `macos control`, `structure gate`, the Windows shard +matrix placeholder. + +The first attempt had one failure, and it did not belong to this unit: `macos 1/2`, on +`tests/server/memory-watchdog.test.ts` -- +*serializes only an allowlisted Bun runtime provenance, omitting it otherwise (#848)* -- at 47.4s +against its own 20s timeout, with 13436 pass, 12 skip, 1 fail on that leg. That test makes eight +full `/api/system/memory` route calls and its own comment records the route costing roughly +600ms per read on shared runners, so it is timing fragile by construction. It passed at +`64b0eca2b0`, which already contained #5355 and #5356, and `917d690ecb` adds only favicon bytes +and a favicon generator. It also failed at `07e2ac9b41`, before any of this landed, and a focused +local run finishes in 377ms. + +Re-running that job at the same SHA turned it green, and the aggregate `ci` check with it, so +exact-head CI for `917d690ecb` is green. The flake is real and still there: the fix is to stop the +test paying for eight route snapshots, not to widen the timeout again. That is separate scope. + +## Left deliberately + +- `docs-site/src/assets/logo-light.png` and `logo-dark.png` stay as they are. They are the artwork + the vector was traced from, not derived assets. +- `og.png` stays, and carries a pre-existing mismatch worth its own scope: the configuration + declares 1200x630 and the committed file is 1536x1024. + diff --git a/devlog/_fin/260921_cross_path_contract/000_plan.md b/devlog/_fin/260921_cross_path_contract/000_plan.md new file mode 100644 index 0000000000..c9ab2d7f07 --- /dev/null +++ b/devlog/_fin/260921_cross_path_contract/000_plan.md @@ -0,0 +1,113 @@ +# Cross-path contract gaps before the next release + +Status: open. Target branch for every lane: `dev`. + +The batch that landed on 2026-09-20 fixed several defects one path at a time. The +audit that followed found the same shape repeating: a policy is correct where it +was written and absent one wrapper away, or a guard that protects a real hazard +also refuses the supported case. This unit closes that class before the release +rather than adding features. + +Each lane is one branch, ordered commits, and one pull request to `dev`. No +stacked child pull requests, no native stacks. A lane owns its files; where two +lanes touch the same subsystem the split is written below so the merge is a union +and not a conflict. + +## L1 — one replay refusal on all three HTTP surfaces + +`src/lib/upstream-retry.ts` answers an ambiguous connection loss with +`upstream_reset_replay_refused` and no `Retry-After`, which means "this may +already have executed, do not send it again". `src/server/chat-native.ts` and +`src/server/responses/passthrough-error.ts` recognise that. The translated Chat +wrapper in `src/server/chat-completions.ts` does not: it preserves only the cyber +policy code and `model_not_found`, assigns `upstreamCode` just when +`classifyError` produced no code, and then adds a default `Retry-After: 2`. A +refusal to replay leaves the proxy as an ordinary rate limit that clients retry. + +Carry the replay verdict as a property of the result the three wrappers share, so +no wrapper re-derives it from a status code. Then decide the status deliberately: +the widely used Python SDK retries 429 by default, so preserving the code while +dropping `Retry-After` does not by itself stop a resend. The acceptance evidence +is the number of physical upstream sends observed through a client with retries +enabled, not a single `fetch`. + +## L2 — the Chat translation inbound loses developer position + +Outbound keeps a `developer` message where the conversation put it +(`src/adapters/openai-chat/messages.ts`). Inbound does not: +`src/chat/inbound.ts` routes both `system` and `developer` into +`systemParts` and joins them into `body.instructions`, so +`U1 → A1 → D2 → U2` becomes `instructions: D2` with `U1 → A1 → U2`. Position +is gone before any adapter sees it, and no outbound fix can restore it. + +This is not a rare internal path. Combo, policy, synthetic effort rows and several +preprocessing routes translate, so the same transcript behaves differently once a +routing feature is on. The Claude inbound already models this correctly by keeping +a mid-conversation instruction as a developer input item +(`src/claude/inbound.ts`, `src/responses/parser.ts`). Reuse that representation +for the mid-conversation case only; a leading system block keeps its current +treatment. + +## L3 — an explicit developer-role setting is ignored natively + +`foldDeveloperRoleToSystem` decides the role on the translated path. The native +Chat passthrough (`src/adapters/openai-chat/passthrough.ts`) forwards the +caller's `messages` untouched and never reads it, so an operator who recorded +"this destination rejects `developer`" still sends `developer` there. Honour the +explicit setting on both paths and leave the unset default alone: the existing +native test that preserves caller messages stays green. + +## L4 — the paginated-history transition, past "enable succeeded" + +The provider-table transition on a paginated `openai` home now completes. The +remaining risk is the state after it. Acceptance is destination preservation, not +a successful sync: existing conversations must not resume against the default +OpenAI endpoint, new conversations must use the injected provider and catalog, +restore must return operator-owned settings and remove only what this project +owns, a user-owned root override must not be taken over, and an admission-token +home must still be refused rather than reported as supported. + +## L5 — tool constraints survive response repair + +`createGrokResponsesSparseTerminalBlockRewrite` rebuilds a terminal output from +collected `output_item.done` events and receives a budget but not this request's +tool selection. The undeclared-tool guard answers a different question — whether a +name was declared — so a request with `tool_choice: none` or a narrowed allow-list +can still receive a call the repair put back. Pass the request scope into the +repair and enforce it there. Keep the failure narrow: one forbidden call must not +discard the ordinary text that accompanied it. The empty-catalog case belongs to +the same rule — compatibility is judged on the final request and the final +response, after every removal, rename and translation. + +## L6 — a client integration that writes a store nobody reads + +A newer client release reads its provider list from a different file than the one +this exporter writes, and the legacy import does not run again once the new file +exists, so an apply that reports success produces no models. Support the store the +running client actually reads, including catalog refresh and disable, or report +the write as ineffective. Deleting the new file to re-trigger a migration is not a +supported remedy. The verification unit is "the client requests the intended +provider", not "the file was written". + +## L7 — one developer-role policy in both documents and the code + +`structure/providers/chat-compat.md` states the role is forwarded as itself on +every destination; `docs-site` states an unset setting sends `system`; the two +code paths differ again. Whoever fixes this area next picks one of them and +reintroduces the regression. Make the three agree after L2 and L3 settle, and +derive the statement from the code where a test can hold it. + +## L8 — one resend budget per logical request + +The ambiguous-resend gate landed with one operator grant per request. The +composition still needs evidence: first send, reset, replacement, disconnect after +`response.created`, then the combo candidate, credential refresh and 429 legs. +Observe two separate numbers — physical sends, and sends of a turn that may already +have executed. The neighbouring retry issues are not closed by this lane and stay +open with their remaining scope recorded. + +## Out of scope + +A lenient finish for a text-only stream with no terminal event is existing +compatibility behaviour with its own regression coverage. Turning every EOF into an +error would be a policy change, not a fix, and is not part of this unit. diff --git a/devlog/_fin/260921_cross_path_contract/010_lane_boundaries.md b/devlog/_fin/260921_cross_path_contract/010_lane_boundaries.md new file mode 100644 index 0000000000..ae2d37c9f3 --- /dev/null +++ b/devlog/_fin/260921_cross_path_contract/010_lane_boundaries.md @@ -0,0 +1,70 @@ +# Lane ownership, ordering and acceptance shape + +The first audit round of `000_plan.md` returned blocking findings: the lanes were +described by symptom without an owned-file set, two lanes overlapped on the send +path, one lane depended on two others without saying so, and two lanes named an +acceptance unit that no test can observe. This document answers those and is the +binding half of the unit. + +## Owned files + +A lane changes files in its own row. A file in another row is read-only for it. +Anything outside every row is open, but a second lane touching it has to say so in +its pull request. + +| Lane | Owns | +|---|---| +| L1 | `src/server/chat-completions.ts` error path, `src/server/chat-native.ts` error path, `src/server/responses/passthrough-error.ts`, the replay-verdict carrier it extracts, and tests for those | +| L2 | `src/chat/inbound.ts`, `src/responses/parser.ts` where the Chat path needs it, and its own tests | +| L3 | `src/adapters/openai-chat/passthrough.ts`, `src/adapters/openai-chat/messages.ts` role selection, and its own tests | +| L4 | `src/codex/history-provider.ts`, `src/codex/inject.ts`, `tests/codex-integration/*` | +| L5 | `src/server/grok-responses-snapshot-repair.ts`, `src/server/responses-undeclared-tool-guard.ts`, `src/server/responses/passthrough-dispatch.ts` call sites, and its own tests | +| L6 | `src/clients/config-export/`, `src/integrations/registry.ts` entry for that client, and its own tests | +| L7 | `structure/providers/chat-compat.md`, `docs-site` provider reference, and the generated binding check | +| L8 | `src/lib/request-execution-budget.ts`, `src/lib/request-resend-gate.ts`, and send-count tests | + +L1 and L8 both live near the send path and are split by question. L1 owns what the +client is told when a replay is refused — code, status, retry header, and the +carrier that stops each wrapper re-deriving it. L8 owns how many sends one logical +request may make and which leg may spend the shared reserve. L8 does not change an +error body; L1 does not change an allowance. + +## Ordering + +L7 lands last. It writes down the single developer-role policy, and that policy is +not settled until L2 fixes where the message sits and L3 fixes which role it +carries. Until both are on `dev`, L7 keeps its branch rebased and its pull request +open. Every other lane is independent and merges in whatever order its evidence +arrives. + +## Acceptance that a test can hold + +Static reading is how a lane reviews itself; hosted CI on the exact head is what +decides. A lane whose acceptance sentence names something no job can observe has +to restate it: + +- L1 and L8 count sends against a recorded fetch, so the number is an assertion and + not an inference. L1 additionally asserts the response the client receives. +- L4 drives a temporary home with fixtures: enable, create, resume, restore. It + asserts the destination recorded for an existing conversation and for a new one, + and it asserts the refusal that an admission-token home still receives. The + refusal path already has coverage; the transition and the post-transition + destinations are the new part. +- L6 cannot prove what a third-party client does at runtime. Its assertion is that + the file the current client release reads carries the intended provider after + enable and refresh, and carries nothing after disable — with the client's own + published schema quoted in the pull request as the reason that file is the one + that matters. If the lane cannot establish the schema, it reports the write as + ineffective instead, which is the honest half of the original instruction. +- L5 asserts on the block rewrite directly: `tool_choice: none`, a narrowed + allow-list, an empty catalog after normalisation, and a forbidden call arriving + beside ordinary text, which must survive. +- L7 asserts that the documented default is derived from the code, so a default + change fails a check rather than only a review. + +## Baseline + +The lanes are cut from `dev` after the September 20 batch, so the paginated +transition and the per-request resend gate are already present. A lane that finds +its premise already satisfied says so in its pull request and narrows to the part +that is not, rather than reimplementing what landed. diff --git a/devlog/_fin/260921_cross_path_contract/090_outcome.md b/devlog/_fin/260921_cross_path_contract/090_outcome.md new file mode 100644 index 0000000000..250d3754a8 --- /dev/null +++ b/devlog/_fin/260921_cross_path_contract/090_outcome.md @@ -0,0 +1,43 @@ +# Outcome + +All eight lanes are on `dev`. The unit closes here; what each lane actually +changed is below, with the parts that were narrowed or left open named rather +than implied. + +| Lane | Landed as | What it changed | +|---|---|---| +| L1 | `556b670251` | The ambiguous-resend refusal now carries the same code and the same retry policy on the translated Chat wrapper, the native Chat route and Responses, from one shared verdict instead of three re-derivations from a status code | +| L2 | `0ab648b4e3` | A mid-conversation instruction keeps its slot through the Chat translation inbound instead of being folded into `instructions`; a leading system block is unchanged | +| L3 | `6e0e912ebd` | A recorded `foldDeveloperRoleToSystem` decides the role on the native Chat route as well; the unset default and the caller-message preservation test are untouched | +| L4 | `a6e8df4ec1` | The paginated-history transition is judged by conversation destination — enable, new conversation, resume, restore — rather than by a successful sync, and an admission-token home is still refused | +| L5 | `b20acc79d2` | The request's tool selection is enforced after a sparse-terminal repair, so a forbidden call cannot re-enter the terminal output, and ordinary text beside it survives | +| L6 | `5d3f5db84a` | The integration writes the provider store the client actually reads, with enable, refresh and disable symmetric, and reports an ineffective write instead of success when the store's schema is not one it knows | +| L7 | `34332bb785` | One developer-role policy across the contract document, the configuration reference and its translations, derived from the adapter so a changed default fails a check | +| L8 | `3a0718c81c` | One resend allowance per logical request across the composed recovery legs, with the physical send count and the possibly-executed send count observed separately | + +## What this unit did not close + +`#5348` is closed because its acceptance is on `dev`. The neighbouring retry +issues about a WebSocket failure stage and single-key 429 handling are not +resolved by L8 and stay open with their remaining scope recorded on the issues +themselves. + +## What the audit round changed + +The first reviewer pass returned blocking findings and two mistaken ones. The +mistaken pair read a stale checkout and reported L4's refusal and L8's gate as +still absent; both had landed the day before, so those lanes narrowed to the +part that was missing instead of reimplementing what was there. The real +findings — no owned-file set, L1 and L8 overlapping on the send path, L7 +depending on L2 and L3 without an order, and two acceptance sentences no job +could observe — are answered in `010_lane_boundaries.md` and were followed. + +## Friction worth remembering + +Two lanes conflicted on `scripts/test-layout/layout.json` and +`tests/fixtures/test-layout-expected.json` because each appended its own +registration line. The resolution is always the union; the file is a registry, +not a narrative. One lane's own tests failed for the two familiar reasons: a +test asserting a field name the type does not carry, and a fixture that set an +unbound fingerprint beside `canApply`, which the parser refuses by contract. +Both were fixed by deriving from the source rather than restating it. diff --git a/devlog/_fin/260921_cross_path_contract/100_verification.md b/devlog/_fin/260921_cross_path_contract/100_verification.md new file mode 100644 index 0000000000..535cefe502 --- /dev/null +++ b/devlog/_fin/260921_cross_path_contract/100_verification.md @@ -0,0 +1,48 @@ +# Verification on the landed tip + +The eight lanes were re-read on `dev` after they landed, from source rather +than from commit messages, to check that each acceptance is actually present. +Six hold as written. Three carry a boundary that the plan did not state, and +they are written down here because each of them is the kind of thing a later +reader would otherwise discover as a surprise. + +**The replay refusal is identical on all three surfaces.** The translated Chat +wrapper, the native Chat route and the Responses error path each preserve +`upstream_reset_replay_refused`, drop `Retry-After`, and send +`x-should-retry: false`. That last header is what makes the intent legible to a +client whose default is to retry a 429. + +**The Chat inbound keeps a mid-conversation instruction in place, with one +ordering rule.** Leading system text still becomes `instructions`. A developer +message that arrives inside an open tool-call batch is held until the batch +closes rather than being spliced between a call and its result, because a +transcript that interleaves them is not one any destination accepts. Outside a +batch the message stays exactly where it arrived. + +**The native Chat route honours an explicit setting only.** `true` rewrites the +developer role to system; `false` and unset return the caller's messages +untouched, which keeps the passthrough contract that route exists for. + +**The repair enforces the request's tool selection on what it reconstructs.** +The rebuilt terminal output excludes a forbidden call and keeps the ordinary +text that arrived beside it. Raw `output_item.done` blocks are not rewritten by +the repair — policing the raw stream belongs to the undeclared-tool guard, and +splitting it that way keeps one owner per question. + +**The client store is shared by enable, refresh and disable.** An unknown schema +is reported rather than merged into. When this project's own block is still in +the legacy file, refresh refuses and disable cleans that file first: the +alternative is a block left in one file while another is written, which is the +state that made the original report hard to diagnose. + +**The resend grant is one shared ledger entry.** Derived and combo budgets +inherit it rather than opening their own, so composing recovery legs cannot +multiply the allowance. The ceiling itself stays operator-configurable; what is +fixed is that there is one of it per logical request. + +## Issue dispositions + +`#5348` is closed: its acceptance is on `dev`. `#4191` and `#5180` stay open +with their remaining scope recorded on the issues — a dead mid-turn transport +still has no fallback, and a provider 429 on a single key still has no cooldown +policy. Neither is what a per-request resend budget decides. diff --git a/devlog/_fin/260921_jev_auto_routing/010_plan.md b/devlog/_fin/260921_jev_auto_routing/010_plan.md new file mode 100644 index 0000000000..bd63a81fb7 --- /dev/null +++ b/devlog/_fin/260921_jev_auto_routing/010_plan.md @@ -0,0 +1,256 @@ +# JEV Auto Routing Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Add one optional `jev-auto` model that asks TypeSafe JEV to choose the initial OpenCodex Combo target and reasoning effort, while preserving every existing model and all existing Combo fallback behavior. + +**Architecture:** Extend Combo with a `jev` strategy. A small native TypeScript decision module builds the bounded JEV state and joint target/effort question, calls the fixed TypeSafe endpoint with the configured JEV credential, validates the answer, and returns either an eligible initial pick or a deterministic fail-open pick. The existing Combo dispatcher remains responsible for eligibility, cooldowns, quota state, concrete routing, retries, and subsequent fallback attempts. A registry-only JEV provider row owns key setup without publishing a routable model. The GUI adds JEV to the existing Combo editor and provides a prefilled `jev-auto` action whose target list remains fully editable. + +**Tech Stack:** Bun, TypeScript, OpenCodex Combo runtime, provider registry/management API, React/Vite GUI, Bun test runner. + +**Spec:** `docs/superpowers/specs/2026-09-21-jev-auto-routing-design.md` + +## Global Constraints + +- Existing public model ids, aliases, picker rows, defaults, and direct routing must remain unchanged. +- `jev-auto` is opt-in and is never synthesized until the operator creates the JEV Combo. +- JEV chooses once per logical model call. Existing Combo logic alone owns later failover. +- Candidate models come only from the configured Combo target allowlist and must pass existing eligibility checks before they are offered to JEV. +- Missing credentials, timeout, redirect, non-2xx, malformed JSON, invalid choices, and empty usable candidate sets fail open to the first existing eligible Combo pick. +- Caller cancellation propagates; it must not be converted into fail-open dispatch. +- TypeSafe calls use `https://api.typesafe.ai/v1/systemone`, model `jev-latest`, a four-second deadline, manual redirect handling, one attempt, and a bounded response body. +- JEV request state is bounded and excludes secrets, raw images, tool arguments, headers, encrypted reasoning, and full conversation history. +- Observability may contain only the selected target, effort, gate/reason, latency, confidence/probability, and numeric usage. It must never contain the JEV key or decision state. +- All TypeSafe coverage is mocked. A live smoke is explicitly deferred until the user supplies a key. +- New test files must be registered in `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`. +- Visible GUI strings must be added to every locale in `gui/src/i18n/`. + +## Review Focus + +- A hostile JEV response cannot select a target or effort outside the eligible, configured choice map. +- A caller abort during the JEV call ends the request as cancellation and never dispatches the fail-open target. +- A JEV outage cannot suppress the request or alter existing direct-model routing. +- The selected effort is clamped/omitted through the existing target capability ladder and caller `service_tier` is removed for the JEV-selected initial child only; fallback children rebuild from the original request under ordinary Combo rules. +- The JEV provider row stores credentials but emits no direct model/catalog row and can never be selected as a Combo target. +- After a JEV-selected target fails retryably, existing cooldown and fallback ordering continue without a second JEV call. + +--- + +## Task 1: Add the JEV Combo strategy and pure decision contract + +**Files:** + +- Create: `src/combos/jev.ts` +- Modify: `src/types/config.ts` +- Modify: `src/combos/types.ts` +- Modify: `src/combos/index.ts` +- Modify: `src/cli/combo.ts` +- Modify: `tests/codex-integration/combos.test.ts` +- Modify: `tests/cli/cli-headless-parity.test.ts` +- Create: `tests/routing/jev-decision.test.ts` +- Modify: `scripts/test-layout/layout.json` +- Modify: `tests/fixtures/test-layout-expected.json` + +**Interfaces produced:** + +```ts +export interface JevCandidate { + key: string; + provider: string; + model: string; + reasoningEfforts: readonly OcxComboDefaultEffort[]; +} + +export interface JevDecision { + targetKey: string; + effort: OcxComboDefaultEffort | null; + gate: "apply" | "missing_key" | "no_choices" | "timeout" | "network" | "redirect" | "http" | "malformed" | "invalid"; + latencyMs: number; + confidence?: number; + chosenProbability?: number; + usage?: Record; +} + +export function buildJevState(body: unknown): Record; +export function buildJevRouteQuestion(candidates: readonly JevCandidate[]): Record; +export function parseJevDecision(payload: unknown, candidates: readonly JevCandidate[]): Pick; +``` + +- [ ] Add focused failing Combo and CLI tests proving `strategy: "jev"` validates, normalizes, round-trips, falls back to configured order in the synchronous picker, and is accepted by `ocx combo set`. Run `bun test tests/codex-integration/combos.test.ts tests/cli/cli-headless-parity.test.ts`; expect assertions to fail because `jev` is rejected or normalized to `failover`. +- [ ] Extend `OcxComboStrategy`, validation text, normalization, Combo exports, and CLI `--strategy` parsing/help with `jev`. Re-run the focused test; expect it to pass. +- [ ] Add failing pure tests for bounded current-user extraction, envelope removal, recent assistant intent, last tool-output tail/name, image presence, literal choice-map construction, known Luna/Sol/Astra profiles, neutral arbitrary-target profiles, valid response parsing, complete probability validation, invalid/out-of-allowlist choices, malformed confidence, and numeric-only usage extraction. Run `bun test tests/routing/jev-decision.test.ts`; expect an import failure because `src/combos/jev.ts` does not exist. +- [ ] Implement only the pure state/question/parser pieces in `src/combos/jev.ts`. Keep state caps aligned with the reference router: 500-character head/tail current ask, 240-character assistant tail, and 520-character tool-output tail. Re-run the new tests; expect all to pass. +- [ ] Register the test file in both test-layout manifests, run `bun test tests/test-layout.test.ts tests/test-layout-tooling.test.ts`, then run `bun run typecheck`. Expect exit code 0. +- [ ] Stage the Task 1 files and commit with `git commit -m "feat: add JEV combo decision contract"`. + +## Task 2: Add registry-backed JEV credential setup and the secure TypeSafe client + +**Files:** + +- Modify: `src/providers/registry/entries-extended.ts` +- Modify: `src/providers/derive.ts` only if the empty-model decision-service row needs a narrow projection adjustment +- Modify: `src/combos/jev.ts` +- Modify: `src/server/management/provider-routes.ts` +- Create: `tests/providers/jev-provider.test.ts` +- Modify: `tests/server/management-provider-validation.test.ts` +- Modify: `tests/providers/provider-registry-parity.test.ts` +- Modify: `scripts/test-layout/layout.json` +- Modify: `tests/fixtures/test-layout-expected.json` + +**Interfaces produced:** + +```ts +export const JEV_PROVIDER_ID = "jev"; +export const JEV_API_URL = "https://api.typesafe.ai/v1/systemone"; +export const JEV_MODEL = "jev-latest"; + +export interface ResolveJevDecisionOptions { + body: unknown; + candidates: readonly JevCandidate[]; + fallback: { targetKey: string; effort: OcxComboDefaultEffort | null }; + config: OcxConfig; + signal?: AbortSignal; + post?: typeof providerOutboundPost; + now?: () => number; +} + +export function resolveJevDecision(options: ResolveJevDecisionOptions): Promise; +``` + +- [ ] Add a failing provider test proving the registry exposes a paid key-auth `jev` preset with the fixed endpoint, no models/default model, and `liveModels: false`; prove `fetchProviderModelsWithAuth` emits no JEV catalog row. Run `bun test tests/providers/jev-provider.test.ts`; expect no preset. +- [ ] Add the `jev` registry entry (`adapter: "jev-decision"`, `preserveCustomDestination: true`, TypeSafe dashboard/docs URL, no model roster) and make only the minimum projection adjustment required. Re-run the provider test; expect it to pass. +- [ ] Add failing client tests using an injected POST boundary. Cover configured key, `${TYPESAFE_API_KEY}`/environment fallback, exact endpoint/model/auth headers/body, four-second timeout, manual redirect refusal, non-2xx, oversized body, invalid JSON, invalid decision, and caller cancellation. Assert returned decisions rather than mock call counts except where endpoint/auth/body are the contract. Run `bun test tests/routing/jev-decision.test.ts`; expect client cases to fail because `resolveJevDecision` is absent. +- [ ] Implement the secure client with `resolveProviderApiKey`, environment fallback, `providerOutboundPost`, `providerRedirectError`, `readBoundedResponseBytes`, `AbortSignal.timeout(4000)`, and one request only. Re-run the client tests; expect all to pass. +- [ ] Add a failing management test for `POST /api/providers/test?name=jev`: missing key returns a sanitized failure; a mocked valid one-choice JEV answer returns connected; upstream body text is never echoed. Run `bun test tests/server/management-provider-validation.test.ts`; expect the static-catalog not-applicable result. +- [ ] Add the narrow JEV connection-probe branch before the generic static-catalog branch and reuse the same bounded client. Re-run the management test, provider test, layout tests, and `bun run typecheck`; expect exit code 0. +- [ ] Stage the Task 2 files and commit with `git commit -m "feat: add TypeSafe JEV provider setup"`. + +## Task 3: Route Combo first picks through JEV without replacing fallback + +**Files:** + +- Modify: `src/combos/jev.ts` +- Modify: `src/server/responses/core-combo.ts` +- Modify: `src/server/responses/core-options.ts` +- Create: `tests/server/server-jev-combo-e2e.test.ts` +- Modify: `scripts/test-layout/layout.json` +- Modify: `tests/fixtures/test-layout-expected.json` + +**Interfaces consumed:** Task 1's strict choice map/parser and Task 2's `resolveJevDecision` client. + +- [ ] Add a failing server test that configures an aliased `jev-auto` Combo, injects a successful JEV answer selecting the second target at `high`, and proves only that target receives the request, with forced/clamped `reasoning.effort` and no caller `service_tier`. Assert the served catalog has one public `jev-auto` row and still contains unchanged direct-model rows. Run `bun test tests/server/server-jev-combo-e2e.test.ts`; expect the first configured target to receive the request. +- [ ] Add a small helper that enumerates currently eligible `jev` targets in configured order without marking them all attempted, asks JEV once, and rebuilds the selected `ComboPick` with only the chosen target in `attempted`. Integrate it immediately after the existing initial `pickWithWait`; keep the loop and `advanceComboAfterFailure` unchanged. Re-run the focused test; expect it to pass. +- [ ] Add failing cases for: missing key fail-open to first eligible at medium; invalid JEV choice fail-open; selected target retryable failure then existing fallback with no second JEV call and with the original caller effort/tier restored; cooled/disabled targets omitted from choices; explicit empty target effort ladder omitted/stripped; caller abort during JEV returns 499 and sends no model request. Run the focused test and inspect each expected failure. +- [ ] Implement the minimum runtime behavior for those cases. Apply the JEV effort and remove `service_tier` only on the selected initial child. If that child fails, rebuild every fallback from the untouched original request with the Combo's ordinary effort/tier behavior. Emit one sanitized structured debug event for the decision. Re-run the focused test plus `bun test tests/routing/combo-management-api.test.ts tests/codex-integration/combos.test.ts`; expect all to pass. +- [ ] Register the new test file, run layout tests and `bun run typecheck`; expect exit code 0. +- [ ] Stage the Task 3 files and commit with `git commit -m "feat: route jev-auto through combo runtime"`. + +## Task 4: Add the editable JEV Auto GUI flow + +**Files:** + +- Modify: `gui/src/combo-workspace-data.ts` +- Modify: `gui/src/components/combo-workspace-controls.tsx` +- Modify: `gui/src/components/combo-workspace-add-modal.tsx` +- Modify: `gui/src/components/ComboWorkspace.tsx` +- Modify: `gui/src/components/combo-workspace-types.ts` +- Modify: `gui/src/pages/Combos.tsx` only if the prefilled-add state belongs at the page boundary +- Modify: `gui/src/components/provider-workspace/ProviderOverview.tsx` +- Modify: `gui/src/components/provider-workspace/ProviderDetails.tsx` +- Modify: `gui/src/pages/Providers.tsx` +- Modify: `gui/src/hash-routing.ts` +- Modify: `gui/src/pages/models-tab.ts` +- Modify: `gui/src/i18n/en.ts` +- Modify: `gui/src/i18n/de.ts` +- Modify: `gui/src/i18n/fr.ts` +- Modify: `gui/src/i18n/ja.ts` +- Modify: `gui/src/i18n/ko.ts` +- Modify: `gui/src/i18n/ru.ts` +- Modify: `gui/src/i18n/tr.ts` +- Modify: `gui/src/i18n/vi.ts` +- Modify: `gui/src/i18n/zh.ts` +- Modify: `gui/src/i18n/zh-TW.ts` +- Modify: `tests/gui/combo-workspace-data.test.ts` +- Create: `gui/tests/jev-auto-combo.test.tsx` + +**Interfaces produced:** + +```ts +export function jevAutoDraft(models: readonly ModelOption[]): ComboItem; +``` + +- [ ] Add failing pure GUI tests proving `jev` parses/serializes without drift and `jevAutoDraft` creates id/alias `jev-auto`, strategy `jev`, adaptive effort mode, and available Astra/Sol/Luna targets in fail-open order Astra → Sol → Luna while leaving the target list editable. Run `bun test tests/gui/combo-workspace-data.test.ts`; expect missing strategy/template failures. +- [ ] Implement the GUI strategy records and pure template builder. Re-run the pure tests; expect them to pass. +- [ ] Add a failing component test proving both the Combo workspace and configured JEV provider overview expose `Create JEV Auto`; the provider action deep-links into the same prefilled add modal. Prove the modal lets the user add/remove/change targets and submits the normal `PUT /api/combos` shape. Also prove the action is disabled or clearly reports a collision when `jev-auto` already exists. Run `cd gui && bun test tests/jev-auto-combo.test.tsx`; expect the actions to be absent. +- [ ] Add the quick action by parameterizing the existing add modal with an initial draft and one hash route owned by the Models/Combos page; do not fork the target editor or create a JEV-only editor. For the `jev` strategy, mark the first row as fail-open and show each row's known effort ladder. Add JEV strategy/target/setup copy to all ten locale modules. Re-run the component and pure tests; expect them to pass. +- [ ] Run `cd gui && bun test tests`, `cd gui && bun run lint`, `cd gui && bun run lint:i18n`, and `cd gui && bun run build`; expect exit code 0 for each. +- [ ] Stage the Task 4 files and commit with `git commit -m "feat(gui): add JEV Auto setup flow"`. + +## Task 5: Add per-target JEV effort allowlists and prove key setup + +**Files:** + +- Modify: `src/types/config.ts` +- Modify: `src/combos/types.ts` +- Modify: `src/server/responses/core-combo.ts` +- Modify: `gui/src/combo-workspace-data.ts` +- Modify: `gui/src/components/combo-workspace-controls.tsx` +- Modify: `gui/src/styles-combos-workspace.css` +- Modify: `gui/src/i18n/*.ts` +- Modify: focused Combo, JEV runtime, GUI, provider, and CLI-login tests + +- [ ] Add failing config and GUI round-trip tests proving an optional non-empty + `target.reasoningEfforts` list survives load/save exactly, rejects malformed or + duplicate values, participates in dirty-state comparison, and is omitted by + older/unrestricted configurations. +- [ ] Add a failing JEV runtime test proving unchecked efforts are absent from + the TypeSafe choice criteria and a configured allowlist is intersected with + the target's current supported ladder rather than broadening it. +- [ ] Implement the smallest typed config/runtime projection. An omitted list + means all advertised efforts; a present list means only its supported + intersection. A present list with no supported member contributes no JEV + target/effort choice. +- [ ] Add a failing component test for per-target effort checkboxes. All + advertised efforts start selected through omission, toggling persists an + explicit subset, the final selected effort cannot be removed, and changing + provider/model resets the override to all. +- [ ] Implement those controls in the existing target editor, with accessible + labels and localized copy; do not create a JEV-only model picker or alter the + ordinary picker. +- [ ] Add behavioral tests proving the JEV provider exposes the ordinary GUI + API-key surface and `ocx login jev` persists a key-backed, credential-only + provider without publishing a model. Avoid a spurious model-catalog probe for + this decision-only provider. +- [ ] Run the focused server/GUI/provider/CLI suites and typecheck. Commit with + `feat: add per-target JEV effort controls` after fresh tests pass. + +## Task 6: Document, review, verify, and publish the PR + +**Files:** + +- Modify: `docs-site/src/content/docs/guides/combos.md` +- Modify: `docs-site/src/content/docs/reference/configuration/routing.md` +- Modify: `structure/runtime.md` +- Modify: `structure/providers-and-adapters.md` +- Modify: `structure/gui-and-management-api.md` +- Modify: `.github/PULL_REQUEST_TEMPLATE.md` only if the existing template cannot represent the required screenshot/evidence; otherwise leave it unchanged +- Add a screenshot only in the repository's accepted documentation/media location if needed for a stable PR-body link + +- [ ] Update canonical docs with JEV key setup, the `jev` strategy, editable target allowlist, `jev-auto` quick-create flow, fail-open/cancellation behavior, one-decision-per-call rule, and the no-live-key testing boundary. Update structure docs for the new runtime/provider/GUI ownership. +- [ ] Run `bun run structure:check`, `bun run privacy:scan`, `bun run typecheck`, `bun run test`, `bun run prepush`, and `cd docs-site && bun install --frozen-lockfile && bun run build`. Save complete outputs in the execution workspace and require exit code 0. +- [ ] Start a disposable local OpenCodex instance with a mocked model target and no TypeSafe key, call `jev-auto`, and verify it reaches the first eligible fail-open target. Use a separate temporary OpenCodex home and ports; never mutate or restart the user's active instance. +- [ ] Launch the built GUI against a disposable local config, create/open the JEV Auto editor, and capture a screenshot showing the JEV strategy plus editable targets. Do not modify the user's running OpenCodex config. +- [ ] Generate the execution skill's whole-branch review package from merge-base `dev` to `HEAD`. Dispatch the required read-only fresh-context reviewer, then verify and fix every valid Critical/Important finding through a new RED→GREEN test before one final full-suite run. +- [ ] Run `git diff --check`, verify `git status --short`, and commit documentation/review fixes with Conventional Commits after fresh tests/builds pass. +- [ ] Push `feat/jev-auto-routing`, create a PR against `dev` using the repository template, include the GUI screenshot and exact test/build evidence, request Codex and Copilot review once, and attach the PR artifact to this task. Do not claim a live TypeSafe decision test. + +## Completion Contract + +- The ordinary picker still contains every pre-existing model unchanged. +- `jev-auto` appears only after explicit GUI/CLI/API creation. +- The JEV key can be configured through the provider GUI, `ocx login jev`, or `TYPESAFE_API_KEY`. +- JEV can choose only the operator-selected eligible targets and each target's operator-selected supported efforts; omitted target effort lists retain the all-advertised default. +- Every JEV failure mode has a tested first-eligible fail-open path; cancellation has a tested fail-closed 499 path. +- Retryable selected-target failure uses existing Combo fallback exactly once per target without another JEV call. +- Root tests/typecheck/privacy/structure/prepush, GUI tests/lint/build, and docs build pass on the final tree. +- The PR targets `dev`, includes the screenshot and verification evidence, and explicitly states that live-key validation is pending. diff --git a/devlog/_fin/260921_jev_auto_routing/020_design.md b/devlog/_fin/260921_jev_auto_routing/020_design.md new file mode 100644 index 0000000000..129c81c8f4 --- /dev/null +++ b/devlog/_fin/260921_jev_auto_routing/020_design.md @@ -0,0 +1,383 @@ +# JEV Auto Routing Design + +## Goal + +Add one optional `jev-auto` model to OpenCodex. Each request sent to that model +is classified by JEV (TypeSafe System One), which chooses one configured target +model and a compatible reasoning effort. Every existing provider and model +remains directly selectable and keeps its current behavior. + +The integration must feel native to OpenCodex: setup and candidate selection +live in the GUI, dispatch reuses the existing Combo machinery, and no separate +Python service or recursive loopback request is required. + +The behavioral reference is +[`0xNatoshi/jev-codex-router`](https://github.com/0xNatoshi/jev-codex-router): +bounded per-turn context extraction, a joint model-and-effort choice, strict +answer validation, standard service tier, fail-open routing, and local decision +telemetry. The implementation is a TypeScript adaptation to OpenCodex's routing +and security boundaries, not a copy of its HTTP relay. + +## User-visible invariants + +1. Installing or enabling JEV does not hide, rename, disable, reorder, or + redirect any existing model. +2. JEV is never made the default model automatically. +3. After the operator creates the JEV Combo, the integration publishes exactly + one additional public selector, `jev-auto`, with display name `JEV Auto`. +4. Selecting any ordinary model bypasses JEV completely. +5. Removing or disabling the JEV Auto combo removes only `jev-auto`; candidate + models remain available individually. +6. Candidate models are edited through the existing Combo target picker. The + initial template is seeded with the available OpenAI Luna, Sol, and Astra + models, but users may add or remove any currently routable OpenCodex model. + +## Options considered + +### External JEV provider sidecar + +Run the reference Python server on loopback, register it as a custom +OpenAI-Responses provider, and have it call OpenCodex again with the selected +model. This is close to the reference deployment but requires a second service, +two lifecycle systems, recursive HTTP routing, loop prevention, and custom GUI +bridging for candidate configuration. + +### Native JEV provider adapter + +Represent JEV as a model provider whose adapter internally re-routes to another +provider. This reuses provider credential UI but makes an adapter own recursive +dispatch and failover, responsibilities already handled by Combos. It also +risks publishing both a canonical provider/model selector and the desired +`jev-auto` alias. + +### Native JEV Combo strategy + +This is the selected design. A Combo already owns an alias, a list of concrete +provider/model targets, target eligibility, retries, quota cooldowns, reasoning +capability calculation, request replay, and GUI editing. The new `jev` strategy +changes only how the first eligible target and effort are chosen. Existing +Combo failure handling owns subsequent attempts. + +## Configuration model + +### Decision-service credential + +Add a registry-backed `jev` decision-service entry for credential ownership and +GUI setup. It has these fixed properties: + +- endpoint: `https://api.typesafe.ai/v1/systemone` +- API model: `jev-latest` +- key authentication +- no live model discovery +- no directly routable language models + +The entry exists to reuse OpenCodex's provider API-key storage, environment +reference resolution, masking, optional OS-keychain storage, and credential +management surfaces. It must never publish a model row or accept a normal model +dispatch. The runtime reads the key only when a Combo with strategy `jev` is +selected. + +`TYPESAFE_API_KEY` remains a supported environment source. A key entered in the +GUI follows the same storage and redaction rules as other provider API keys. +Management DTOs expose only credential presence and health, never the value. + +### JEV Combo + +Extend `OcxComboStrategy` with `jev`. A normal Combo record remains the source +of truth: + +```json +{ + "combos": { + "jev-auto": { + "alias": "jev-auto", + "strategy": "jev", + "targets": [ + { + "provider": "openai", + "model": "gpt-5.6-luna", + "reasoningEfforts": ["low", "medium"] + }, + { "provider": "openai", "model": "gpt-5.6-sol" }, + { "provider": "openai", "model": "gpt-6-astra" } + ], + "reasoningEffortMode": "adaptive" + } + } +} +``` + +The GUI template creates this record only after an explicit user action. It +filters unavailable seed targets rather than creating broken references. The +ordinary Combo editor remains authoritative after creation. + +Each target may optionally persist a non-empty `reasoningEfforts` allowlist. +Omitting it preserves the original behavior and offers every reasoning effort +advertised by that target. When present, JEV receives only the intersection of +that allowlist and the target's current advertised ladder. A stale allowlist +must never broaden capability or silently turn into an unrestricted choice. + +Target order has one extra meaning for this strategy: the first eligible target +is the fail-open target when JEV is unavailable or returns an invalid answer. +The GUI labels this clearly. For the reference triptych template, Astra is +placed first for fail-open parity even if the candidate list is displayed in a +friendlier order. + +No per-model capability prose is persisted in the first version. Known Luna, +Sol, and Astra targets receive the reference capability profiles. Other targets +receive neutral criteria derived from their selector, display name, declared +input modalities, context window, and supported reasoning ladder. Richer +operator-authored model profiles are intentionally deferred until their schema +and portability contract are decided. + +## Runtime architecture + +### Activation boundary + +Only a request resolving to a Combo whose strategy is `jev` imports and invokes +the JEV selector. Normal routes and other Combo strategies execute no JEV code, +start no timers, and perform no decision-service I/O. + +The JEV selector is a leaf module under `src/combos/`. It receives an already +validated Combo, the current Responses body, and concrete eligible targets. It +does not import the server composition root or dispatch requests itself. + +### Per-turn flow + +```text +Codex request model=jev-auto + -> existing Combo identification and admission + -> calculate currently eligible targets + -> derive each target's supported effort ladder + -> extract bounded decision state from the Responses request + -> one HTTPS call to TypeSafe System One + -> validate the selected target+effort pair + -> existing Combo child dispatch to that concrete provider/model + -> existing Combo preflight, retry, quota, and response relay +``` + +The selection happens once per incoming model call, including tool-result +continuations. A failed concrete attempt does not spend another JEV decision: +the existing Combo loop tries remaining eligible targets in configured order. + +### Decision state + +Port the bounded extraction contract from the reference implementation: + +- current user request with OpenCodex/system envelope blocks removed +- bounded recent assistant intent +- the most recent tool-result digest, without tool arguments +- whether image input is present +- request/item counts and step type + +The full conversation, credentials, provider headers, encrypted reasoning +payloads, tool arguments, and raw image bytes never enter the decision request. +All strings and aggregate payload size have explicit limits. Oversized or +unrecognized input degrades to fail-open instead of being truncated without a +marker or sent in full. + +### Choice contract + +Build one TypeSafe `choice` question whose criteria are the Cartesian product +of each eligible target and its supported reasoning efforts. A target that +advertises no reasoning control contributes one model-only choice. + +Each criterion uses an opaque local choice id. Provider names and model ids are +values in the criterion, never executable instructions. A response is accepted +only when: + +- the answer contains the expected question, +- the selected choice id belongs to the exact request-specific candidate set, +- optional probabilities are finite, bounded, complete, sum within tolerance, + and agree with the winning choice, +- optional confidence is finite and within `[0, 1]`. + +Confidence and probability distribution are telemetry only. They never +override a valid choice. + +### Applying the choice + +The chosen concrete target is dispatched through the existing Combo child +request path. JEV's effort replaces any effort attached to `jev-auto` for that +child only and is validated against the target's resolved ladder. The child is +forced to the normal/default service tier; JEV Auto does not request Fast mode. + +The original request body remains the replay source for fallback attempts. No +JEV metadata, API key, or decision response is inserted into model-visible +input. + +## Failure behavior + +JEV Auto is fail-open at the decision boundary: + +- missing key +- timeout, DNS, TLS, or network failure +- non-2xx TypeSafe response, including exhausted credits +- malformed JSON +- missing, unknown, or inconsistent choice +- no safe extractable decision state + +All use the first currently eligible target. The fail-open effort is `medium` +when supported, otherwise that target's declared default/nearest supported +effort, otherwise no explicit effort. + +If no target is eligible, the existing Combo-unavailable response is returned. +Once a target is chosen, existing Combo behavior remains authoritative for +provider errors, quota cooldowns, retry ordering, stream preflight, committed +output, and final error delivery. + +The TypeSafe call has a four-second timeout and `redirect: "error"`. It is never +retried within the same model call. Client cancellation and server shutdown +abort it through the request signal. + +## Security and privacy + +- The TypeSafe endpoint is registry-fixed HTTPS. User config cannot redirect + the JEV credential to another origin. +- The API key is resolved immediately before the request and is never copied + into logs, request metadata, Combo state, or management DTOs. +- Error text is bounded and sanitized before logging or returning status. +- Decision logs contain selectors, effort, timing, gate, and numeric usage only. + They do not retain extracted prompt text. +- The GUI follows existing credential-consent and CSRF rules. +- JEV cannot select a target outside the configured, currently eligible target + set, even if the service returns an arbitrary string. +- A JEV Combo cannot target itself or another path that resolves recursively to + the same Combo. + +## GUI design + +### Setup + +Add a `JEV` row to the provider catalog. Its setup pane accepts the TypeSafe API +key, links to the TypeSafe console/documentation, tests only the fixed decision +endpoint, and reports configured/missing/invalid without showing the key. + +After successful setup, offer `Create JEV Auto`. This creates the Combo template +but does not select it as the default model and does not change global model +visibility. + +### Candidate editing + +Add `JEV` to the existing Combo strategy control. Reuse the current target +editor and model inventory; do not create a second model picker. The editor: + +- marks the first eligible target as the fail-open target, +- shows each target's available reasoning efforts and lets the user select the + exact non-empty subset JEV may choose, +- treats an omitted subset as "all advertised efforts" for backward + compatibility and resets that default when the target model changes, +- prevents direct or indirect self-reference, +- warns when a target is disabled, missing, or has no usable route, +- permits saving only when at least one concrete target is valid. + +The resulting catalog contains one `JEV Auto` row with selector `jev-auto`. +Candidate models continue to appear in their original provider groups. + +### Observability + +The Combo detail view shows the latest decision state without prompt content: +selected target, selected effort, decision latency, gate (`apply` or fail-open +reason), and timestamp. Request logs record the same fields and identify the +served provider/model through existing attempt records. + +## Compatibility and rollout + +- Existing Combo records and strategies remain valid without migration. +- Configurations from a newer build that contain strategy `jev` degrade by + disabling only that Combo on an older build; provider/model configuration is + preserved. +- Disabling or deleting the JEV decision-service entry leaves the Combo record + intact but makes requests fail-open. +- Disabling or deleting the Combo removes `jev-auto` on the next normal catalog + convergence. +- No system service, Python runtime, loopback port, OpenCodex bind change, or + automatic migration is introduced. + +## Expected implementation boundaries + +- `src/types/config.ts` and config schema: `jev` Combo strategy and validation. +- `src/combos/`: bounded state extraction, TypeSafe client, decision validation, + and strategy-aware initial selection. +- `src/server/responses/core-combo.ts`: one async initial-selection seam and + application of the selected effort; existing dispatch/retry remains intact. +- provider registry and management API: fixed JEV credential owner and bounded + key-health test. +- `gui/src/components/combo-workspace-*`: strategy option, default template, + fail-open labeling, and candidate editing. +- provider catalog/auth UI: JEV key setup and `Create JEV Auto` action. +- request-log DTO/UI: secret-free decision metadata. +- docs and structure ownership notes required by the touched source areas. + +No broad adapter refactor, generic AI-router framework, external process +manager, or unrelated Combo behavior change belongs in this PR. + +## Test design + +### Pure decision tests + +- bounded extraction for text, images, tool continuations, envelope-only input, + malformed items, and oversized state +- request-specific criterion generation for mixed reasoning ladders +- valid choice acceptance and rejection of unknown, incomplete, non-finite, or + inconsistent answers +- known reference profiles versus neutral metadata-derived profiles +- deterministic fail-open target and effort selection + +### Runtime tests + +- ordinary models and non-JEV Combos perform no TypeSafe request +- `jev-auto` dispatches exactly the selected provider/model and effort +- incoming model effort and Fast preference cannot override the JEV decision +- missing key and every bounded upstream failure class dispatch fail-open +- JEV is called once when the chosen model fails and normal Combo fallback runs +- self-reference and unavailable targets never enter the criteria +- cancellation aborts an in-flight decision call +- request logs contain decision metadata and no extracted text or key material + +All TypeSafe traffic is mocked. Tests require no real JEV key. + +### Management and GUI tests + +- key values are write-only and redacted from every DTO/error path +- the fixed endpoint cannot be overridden +- setup creates one disabled-until-requested `jev-auto` catalog addition and + never changes the default model +- target editing round-trips exact provider/model ids and preserves unrelated + Combo fields +- target effort editing round-trips an exact non-empty subset and JEV never + receives unchecked or newly unsupported efforts +- the JEV API key can be stored through the provider GUI and `ocx login jev` +- removal affects only `jev-auto` +- keyboard, focus, labels, loading, and error states follow existing provider + and Combo accessibility patterns + +### Verification gates + +- focused Combo, routing, management, catalog, request-log, and GUI tests +- `bun run typecheck` +- `bun run test` +- `bun run privacy:scan` +- `bun run structure:check` +- `bun run prepush` +- local no-key smoke proving `jev-auto` reaches its fail-open target while a + directly selected model bypasses JEV + +A real decision smoke is deferred until the user supplies a TypeSafe key and is +reported separately from mocked and no-key coverage. + +## Acceptance criteria + +- Existing model/provider behavior and picker availability are unchanged. +- Enabling the integration adds exactly one opt-in `jev-auto` selector. +- GUI setup stores or references the TypeSafe key without exposing it. +- GUI users can choose the concrete models JEV is allowed to select. +- GUI users can choose the exact advertised efforts JEV is allowed to select + for each target, while older configs with no target allowlist still mean all. +- Every JEV call chooses only from the current eligible candidates and jointly + selects a compatible effort. +- Missing or broken JEV fails open predictably without blocking a turn. +- Existing Combo retry, quota, streaming, continuation, and cancellation + behavior remains authoritative after selection. +- No external JEV server or additional local port is required. +- Relevant focused and full verification gates pass before the PR is opened. diff --git a/devlog/_fin/260921_remaining_seams/000_plan.md b/devlog/_fin/260921_remaining_seams/000_plan.md new file mode 100644 index 0000000000..84aa1d2a45 --- /dev/null +++ b/devlog/_fin/260921_remaining_seams/000_plan.md @@ -0,0 +1,104 @@ +# The seams the first batch left open + +Status: open. Target branch for every lane: `dev`. + +The cross-path unit fixed each contract where it was written. Re-reading the +integrated tree found four places where those fixed policies stop one layer +short: an endpoint the refusal never reached, a name that changes between the +check and the thing checked, a match that accepts more than it was given, and a +pair of builders that read different halves of the same provider setting. + +Each lane is one branch, ordered commits, and one pull request to `dev`. + +## Owned files + +| Lane | Owns | +|---|---| +| N1 | `src/server/claude-messages.ts` error path and its tests | +| N2 | `src/server/responses-request-tool-scope.ts`, the scope call site in `src/server/responses/passthrough-delivery.ts`, `src/responses/muse-tool-name-alias.ts` where identity is carried, and the composition test | +| N3 | `src/adapters/openai-chat/passthrough.ts`, the shared wire policy it and `src/adapters/openai-chat.ts` both call, and its tests | +| N4 | `src/integrations/merge.ts` selector parsing and its tests | + +N2 owns the scope module outright; N1 and N3 do not touch it. + +## N1 — the refusal stops at the Claude Messages wrapper + +`src/server/chat-completions.ts`, `src/server/chat-native.ts` and +`src/server/responses/passthrough-error.ts` now agree: an ambiguous connection +loss answers with `upstream_reset_replay_refused`, no `Retry-After`, and +`x-should-retry: false`. `src/server/claude-messages.ts` rebuilds the error +envelope itself. It keeps only the message string, runs the generic +`resolveClientRetryAfter`, and emits an Anthropic error with `Content-Type` and +`Retry-After` — so the refusal reaches the caller as an ordinary retryable +rate limit. Anthropic's own client reads `x-should-retry` before the status +code, so the header is the part that actually stops the resend. + +Read the shared verdict here rather than re-deriving it from the message text, +and carry the code, the header and the suppressed `Retry-After` through. Two +behaviours must survive: the transient-5xx to 529 mapping the Claude client +depends on for backoff, and an ordinary provider 429, which keeps the retry +policy it has today. The acceptance is the header and code observed at +`/v1/messages`, not the internal Responses result. + +## N2 — identity has to survive the rename + +A long client tool name is sent upstream under a short alias, and +`tool_choice` is rewritten to that alias with it. On the way back the payload +rewrites restore the client name first, and only then does the snapshot repair +check the call against the scope built from the outbound body, which still +spells the selector as the alias. The restored name is not the alias, so an +allowed call is removed from the reconstructed terminal output and the turn ends +incomplete. + +The same module accepts too much in the other direction. A call is matched by +any of its spellings — bare name, `namespace__name`, `namespace.name` — against +a set holding the selector's spellings, so `alpha.lookup` and `beta.lookup` +both offer bare `lookup` and match each other. The selection set is a set of +strings, so a `custom` and a `function` tool of the same name are not separated +either; the fix is to distinguish a verified conversion from a coincidence of +names, not to refuse every kind mismatch. + +Carry the correspondence between the original identity, the wire alias and the +restored identity from the request, and have the scope read that correspondence. +`src/responses/namespace-tool-compat.ts` already reasons about selector kinds +and dotted-alias ambiguity; reuse it rather than growing a second, looser name +set. The regression test has to run the real order — a tool name past the length +limit, a named or allowed-tools selector, alias on the way out, restore on the +way back, sparse terminal reconstruction — and end with the original name and +call id intact. Keep the negative case: a tool the request did not select is +still refused after restoration. + +## N3 — two builders, one provider setting + +The translated builder turns `reasoningWireFormat: "gateway-object"` with an +effort of `none` into the gateway's object form, and omits the effort entirely +for a tool-bearing request when the model is listed in +`omitReasoningEffortWithToolsModels`. The native Chat passthrough reads neither, +so the same provider and model behave differently depending on whether a routing +feature sent the request through translation. + +Apply the explicit settings through one small policy both builders call, after +the provider is resolved. Do not route native requests through translation to +get it: the native path exists to preserve Chat-only fields such as `n`, audio +and logprobs, and losing those is a worse regression than the one being fixed. +An unset setting keeps today's native behaviour. The test compares the final +request body captured on both paths for the same input, not the status code. + +## N4 — a selector path must not change meaning + +The integration merge grammar gained a conjunction form, `[field=value,field=value]`, +because one field is not always an identity. The single-criterion form allows a +comma inside the value, so a path already written into an ownership record — for +example one whose value itself contains `,` and `=` — can parse as a conjunction +under the new rule and select a different element. + +No record in that shape has been found, so this is a migration hazard rather +than a reported loss. Close it deliberately: version the grammar, structure the +selector, or define an escape, and cover it with a test that reads a record +written under the older rule and asserts it still names the same element. + +## Out of scope + +The paginated-history work and the client provider store landed and are not +reopened here. The two retry issues left open after the first batch keep their +recorded scope; neither is a lane in this unit. diff --git a/devlog/_fin/260921_remaining_seams/090_outcome.md b/devlog/_fin/260921_remaining_seams/090_outcome.md new file mode 100644 index 0000000000..6ecdae8345 --- /dev/null +++ b/devlog/_fin/260921_remaining_seams/090_outcome.md @@ -0,0 +1,38 @@ +# Outcome + +All four lanes are on `dev`, and the seams they were opened for are closed. + +| Lane | Landed as | What it changed | +|---|---|---| +| N1 | `2cc11b780a` | The routed Claude Messages error path now carries the replay-refusal code, `x-should-retry: false` and no `Retry-After`, so a refusal no longer reaches that endpoint as an ordinary retryable rate limit. The transient-5xx to 529 mapping the client relies on for backoff and the policy for a genuine provider 429 are unchanged | +| N2 | `ebaf78a46c` | The scope reconstructs each exact outbound identity through the namespace and wire aliases before judging a call, so an allowed call restored from its alias survives sparse-terminal reconstruction. Authorization keys carry kind, namespace and name, so a shared bare name no longer grants another namespace or another kind | +| N3 | `03ab5bfb11` | `applyExplicitChatReasoningWirePolicy` owns the gateway-object form and the tool-bearing effort omission, and both the translated and native builders call it. It is a no-op when neither setting is recorded, and the native path still copies its preserved Chat-only fields verbatim | +| N4 | `bd4822bcea` | A conjunction selector now needs an explicit marker, so a legacy single-criterion value containing a comma keeps its old meaning. The whole recorded path is validated before traversal, and an unreadable selector stays attached to the file its record names as unsafe rather than moving the operation elsewhere | + +## Found while waiting + +A manual full-matrix run on one lane's branch exposed a Windows-only defect that +pull-request CI never ran: the desktop release-asset test took a basename with +`path.split("/")`, which is not a separator on Windows, so it compared whole +`C:\…` paths against asset names and five shards failed. Fixed in +`e2453085b9` by asking the platform for the last segment. The release matrix +would have hit it. + +## Verification + +Each lane merged on its own exact head with every requested job green. The four +landed seams were then re-read on the integrated tip and all five acceptance +statements hold at source level. + +Two CI conditions shaped the pace and are worth recognising next time. Duplicate +runs for one head appear regularly and one of them is cancelled by concurrency; +a cancellation is not evidence in either direction, and the remedy is to re-run +the failed jobs of the real matrix rather than to read the rollup. Hosted macOS +capacity was saturated for much of this batch, with single jobs queued for over +an hour, which is why the last two merges trailed the rest. + +## Not in this unit + +The two retry issues left open after the first batch keep their recorded scope: +a mid-turn transport death still has no fallback, and a provider 429 on a single +key still has no cooldown policy. Neither is decided by anything landed here. diff --git a/devlog/_fin/260923_command_code_mimo_reaudit/010_plan.md b/devlog/_fin/260923_command_code_mimo_reaudit/010_plan.md new file mode 100644 index 0000000000..27fff00385 --- /dev/null +++ b/devlog/_fin/260923_command_code_mimo_reaudit/010_plan.md @@ -0,0 +1,85 @@ +# 260923 Command Code / MiMo re-audit: tool-call text leak, wire compatibility, catalog + +## Problem + +Users report that Xiaomi MiMo models routed through Command Code show tool calls as plain +assistant text in Codex. Separately, the shipped Command Code facts (model fixture, reasoning +ladders, wire selection) drifted from the live Provider API, which now publishes a per-model +`supported_endpoints` list. + +## Evidence (collected 2026-09-23, raw captures in `.tmp/cc-audit/`, not committed) + +| Question | Surface | Finding | +|---|---|---| +| Does a simple MiMo tool call work? | direct `/alpha/generate`, `/provider/v1/chat/completions`, proxy chat + responses | Yes for single, parallel and multi-turn calls on `xiaomi/mimo-v2.6-flash`, `-pro`, `v2.5-pro`. | +| Where does the text come from? | real `codex exec -m command-code/xiaomi-mimo-v2.6-flash -c model_reasoning_effort=high` (2 of 2 long runs leaked) and a direct replay of the captured turn | Upstream order: `tool-input-start(exec)` → `tool-input-delta*` → `text-start` → `text-delta "RAW JS…"` → `text-end` → `tool-input-end` → `tool-call{toolName:exec,input:"RAW JS",invalid:true}` → `tool-error` → `finish-step(tool-calls)`. MiMo writes Codex's freeform `exec` body as raw JavaScript, the gateway's JSON parse of the `{input:string}` schema fails, and the gateway re-emits the model's native XML as text. The native call still executes downstream; the duplicate text is what the user sees. | +| Other wires | same captured turn on Chat and Responses | structured calls, no marker text. | +| MiMo 2.7 | Command Code live catalog, Xiaomi docs/release note, OpenRouter, Zen, Vercel AI Gateway | Not present anywhere checked. Newest ids are `mimo-v2.6-pro`, `mimo-v2.6-flash`, `mimo-v2.6-pro-ultraspeed` (Command Code added them 2026-09-22). | +| Command Code endpoints | live `/provider/v1/models` (77 rows) + docs | 9 `claude-*` ids are `/messages` only; 7 ids are `/chat/completions` only; the rest serve Chat and Responses. MiMo v2.6 rows: Chat + Responses. | +| Key preset on Claude | direct POST | `/chat/completions` → 400 `must be called via /provider/v1/messages`; `/provider/v1/messages` routes (403 plan gate on this account), Bearer and x-api-key both accepted. | +| Other clients (Aside research, `.tmp/cc-audit/client-handling.md`) | GitHub issues | Same leak without any native call: anomalyco/opencode#43385, patlux/pi-commandcode-provider#110 (Command Code: mimo 2.5, 2.6 flash, qwen 3.8 omni flash, GLM 5.3 flash), QwenLM/qwen-code#10692, XiaomiMiMo/MiMo#44 (worse with thinking high). Proposed fix everywhere is a text fallback parser. Vercel AI SDK `repairToolCall` cannot see text leaks. | +| MiMo grammar (Aside research, `.tmp/cc-audit/mimo-tool-format.md`) | HF chat templates, vLLM/SGLang parsers | `V`; strings raw, other types JSON; freeform input is the raw body with no parameter tags. | +| Catalog drift | live catalog vs `tests/fixtures/commandcode-models.json`; commandcode.ai profile payloads | fixture 59 rows vs live 77 (20 new, 2 retired); ladder corrections for `deepseek-v4.1-flash`, `Qwen3.8-Flash`, `muse-spark-1.3-contributor`; new ladders for 12 ids; Muse 1.3 profile URLs point at non-model routes; the refresh parser matches prose the pages no longer contain. | + +## Architect consultation + +Architect (gpt-6-sol, read-only) proposed D1-D4. Main dispositions: + +- D1 accepted and widened after the Aside research below: drop a text block only when it exactly duplicates the following native call; salvage a complete block only for a declared tool when no native call exists (architect advised rejecting salvage; external reports show the text-only form is the common failure and the undeclared-tool guard plus the declared-name check bound the risk). +- D2 accepted: provider-scoped `claude-` prefix pin to `anthropic` for `commandcode`, shared by the runtime resolver, config validation and the captured fast-policy authority. +- D3 accepted: refresh the static table and fixture, and repair the refresh parser to read the serialized profile payload, keeping the static row on any ambiguity. +- D4 rejected for this unit: `upstreamProtocolForAdapter` groups cursor, devin, kiro and command-code under the chat translation family on purpose; a distinct label needs Lab observation changes that no user path exercises. Recorded as a follow-up. + +## Diff-level plan + +wp2 — MiMo tool-call text handling (`src/adapters/command-code.ts`, new sibling `src/adapters/command-code-tool-text.ts`) +1. `buildRequest` attaches `commandCodeDeclaredTools` (wire name → `{ freeform, schema }` from `OcxTool.freeform` and `parameters`) to the `AdapterRequest` (new optional field in `src/adapters/base.ts`, same pattern as `convertedMuseToolNameAliases`; spread copies such as the effort-downgrade retry keep it). `fetchResponse` maps every returned `Response` to it in a module `WeakMap`; `parseStream` reads it. No declared tools → no salvage. +2. `parseStream` tracks open tool inputs by id (`tool-input-start` → name, cleared on `tool-input-end`/`tool-call`) and text blocks by `text-start`/`text-end` id. A block is held while its whitespace-trimmed lead is a prefix of, or starts with, ``; a divergent prefix flushes immediately and the block streams normally. Each held block records the set of input ids open when it started. Held bytes are reserved in the translator budget; a 64 KiB cap releases the block as text. +3. Parsing follows the official MiMo/Qwen3-Coder grammar (SGLang `MiMoDetector`, vLLM `mimo` → Qwen3 engine): `BODY`; BODY with `V` pairs → object (string schema types raw, one wrapping newline trimmed; integer/number/boolean/null/object/array JSON-decoded per the declared schema, a value that does not decode to its declared type makes the block non-salvageable); BODY without parameter tags → the raw freeform string (a stray trailing ``, as captured, is tolerated). +4. Dedupe: on `tool-call`, a held block is dropped only when it parses completely, its NAME equals the call's toolName, the call's id is in the block's recorded open-input set (or the set was empty), and the decoded value equals the call input exactly (string vs trimmed string; object deep-equal). A non-matching call leaves a block held while that block's recorded input ids are still unresolved; each arriving call rules out its own id, and the block is released as text only after every recorded candidate id has been ruled out (or immediately on a mismatch when it recorded none). +5. Salvage: at `finish-step`/`finish`/stream end, an unmatched held block becomes a synthetic call (`tool_call_start`/`delta`/`end`, id `call_ocx_`) only when it parses completely, names a declared tool, and its arguments fit that tool: a freeform tool takes a parameter-free body as its raw input (the same raw form the native path already relays); a function tool takes a parameter object that contains every `required` key and only declared keys, serialized as JSON. A finish reason of `stop` is reported as `tool-calls`. Anything else is released as text. +6. Tests (`tests/providers/command-code-tool-text.test.ts`, registered in `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`): captured event order (text dropped, one call), two interleaved tool inputs (each block paired by id), split `` deltas, name/param mismatch and substring-only match (released), text-only freeform block (salvaged raw), text-only function block with typed params (salvaged JSON), missing required key or undeclared key or undeclared name (released), a typed parameter that does not decode (released), call B arriving before the call matching block A (A still dropped), ordinary text untouched, overflow release, and a salvaged call id paired with its tool result on the next request (`wireMessages`). One end-to-end case drives a salvaged function call through the Responses bridge. + +wp3 — catalog re-aggregation +1. `tests/fixtures/commandcode-models.json` ← live snapshot (77 rows including `supported_endpoints`). +2. `src/providers/command-code-efforts.ts`: correct 3 rows, add rows backed by profile payloads, fix Muse profile URLs, teach `parsedProfileEfforts` to decode the serialized payload for the requested model (static row kept on ambiguity). +3. Tests: `tests/providers/command-code-provider.test.ts` (ladders, URLs, parser), `tests/providers/commandcode-provider.test.ts` (fixture count). + +wp4 — compatibility +1. `src/types/wire.ts`: add provider-scoped prefix pins (`commandcode`: `claude-` → `anthropic`) beside the exact-id table; `isWirePinnedModel` and `pinnedWireAdapter` consult both, so `src/server/adapter-resolve.ts:35`, `src/providers/resolved-model-policy.ts:284`, `src/config/provider-validation.ts:373` and `src/config/schema/leaf-validators.ts:493` follow automatically. Export `captureWireAdapterHardPinPrefixes(providerName)` returning a frozen `Record`, re-exported from the `src/types.ts` barrel (`tests/config/types-barrel-identity.test.ts`). +2. `src/providers/fastwire.ts`: optional `hardPinPrefixes` on `FastPolicyAuthority`, applied after exact pins and before overrides; `src/providers/service-tier.ts:115` captures it; the no-provider authority (`service-tier.ts:155`) and the synthetic one (`:207`) keep it empty. +3. Tests: `tests/server/adapter-resolve.test.ts` (claude id → anthropic even with a Chat `modelAdapters` entry, MiMo stays chat, other providers unaffected), `tests/routing/fastwire-policy.test.ts` (prefix pin through policy resolution), and a new sibling `tests/config/config-commandcode-claude-pin.test.ts` (registered in layout) because `tests/server/config.test.ts` sits at its 3,828-line cap. +4. Docs: `docs-site/src/content/docs/guides/providers.md` Command Code paragraph and `reference/adapters.md` Command Code section state the real wires (key: Chat, `claude-*` on Messages; OAuth: `/alpha/generate` NDJSON) and the MiMo text handling; translations checked for contradiction. `structure/` owners for `src/adapters/` and `src/types/` reviewed via `bun run structure:check` and updated if they name the touched contracts. + +## Acceptance + +- The captured failing event order produces one `exec` call and zero text in a focused test (red before wp2, green after). +- `resolveWireProtocolOverride("commandcode", "claude-opus-5-5", keyProvider).adapter === "anthropic"`, MiMo ids stay `openai-chat`. +- Fixture and effort table match `.tmp/cc-audit/live-models.json` and `.tmp/cc-audit/B/report.md`. +- `bun test` on the touched files, `bun run typecheck`, `bun run structure:check`, `tests/test-layout.test.ts`, `tests/ci-workflows/file-size-ratchet.test.ts` pass. + +## Out of scope / residual + +- Dotted flat tool names (`functions.exec_command`) echoed without prefix by MiMo (synthetic probe only). +- Intermittent upstream `The connection was closed` 502s on `/alpha/generate`. +- Lab protocol label (D4). +- Profile-declared vision for new ids (needs route-specific image proof before `COMMAND_CODE_IMAGE_MODELS`). + +## wp3 amendments (P, 2026-09-23) + +The W3 worker draft (`.tmp/cc-audit/wp3.patch`) is applied in B with three corrections found in review: +restore the rationale comments it deleted (exact-id key rule and the measured index map), fix the +remaining Muse 1.2/1.1 profile URLs (`meta-muse-spark-1.2` redirects 302; `muse-spark-1-2`, +`muse-spark-1-2-contributor`, `muse-spark-1-1` serve the payload), and raise the refresh bound from +256 KiB to 512 KiB because live profile pages now measure 240-259 KB. Every captured ladder matched +the payload parser (`.tmp/cc-audit/chk/ladders.ts`); live refresh reproduces the committed rows for +gpt-5.6-luna, GLM-5.2/5.3, deepseek-v4-flash and gemini-3.7-flash. + +Audit fold (reviewer, wp3): the payload parser also rejects a record whose indexed keys decode to the +same field name twice, with a test; the 512 KiB bound gets a test with a page above 256 KiB. + +Audit fold (reviewer, wp4): the `claude-` prefix pin applies only when the provider's baseUrl is +Command Code's Provider API endpoint, so a custom provider that reuses the `commandcode` name for +another destination keeps its own wire and its Chat overrides; every consumer passes the provider +config. The unused `ResolvedFastPolicy.hardPinned` field is dropped, and the guide says the Messages +route authenticates with `x-api-key` (Command Code accepts it). diff --git a/devlog/_fin/260923_command_code_mimo_reaudit/020_wp2_result.md b/devlog/_fin/260923_command_code_mimo_reaudit/020_wp2_result.md new file mode 100644 index 0000000000..207b7354ff --- /dev/null +++ b/devlog/_fin/260923_command_code_mimo_reaudit/020_wp2_result.md @@ -0,0 +1,40 @@ +# wp2 result — MiMo tool-call text on Command Code + +Commit `48514ccadc` (`fix(command-code): stop MiMo tool-call markup from reaching the client as text`). + +## What changed + +- `src/adapters/command-code-tool-text.ts` parses MiMo's native grammar + (`V`, raw freeform bodies, + the gateway's stray ``), matches a block against a native call exactly, and builds + restored arguments only when they fit a declared tool (required keys present, no undeclared keys, + typed values decode). +- `CommandCodeToolTextFilter` holds a text block only while it opens with ``; a block + pairs with the tool inputs open when it started and is released only after all of them are ruled + out. Held bytes are reserved in the translator budget and capped at 64 KiB. +- `src/adapters/command-code.ts` feeds `text-start`/`text-delta`/`text-end`/`tool-input-start` + into the filter, runs the duplicate check before relaying each native call, restores unmatched + blocks at `finish` (reporting `tool_calls`), and releases held text on an error finish or an + `error` event. The adapter reads the catalog of the request it last built, because the server + builds one adapter per routed request and parses a guarded wrapper of the upstream response. + +## Evidence + +- Red/green: on the pre-change adapter, the captured event order, both restore cases and the + Responses bridge case fail (4/4); on `48514ccadc` all 17 cases in + `tests/providers/command-code-tool-text.test.ts` pass. +- Focused suites: command-code-tool-text, command-code-provider, the three adapter conformance + files and both layout guards — 144 pass, 0 fail. + +## Direction change during the cycle + +The plan first carried the declared catalog on `AdapterRequest` and a `WeakMap`. The +Responses bridge test showed the server hands `parseStream` a wrapped response, so the map never +hit; the shared `AdapterRequest` field was removed and the per-instance catalog is the only source. + +## Pre-push review fold (wp5) + +A final review found that a block resolved by `toolCall` stayed `held` in the open-block map, so +text arriving before its `text-end` was retained and never released. Resolved blocks now switch to +streaming, and a test asserts the translator budget returns to zero. Restored arguments also reject +unsafe integers and values outside a declared `enum`, `const` or numeric bound. diff --git a/devlog/_fin/260923_command_code_mimo_reaudit/030_wp6_missing_function_close.md b/devlog/_fin/260923_command_code_mimo_reaudit/030_wp6_missing_function_close.md new file mode 100644 index 0000000000..c1007e8dad --- /dev/null +++ b/devlog/_fin/260923_command_code_mimo_reaudit/030_wp6_missing_function_close.md @@ -0,0 +1,103 @@ +# wp6 — MiMo markup without `` (loop-spec, C2) + +Loop-spec: class C2 (one adapter leaf + its focused tests + two doc sentences). Tool/credential +scope: local repo, gh for PR/CI/merge (authorized by the user: "pr 넣은 후에 머지해줘"). Write scope: +`src/adapters/command-code-tool-text.ts`, `tests/providers/command-code-tool-text.test.ts`, +`structure/providers-and-adapters.md`, `docs-site/src/content/docs/reference/adapters.md`, this unit. +Budget: one PABCD cycle; wall-clock bound: this session. No live service restart. + +## Previous D (quoted) + +wp2/wp5 (`020_wp2_result.md`): "`src/adapters/command-code-tool-text.ts` parses MiMo's native grammar +(`V`, raw freeform bodies, +the gateway's stray ``)". Direction kept: fix the parser, keep dedupe/restore gates. + +## Evidence + +- 2026-09-23 13:04 KST, codex subagent on `command-code/xiaomi-mimo-v2.6-flash`, live proxy `206fbc6b3f` + (contains #5611): visible text `RAW JS` while the + native `exec` call ran. No ``. +- Probe `.tmp/mimo-research/probe.ts` on `e9643875f0`: `parseToolCallMarkup` returns `undefined` for + `…`, `…`, the newline-wrapped variant and a params body without + ``; only the canonical close parses. Root cause: `WRAPPER` (line 44) requires + `\s*`. + +## Architect consultation + +Architect: gpt-6-sol read-only subagent `01a0cc77-0ca8-7bb1-bdbd-35a27bcb4ca0` (Aquinas). + +| ID | Proposal | Main disposition | +|---|---|---| +| D1 | Accept `BODY` or `BODY`; raw body = no `` from raw; keep rejecting nested tags, outside prose, mixed bodies, missing `` | Accepted | +| D2 | A raw body that literally ends in ``/`` is ambiguous; prefer canonical reading, rely on exact native-input match for dedupe, document residual for text-only restore | Accepted; residual recorded below | +| D3 | No change to `markupMatchesInput`, `matchNative`, `salvagedArguments`, finish gating | Accepted | +| D4 | Tests in the existing file (588 lines, uncapped; 2000-line default threshold) | Accepted | +| D5 | Sync MiMo descriptions in docs/structure | Amended: the guide and reference sentences stay true; add one clause to `structure/providers-and-adapters.md` (source owner) and `reference/adapters.md`; translations unaffected (no contradiction) | + +## Diff-level plan + +1. `src/adapters/command-code-tool-text.ts` + - `WRAPPER` → `/^\s*\s]+)>([\s\S]*)<\/tool_call>$/`; after the match, remove one + trailing `\s*\s*` from the body when present (canonical reading first). + - Nested ``/`` when the body has no ``, `…`, newline-wrapped raw body → parsed; + params body without `` (per the audit amendment below), missing ``, trailing prose + after ``, text before ``, nested ``) → exactly one `exec` call, zero text. + - adapter: text-only variant (no native call) → restored `exec` call for declared freeform tool; same with an + undeclared name → released as text. + - adapter: native call whose input differs from the variant body → text released (no false drop). +3. Docs: one clause each in `structure/providers-and-adapters.md` line 58 and `docs-site/.../reference/adapters.md`. + +## Acceptance (activation scenarios) + +- c-1: the captured-order test fails on `e9643875f0` (text leaks) and passes after the change. +- c-2: restore / undeclared / mismatch / negative parse tests pass; the mismatch case proves the drop path + does not fire on a different input. +- c-3: `bun test tests/providers/command-code-tool-text.test.ts tests/providers/command-code-provider.test.ts` + + adapter conformance files, `bun run typecheck`, `bun test tests/test-layout.test.ts`, + `bun test tests/ci-workflows/file-size-ratchet.test.ts`, `bun run structure:check`. +- c-4: PR to `dev`, exact-head required CI green, squash merge, merge commit on `origin/dev`. + +## Residual + +- D2 ambiguity: a freeform body whose real last characters are `` or `` loses them + on text-only restore (the pre-existing `` rule already had this). Dedupe is exact-match only. +- A block missing `` (truncated stream) stays text by design. +- Routing the OAuth preset over `/provider/v1` is not changed here; see the research note when it lands. + + +## Reflection (same architect) + +Verdict MISALIGNED on D4 only; folded: +- add a nested `` negative parse case beside nested `` keeps `x` as input; + `x` keeps the canonical reading's stray-strip (documented residual); + a native call whose input ends in `` against the close-less echo is released as text, never + dropped (exact-match dedupe), which pins the D2 residual as a visible text leak rather than lost input. +D1-D3, D5 mapped ALIGNED. + + +## Research sidecar (devin/swe-2, Aside + git + npm `command-code@1.64.0`) + +Report: `.tmp/mimo-research/report.md` (scratch). The OAuth preset has posted to `/alpha/generate` since +`4505210d23`; no commit routed it over `/provider/v1` or reverted such a route (the only Command Code revert, +`a312f75747`, is the quota probe). The OAuth bearer does work on `/provider/v1/chat/completions` and +`/responses` for MiMo with structured calls, but switching needs a per-model base URL and loses +`/alpha/generate`-only fields; recorded as a follow-up. SGLang `MiMoDetector` and vLLM `mimo` both require +``; no request option suppresses the gateway echo. Recommendation adopted: tolerate the missing +close only on the parameter-free (freeform) path. + +## Audit (gpt-6-sol 01a0cc7a, NEAR-PASS) dispositions + +1. Close-less raw body literally ending in `` on text-only restore — folded as a pinned test of the + canonical reading; residual kept (a freeform JS body ending in that literal is not valid JS). +2. Trimmed raw comparison in dedupe — rebutted: a match only drops the echoed text; the native call and its + own input are relayed unchanged, so no input can be altered or lost. +3. (non-blocking) Inner `` kept by the greedy wrapper — folded: a body containing `` + is rejected; regression test added. +Amended step 1: `` may be omitted only when the body has no `
    `. +- Script: recommend() also moves the non-recommended cards into `.lp-dl-more-grid` (null-safe) and removes `hidden` from the details; delete the four `*.sha256` patterns. + +`docs-site/src/styles/custom.css` +- `.lp-announce`: `white-space: nowrap`, drop `text-wrap: balance`, the <30rem override and `.lp-announce-tag`; drop the ko keep-all on `.lp-announce-text` (keep it for `.lp-download`). +- `.lp-dl-alt` gets `align-self: flex-start` (the actions column stretches children). Summary joins the focus-visible outline rule. +- Remove `.lp-dl-sha`, `.lp-dl-sep`, `.lp-dl-links` rules; when editing the shared selector lists keep the `.lp-dl-alt` colour and hover rules. `.lp-dl-actions` reserves one link line for every card (min-height = button + gap + one link line) so buttons share a row in the three-column fallback and the two-column disclosure. +- `.lp-download:has(.lp-dl-more:not([hidden])) .lp-dl-grid { grid-template-columns: minmax(0, 28rem); }`. +- `.lp-dl-more:not([hidden])` block, summary chevron (`list-style:none`, hide webkit marker, `::after` rotate on [open] inside the no-preference motion block), `.lp-dl-more-grid` two columns max 56rem, one column ≤48rem; summary added to the focus-visible outline rule. + +SoT sync: structure/ has no docs-site landing owner (checked in C with rg); the PR body is the record. + +## Architect consultation + +- Handle 01a0cc9b-10e4-76f0-9d6b-dd16dd3a8e5f; proposal D1–D8 plus a rejected CSS-only alternative (focus-order reason). +- Dispositions: D1 accept (summary label ko amended to "다른 플랫폼 더보기" to match the user's "더보기"); D2 accept; D3 accept; D4 accept; D5 accept; D6 accept; D7 accept as proposed (after reflection gap 1): the .deb link stays after the AppImage button and every card's actions reserve one link line; D8 accept. +- Reflection on revision 1: ALIGNED, D1-D8 mapped; four gaps, all folded into revision 2: (1) .deb link moved back after the button for focus order, visible label stated; (2) keep .lp-dl-alt colour/hover while deleting .lp-dl-sha from shared lists; (3) the unknown-platform path is exercised through a real UA override (Chrome --user-agent + --dump-dom) instead of script-stripped HTML; (4) verifier wording corrected — isScannedPath(custom.css)=true with THRESHOLD 2000, Landing.astro not scanned. +- SoT: structure/INDEX.md routes docs-site/ to structure/ops/docs-and-release.md; C checks whether it describes the landing download surface and patches it only if it does. + + +## Audit (A) + +- Reviewer 01a0cc9f: NEAR-PASS. Blockers folded in revision 3: (1) iPad desktop-mode row downgraded to unobserved with reason, Windows UA row added; (2) fr keys verified by grepping dist/fr/index.html, and callers of t('New') / the old pill key checked before deletion; (3) D7 disposition made consistent (link after the button, one-line reserve). Notes folded: .lp-dl-alt align-self, summary focus outline, open-state screenshot shows demoted buttons. diff --git a/devlog/_fin/260923_desktop_download_focus/010_evidence.md b/devlog/_fin/260923_desktop_download_focus/010_evidence.md new file mode 100644 index 0000000000..11f3e38d96 --- /dev/null +++ b/devlog/_fin/260923_desktop_download_focus/010_evidence.md @@ -0,0 +1,44 @@ +# Evidence — wp1 check + +Commits: c6819b6fc0 (landing focus), 75f8ca77d7 (single-card reserve). Build and checks ran at 75f8ca77d7 against `astro preview` on :4329 serving docs-site/dist. + +## Activation matrix (CDP Emulation.setUserAgentOverride with UA metadata, touch emulation for iPad) + +| Case | grid cards | disclosure cards | details hidden | hero label | +|---|---|---|---|---| +| windows (UA-CH platform Windows) | windows | macos, linux | false | Download for Windows | +| linux x86 (architecture x86) | linux | macos, windows | false | Download for Linux | +| linux arm (architecture arm) | macos, windows, linux | — | true | Download | +| iPad desktop mode (Mac UA, maxTouchPoints 5) | macos, windows, linux | — | true | Download | +| macos | macos | windows, linux | false | Download for macOS | +| Android (Chrome --user-agent, --dump-dom) | macos, windows, linux | — | true | Download | +| no JavaScript (curl) | macos, windows, linux | — | true (hidden) | Download | + +Every case: 0 occurrences of "SHA-256"; pill text "Desktop beta"; with JS the dmg href resolves to releases/download/v2.63.0/OpenCodex-2.63.0-macos.dmg, without JS it stays releases/latest. + +Test-method note: `--user-agent` alone does not change `navigator.userAgentData.platform`, so a Windows UA run first detected macOS; the CDP run with UA metadata is the real Windows evidence. + +## Render observation (headless Chrome, agbrowse CDP 9333) + +- 1440 light: one-line "Desktop beta" pill; macOS card alone (28rem) with the recommended badge; "Other platforms" summary; no empty reserve under the button after 75f8ca77d7. +- Disclosure open: Windows and Linux in two columns with outline buttons, Linux keeps ".deb package". +- 1440 dark: same layout, contrast holds. +- 320px pills (ko ru fr tr ja zh-tw): single line, no overflow. +- 390px ko: download section breaks Korean between words. +- Screenshots: /Users/jun/.browser-agent/screenshots/screenshot_1790139667580.png, /Users/jun/.browser-agent/screenshots/screenshot_1790139697444.png, /Users/jun/.browser-agent/screenshots/screenshot_1790139672949.png, /Users/jun/.browser-agent/screenshots/screenshot_1790139700162.png, /Users/jun/.browser-agent/screenshots/screenshot_1790139705932.png, /Users/jun/.browser-agent/screenshots/screenshot_1790139720186.png (uploaded to pr-assets for the PR body). + +## Gates + +- `cd docs-site && bun run build` → exit 0, 497 pages, internal links 65,499 checked. +- `rg 'App de bureau bêta|Autres plateformes' docs-site/dist/fr/index.html` → 2 matching lines; the old fr pill text and "SHA-256" → 0 in dist/fr and dist/index. +- cxc receipt test → `bun test` 5 files: 94 pass, 0 fail (receipt .codexclaw/evidence/01a0cc7c-73a3-7582-ad6d-17d34211ba09/test-receipt.json). +- `bun run privacy:scan` → passed; `git diff --check` clean. +- SoT: structure/ops/docs-and-release.md (docs-site owner) does not describe the landing download surface; no patch needed. +- Unobserved: none of the planned conditional paths remain unobserved (the iPad and Linux-arch rows were exercised through CDP emulation). + + +## Follow-up: centred layout (user steering during C) + +- custom.css: `.lp-download` centres its heading, subtitle, version line, the single detected card (`justify-content: center`), the Other platforms summary, the disclosure grid (`margin-inline: auto`) and the terminal row; card copy stays left-aligned (`.lp-dl-card { text-align: start }`). +- Rebuilt (exit 0) and re-observed at 1440 light (closed and open), 1440 dark, and 390px ko with the disclosure open: every block centred, no overlap. + diff --git a/devlog/_fin/260923_log_served_model_echo/000_plan.md b/devlog/_fin/260923_log_served_model_echo/000_plan.md new file mode 100644 index 0000000000..e69b2fee44 --- /dev/null +++ b/devlog/_fin/260923_log_served_model_echo/000_plan.md @@ -0,0 +1,96 @@ +# Log served-model echo — plan + +loop-spec: C2 single work-phase (wp1), goal = remove the false reroute arrow for Anthropic rows at its source. + +## Problem + +The Logs model column renders `claude-opus-5-5 → anthropic/claude-opus-5-5` for every +Anthropic-routed request. Live row shape (usage.jsonl, 2026-09-23): + +``` +provider=anthropic model=claude-opus-5-5 requestedModel=anthropic/claude-opus-5-5 +resolvedModel=claude-opus-5-5 wireModel=claude-opus-5-5 servedModel=anthropic/claude-opus-5-5 +``` + +## Root cause + +- `src/server/responses/core-normalize.ts` keeps the Codex-facing selector for Anthropic + routes (`parsed._responseModelId`), and every delivery path writes it into `response.model`. +- `src/server/request-log.ts` `applyResponseLogMetadata` reads `response.model` from that + client-facing payload and stores it as `servedModel` — ocx's own echo, recorded as if the + upstream had reported it. +- `gui/src/pages/logs-model-title.ts` `isModelRerouted` compares `servedModel` with + `wireModel ?? model` and draws the arrow. #5609 made the wire model explicit; the echo + itself predates it. + +## Changes (diff level) + +| File | Change | +|---|---| +| src/usage/log.ts | export `isClientSelectorEcho(source, served)`; `modelIdentityLogFields` and `normalizeUsageEntry` drop an echo `servedModel` (and a `resolvedModel` that only repeated it) | +| src/server/request-log.ts | internal `responseModelEcho` on RequestLogContext; `applyResponseLogMetadata` ignores an echo payload model | +| src/server/responses/core-normalize.ts | set `logCtx.responseModelEcho` in the block that already records `wireModel` | +| tests/server/response-model-identity.test.ts | bridged JSON/SSE and passthrough rows never carry the echo as servedModel | +| tests/usage/request-log-served-model.test.ts | capture skip, real reroute still recorded, legacy row repaired on read | + +Field chain for `responseModelEcho`: creation core-normalize → consumed by applyResponseLogMetadata +and modelIdentityLogFields → never serialized (internal, like `preserveResolvedModelFromRoute`). + +## Scope + +IN: capture-time and read-time served-model identity. OUT: GUI rendering (it is correct once the +data is), response.model contract to clients (unchanged), pricing. + +## Acceptance + +1. Anthropic bridged JSON/SSE: `logCtx.servedModel` is not the namespaced selector. +2. A real upstream model difference (e.g. `claude-opus-5-1`) is still recorded as servedModel. +3. A legacy persisted row with `servedModel = provider/wire` normalizes without servedModel. +4. Existing served-model sanitization test still passes. +5. `bun run typecheck`, focused files, exact-head CI green; PR merged into dev. + + +## Architect consultation (devin/swe-2, agent 01a0cc84-dad4-7962-8dc1-f5a9b5ebf040) + +- D1 responseModelEcho on RequestLogContext only — ACCEPT. Combo parents inherit it via + `Object.assign(logCtx, childLog, …)` in core-combo.ts. +- D2 predicate — AMEND, adopted: also match the persisted `requestedModel`, because a bare combo + selector sets neither `requestedAlias` nor a slash form, so legacy combo rows are repairable only + through `requestedModel`. Final predicate: served differs from `wireModel ?? model` and equals + one of `responseModelEcho`, `requestedAlias`, `requestedModel`, or `provider/(wireModel ?? model)`. +- D3 capture skip in applyResponseLogMetadata — ACCEPT. It is the only body-derived writer; the + `openai-model` header writer in passthrough-delivery.ts is a real observation and stays. +- D4 read-time repair — ACCEPT with the D2 amendment. +- Noted, out of scope: passthrough-dispatch.ts:623 hands the selector to `notifyResponseComplete` + (recall, not logs). On adapter paths the true upstream model is not observable at all, so a genuine + Anthropic-side reroute now shows as no arrow instead of a false one; the code comment says so. + +Explorer (devin/swe-2, agent 01a0cc85-0426-78a2-931c-31ea0d27a739): no consumer requires servedModel +on Anthropic rows; the echo also produced a duplicate `anthropic/claude-opus-5-5` option in the +Logs model filter (logs-filter.ts:126,169), which the same fix removes. Pricing and CLI read `model`. +Test note: tests/usage/request-log.test.ts is at its file-size cap (2075), so new assertions go to +request-log-served-model.test.ts and response-model-identity.test.ts. + + +## Reflection (same architect): ALIGNED, gaps folded into acceptance + +6. Combo parent: a row with `provider = combo`, `requestedModel = mycombo` and an inherited + echo `servedModel = mycombo` keeps no servedModel (capture via inherited responseModelEcho, + read via requestedModel). +7. Legacy bare-selector row (`servedModel === requestedModel`, provider slash form absent) normalizes + without servedModel. +8. Acceptance 2 is a capture-level unit: `applyResponseLogMetadata` with a non-echo model still sets + servedModel. Adapter paths cannot observe a real upstream reroute; only passthrough can. + + +## Audit round 1 (devin/swe-2 auditor 01a0cc89-541e-7ca3-9dda-149704c31ce5): FAIL, 1 blocker — folded + +Blocker: src/server/request-log.ts is 1999 lines and untracked by the ratchet (THRESHOLD 2000 in +scripts/file-size-ratchet.ts:4); any net addition fails NEW_OVERSIZED. + +Fold: extract the served-model write out of `applyResponseLogMetadata` into +`recordObservedServedModel(target, value)` in src/usage/log.ts, beside `sanitizeServedModel` and +`modelIdentityLogFields`, which already own served-model identity. The five-line block becomes one +call, which pays for the `responseModelEcho` field and its comment. request-log.ts must end ≤ 1999 +lines (checked in C with `wc -l` and the file-size ratchet test). + diff --git a/devlog/_fin/260923_log_served_model_echo/010_done.md b/devlog/_fin/260923_log_served_model_echo/010_done.md new file mode 100644 index 0000000000..ada3c10676 --- /dev/null +++ b/devlog/_fin/260923_log_served_model_echo/010_done.md @@ -0,0 +1,22 @@ +# Log served-model echo — done + +The Logs model column showed `claude-opus-5-5 → anthropic/claude-opus-5-5` on every Anthropic row +because the request logger stored ocx's own client-selector echo from `response.model` as the +upstream served model. `recordObservedServedModel` in src/usage/log.ts now refuses that echo at +capture, and `modelIdentityLogFields` drops it from rows persisted earlier, so hydrated history +renders `claude-opus-5-5` alone and the model filter loses its duplicate option. + +Evidence: four new/strengthened assertions fail on the pre-fix source and pass after it +(tests/usage/request-log-served-model.test.ts, tests/server/response-model-identity.test.ts). +Typecheck, request-log, file-size ratchet, test layout, structure SSOT, GUI logs title/filter tests +and privacy scan pass. request-log.ts ends at 1996 lines (auditor blocker: 2000 threshold). + +Did not improve / limits: on adapter paths the upstream's real model is not observable, so an +Anthropic-side reroute shows no arrow rather than a false one. passthrough-dispatch.ts still hands +the selector to `notifyResponseComplete` (recall, not logs). A local `test:changed` expanded to the +full suite and failed only in the ~/.codex worktree test-home guard; hosted CI is the broad verdict. + +Subagents (devin/swe-2): architect (proposal + ALIGNED reflection), explorer (consumer map), +auditor (FAIL → folded → PASS). The swe-2 test worker hit Devin `resource_exhausted` before writing; +the main agent wrote the tests. + diff --git a/devlog/_fin/260923_mimo_surface_audit/000_audit_roadmap.md b/devlog/_fin/260923_mimo_surface_audit/000_audit_roadmap.md new file mode 100644 index 0000000000..276c1c13b6 --- /dev/null +++ b/devlog/_fin/260923_mimo_surface_audit/000_audit_roadmap.md @@ -0,0 +1,66 @@ +# 260923 MiMo surface audit — findings and roadmap + +## Problem + +After #5611 and #5637 fixed Command Code MiMo tool-call text, the user asked for an exhaustive pass over +every other Xiaomi MiMo surface. Xiaomi shipped the V2.6 family (`mimo-v2.6-pro`, `-pro-ultraspeed`, +`-flash`) and announced that `mimo-v2.5` / `mimo-v2.5-pro` stop working on 2026-10-21 10:00 Beijing time +with no automatic redirect ([deprecation notice](https://mimo.mi.com/docs/en-US/updates/deprecate)). Most +of opencodex still stops at V2.5. + +## Method + +Three read-only gpt-6-sol auditors over disjoint slices (reports in `.tmp/mimo-audit/{A-runtime,B-catalog,C-docs}.md`, +scratch): A runtime/adapters, B catalog/metadata/pricing against live upstream, C docs/structure/locales. +Main verified each finding in source (path:line below) and pulled the current models.dev record +(`https://models.dev/api.json`, fetched 2026-09-23, scratch copy `.tmp/mimo-audit/models.dev.json`). + +## Surface inventory + +| Surface | Where | V2.6 today | +|---|---|---| +| Xiaomi Anthropic preset `xiaomi` | `src/providers/registry/entries-extended.ts:1142` | default `mimo-v2.5-pro`, no roster | +| Xiaomi Chat preset `xiaomi-mimo` | `entries-extended.ts:1148-1160` | default/roster `mimo-v2.5` only | +| Xiaomi token plan `mimo` | `entries-extended.ts:1188-1209` | default `mimo-v2.5-pro`, roster V2.5 only | +| MiMo Free `mimo-free` | `src/adapters/mimo-free.ts`, `entries-extended.ts:1162-1177` | opaque `mimo-auto` (correct) | +| Command Code OAuth/key | `src/adapters/command-code.ts`, `command-code-tool-text.ts`, `command-code-efforts.ts` | live ids; markup filter gated to V2.6 only | +| OpenCode Go | `src/providers/registry/entries-core.ts:876-885`, `model-seeds.ts:272-274` | toggle/vision tables V2.5 only | +| OpenCode Zen | `model-seeds.ts:401-417` | image hint for V2.5 free only | +| Cline Pass (static catalog) | `model-seeds.ts:944-1012` | V2.5 only | +| DigitalOcean | `model-seeds.ts:881` | V2.5 Pro only | +| Bundled metadata | `scripts/model-metadata.source.json` → `src/generated/model-metadata.ts` | no V2.6 rows anywhere | +| Docs | `docs-site/.../guides/providers.md` (+7 locales), `reference/adapters.md` | see C findings | + +## Findings and dispositions + +| ID | Class | Finding (verified evidence) | Disposition | +|---|---|---|---| +| A-01 | defect | `mimo-free.ts:225` `buildRequest` calls `getMimoJwt()` without the caller's abort signal; an aborted first turn waits for the bootstrap (up to 15 s). | Fix in wp3 (020) | +| A-02 | defect | `mimo-free.ts:162-180` shares one bootstrap promise created with the first caller's signal; aborting that caller fails every concurrent waiter. | Fix in wp3 (020) | +| A-03 | defect | `command-code.ts:600` enables MiMo markup dedupe/restore only for `xiaomi/mimo-v2.6-*`; Command Code still serves `xiaomi/mimo-v2.5`/`-pro` (fixture) and third-party reports show the same markup leaking there (patlux/pi-commandcode-provider#110). | Fix in wp3: gate on the `xiaomi/mimo-` family | +| B-CAT-01 | defect (to prove) | Command Code presets have no static MiMo ids, so a cold/failed discovery cannot decode `command-code/xiaomi-mimo-v2.6-pro`. | wp3: red test first; fix only if red | +| B-CAT-02 | defect | No V2.6 price anywhere: `resolveMatchedPrice` returns null for Xiaomi/OpenRouter/Go/Command Code V2.6. | Fix in wp2 via models.dev rows | +| B-CAT-03 | stale | Xiaomi presets default to and list only V2.5; V2.5 dies 2026-10-21. | Fix in wp2 (new defaults + roster; saved configs untouched) | +| B-CAT-04 | cleanup | Expired first-party `xiaomi/mimo-v2-{flash,omni,pro}` rows stay in metadata. | Rejected: no preset advertises them; removing them would unprice historical usage rows. | +| B-CAT-05 | inconsistency | No V2.6 context/output/modality facts. | Fix in wp2 with the metadata rows | +| B-EXT-06 (main) | stale | OpenCode Go thinking-toggle/vision tables, Cline Pass static catalog and context/image tables stop at V2.5; models.dev lists V2.6 on both. | Fix in wp2 | +| C-DOC-01 | inconsistency | `guides/providers.md:588-589` (+7 locales) calls Xiaomi Anthropic-only although a Chat preset exists. | Fix in wp4 (030) | +| C-DOC-02 | stale | 7 locale guides still carry the pre-#5611 Command Code paragraph. | Fix in wp4 | +| C-DOC-03 | inconsistency | `reference/adapters.md:214-218` omits the clean-finish condition for restoration. | Fix in wp4 | + +## Not changed (recorded) + +- Command Code MiMo effort ladder: no published ladder; needs a live `/alpha/generate` probe. Follow-up. +- Gateway image support for V2.6 on Command Code, Zen free and Cline Pass is unverified: first-party modality does not prove a gateway forwards images, and positive image hints need a route probe (`model-seeds.ts:327-387` policy). Static tables leave V2.6 out of their image sets; the sidecar keeps images working. +- DigitalOcean V2.6: no catalog evidence found. Not applicable until listed. +- Migrating saved configs off V2.5 before 2026-10-21: explicit user choices stay; recorded as a dated follow-up. +- Routing the Command Code OAuth preset over `/provider/v1`: follow-up from #5637. + +## Work-phase map (dependency order) + +1. wp1 — this audit and roadmap (docs only). +2. wp2 — catalog facts: metadata rows + regenerate, Xiaomi presets, OpenCode Go, Cline Pass (`010_wp2_catalog_v26.md`). +3. wp3 — runtime: MiMo Free abort, Command Code family gate, cold-start decode (`020_wp3_runtime.md`). +4. wp4 — docs in English and 7 locales, then PR, CI, merge (`030_wp4_docs_delivery.md`). + +One branch `codex/mimo-surface-audit`, ordered commits, one PR to `dev`. diff --git a/devlog/_fin/260923_mimo_surface_audit/010_wp2_catalog_v26.md b/devlog/_fin/260923_mimo_surface_audit/010_wp2_catalog_v26.md new file mode 100644 index 0000000000..d93274986e --- /dev/null +++ b/devlog/_fin/260923_mimo_surface_audit/010_wp2_catalog_v26.md @@ -0,0 +1,63 @@ +# wp2 — V2.6 catalog facts (diff-level) + +Loop-spec: C3 (registry data, vendored metadata snapshot, one registry field on two presets; no runtime logic). +Write scope: files below. Budget: one cycle. + +## Changes + +1. `scripts/model-metadata.source.json`: add V2.6 rows only to bundles the generator reads + (`allowedProviders` = derived aliases ∪ `COST_VENDOR_BUNDLES`, `scripts/generate-model-metadata.ts:42-53`), copied + from models.dev (2026-09-23) in the neighbouring V2.5 row shape, `input` limited to `text`/`image`: + - `xiaomi` (vendor price bundle): `mimo-v2.6-pro` 1,048,576/131,072 image 0.435/0.87/0.0036; + `mimo-v2.6-pro-ultraspeed` 4.35/8.7/0.036; `mimo-v2.6-flash` 0.14/0.28/0.0028. + - `openrouter`: `xiaomi/mimo-v2.6-{pro,pro-ultraspeed,flash}` (same prices). + - `opencode-go`: `mimo-v2.6-pro` (cacheRead 0.003625), `mimo-v2.6-flash`; `input: ["text"]` until the route is probed. + Not added: kilo/nano-gpt/vercel/opencode(Zen) — unmapped bundles, rows would be inert. + Regenerate `src/generated/model-metadata.ts` (`bun scripts/generate-model-metadata.ts`). +2. `src/providers/registry/entries-extended.ts`: + - `xiaomi` and `xiaomi-mimo`: `jawcodeBundle: "xiaomi"` so first-party presets read context/output/modalities/price + from the vendor bundle. Chain: `deriveJawcodeAliases` (`src/providers/derive.ts:629`) → generator alias map (regenerated) + → `resolveMetadataProvider` → `model-hints.ts` and `cost.ts`. Token plan (`mimo`) stays unmapped: no plan-specific facts are + claimed, and its estimates keep coming from the model-level vendor fallback (pay-as-you-go equivalent), as for V2.5. + - `xiaomi`: `defaultModel: "mimo-v2.6-pro"`. + - `xiaomi-mimo`: `defaultModel: "mimo-v2.6-flash"`, + `models: ["mimo-v2.6-flash", "mimo-v2.6-pro", "mimo-v2.6-pro-ultraspeed", "mimo-v2.5"]`. + - `mimo` (token plan, roster per models.dev `xiaomi-token-plan-*`): `defaultModel: "mimo-v2.6-pro"`, + `models: ["mimo-v2.6-pro", "mimo-v2.6-flash", "mimo-v2.5-pro", "mimo-v2.5"]`; `noVisionModels` unchanged. + - Comments record the 2026-10-21 V2.5 deprecation. +3. `src/providers/registry/model-seeds.ts`: + - `OPENCODE_GO_THINKING_TOGGLE_MODELS` += `mimo-v2.6-pro`, `mimo-v2.6-flash` (vendor toggle family; preemptive, commented). + - `CLINE_PASS_MODELS` += `cline-pass/mimo-v2.6-pro`, `cline-pass/mimo-v2.6-flash` ahead of V2.5; context 1,048,576; + image support unverified on the route → not in `CLINE_PASS_IMAGE_MODELS`. +4. Saved configs: registry seeds change only defaults for new providers; a persisted `defaultModel`/`models` is not rewritten. + +## Tests + +- `tests/codex-integration/model-metadata-sync.test.ts` (regeneration byte-equal). +- `tests/usage/usage-cost.test.ts`: priced estimates for `xiaomi-mimo`/`mimo-v2.6-flash`, `openrouter`/`xiaomi/mimo-v2.6-pro`, + `command-code`/`xiaomi/mimo-v2.6-pro` (vendor-prefix fallback) — null before. +- Catalog hint test: `xiaomi-mimo`/`mimo-v2.6-flash` resolves 1,048,576 context and image input; `xiaomi`/`mimo-v2.5-pro` text-only. +- `tests/providers/mimo-token-plan-provider.test.ts`: defaults/rosters for the three presets; a saved `defaultModel: "mimo-v2.5"` survives. +- Existing Cline Pass / Go / alias / preset-count tests found by rg; update counts by derivation, not by restating numbers. + +## Reflection (B auditor, MISALIGNED → folded) + +Unmapped bundles, models.dev key names, Go cacheRead 0.003625, zero-price rows returning null, Zen image evidence, +and the missing Chat ultraspeed id were folded above. B-CAT-04 rejection confirmed sound. + +## Audit fold (gpt-6-sol 01a0cca0, NEAR-PASS) + +- OpenCode Go V2.6 image capability is unverified on that route: its metadata rows carry `input: ["text"]` and both ids join + the Go `noVisionModels` list (`entries-core.ts:876`) so images go through the sidecar; a route probe is a follow-up. +- The exact alias map in `tests/providers/provider-registry-parity.test.ts` gains `xiaomi` and `xiaomi-mimo` → `xiaomi`. +- New catalog-hint cases go to a sibling test file registered in `scripts/test-layout/layout.json` and + `tests/fixtures/test-layout-expected.json` (`codex-catalog.test.ts` is ~11 lines under its cap). +- Saved `defaultModel`/`models` survive enrichment (`src/providers/derive.ts:523`); no MiMo rename rule exists. + +## Re-audit (gpt-6-sol 01a0cca0: 010 NEAR-PASS, 020 PASS) + +Residual accepted for Go vision: a persisted Go `noVisionModels` list is filled all-or-nothing (`src/providers/derive.ts:551`), so +an existing config does not learn the V2.6 entries. With V2.6 metadata text-only, the catalog advertises text-only for those +rows and the app blocks image attachment instead of sending an image the route may reject; new configs get the sidecar. +A guarded list repair would apply to every provider's all-or-nothing lists and is a separate unit. No opencode-go key is +configured locally, so the route probe that would settle native image support stays a follow-up. diff --git a/devlog/_fin/260923_mimo_surface_audit/020_wp3_runtime.md b/devlog/_fin/260923_mimo_surface_audit/020_wp3_runtime.md new file mode 100644 index 0000000000..9c69589bad --- /dev/null +++ b/devlog/_fin/260923_mimo_surface_audit/020_wp3_runtime.md @@ -0,0 +1,28 @@ +# wp3 — MiMo runtime fixes (diff-level) + +1. `src/adapters/mimo-free.ts` (A-01, A-02): + - The shared bootstrap runs with the timeout only (`fetchJwt()` without a caller signal). + - `getMimoJwt(signal?)` returns `abortable(inFlightJwt, signal)`: a per-caller race that rejects with the caller's + abort reason without cancelling the shared promise; cache still written only on success. + - `buildRequest(parsed, incoming)` passes `incoming.abortSignal`. + Tests (`tests/providers/mimo-free-provider.test.ts`): abort during initial bootstrap rejects promptly and sends no + inference; two concurrent waiters, abort the first, the second still receives and caches the JWT. Both red before. +2. `src/adapters/command-code.ts:600` (A-03): gate becomes `/^xiaomi\/mimo-/i` on the canonical id. + Test (`tests/providers/command-code-tool-text.test.ts`): captured order on `xiaomi/mimo-v2.5-pro` drops the echo + (red before); a non-MiMo model keeps markup text untouched. +3. Cold-start decode (B-CAT-01): add a red test in `tests/providers/command-code-provider.test.ts` that routes + `command-code/xiaomi-mimo-v2.6-pro` with an empty discovery cache. Only if red: seed the five MiMo ids in the Command + Code registry entries through an existing model-keyed or `models` field; otherwise record "not reproducible". +4. Docs sync for these behaviours happens in wp4. + +## Audit fold (gpt-6-sol 01a0cca0, NEAR-PASS) + +- Abortable wait: reject immediately when the caller signal is already aborted; attach one `abort` listener and remove it + when either the shared bootstrap or the abort settles; the shared promise always has a rejection handler so a bootstrap + failure after every waiter left is not unhandled; cache is written only on success. +- Replace the existing test at `tests/providers/mimo-free-provider.test.ts:273` that asserts the caller signal reaches the + bootstrap `fetch`; add cases: already-aborted caller, every caller aborts then the bootstrap fails (no unhandled + rejection, next call bootstraps again), and a later successful retry. +- Cold-start decode, if red: seed decode ids through `modelContextWindows` on both Command Code entries with the live + fixture's 1,048,576 windows for the V2.6 ids (a verified model-keyed fact), never the `models` roster; test the + degraded catalog (no static roster rows appear) together with routing. diff --git a/devlog/_fin/260923_mimo_surface_audit/030_wp4_docs_delivery.md b/devlog/_fin/260923_mimo_surface_audit/030_wp4_docs_delivery.md new file mode 100644 index 0000000000..619eaf18e4 --- /dev/null +++ b/devlog/_fin/260923_mimo_surface_audit/030_wp4_docs_delivery.md @@ -0,0 +1,13 @@ +# wp4 — docs and delivery (diff-level) + +1. English `docs-site/src/content/docs/guides/providers.md`: + - C-DOC-01: the Anthropic-compatible example names the `xiaomi` preset and points to `xiaomi-mimo` for Chat. + - Command Code paragraph: MiMo markup handling covers the whole MiMo family. + - Xiaomi preset text states the V2.6 defaults and the 2026-10-21 V2.5 deprecation. +2. `reference/adapters.md` (C-DOC-03): restoration only after a clean stop/tool-call finish; abnormal finishes leave text. +3. Seven locales (fr, ja, ko, ru, tr, zh-cn, zh-tw) `guides/providers.md`: C-DOC-01 sentence and the current Command Code + paragraph (C-DOC-02) translated from the English source; one locale per worker, token parity checked by main + (`commandcode`, `command-code`, `claude-*`, `/provider/v1/messages`, `/alpha/generate`, preset ids). +4. `structure/providers-and-adapters.md`: MiMo gate wording; `structure/` owners reviewed by `bun run structure:check`. +5. Delivery: gates (focused suites of wp2-wp4, typecheck, layout, file-size ratchet, structure:check, privacy:scan, + `git diff --check`), PR to `dev` with the template, exact-head CI, squash merge, verify on `origin/dev`. diff --git a/devlog/_fin/260923_mimo_surface_audit/040_result.md b/devlog/_fin/260923_mimo_surface_audit/040_result.md new file mode 100644 index 0000000000..e2a9237fc8 --- /dev/null +++ b/devlog/_fin/260923_mimo_surface_audit/040_result.md @@ -0,0 +1,17 @@ +# Result — MiMo surface audit + +| Finding | Outcome | Commit | Proof | +|---|---|---|---| +| B-CAT-02/03/05, B-EXT-06 | V2.6 metadata rows (xiaomi, openrouter, opencode-go text-only), `xiaomi`/`xiaomi-mimo` read the xiaomi bundle and default to V2.6, token-plan roster gains V2.6, Go toggle/sidecar lists and Cline Pass catalog gain V2.6 | `b99edfa9a8` | `tests/providers/mimo-v26-catalog.test.ts` 7 red on `0f9254b564`, green after | +| A-01/A-02 | MiMo Free bootstrap bound to its timeout only; each request aborts its own wait; `buildRequest` passes the request signal | `0b5f615fba` | 3 new cases in `tests/providers/mimo-free-provider.test.ts` red (one hangs) on the old code | +| A-03 | Command Code markup filter covers every `xiaomi/mimo-` model | `3bfd160b43` | V2.5 case in `tests/providers/command-code-tool-text.test.ts` red before | +| B-CAT-01 | `COMMAND_CODE_MIMO_CONTEXT_WINDOWS` on both Command Code presets makes MiMo slugs decode on a cold start without a roster | `3bfd160b43` | cold-start decode case red before (`xiaomi-mimo-v2.6-pro` sent verbatim) | +| C-DOC-01/02/03 | English guide and reference, seven locale guides, `structure/providers-and-adapters.md` | wp4 docs commit | token parity across 8 guides; docs-provider-* suites pass | +| B-CAT-04 | Rejected (historical pricing) | — | — | + +Follow-ups: route probes for V2.6 image input on OpenCode Go, Zen free, Cline Pass and Command Code; a Command Code +V2.6 effort ladder probe; migrating saved V2.5 defaults before 2026-10-21 if users ask; a guarded repair for +all-or-nothing registry lists such as `noVisionModels`; routing the Command Code OAuth preset over `/provider/v1`. + +What did not improve: nothing here proves live gateway behaviour; every V2.6 capability beyond first-party +metadata stays conservative (sidecar) until probed. diff --git a/devlog/_fin/260923_release_2_64/000_plan.md b/devlog/_fin/260923_release_2_64/000_plan.md new file mode 100644 index 0000000000..579ec410cb --- /dev/null +++ b/devlog/_fin/260923_release_2_64/000_plan.md @@ -0,0 +1,68 @@ +# 260923 release 2.64 — plan + +## Objective + +Ship the verified `dev` tree as preview `2.64.0-preview.20260923` and stable `2.64.0` +after closing the two items that kept the previous readiness answer at "not yet": + +1. The critical Dependabot alert on `desktop/src-tauri` (GHSA-c9pr-q8gx-3mgp, + `tauri-plugin-shell` below 2.2.1). +2. Missing security-review records for the CI, release and account-routing changes + merged by the parallel batch (#5471, #5456, #5653, #5469, #5024, #5654, #5655). + +The owner authorized PR creation, admin squash merges to `dev`, promotion merges to +`main` and `preview`, and release dispatch for this round. + +## Starting state (2026-09-23 07:50Z) + +| Ref | Commit | Version | Evidence | +|---|---|---|---| +| `dev` | `fa81e5a2a7` | 2.64.0 in all four version sources | lane=all run 35828289232, every job success, privacy gate skipped by design | +| `main` | `96b1406cb6` | 2.63.0 | npm `latest`, release v2.63.0 | +| `preview` | `5bec58cdda` | 2.63.0-preview.20260923 | npm `preview` | +| Dependabot #5525 | `d6dea8f246` (base `main`) | tauri-plugin-shell =2.2.1 | applies to `dev` cleanly, merge tree `11d41e0b70` | + +Open Dependabot alerts on `desktop/src-tauri/Cargo.lock`: critical `tauri-plugin-shell`, +medium `serde_with`, `time`, `glib`. Only the critical one is in scope. `glib` 0.20 +needs a GTK binding upgrade that the pinned Tauri line does not take; `serde_with` and +`time` are transitive and are recorded as residuals for a later dependency round. + +## Constraints + +- No local test, typecheck, build, install, cargo or ocx run. Hosted CI at the exact + head is the only execution evidence. Helper scripts that read state (version-source + check, merge-tree, gh reads) or rewrite the four version sources + (`release-version-sources.ts sync`) are allowed; neither executes the product. +- Skipped, cancelled, missing or older-head results are not success. A Windows job that + fails once on a known runner stall is rerun once; a repeat is a defect. +- No timeout increase, platform skip, weakened assertion, or ratchet cap raise. +- Security analysis stays in scratch space outside this repository's tracked tree. This + unit records only that each review happened and how findings were dispositioned. +- Pushes use `--no-verify`; merges use `--admin` with `--match-head-commit`. + +## Work-phase map (dependency order) + +| Phase | Doc | Consumes | Produces | +|---|---|---|---| +| wp1 | this unit | current state | locked roadmap | +| wp2 | [010](010_wp2_prerelease_items.md), [011](011_wp2_privacy_gate_complement.md), [012](012_wp2_request_owned_main_cursor.md) | wp1 | `tauri-plugin-shell` 2.2.1 on `dev`; review records; fixes for the two confirmed findings | +| wp3 | [020](020_wp3_dev_candidate.md) | wp2's final `dev` SHA | fixed candidate SHA with a fully green lane=all run | +| wp4 | [030](030_wp4_release.md) | wp3's candidate | dev pre-move, promotions, both releases, channel verification | + +Each phase closes with something checkable from GitHub alone: a merged PR with its +exact-head run, a dev run ID, release run IDs and registry state. + +## Verifiers + +| Script (scratch) | Reads | Proves | +|---|---|---| +| `check-wp1.sh` | this unit | numbered docs, no private review detail, no absolute user paths | +| `check-wp2.sh` | PRs, `origin/dev`, Dependabot API | the three wp2 PRs merged at their verified heads with green exact-head runs, `Cargo.toml` pins `=2.2.1` on `dev`, review and second-review reports present | +| `check-wp3.sh` | the candidate run | every job `success` except the privacy gate skip, run head equals candidate | +| `check-wp4.sh` | npm registry, GitHub releases, `latest.json` | channel versions, release assets, updater signatures | + +## Terminal outcomes + +DONE when wp4's verification passes. BLOCKED on a confirmed security blocker that +cannot be fixed inside this round or on a repeated CI defect. UNSAFE if a release gate +would have to be bypassed. NEEDS_HUMAN on a policy decision this plan does not cover. diff --git a/devlog/_fin/260923_release_2_64/010_wp2_prerelease_items.md b/devlog/_fin/260923_release_2_64/010_wp2_prerelease_items.md new file mode 100644 index 0000000000..d5881da0d3 --- /dev/null +++ b/devlog/_fin/260923_release_2_64/010_wp2_prerelease_items.md @@ -0,0 +1,86 @@ +# 010 — wp2: pre-release items + +## A. tauri-plugin-shell 2.2.1 on dev + +Dependabot opened #5525 against `main`, the default branch. `dev` is the integration +branch, so the same commit is carried to `dev` and reaches `main` through promotion. + +Branch `codex/260923-tauri-plugin-shell-2.2.1` from `origin/dev` in a scratch worktree: + +```bash +git fetch origin pull/5525/head:refs/remotes/origin/pr-5525 +git switch -c codex/260923-tauri-plugin-shell-2.2.1 origin/dev +git cherry-pick -x d6dea8f246944677c8ce80264c66095b562e3deb +``` + +Resulting diff (exactly two files, four lines): + +```diff +--- a/desktop/src-tauri/Cargo.toml ++++ b/desktop/src-tauri/Cargo.toml +-tauri-plugin-shell = "=2.2.0" ++tauri-plugin-shell = "=2.2.1" +--- a/desktop/src-tauri/Cargo.lock ++++ b/desktop/src-tauri/Cargo.lock + name = "tauri-plugin-shell" +-version = "2.2.0" ++version = "2.2.1" + source = "registry+https://github.com/rust-lang/crates.io-index" +-checksum = "bb2c50a63e60fb8925956cc5b7569f4b750ac197a4d39f13b8dd46ea8e2bad79" ++checksum = "69d5eb3368b959937ad2aeaf6ef9a8f5d11e01ffe03629d3530707bbcb27ff5d" +``` + +The lock hunk was produced by the dependency tool, not by hand, and the dependency list +of the package is unchanged, so no other lock entry moves. + +PR to `dev`, filled from the repository template, with a `Co-authored-by` trailer for +the Dependabot author because the description names the carried PR. Push with +`--no-verify`, then dispatch the full lane on the PR branch: + +```bash +gh workflow run ci.yml --ref codex/260923-tauri-plugin-shell-2.2.1 -f lane=all +``` + +A pull-request event alone would also run `desktop shell` (`desktop/**` matches both the +`ci` and `native` filters in `.github/workflows/ci.yml`), but the dispatched lane=all run +also builds the macOS bundle and the widget, which link the same crate graph. + +Acceptance: + +- `desktop shell` (`cargo fmt --check`, `cargo clippy -D warnings`, `cargo test`), + `platform-macos` and `widget` jobs succeed at the exact PR head in the lane=all run, + and the `ci` aggregate succeeds. +- Merge with `gh pr merge --admin --squash --match-head-commit ` after a clean + `git merge-tree` against the current `origin/dev`. +- After merge, `git show origin/dev:desktop/src-tauri/Cargo.toml` pins `=2.2.1`. +- #5525 is closed with a note once `main` carries the bump (wp4), because Dependabot + targets `main` and would otherwise stay open. + +## B. Security reviews of the unreviewed batch + +`MAINTAINERS.md` asks for explicit security review of changes to GitHub Actions workflows, +release automation and credential handling. Seven merged PRs had no review record: + +| Review | PRs | Surface | +|---|---|---| +| S1 | #5471 | PR quality gate script run by a `pull_request_target` workflow | +| S2 | #5456, #5653 | Bun batch runner; `ci.yml` and `release.yml`, release preflight | +| S3 | #5469 | privacy-scan gating in `ci.yml` | +| S4 | #5024, #5654, #5655 | request-owned account routing; remote workspace helper protocol | + +Each review is read-only against the merged code on `dev` and is written to scratch +space, not to this unit. Disposition rules: + +- A blocker or major finding counts only after a second, independent reviewer reproduces + it from source (file and line, concrete trigger). A finding the second reviewer cannot + reproduce is rebutted with the reason recorded in scratch. The second reviewer also + states whether the batch introduced it and whether it reaches a release artifact. +- A confirmed blocker or major gets a focused fix PR to `dev`, designed at diff level in + its own numbered doc (011, 012, ...), reviewed the same way and merged at a green exact + head before the candidate is fixed in wp3. If a fix cannot be made inside this round, + the round stops as BLOCKED rather than releasing. Once a fix has shipped, its doc is the + public record; until then the doc describes only the change and its tests. +- Minor and informational findings are recorded for follow-up and do not gate the release. + +This unit's D summary states only which reviews ran and whether any finding gated the +release; details of an unfixed weakness never enter the tracked tree. diff --git a/devlog/_fin/260923_release_2_64/011_wp2_privacy_gate_complement.md b/devlog/_fin/260923_release_2_64/011_wp2_privacy_gate_complement.md new file mode 100644 index 0000000000..1c8dd782d2 --- /dev/null +++ b/devlog/_fin/260923_release_2_64/011_wp2_privacy_gate_complement.md @@ -0,0 +1,85 @@ +# 011 — wp2: privacy scan on every pull request that `gates` skips + +## Problem + +`privacy:scan` runs in two jobs of `.github/workflows/ci.yml`. `gates` runs it on every event +it runs for, and `gates` is skipped on a pull request whose paths miss the `ci` filter. +`privacy-gate` covers that gap only when the `privacy` filter matches, and that filter lists +`devlog/**` and `ci.yml` alone. A pull request touching only paths outside both filters +(for example `docs-site/**`, `structure/**`, `native/**`, `.github/actions/**`, +`.github/release.yml`, `.github/CODEOWNERS`, the pull request template, or root markdown +other than `README.md`) therefore runs no scan, and the `ci` aggregate still concludes +success. + +The gap predates #5469, which closed it for `devlog/**` only. It does not reach an npm or +GitHub release: `release.yml` requires a push-event `ci.yml` success on the exact release +SHA, where `gates` scans the whole tree. It does reach GitHub Pages, because +`deploy-docs.yml` publishes `docs-site` on a `main` push without waiting for that scan. + +## Change + +Make `privacy-gate` the exact complement of `gates` on pull requests, so the scan's coverage +stops depending on an enumerated path list. The `privacy` filter then selects nothing and is +removed with its plumbing. + +`.github/workflows/ci.yml`: + +```diff +- privacy: +- - 'devlog/**' +- - '.github/workflows/ci.yml' +``` + +(with the comment block above it, which describes the removed filter), the `privacy` output of +`changes`, the `PRIVACY_SCOPE` validation in the `scope` step, and the `CHANGES_PRIVACY` env +of the aggregate. + +```diff + privacy-gate: + name: privacy gate + needs: changes +- if: github.event_name == 'pull_request' && needs.changes.outputs.ci != 'true' && needs.changes.outputs.privacy == 'true' ++ if: github.event_name == 'pull_request' && needs.changes.outputs.ci != 'true' +``` + +```diff +- privacy=not-requested +- if [ "$scoped" = not-requested ] && [ "$CHANGES_PRIVACY" = "true" ]; then +- privacy=requested +- fi ++ privacy=not-requested ++ if [ "$scoped" = not-requested ]; then ++ privacy=requested ++ fi +``` + +The aggregate already sets `scoped=not-requested` exactly when the event is `pull_request` +and `CHANGES_CI` is not `true`, which is the job's new condition, so both sides keep deriving +the same expectation. Comments above `privacy-gate` and the aggregate derivation are reworded +to state the complement rule. + +Tests (`tests/ci-workflows/`): + +- `ci-privacy-gate.test.ts`: the event × `ci` matrix expects exactly one scanner for every + combination: `gates` off pull requests or when `ci` is true, `privacy-gate` otherwise. + The executed-aggregate cases cover a docs-only and a no-filter pull request requiring + `privacy-gate` success, `skipped`/`failure`/`cancelled` failing by name, and a second + scan still rejected when `ci` is true. The filter and malformed-output cases for the removed + `privacy` output are replaced by an assertion that no job or step reads it. +- `ci-review-lanes.test.ts` and any other test that executes the aggregate: a pull request + with `CHANGES_CI=false` now requires `privacy-gate`; fixtures are updated to include it, + never by loosening the aggregate. + +No file-size cap is raised; if a test file would exceed its cap, the new cases move to a +sibling registered in `scripts/test-layout/layout.json` and +`tests/fixtures/test-layout-expected.json`. `structure/ops/cross-platform-ci.md` is updated +where it describes the privacy gate. + +## Acceptance + +- The PR changes `ci.yml`, so its own pull-request run sets `ci` true and scans in `gates`, + not in `privacy gate`. The complement is proven by the executed tests above, which run + the checked-in `if:` expressions and aggregate shell. +- `gates`, `structure gate` (the PR edits `structure/ops/`), and the `ci-privacy-gate` and + `ci-review-lanes` tests pass at the exact PR head; the `ci` aggregate succeeds. +- An independent reviewer confirms that no event loses a scan it had before. diff --git a/devlog/_fin/260923_release_2_64/012_wp2_request_owned_main_cursor.md b/devlog/_fin/260923_release_2_64/012_wp2_request_owned_main_cursor.md new file mode 100644 index 0000000000..1a7d29d36f --- /dev/null +++ b/devlog/_fin/260923_release_2_64/012_wp2_request_owned_main_cursor.md @@ -0,0 +1,97 @@ +# 012 — wp2: request-owned main stays out of shared active state + +## Problem + +Unreleased on `dev` since #5024: a request that carries its own main credential makes the +stored `main` account an ordinary pool candidate for that request +(`CodexAccountUsabilityOptions.requestOwnedMainCredential`). #5654 stopped three writes in +`resolveCodexAccountForThreadDetailed` from recording such a pick as the shared active account, +through a local `sharesActiveSelection` closure in `src/codex/routing.ts`. The same resolve still +reaches other writers of shared active state with the request's `selectionOptions`: + +| Site | Writer | Path | +|---|---|---| +| `src/codex/routing/selection.ts` `pickUnboundStrategyAccount`, round-robin and fill-first/reset-first branches | `rememberActiveCodexAccount` | new unbound session under a non-quota strategy | +| `src/codex/routing/selection.ts` `applyQuotaAutoSwitch` | shared active write inside the helper (persisted) | default `quota` strategy crossing the switch threshold | +| `src/codex/routing/selection.ts` `applyFailureFailover` | shared active write inside the helper | failover streak on the active account | +| `src/codex/routing.ts` priority preemption | `rememberActiveCodexAccount(preempted)` | a higher tier becomes selectable | +| `src/codex/routing.ts` bound-thread quota re-evaluation | `promoteActiveCodexAccount(cooler)` | bound thread moves to a cooler account | +| `src/codex/routing.ts` expired transient hold | `promoteActiveCodexAccount(expiredDetour)` | bound thread adopts its detour | + +When any of them picks `main` for a request that owns the main credential, later requests that +do not carry that credential read `main` as the effective (or persisted) active account. +No credential moves between callers: only the account id is recorded. + +## Change + +One rule, one helper, applied at every shared-state write reachable from a request-owned +selection. + +`src/codex/routing/selection.ts` exports: + +```ts +/** + * A main that is live only through this request's own credential serves this request alone. + * Recording it as the shared active account would route later requests through a credential + * they do not carry (see CodexAccountUsabilityOptions.requestOwnedMainCredential). + */ +export function sharesActiveSelection( + accountId: string, + selectionOptions?: CodexAccountUsabilityOptions, +): boolean { + return !(accountId === MAIN_CODEX_ACCOUNT_ID && selectionOptions?.requestOwnedMainCredential === true); +} +``` + +and guards with it: + +- `pickUnboundStrategyAccount`: both `if (commitSharedActive)` blocks become + `if (commitSharedActive && sharesActiveSelection(picked, selectionOptions))`. +- `applyQuotaAutoSwitch` and `applyFailureFailover`: every write of shared active state is + skipped when `sharesActiveSelection(target, selectionOptions)` is false. The returned account + is unchanged, so the request is still served by its own credential. + +`src/codex/routing.ts` (1618 lines against a 1626 cap; the change must not grow it past the cap): + +- Delete the local `sharesActiveSelection` closure and its comment; import the helper from + `./routing/selection`; the three existing call sites pass `selectionOptions`. +- Add the same condition to the existing `if` guarding `promoteActiveCodexAccount(cooler)`, + `promoteActiveCodexAccount(expiredDetour)` and `rememberActiveCodexAccount(preempted)`, + editing the condition in place. + +Post-response failover (`recordCodexUpstreamOutcome` quota-refusal branches and the account +exclusion path) promotes `meta.promoteAccountId` or `pickAlternateCodexAccount(...)` without +request selection options. The implementation traces where `meta.promoteAccountId` is set; if a +request-owned retry can place `main` there, the same rule is applied by carrying the request's +ownership into that metadata, and if it cannot, the PR states the reason with file and line. + +Thread affinity, round-robin ring bookkeeping and the account that serves the request are +unchanged. + +## Regression tests + +`tests/codex-integration/codex-pool-rotation.test.ts` (no file-size cap), next to the existing +`getEffectiveActiveCodexAccountId` assertions, each resolving through +`resolveCodexAccountForThreadDetailed` with +`{ requestOwnedMainCredential: true, isMainAccountTokenLive: () => true }` on a pool whose +operator-selected active account is a stored account: + +- default `quota` strategy with the active account over its switch threshold and `main` the + cooler candidate: the request resolves to `main`, while `config.activeCodexAccountId` and + `getEffectiveActiveCodexAccountId(config)` still name the operator's account; +- round-robin and fill-first new sessions whose next pick is `main`: same assertions; +- control: each scenario without `requestOwnedMainCredential` (stored main live) does move + the active account to `main`, proving the new cases are not passing because nothing moves. + +Preemption and the bound-thread paths get a case each when the fixture can reach them with a +request-owned selection; otherwise the PR names why they are unreachable for such a request. + +## Acceptance + +- The PR's pull-request run executes this file (`src/**` and `tests/**` match the `ci` filter) + and every requested job succeeds at the exact head; `file-size ratchet` passes with no cap + change. +- An independent reviewer enumerates every writer of `runtimeActiveCodexAccountId` and + `config.activeCodexAccountId` (`git grep -n -E 'rememberActiveCodexAccount|promoteActiveCodexAccount|setActiveCodexAccount|activeCodexAccountId =' -- src`) + and confirms each is guarded or unreachable from a request-owned main selection, and that + the new tests fail without the change. diff --git a/devlog/_fin/260923_release_2_64/020_wp3_dev_candidate.md b/devlog/_fin/260923_release_2_64/020_wp3_dev_candidate.md new file mode 100644 index 0000000000..994ea993d5 --- /dev/null +++ b/devlog/_fin/260923_release_2_64/020_wp3_dev_candidate.md @@ -0,0 +1,41 @@ +# 020 — wp3: dev candidate + +The candidate is the `dev` SHA after wp2's last merge. It is fixed before the version +pre-move, so the pre-move PR never changes the tree that ships. + +Dispatch the full lane on `dev` and bind it to the exact SHA: + +```bash +git fetch origin dev +CAND=$(git rev-parse origin/dev) +gh workflow run ci.yml --ref dev -f lane=all +gh run list --workflow ci.yml --branch dev --event workflow_dispatch --limit 3 \ + --json databaseId,headSha,status,conclusion +``` + +Take the run whose `headSha` equals `CAND`. If `dev` moves before the dispatch resolves, +the candidate is the run's head, and every later step uses that SHA. + +Once the candidate is bound, dispatch the dev pre-move of [030](030_wp4_release.md) §1 so its +pull request runs its own checks in parallel with the candidate run. The pre-move never changes +the candidate: the candidate is a fixed SHA, and the pre-move PR is merged only after both its +own exact-head checks and this run have finished green. + +Acceptance: every job of that run has conclusion `success`, except `privacy gate`, +which is skipped by design on `workflow_dispatch`, and the `ci` aggregate is `success`. +A job counts at its latest attempt only. + +Failure handling: + +- A Windows job failing once with a known runner-stall signature (`spawnSync ETIMEDOUT`, + a 480 s batch timeout where each file passes alone, `EPERM` on temp cleanup) is rerun + once with `gh run rerun --job ` after the run completes. +- The same case failing twice is a defect: a focused fix PR to `dev`, reviewed, merged at + a green exact head, then a new lane=all dispatch on the new candidate. +- Any non-Windows failure is a defect from the first occurrence. + +Runners: the owner's standing instruction for release rounds is that release-path runs get +the runners and other runs are cancelled by hand, one at a time, never by script. While this +run and the release runs are active, other queued or in-progress runs are cancelled +individually after reading each run's workflow, branch and event; runs on `main`, `preview`, +the candidate run, this round's own PR runs and `Release` runs are never cancelled. diff --git a/devlog/_fin/260923_release_2_64/030_wp4_release.md b/devlog/_fin/260923_release_2_64/030_wp4_release.md new file mode 100644 index 0000000000..ff81f5385e --- /dev/null +++ b/devlog/_fin/260923_release_2_64/030_wp4_release.md @@ -0,0 +1,116 @@ +# 030 — wp4: release + +Order is fixed by `scripts/version-line.ts` `assertReleasable`: a candidate must strictly +outrank every existing tag, so the preview of core 2.64.0 is published before the stable +2.64.0. This is a gate, not a convention: once `v2.64.0` exists, `2.64.0-preview.20260923` +no longer outranks the tag set and its publish job refuses. Both channels ship the wp3 +candidate tree. + +The candidate is the `dev` SHA verified in wp3, taken before the pre-move below. Its four +version sources already read 2.64.0, so the `main` promotion tree is byte-identical to the +verified tree and needs no metadata commit (precedent: candidate `a077087b74` was taken +before the 2.63.0 pre-move #5601). + +## 1. Dev pre-move + +`release.yml` refuses to publish unless `origin/dev` outranks the release version +(`version-line.ts assert-ahead`). Move `dev` to 2.65.0 first: + +```bash +gh workflow run dev-version-bump.yml --ref main -f intended-version=2.64.0 -f mode=pre-move +``` + +The workflow opens a PR changing only the four version sources to 2.65.0. It is dispatched +as soon as wp3 binds the candidate, so its checks run alongside the candidate run. Confirm +the diff is exactly `package.json`, `desktop/src-tauri/tauri.conf.json`, +`desktop/src-tauri/Cargo.toml` and the `opencodex-desktop` entry of +`desktop/src-tauri/Cargo.lock`, wait until every requested check at its exact head has +succeeded, then admin squash merge it with `--match-head-commit`. + +## 2. Promotion PRs + +Both promotions start at the candidate and merge the branch tip with the `ours` strategy, +so the promoted tree is exactly the candidate (precedent #5603 and #5602). + +```bash +git switch -c codex/260923-release-preview-2.64.0 "$CAND" +git merge -s ours --no-edit origin/preview -m "release: promote the verified 2.64.0 preview tree to preview" +# the only writer of the four version sources (scripts/release-version-sources.ts): +# package.json "version" +# desktop/src-tauri/tauri.conf.json "version" +# desktop/src-tauri/Cargo.toml [package] version +# desktop/src-tauri/Cargo.lock [[package]] opencodex-desktop version +bun scripts/release-version-sources.ts sync 2.64.0-preview.20260923 +git commit -am "release: prepare 2.64.0-preview.20260923 version metadata" +bun scripts/release-version-sources.ts check 2.64.0-preview.20260923 + +git switch -c codex/260923-release-main-2.64.0 "$CAND" +git merge -s ours --no-edit origin/main -m "release: promote the verified 2.64.0 tree to main" +bun scripts/release-version-sources.ts check 2.64.0 +``` + +Checks before opening: `git diff --stat $CAND codex/260923-release-main-2.64.0` is empty, +and the preview branch differs from `CAND` only in the four version lines. Push with +`--no-verify`, open PRs to `preview` and `main` from the template, and merge each with +`gh pr merge --merge --admin --match-head-commit ` (a merge commit, never squash, +so the candidate stays an ancestor of both release branches). + +Owner steering for this round: merge these two promotion PRs immediately after confirming +their head and base, while their PR checks are pending. This was done for #5670 and #5671. +Their release-branch push CI and Service lifecycle runs still gate publication below. + +## 3. Release-branch CI + +`release.yml` requires, for the exact release SHA: + +- a successful `ci.yml` run with event `push` on that branch (a PR run does not qualify); +- a successful Service lifecycle run, because `package.json` and `desktop/**` changed + since the previous tag. + +Read each run's jobs at the merge SHA. Failures follow wp3's rerun and defect rules. + +## 4. Dispatch + +```bash +gh workflow run release.yml --ref preview -f version=2.64.0-preview.20260923 -f tag=preview \ + -f expected-sha= -f dry-run=false +# after the preview release run succeeds: +gh workflow run release.yml --ref main -f version=2.64.0 -f tag=latest \ + -f expected-sha=
    -f dry-run=false +``` + +The release preflight fails fast on a version-source mismatch, an existing tag or release, +an npm version already present, or a tag-ordering violation. A job that fails after npm +acknowledged publication is completed by re-dispatching with the same version and +expected SHA plus `resume-after-npm-publish=true`; the version is never republished. +Other failed jobs are rerun individually. + +Push-event CI on `main` and `preview` does not run the Windows shards; the wp3 lane=all +run is the Windows evidence for this tree, which is why both promotions carry the +candidate tree unchanged apart from the preview version line. + +## 5. Verification + +```bash +curl -s https://registry.npmjs.org/@bitkyc08%2fopencodex # dist-tags.latest / .preview +gh release view v2.64.0 --json assets,isPrerelease,targetCommitish +gh release view v2.64.0-preview.20260923 --json assets,isPrerelease,targetCommitish +curl -sL https://github.com/lidge-jun/opencodex/releases/latest/download/latest.json +``` + +Acceptance: `latest` = 2.64.0 and `preview` = 2.64.0-preview.20260923 on npm; both GitHub +releases exist with the same asset count as v2.63.0 (25); `latest.json` reports 2.64.0 +with a signature for every platform entry. Registry propagation lag is waited out, not +worked around. + +A green release run can still end with registry verification `pending` (the post-publish +smoke retries six times and then reports pending rather than failing). The release-outcomes +rows of each run and a direct registry read, not the run conclusion alone, decide the +channel state. + +## 6. Close-out + +- Close #5525 with a note that `main` now carries `tauri-plugin-shell` 2.2.1 through the + 2.64.0 promotion. +- Record residual medium alerts (`serde_with`, `time`, `glib`) for a dependency round. +- D summary in `050_done.md`. diff --git a/devlog/_fin/260923_release_2_64/050_done.md b/devlog/_fin/260923_release_2_64/050_done.md new file mode 100644 index 0000000000..8ca96ac2aa --- /dev/null +++ b/devlog/_fin/260923_release_2_64/050_done.md @@ -0,0 +1,67 @@ +# 050 — 2.64.0 release outcome + +## Result + +The 2.64.0 release round is complete. The verified `dev` candidate +`a1131f521b644c09f43c924a615ea48dfca5b607` shipped to both channels. + +| Channel | Version | Promotion merge | Release run | npm dist-tag | +|---|---|---|---|---| +| preview | 2.64.0-preview.20260923 | `836321b33e` (#5670) | [35843685057](https://github.com/lidge-jun/opencodex/actions/runs/35843685057) | `preview` | +| stable | 2.64.0 | `4cb43cb0a8` (#5671) | [35847101363](https://github.com/lidge-jun/opencodex/actions/runs/35847101363) | `latest` | + +Both release runs completed successfully. Both GitHub releases are public, with 25 +assets each. `releases/latest/download/latest.json` serves `2.64.0` and has a +signature for each of its five platform entries. Direct npm registry reads after +propagation reported `latest=2.64.0` and +`preview=2.64.0-preview.20260923`. + +## Pre-release close-out + +- Critical desktop dependency update: #5661 merged as `a1131f521b`. + `tauri-plugin-shell` is pinned to `=2.2.1` in Cargo.toml and Cargo.lock. + The [full-platform branch run](https://github.com/lidge-jun/opencodex/actions/runs/35835959962) + finished with 39 successful jobs; a privacy gate skip was expected for dispatch. + #5525 closed after the update reached `main`. +- Privacy scan complement: #5662 merged as `da662a30ee`. A pull request skipped by + `gates` now runs the dedicated `privacy gate` job. The exact-head PR CI passed. +- Request-owned main selection: #5663 merged as `f2e8045140`. A request's own + main credential no longer changes shared active-account state. The exact-head + PR CI passed after tests were moved to a registered sibling file to satisfy the + file-size ratchet. +- Reviews of #5471, #5456/#5653, #5469 and #5024/#5654/#5655 were performed. + The two confirmed findings were fixed in #5662 and #5663 before the candidate + was selected. Unreleased review details were kept out of this tracked unit. + +## Verification and operations + +- The `dev` candidate passed [lane=all run 35840680817](https://github.com/lidge-jun/opencodex/actions/runs/35840680817): + 39 successful jobs and the dispatch-only privacy gate skipped as designed. +- The dev pre-move #5666 merged as `685321e297`, taking `dev` to 2.65.0 + before either publication. Its PR Cross-platform CI and Service lifecycle passed. +- The preview promotion's push Cross-platform CI + [35843639351](https://github.com/lidge-jun/opencodex/actions/runs/35843639351) + and Service lifecycle [35843639372](https://github.com/lidge-jun/opencodex/actions/runs/35843639372) + passed on `836321b33e`. The Linux `test 1/4` batch timed out once only when + twelve files ran together; the one job passed on its second attempt. All other + requested jobs succeeded. +- The stable promotion's push Cross-platform CI + [35843612570](https://github.com/lidge-jun/opencodex/actions/runs/35843612570) + and Service lifecycle [35843612468](https://github.com/lidge-jun/opencodex/actions/runs/35843612468) + passed on `4cb43cb0a8`. +- The owner directed the two promotion PRs to merge before their PR checks + finished. Their release-branch push checks succeeded before publication, + as enforced by `release.yml`. +- The first preview publish attempt reached its CI gate before the preview push + run passed; only the failed publish job was rerun. The stable packaging run + started before the preview retry, so it was cancelled to preserve the required + preview-before-stable version order, then the stable release was dispatched + again. Publication was not repeated for either version. +- Local tests, typecheck and build: NOT RUN. Hosted CI above is the execution + evidence. + +## Residuals + +Three medium Dependabot alerts remain in the desktop lockfile: +`serde_with`, `time` and `glib`. They are separate dependency work. +No release blocker remains from this round. diff --git a/devlog/_fin/260925_release_2660_prs/000_plan.md b/devlog/_fin/260925_release_2660_prs/000_plan.md new file mode 100644 index 0000000000..c14de62a27 --- /dev/null +++ b/devlog/_fin/260925_release_2660_prs/000_plan.md @@ -0,0 +1,144 @@ +# 260925 release 2.66.0 — plan + +## Reader summary + +`dev` at `76db92a4cd` carries 34 commits since v2.65.0 and one regression: #5806 and #5820 +together fail `tests/responses/protocol-direct-encoders-chat.test.ts`, which also turns every PR +based on current `dev` red. The owner asked (2026-09-25) to fix that, land eleven reviewed PRs, +and release 2.66.0. This unit fixes the regression first, lands the PRs in four groups, then +binds a candidate, pre-moves `dev` to 2.67.0, promotes and publishes. + +## Loop spec + +- Archetype: satisfy-spec, multi-cycle HOTL; one PABCD cycle per work-phase; wp1 is this + docs-only roadmap. +- Goal: the regression fixed on `dev`; #5006, #5754, #5835, #5837, #5826, #5838, #5839, #5757, + #5778, #5776, #5780 merged at green exact heads or dropped with a recorded reason; v2.66.0 and + v2.66.0-preview published and verified. +- Non-goals: deferred PRs (#5790, #5147, #5831, #5758, #5836, #5756, other drafts); raising any + file-size cap; direct pushes to `dev`/`main`/`preview`; security write-ups in tracked + directories; rewriting protected history. +- Verifier: per PR, the exact-head check list (`gh pr checks`, run jobs read at the head SHA), + focused local tests on the merge result, `bun run typecheck`; for the release, the lane=all run + on the candidate, push-event CI and Service lifecycle at the promoted SHAs, `release.yml` + outcome rows, npm dist-tags, `gh release view`, `latest.json`. +- Stop condition: 2.66.0 published and verified, or a terminal outcome below. +- Terminal outcomes: DONE; a PR that cannot be made green after root-cause work, or fails security + review, is dropped with reason and the release proceeds; UNSAFE if merged code has an unfixable + high-severity issue; BLOCKED only for a release gate that cannot pass; NEEDS_HUMAN for an owner + product decision. +- Resource bounds: no token or wall-clock budget set by the owner. Tool scope: gh (PR creation, + pushes to same-repo branches and fork branches with `maintainerCanModify`, fork workflow-run + approval, close/reopen for fresh merge refs, admin squash merges with `--match-head-commit`, + dispatch of `ci.yml`, `dev-version-bump.yml`, `release.yml`), git in scratch worktrees under + `/tmp`, local bun tests. Kimi subagents for architect, audit, security review and discovery. +- Memory artifact: this unit, the goalplan under + `.codexclaw/goalplans/land-the-owner-selected-pre-release-pr-set-into/`, `070_done.md` at the end. + +## Phase map (dependency order) + +| wp | Doc | Change | Depends on | +|---|---|---|---| +| wp1 | 000 (this) | roadmap | — | +| wp2 | [010](010_wp2_direct_chat_heartbeat.md) | direct Chat heartbeat parity fix | wp1 | +| wp3 | [020](020_wp3_5006_5754.md) | #5006, #5754 | wp2 | +| wp4 | [030](030_wp4_5835_5837_5826.md) | #5835, #5837, #5826 | wp2 | +| wp5 | [040](040_wp5_security_5838_5839_5757.md) | security review + #5838, #5839, #5757 | wp2 | +| wp6 | [050](050_wp6_5778_5776_5780.md) | #5778, #5776, #5780 | wp2 | +| wp7 | [060](060_wp7_release.md) | candidate, pre-move, promotion, publish | wp3-wp6 | + +wp2 is the foundation: until it lands, every PR's CI carries the `dev` failure. wp3-wp6 are +independent of each other (disjoint files, clean pairwise merge trees) and run in order. + +## Cross-cutting rules + +1. Exact-head evidence: a PR merges only when every expected check at its current head ran and + succeeded. Skipped counts only when the workflow's path conditions skip it for that diff. + Fork runs are approved by main only after main has read the diff. +2. Fresh merge refs: a PR whose green run predates wp2 gets a new `pull_request` event (a push, + or close/reopen) so CI tests it against the fixed `dev`. Old runs are not re-run: a re-run + reuses the stale merge ref. #5006 is the exception: it has a full green run at its head with + no `dev` overlap beyond additive layout entries, so its union risk is covered by local tests on + the merge result plus the wp7 candidate run. +3. File-size ratchet: no cap increases. Overflow moves into a registered sibling file. +4. After each merge the next PR is re-checked against the new `dev` tip (`git merge-tree`) and + union risks (layout maps, counts, locale catalogs) are re-run locally. +5. A merge that turns `dev` red freezes further merges until a repair PR lands. +6. Maintainer integration (MAINTAINERS.md 2026-09-06) is recorded on each PR with the exact-head + evidence. Security reviews stay in `.tmp/`; the PR carries only the verdict. +7. Carrying someone else's work onto a maintainer branch adds a `Co-authored-by` trailer. + +## Starting state (2026-09-25 ~11:00Z) + +| Ref | Commit | Note | +|---|---|---| +| `dev` | `76db92a4cd` | 2.66.0; last full Cross-platform CI success at `ed181a0d0c` | +| `main` | `87a78e5f26` | v2.65.0 | +| `preview` | `d4c26e2b09` | v2.65.0-preview.20260925 | + +Local checks on `76db92a4cd`: typecheck, `structure:check`, `skill:surface:check`, file-size +ratchet, repo hygiene, layout and Lab boundary tests pass; `protocol-direct-encoders-chat` fails +(47 pass, 1 fail). + + +## Architect consultation + +Architect: Kimi `kimi/kimi-for-coding-highspeed`, handle Nash (`01a0d83c-3a51-7780-befd-9e5fde34d22b`), +proposal D1-D7 received 2026-09-25. Main dispositions: + +- D1 order (regression fix first, then wp3-wp6, release last): accepted. +- D2 fresh CI: accepted with amendment. Same-repo branches (#5835, #5837, #5838, #5839): merge + `origin/dev` into the branch and push without force, but only after re-reading the head: the + author (Ingwannu, a maintainer) pushed new heads to #5837 (`de26479105`) and #5838 + (`09c6993a15`) during wp1, so each wp re-reads heads at P and leaves an actively moving branch to + its author unless the author stops. Architect's reading that a maintainer push does not reset the + readiness gate (`authorHasPushPermission`, `.github/scripts/pr-quality.cjs`) is recorded; if a PR + is drafted anyway, `gh pr ready` precedes the merge (2.65 precedent). Fork branches with + `maintainerCanModify`: a merge-from-`dev` push by main is the certain way to a fresh merge ref; + close/reopen is the fallback when a push is undesirable. +- D3 heartbeat fix: accepted in substance; delivery amended to the owner's existing branch (010 + amendment), which already includes the relayed-counter change D3 asks for. +- D4 #5778 ratchet: accepted (050). +- D5 #5839 drop criteria: accepted; the setsid descendant escape and the Windows regression are + the drop criteria, async scan and env allowlist are recommended (040). +- D6 release sequencing and runner policy: accepted (060). +- D7 union checks: accepted except the claim that `dev` itself fails the ratchet on + `src/adapters/openai-chat.ts`: rejected. `dev` has 822 lines against a cap of 822 and + `./tests/ci-workflows/file-size-ratchet.test.ts` passes on `76db92a4cd`; the 823 seen in PR runs + comes from merge refs built before #5822. + +## Audit round 1 (Kimi auditor Meitner, verdict FAIL) — dispositions + +1. BLOCKER "pre-move must use 2.67.0": rebutted. `.github/workflows/dev-version-bump.yml:27-29` + defines `intended-version` as "Version about to be released (pre-move)", and + `scripts/bump-dev-version.ts` derives the next line from it; 2.64 dispatched 2.64.0 and `dev` + moved to 2.65.0. 060's `intended-version=2.66.0` moves `dev` to 2.67.0 as intended. +2. MAJOR "#5006 lacks a maintainer approval at the head": rebutted. Ingwannu APPROVED at + `f5516ca15f` (the current head) on 2026-09-25T04:00:42Z. +3. MAJOR "security verdicts are not a PR sign-off": folded. 040 now makes a maintainer security + sign-off on the current head (an approving review by the owner citing the independent review + verdict, no exploit detail) a precondition of each merge. +4. MAJOR "#5778/#5776 still CHANGES_REQUESTED": folded. 050 makes an owner approving review on the + current head, listing the verified dispositions, a precondition; the same applies to #5780 and + to #5754 (020), whose approvals name older commits. +5. MAJOR "`PV` must be one fixed string": folded. 060 computes `PV` once at the start of wp7, + records it in the D summary, and every later command reuses it. +6. MINOR (commit reachability): both commits are on the branch (`dd80e03ca3` parent of + `ac3ea085e6`); B checks `git log origin/dev..HEAD` shows exactly those two after cherry-pick. +7. MINOR (runner policy missing in 060): rebutted; 060 section 1 carries the "Runners:" paragraph. +8. MINOR: no action. + +## Execution finding (wp3/wp4, 2026-09-25 ~12:10Z) — amends cross-cutting rule 2 + +Close/reopen does not rebuild the merge ref. The reopened runs checked out the old merge commits +(#5835 run 36131541136: `Merge 61dfd0225a into 76db92a4cd`; #5757: `... into 88b9da8c51`), so +they failed exactly as before. A fresh merge ref needs a new head: main uses GitHub's update-branch +(`gh pr update-branch `, a server-side merge of `dev` into the head branch; forks only with +`maintainerCanModify`), then approves fork runs after reading the diff. The PR description or a +comment names the merge commit so the author knows the branch moved. + +#5754 (wp3) failed its own head's contract test +`tests/codex-integration/bearer-admission-routed-provider.test.ts` "search-runTurn": the PR +deliberately plans the OpenAI search helper for runTurn adapters, and that test still asserted the +old no-helper contract; its CI had never run tests (fork approval pending). It moves to a new +work-phase wp8 (test contract update on the fork branch, then land); wp7 waits for wp8. diff --git a/devlog/_fin/260925_release_2660_prs/010_wp2_direct_chat_heartbeat.md b/devlog/_fin/260925_release_2660_prs/010_wp2_direct_chat_heartbeat.md new file mode 100644 index 0000000000..64f389efb3 --- /dev/null +++ b/devlog/_fin/260925_release_2660_prs/010_wp2_direct_chat_heartbeat.md @@ -0,0 +1,124 @@ +# 010 — wp2: direct Chat encoder heartbeat parity + +## Problem + +`dev` at `76db92a4cd` fails +`tests/responses/protocol-direct-encoders-chat.test.ts` > "direct Chat encoder stream lifecycle > +stall watchdog fails the turn like the bridge" with `SyntaxError: JSON Parse error: Unexpected EOF` +at the test's `normalizeFrames` (line 61). Neither parent PR failed on its own head: + +- #5806 (`e22209424d`) made the legacy Responses-to-Chat converter relay each typed + `response.heartbeat` as the SSE comment `: opencodex heartbeat\n\n` + (`src/chat/outbound.ts:615-623`, ADR-5805). +- #5820 (`0f4c8d4a0f`) added the direct Chat encoder (PF-09), whose writer maps + `heartbeat` to `ensureRole` only (`src/protocols/encoders/chat.ts:125`), so it sends no + byte on wire silence. + +The stall test drives both paths through heartbeat ticks. The legacy stream now carries comment +frames, `normalizeFrames` parses the empty `data` of a comment block as JSON and throws. Beyond +the test, the direct path does not deliver the keepalive ADR-5805 promises to Chat clients; the +direct encoder is off by default (`protocols.rollout.directEncoders`), so only opted-in +installs are affected. + +The same failure is the only CI failure on #5838, #5837, #5835 and #5826 at their current heads, +and on draft #5836, because each is based on the current `dev`. + +## Change + +### MODIFY `src/protocols/encoders/chat.ts` + +Replace the heartbeat alias with the same comment the converter emits, through the sink's +keepalive channel (a keepalive does not count as wire activity, matching ADR-5805 "does not +reset a semantic-progress watchdog"), gated like the Messages writer on termination and demand +(`src/protocols/encoders/messages.ts:207-210`): + +```diff ++/** The converter's typed-heartbeat relay (ADR-5805); comment lines produce no Chat event. */ ++const CHAT_HEARTBEAT_COMMENT = ": opencodex heartbeat\n\n"; +@@ + return { + start: ensureRole, +- heartbeat: ensureRole, ++ heartbeat() { ++ ensureRole(); ++ if (terminated || failed || sink.desiredSize() <= 0) return; ++ sink.emitKeepalive(CHAT_HEARTBEAT_COMMENT); ++ }, +``` + +If the demand gate makes the direct stream diverge from the legacy one under the parity test +(the legacy converter enqueues without a demand check), B drops the `desiredSize` clause and +records why; the terminal/failed guard stays. + +### MODIFY `tests/responses/protocol-direct-encoders-chat.test.ts` + +`normalizeFrames` keeps comment-only blocks as a comparable marker instead of parsing them, +so parity covers keepalives too: + +```diff + function normalizeFrames(text: string): unknown[] { + return text.split("\n\n").filter(block => block.trim().length > 0).map(block => { +- const data = block.split("\n").filter(line => line.startsWith("data:")).map(line => line.slice(5).trim()).join(""); ++ const lines = block.split("\n"); ++ // SSE comment blocks (the heartbeat relay) carry no event; compare them verbatim. ++ if (!lines.some(line => line.startsWith("data:"))) return lines.join("\n"); ++ const data = lines.filter(line => line.startsWith("data:")).map(line => line.slice(5).trim()).join(""); +``` + +and the stall test asserts the keepalive is present on the direct stream: + +```diff + expect(directFrames).toEqual(legacyFrames); ++ expect(directFrames).toContain(": opencodex heartbeat"); + expect(JSON.stringify(directFrames.at(-1))).toContain("upstream_stall_timeout"); +``` + +### MODIFY `structure/transports/streaming-health.md` (SoT sync) + +After the paragraph that describes the converter's comment relay (line 36-40), add one sentence: +the direct Chat encoder (PF-09, `src/protocols/encoders/chat.ts`) emits the same comment on its +wire-silence tick through the keepalive channel, so enabling `directEncoders` keeps ADR-5805. + +## Acceptance + +| Row | Activation | Observable proof | +|---|---|---| +| A1 red | test change only, encoder unchanged | stall test fails: direct frames lack the heartbeat marker | +| A2 green | encoder change applied | `bun test ./tests/responses/protocol-direct-encoders-chat.test.ts` 48 pass 0 fail | +| A3 neighbours | same | `bun test ./tests/chat ./tests/protocols` (existing dirs) and `bun run test:changed` pass | +| A4 types/structure | same | `bun run typecheck` exit 0, `bun run structure:check` exit 0 | +| A5 hosted | PR to `dev` | every required check at the exact head `success`; admin squash merge with `--match-head-commit` | + +## Delivery + +Branch `codex/260925-direct-chat-heartbeat` from `origin/dev`, one commit, maintainer PR from +the template, maintainer integration recorded in the PR body (MAINTAINERS.md 2026-09-06 rule). + + +## Amendment after architect consultation (supersedes "Change" and "Delivery" above) + +The owner already wrote this fix on 2026-09-25 on the unpushed local branch +`fix/protocols-fu-encoder-heartbeat` (two commits on top of #5820): + +- `dd80e03ca3` fix(protocols): relay heartbeat keepalives from the direct encoders — + `src/protocols/encoders/chat.ts` heartbeat emits `: opencodex heartbeat` through + `sink.emitKeepalive` after `ensureRole` (guarded by `terminated`); + `src/protocols/encoders/adapter-events.ts` stops counting keepalives as relayed events + (`if (activity) relayed(observation)`), matching the bridge; parity tests in + `protocol-direct-encoders-chat.test.ts` and `protocol-direct-encoders-messages.test.ts` + compare comment-only blocks and cover heartbeats on both wires. +- `ac3ea085e6` docs(structure): `structure/data-planes/protocol-paths.md` and + `structure/transports/responses.md` note the direct-encoder keepalives. + +This is broader than the draft above (it also closes the relayed-event over-count on every +direct encoder) and is the owner's own work, so wp2 ships it: cherry-pick both commits onto +`origin/dev` as branch `codex/260925-direct-encoder-heartbeat`, authorship preserved. The draft +diff above was validated independently in a scratch worktree (red: parity mismatch; green: 48 pass; +`test:changed` 6971 pass / 0 fail; typecheck 0) and serves as the cross-check only. + +An unrelated fork branch `luvs01/fix-encoder-keepalives` (`d5c4b3046a`, `2ff29a9771`, no PR) +fixes the same two points; it is not used, so it carries no co-author obligation. + +Acceptance rows A1-A5 stand, applied to the cherry-picked branch; A2 also runs +`./tests/responses/protocol-direct-encoders-messages.test.ts`, and A4 adds +`bun run structure:check` for the two structure edits. diff --git a/devlog/_fin/260925_release_2660_prs/020_wp3_5006_5754.md b/devlog/_fin/260925_release_2660_prs/020_wp3_5006_5754.md new file mode 100644 index 0000000000..0abdd7a0aa --- /dev/null +++ b/devlog/_fin/260925_release_2660_prs/020_wp3_5006_5754.md @@ -0,0 +1,56 @@ +# 020 — wp3: #5006 and #5754 + +Both are contributor PRs from forks with `maintainerCanModify: true`. Merge trees against +`origin/dev` (`76db92a4cd`) are clean; the only files either shares with commits on `dev` since +its base are the two layout maps (additive entries) and, for #5754, additive text in +`docs-site/src/content/docs/reference/configuration/server.md` and +`structure/providers-and-adapters.md`. Neither touches a file tracked by +`tests/fixtures/file-size-baseline.json`, a hand-maintained count, `src/cli/capabilities.ts` or +`skills/ocx/`. + +## #5006 — accept an account alias and "auto" in the Codex pool verbs + +- Head `f5516ca15f`, base 33 commits behind `dev`. Every Cross-platform CI job ran and passed at + that head (test 1/4-4/4, gates, keyring, docker, docs, npm-global, desktop shell); macOS, + Windows, structure and privacy legs were skipped by the workflow's path conditions. CodeRabbit + threads all resolved. +- Union check (the head's CI predates 33 `dev` commits): build the merge result in a scratch + worktree and run `bun test ./tests/cli/cli-account-alias-target.test.ts + ./tests/cli/cli-account-pool-verbs.test.ts ./tests/test-layout.test.ts + ./tests/test-layout-tooling.test.ts` plus `bun run typecheck` on it. Green -> merge. +- No push to the branch: a new commit resets the contributor readiness gate to draft. +- Merge: `gh pr merge 5006 --squash --admin --match-head-commit f5516ca15f...` with the + maintainer-integration note (MAINTAINERS.md 2026-09-06) in a PR comment. + +## #5754 — hosted web search for runTurn adapters + +- Head `9ba90d86fc`. Only `pull_request_target` checks ran; Cross-platform CI run + `36102264328` and React Doctor run `36102264301` are `action_required` (fork approval). + The earlier maintainer approval names `b6adb964d7`, which is not an ancestor of the head, so + no review covers the current head. +- Steps after wp2 lands on `dev`: + 1. Main reviews the full current diff (`src/web-search/run-turn-loop.ts` new, + `src/server/responses/run-turn-execution.ts`, `request-sidecar-auth.ts`, + `sidecar-execution.ts`). `request-sidecar-auth.ts` is on the credential path, so an + independent Kimi reviewer checks that the synthetic tool loop reuses existing sidecar auth + and adds no credential destination. + 2. Fresh CI on the fixed `dev`: close and reopen the PR (a `reopened` event builds a new + merge ref without a commit), then approve the new fork runs + (`gh api -X POST repos/lidge-jun/opencodex/actions/runs//approve`) after reading the diff. + The old `action_required` runs are left alone: approving them would test a stale merge ref. + 3. Every required check at the head `success` (structure gate included, since structure files + changed), then squash merge with `--match-head-commit` and the maintainer-integration note. + +## Acceptance + +`gh pr view --json state,mergeCommit` shows MERGED for both, each with the exact-head +evidence recorded in the PR comment. A PR that cannot be made green after root-cause work is +dropped from this release with the reason recorded here. + + +## Audit round 1 amendments + +- #5006: Ingwannu APPROVED at the current head `f5516ca15f`; no new review needed unless the head moves. +- #5754: the only approval names `b6adb964d7`. Precondition of the merge: main reviews the current + head and submits an approving review as the owner (`gh pr review 5754 --approve --body ...`) that + names the head SHA and the exact-head CI run. diff --git a/devlog/_fin/260925_release_2660_prs/030_wp4_5835_5837_5826.md b/devlog/_fin/260925_release_2660_prs/030_wp4_5835_5837_5826.md new file mode 100644 index 0000000000..d5ff26dc5f --- /dev/null +++ b/devlog/_fin/260925_release_2660_prs/030_wp4_5835_5837_5826.md @@ -0,0 +1,50 @@ +# 030 — wp4: #5835, #5837, #5826 + +All three merge cleanly into `origin/dev`, touch disjoint files, and touch nothing tracked by +the file-size ratchet or a hand-maintained count. At their current heads each has exactly one +failing test, `tests/responses/protocol-direct-encoders-chat.test.ts` "stall watchdog fails the +turn like the bridge", which is the `dev` defect wp2 fixes. So wp4 starts after wp2 merges. + +| PR | Author / head repo | Branch | Head | Open findings | +|---|---|---|---|---| +| #5835 bounded response resources | Ingwannu (maintainer) / `lidge-jun/opencodex` | `fix/bounded-body-retention` | `61dfd0225a` | none; zero review threads | +| #5837 linear fragmented stream work | Ingwannu (maintainer) / `lidge-jun/opencodex` | `perf/stream-parser-scaling` | `ffc203dad7` | 5 unresolved threads, 2 correct | +| #5826 native Claude tiers 1M | FredAmartey / fork, `maintainerCanModify: true` | `fix/claude-1m-tier-default` | `dac73419ad` | none; 1 thread resolved | + +## #5837 findings disposition + +- ADR ownership marker (Codex threads on `ADR-0102-incremental-stream-accounting.md:3/5`, + CodeRabbit on `:3`): correct. `structure/AGENTS.md` makes the `> Decision record:` line in + the owning section the ownership link. Fix: in + `structure/dashboard-and-usage.md`, the usage-accounting section carries + `> Decision record: [ADR-0102](decisions/ADR-0102-incremental-stream-accounting.md)` and the ADR + header keeps `- Contract owner:`, matching ADR-5805's shape. B verifies the exact convention + against an existing ADR pair before editing. +- Steady-state refresh wording (CodeRabbit on `structure/dashboard-and-usage.md:348`): correct; + `readUsageEntriesIncrementally` still hashes the retained region on each refresh. Reword to + "each refresh verifies the retained region and parses only appended bytes; matching bounds skip + the second hash". +- Title form (Codex on `:1`): already `decision recorded under "Usage accounting"`; resolve as + addressed. +- Size budget (Codex on `structure/dashboard-and-usage.md:352`): `structure gate` passed at the + head; re-run `bun run structure:check` after the edit and resolve with that output. + +## Steps + +1. After wp2 merges, bring each branch onto the new `dev`: #5835 and #5837 live in this + repository, so main merges `origin/dev` into each branch from a scratch worktree + (`git merge --no-edit origin/dev`), adds the #5837 doc fixes as one commit, and pushes with a + plain (non-force) push so a concurrent author push fails the push instead of being overwritten. + #5826 is a fork: close and reopen for a fresh merge ref, then approve its fork runs. +2. Local union proof on each merge result: `bun run typecheck`, the PR's own test files, + `./tests/test-layout.test.ts`, `./tests/test-layout-tooling.test.ts`, + `./tests/responses/protocol-direct-encoders-chat.test.ts`, and `bun run structure:check` for + #5835/#5837 (they edit `structure/`). +3. Every required check at the new head `success`, then squash merge with + `--match-head-commit` and the maintainer-integration note. + +## Acceptance + +All three MERGED at green exact heads, #5837's two correct findings fixed and all five threads +resolved with a reply naming the fixing commit or the evidence. + diff --git a/devlog/_fin/260925_release_2660_prs/040_wp5_security_5838_5839_5757.md b/devlog/_fin/260925_release_2660_prs/040_wp5_security_5838_5839_5757.md new file mode 100644 index 0000000000..243510003d --- /dev/null +++ b/devlog/_fin/260925_release_2660_prs/040_wp5_security_5838_5839_5757.md @@ -0,0 +1,95 @@ +# 040 — wp5: #5838, #5839, #5757 (security review) + +MAINTAINERS.md requires explicit security review for authentication, credential handling and +other security-boundary changes. Each PR got an independent Kimi security reviewer +(`kimi/kimi-for-coding-highspeed`) on 2026-09-25. Reports live in gitignored scratch +(`.tmp/260925/sec_.md`) and are never copied into this unit. This doc records only the +verdicts and the dispositions of findings that are already public in each PR's review threads. + +| PR | Author / head repo | Head | Reviewer verdict | Public unresolved threads | +|---|---|---|---|---| +| #5838 durable management mutations | Ingwannu / `lidge-jun/opencodex` `fix/durable-mutation-atomicity` | `e426ed5d42` | PASS-WITH-FIXES | 2, both correct | +| #5839 bound Cursor foreground shell lifetime | Ingwannu / `lidge-jun/opencodex` `fix/cursor-shell-lifecycle` | `d5af1545ec` | PASS-WITH-FIXES | 4, all correct or directionally correct | +| #5757 authenticate local Claude-intercept proxy clients | luvs01 / fork `fix/claude-intercept-proxy-auth-5751` | `be0e20699b` | PASS (one optional logging note) | 0 | + +## #5838 — fixes before merge + +1. `src/server/management/provider-patch-transaction.ts:49` (Codex P2): do not roll the live + config back once `persistConfigUnlocked` has atomically published `config.json`. B reads + `saveConfigPreservingClaudeCode` and `src/config/live-reconcile.ts` to find the publication + point, then makes the transaction distinguish "failed before publish" (roll back) from "failed + after publish" (keep live = disk, surface the error). Regression test: inject a throw after + publication and assert live config equals the written file. +2. `provider-patch-transaction.ts:32` (CodeRabbit): rollback restores descriptors but not key + order; restore the original insertion order (rebuild the providers object from the snapshot's + key list) so fallback-default selection cannot change after a failed save. Regression test: + failed save leaves `Object.keys(providers)` identical. + +## #5839 — fixes before merge, or drop + +1. `src/adapters/cursor/native-foreground-shell.ts:83` (Codex P1): the new unconditional Windows + refusal breaks the existing Windows success cases in + `tests/providers/cursor/cursor-native-exec-shell.test.ts`. B first establishes whether Windows + foreground shells worked before this PR (read `shellStreamExec` on `origin/dev`). If they did, + the refusal is a user-visible regression and must be replaced by a Windows-capable owner + (job-object or `taskkill /T` tree kill) or scoped so Windows keeps the old path. +2. `:107` (Codex P1): descendants that leave the process group (`setsid`) escape teardown. + Minimum acceptable fix: teardown confirmation fails closed (reports the shell as not cleaned up + and keeps the transport's cleanup pending) when a descendant of the original PID survives. +3. `:66` (CodeRabbit): the 25 ms synchronous `/proc` scan; make it asynchronous or bounded. +4. `:84` (CodeRabbit): document the new limits (1 MiB output cap, Windows behavior). + +Decision rule: #5839 is a hardening PR, and `dev` without it keeps today's behavior. If +fixes 1-2 cannot be completed and proven on Linux and macOS CI plus a Windows CI leg in this +round, #5839 is dropped from 2.66.0 with the reason recorded here and on the PR. It is not merged +with a known Windows regression. + +## #5757 + +No required fixes. Its only CI failure (`file-size ratchet: repository`, `src/adapters/openai-chat.ts` +823 vs 822) comes from its stale merge ref: `dev` brought that file to 822 in #5822. Fresh CI +on the fixed `dev` (close/reopen, approve fork runs) should clear it. The CHANGES_REQUESTED +review must be re-reviewed against the head: main confirms the two blockers named in that review +are addressed (token read per CONNECT, lstat/O_NOFOLLOW/owner checks) and records a maintainer +security sign-off comment on the PR, quoting the scratch verdict without exploit detail. + +## Steps (per PR) + +Bring onto the fixed `dev` (same-repo: merge `origin/dev` in a scratch worktree, plain push; +fork: close/reopen and approve runs), apply fixes as separate commits, local proof (typecheck, +the PR's tests, layout tests, `protocol-direct-encoders-chat`), exact-head CI green including +Windows legs where the diff reaches Windows code, record the security sign-off comment, squash +merge with `--match-head-commit`. + + +## Audit round 1 amendment — merge precondition + +Before each of #5838, #5839, #5757 merges, main submits an approving review on the current head +(`gh pr review --approve`) as the maintainer security sign-off. The body states: independent +security review performed on (reviewer model and date), verdict, and how each public +review thread was resolved (fixing commit or reasoned reply). It carries no exploit detail and +nothing that is not already public. No sign-off, no merge. + +Heads in the table above are as of wp1 (2026-09-25 ~11:00Z). During wp1 the author pushed #5838 to +`09c6993a15` (and #5837 to `de26479105`); wp5's P re-reads every head, and each sign-off names +the head it actually reviewed. The D summary for this unit is `070_done.md`. + +## wp5 P re-verification (2026-09-25 ~12:50Z) + +Heads: #5838 `7d33b4577d`, #5839 `d8e3deae3a` (after update-branch), #5757 `f299c6c598`. +Second-round independent security reviews (scratch `.tmp/260925/sec__r2.md`): + +- #5838: PASS. Both required fixes landed with tests + (`tests/server/management-provider-atomicity.test.ts`): no live rollback once `config.json` is + published, and rollback restores provider key order. Two newer CodeRabbit threads are correct + but describe gaps `dev` already had before this PR; they are accepted as follow-ups and + tracked on the PR's review threads. +- #5839: PASS. The author replaced the escapable process-group owner with a fail-closed stub: + Cursor foreground native shells (`shellArgs`, `shellStreamArgs`) are refused before spawn on every + platform until a kernel-backed descendant owner exists (ADR-0122). Background shells and other + native operations are unchanged; `nativeLocalExec` stays explicit and default-off, so only + opted-in installs see the change. Both drop criteria are resolved (no escape, no Windows-only + regression). This is a user-visible behavior change for opted-in users and goes in the release + notes. Remaining CodeRabbit thread (`native-foreground-shell.ts:21`, fallback redirect text + without a hint): B aligns it with the existing default bridge wording. +- #5757: PASS (first round); CI green at `f299c6c598`. diff --git a/devlog/_fin/260925_release_2660_prs/050_wp6_5778_5776_5780.md b/devlog/_fin/260925_release_2660_prs/050_wp6_5778_5776_5780.md new file mode 100644 index 0000000000..2c376d069a --- /dev/null +++ b/devlog/_fin/260925_release_2660_prs/050_wp6_5778_5776_5780.md @@ -0,0 +1,58 @@ +# 050 — wp6: #5778, #5776, #5780 + +All three are luvs01 fork PRs with `maintainerCanModify: true`. Each merges cleanly into +`origin/dev` and the three merge cleanly together. Every head's only CI failure is +`file-size ratchet: repository` in test 2/4. + +| PR | Head | Review state | Ratchet cause | +|---|---|---|---| +| #5778 retry single targets only after cooldown | `bc86f4eb28` | CHANGES_REQUESTED (Ingwannu) on `1e64a046`; addressed by `d0e08ae7cc` | own: raises `tests/server/server-combo-failover-e2e.test.ts` cap 4166 -> 4218 in `tests/fixtures/file-size-baseline.json`, plus the stale `src/adapters/openai-chat.ts` 823/822 | +| #5776 report external Codex ownership | `aa40e3ae` | CHANGES_REQUESTED (Ingwannu) on `bc718d94`; addressed by `3e9eef07`, `aa40e3ae` | stale merge ref only (`openai-chat.ts`, fixed on `dev` by #5822) | +| #5780 remove only recorded catalog backups | `10696b8a2` | APPROVED (Ingwannu) on `0b309a39`; two later commits change production paths | stale merge ref only | + +## Review dispositions (verified by main at P of wp6 against the head diff) + +- #5778: the request was to gate the single-target replay on this failure's own cooldown + record and add a stale-generation regression. Present at `src/combos/failover.ts:205` + (`coolComboTarget` returns boolean), `src/combos/resolve.ts:341` (`onCooldownRecorded`), + `src/server/responses/core-combo.ts:761-780` and `:811` (`failedTargetCooled` gate), and + `tests/server/server-combo-cooldown-recording.test.ts` (reconciled-away target regression). +- #5776: catch path re-observes ownership before `injection_refused` + (`src/codex/desktop-switches.ts:222-225`); CLI guidance in `src/cli/system-command.ts` and + `src/cli/agent.ts` no longer sends `ownership_undetermined` to `ocx sync`; docs updated. +- #5780: the two post-approval commits initialize ownership metadata before publication and + record ownership before temp cleanup (`src/codex/internal/catalog-writer.ts:222-225`, + `src/codex/catalog/retained-sync.ts`); main reviews them as the re-approval. + +Main records these on each PR as the owner's review (an approving review on the current head +with the verification list). The earlier change requests stay visible; they are answered by the +commits named above, which MAINTAINERS.md accepts as "resolved". + +## #5778 ratchet fix (no cap increase) + +1. Restore `tests/fixtures/file-size-baseline.json` to `dev`'s value for + `tests/server/server-combo-failover-e2e.test.ts` (4166). +2. Move the new single-target cases the PR added to that file (about 58 lines) byte for byte into + `tests/server/server-combo-cooldown-recording.test.ts`, which the PR already registers in both + `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`; carry over any + helper imports they need. The e2e file must end at or below 4166 lines on the merge result. +3. Run both files alone and together; a moved case that changes colour is investigated as a + state-dependency in the case (AGENTS.md), never fixed by weakening it. + +Delivery: a commit on the fork branch if the readiness gate tolerates a maintainer push +(verified in `.github/scripts` at wp6 P); otherwise a maintainer carry PR from +`codex/260925-carry-5778` with `Co-authored-by: luvs01` and the original PR closed as carried. + +## Steps + +Fresh CI on the fixed `dev` (fork: close/reopen or the #5778 push, then approve runs after +reading the diff), local proof on each merge result (typecheck, the PR's tests, +`./tests/ci-workflows/file-size-ratchet.test.ts`, layout tests), every required check at the +exact head green, owner review recorded, squash merge with `--match-head-commit`. + + +## Audit round 1 amendment — merge precondition + +#5778, #5776 and #5780 each merge only after main submits an approving review on the current head +(`gh pr review --approve`) listing the verified dispositions above and the head SHA. Ingwannu's +earlier change requests remain on record; the approving review answers them with the commits named. diff --git a/devlog/_fin/260925_release_2660_prs/060_wp7_release.md b/devlog/_fin/260925_release_2660_prs/060_wp7_release.md new file mode 100644 index 0000000000..771f4aaca0 --- /dev/null +++ b/devlog/_fin/260925_release_2660_prs/060_wp7_release.md @@ -0,0 +1,132 @@ +# 060 — wp7: release 2.66.0 + +Procedure follows `devlog/_fin/260923_release_2_64/020_wp3_dev_candidate.md` and +`030_wp4_release.md`, which 2.65.0 also used (round `260925_release_2650_prs`). Only the +version values and SHAs differ. Starting state for this round: + +| Ref | Commit | Version | +|---|---|---| +| `main` | `87a78e5f26` | 2.65.0 (npm `latest`, release v2.65.0, 25 assets) | +| `preview` | `d4c26e2b09` | 2.65.0-preview.20260925 (npm `preview`) | +| `dev` | moves during wp2-wp6 | 2.66.0 in all four version sources since #5785 | + +## 1. Candidate + +After wp6's last merge: `CAND=$(git rev-parse origin/dev)`, then +`gh workflow run ci.yml --ref dev -f lane=all` and bind the run whose `headSha == CAND`. +Acceptance: every job `success` at its latest attempt, `privacy gate` skipped by design on +`workflow_dispatch`, aggregate `ci` success. Known Windows runner-stall signatures +(`spawnSync ETIMEDOUT`, 480 s batch timeout with each file passing alone, `EPERM` on temp +cleanup) get one `gh run rerun --job`; a second identical failure, or any non-Windows +failure, is a defect: focused fix PR to `dev`, merged green, new candidate and new run. + +Runners: while the candidate, promotion and release runs are active, competing queued or +in-progress runs are cancelled by hand one at a time after reading workflow, branch and event. +Never cancel runs on `main`, `preview`, the candidate run, this round's PR runs, or `Release`. + +## 2. Dev pre-move to 2.67.0 + +```bash +gh workflow run dev-version-bump.yml --ref main -f intended-version=2.66.0 -f mode=pre-move +``` + +Dispatched once the candidate is bound. The PR it opens must change exactly the four version +sources (`package.json`, `desktop/src-tauri/tauri.conf.json`, `desktop/src-tauri/Cargo.toml`, +the `opencodex-desktop` entry of `desktop/src-tauri/Cargo.lock`) to 2.67.0. Merge with +`gh pr merge --squash --admin --match-head-commit ` after its exact-head checks and the +candidate run are both green. The candidate SHA does not change. + +## 3. Promotion PRs + +```bash +PV=2.66.0-preview. +git switch -c codex/260925-release-preview-2.66.0 "$CAND" +git merge -s ours --no-edit origin/preview -m "release: promote the verified 2.66.0 preview tree to preview" +bun scripts/release-version-sources.ts sync "$PV" +git commit -am "release: prepare $PV version metadata" +bun scripts/release-version-sources.ts check "$PV" + +git switch -c codex/260925-release-main-2.66.0 "$CAND" +git merge -s ours --no-edit origin/main -m "release: promote the verified 2.66.0 tree to main" +bun scripts/release-version-sources.ts check 2.66.0 +``` + +Checks before opening: `git diff --stat $CAND codex/260925-release-main-2.66.0` is empty and +the preview branch differs from `CAND` only in the four version lines. Push with +`--no-verify`, open both PRs from the template, merge each with +`gh pr merge --merge --admin --match-head-commit ` (merge commit, never squash, so the +candidate stays an ancestor of both release branches). + +## 4. Release-branch CI + +`release.yml` requires, at each exact release SHA, a successful push-event `ci.yml` run on that +branch and a successful Service lifecycle run (`package.json` changed since the previous tag). +Read each run's jobs at the merge SHA; failures follow section 1's rerun and defect rules. + +## 5. Dispatch + +```bash +gh workflow run release.yml --ref preview -f version="$PV" -f tag=preview \ + -f expected-sha= -f dry-run=false +# after the preview release run succeeds (tag ordering: preview of a core before its stable): +gh workflow run release.yml --ref main -f version=2.66.0 -f tag=latest \ + -f expected-sha=
    -f dry-run=false +``` + +A job that fails after npm acknowledged publication is completed by re-dispatching with the same +version and expected SHA plus `resume-after-npm-publish=true`; a version is never republished. + +## 6. Verification + +```bash +curl -s https://registry.npmjs.org/@bitkyc08%2fopencodex # dist-tags.latest / .preview +gh release view v2.66.0 --json assets,isPrerelease,targetCommitish +gh release view "v$PV" --json assets,isPrerelease,targetCommitish +curl -sL https://github.com/lidge-jun/opencodex/releases/latest/download/latest.json +``` + +Acceptance: npm `latest` = 2.66.0 and `preview` = `$PV`; both GitHub releases exist with 25 +assets (same as v2.65.0); `latest.json` reports 2.66.0 with a signature for every platform. The +release-outcomes rows of each run plus a direct registry read decide the channel state; a green +run with registry verification `pending` is waited out, not worked around. + + +## Audit round 1 amendment — fixed preview version + +At the start of wp7, compute once and record in the D summary: + +```bash +PV="2.66.0-preview.$(TZ=Asia/Seoul date +%Y%m%d)" +``` + +Every later command (version sync, check, release dispatch, verification) uses that recorded +string, even if the publish crosses midnight KST. + +## wp7 P (2026-09-25 ~13:45Z) + +- `PV=2.66.0-preview.20260925` (fixed now; every later command reuses it). +- Candidate `CAND=82cb66e82da2f4bbcd094086ad2970d19c1612cd` (`dev` after #5778, the last in-scope merge; #5754 from wp8 + is already in it). lane=all run `36142367892` (`workflow_dispatch`, headSha = CAND). +- No `v2.66*` tag and no npm `2.66.0` yet; `dev` carries 2.66.0; `main` `87a78e5f26`, `preview` `d4c26e2b09`. +- #5839 now refuses Cursor foreground native shells for `nativeLocalExec` opt-in installs; its squash subject + ("fail closed without shell containment") carries that into the generated notes. + +## wp7 audit round 1 (gpt-6-sol, FAIL) — dispositions + +1. BLOCKER "candidate run still in progress": a sequencing gate the plan already has (section 1); + no dispatch happens before every job of `36142367892` concludes `success` and push-event CI plus + Service lifecycle succeed at each promotion SHA. Not a plan change. +2. MAJOR "pre-move executes older automation": folded. `origin/main:.github/workflows/dev-version-bump.yml` + predates the #5786 hardening and runs `dev` code with a persisted write credential. The pre-move + is prepared by hand instead, reproducing the workflow's steps from a scratch worktree at the + candidate: `bun scripts/bump-dev-version.ts 2.66.0 package.json` (decides 2.67.0), + `bun scripts/release-version-sources.ts sync 2.67.0` then `check 2.67.0`, prove no `v2.66.0` / + `v2.67.0` tag and no npm `2.66.0`, `bun test tests/ci-workflows/release-version-line.test.ts`; + the diff must be exactly the four version sources. PR `codex/260925-dev-pre-move-2.67.0` to + `dev`, merged at a green exact head before the release dispatch. The workflow hardening reaches + `main` through this release's promotion. +3. Commands for dispatch, `expected-sha` (the branch's merge commit, full 40 chars) and + preview-before-stable ordering confirmed as written. + Round 2 (NEAR-PASS) residuals folded: this manual procedure supersedes the section 2 + `gh workflow run dev-version-bump.yml` command for this round, and `git fetch --force --tags origin` + runs before `release-version-line.test.ts` (the test returns early without tags). diff --git a/devlog/_fin/260925_release_2660_prs/070_done.md b/devlog/_fin/260925_release_2660_prs/070_done.md new file mode 100644 index 0000000000..edd60d2a34 --- /dev/null +++ b/devlog/_fin/260925_release_2660_prs/070_done.md @@ -0,0 +1,65 @@ +# 070 — done: 2.66.0 release round + +## Outcome + +2.66.0 shipped from `dev` candidate `82cb66e82da2f4bbcd094086ad2970d19c1612cd` as preview `2.66.0-preview.20260925` and stable +`2.66.0`. Every PR in scope landed; none was dropped. `dev` now carries 2.67.0. + +## What landed on `dev` + +| Order | PR | Merge commit | Notes | +|---|---|---|---| +| wp2 | #5847 direct-encoder heartbeat keepalives | `86e386b821` | owner's local fix; closed the #5806 x #5820 union regression | +| wp3 | #5006 Codex pool alias and `auto` | `6975fc37fe` | green at head; union re-proved locally | +| wp4 | #5837 linear fragmented stream work | `34ef9b12d6` | review findings fixed by the author | +| wp4 | #5835 release bounded response resources | `e34cb3db3a` | | +| wp4 | #5826 native Claude tiers 1M | `bcdbfc6ba7` | | +| wp5 | #5838 durable management mutations | `b0efef5db0` | security sign-off; two dev-parity follow-ups | +| wp5 | #5757 Claude-intercept proxy authentication | `8182562648` | security sign-off | +| wp8 | #5754 hosted search for runTurn adapters | `db12855fab` | stale contract test updated by maintainer commit `223c46bbcc` | +| wp6 | #5776 external Codex ownership reporting | `4183609d55` | structure budget fix | +| wp5 | #5839 Cursor foreground shell fail-closed | `929d4ff8e1` | security sign-off; behavior change below | +| wp6 | #5780 remove only recorded catalog backups | `9c28acf6a1` | structure budget and wording fixes | +| wp6 | #5778 single-target retry after cooldown | `82cb66e82d` | ratchet cap restored, cases moved | +| wp7 | #5852 dev pre-move to 2.67.0 | `ba3b3c56fa` | prepared by hand (see 060) | + +Promotions: #5854 `preview` `0c37e74002`, #5855 `main` `e70b3d86fb` (merge commits from the candidate with +`-s ours`; main tree identical to the candidate, preview differs only in the four version lines). + +## Evidence + +- Candidate: Cross-platform CI lane=all run `36142367892` (`workflow_dispatch`, attempt 1): 39 success, privacy gate skipped by design. +- Promotion SHAs: push-event Cross-platform CI `36145445621` (preview) and `36145464653` (main) success; + Service lifecycle `36145445278` and `36145464796` success. +- Release runs: preview `36148350474` success; stable `36150657310` success. Both runs ended with npm registry verification pending (publish acknowledged, read-back not yet visible); a direct registry read at 15:21Z showed `latest` = 2.66.0 and `preview` = 2.66.0-preview.20260925. GitHub releases `v2.66.0` and `v2.66.0-preview.20260925` each carry 25 assets; `latest.json` reports 2.66.0 with signatures for darwin-aarch64, darwin-x86_64, windows-x86_64, linux-x86_64 and linux-x86_64-deb. +- Every PR merged at an exact head whose required checks ran and passed (details in each PR's integration comment + or owner review). Security sign-offs live on #5838, #5757, #5839 as approving reviews; the reviews themselves + stayed in scratch. + +## Behavior change to call out + +#5839: Cursor foreground native shells (`shellArgs`, `shellStreamArgs`) are now refused before spawn on every +platform until a kernel-backed descendant owner exists (ADR-0122). It affects only installs that opted into +`nativeLocalExec` (default off); background shells and other native operations are unchanged. + +## What did not go to plan + +- Close/reopen does not rebuild a PR's merge ref; seven reopened runs re-tested the stale pre-#5847 merge. Fresh + CI needs a new head (`gh pr update-branch` or a push). Cost: one CI wave. +- #5754's earlier CI had never run tests (fork approval pending), so a contract test it contradicts was only found + by the local union check. +- Docs-only budget overruns (`structure/config.md`, `structure/catalog.md` at 600 lines) appeared only after + `dev` merged into #5776 and #5780; the per-PR structure gate cannot see them earlier. +- A maintainer test move (#5778) carried a duplicated helper block that bun tolerated and `tsc` never saw, + because `tsconfig.json` includes only `src`. The independent audit caught it. +- Kimi hit its 5-hour quota mid-round; later subagents ran on gpt-6-luna and then gpt-6-sol at the owner's request. +- A full `test:changed` inside `~/.codex/worktrees` fails fixture cleanup by design (real Codex home guard); + local suites ran in a `/tmp` worktree at the same commit. + +## Follow-ups (not in this release) + +- Two accepted review follow-ups on #5838 and one on #5780 are tracked in those pull requests' review threads; + details stay there until they are fixed. +- `main`'s `dev-version-bump.yml` is now the hardened #5786 version after this promotion; the next pre-move + can use the workflow again. +- Deferred PRs from the readiness review remain open: #5790, #5147, #5831, #5758, #5836, #5756. diff --git a/devlog/_fin/260926_release_2670/000_plan.md b/devlog/_fin/260926_release_2670/000_plan.md new file mode 100644 index 0000000000..4189d6a4d0 --- /dev/null +++ b/devlog/_fin/260926_release_2670/000_plan.md @@ -0,0 +1,58 @@ +# 260926 release 2.67.0 — plan + +## Reader summary + +`dev` at `08fd8a6284` carries 55 commits since v2.66.0: the twelve-PR sweep #5858, the +Windows cleanup fixes #5863, the main-account 98% hard lock for either window (#5870) and the +single-line quota strip (#5872). Its version line is already 2.67.0 (#5852). The owner asked +(2026-09-26) for cross-platform CI regression verification and a release, with unlimited kimi +subagents, and for the local checkout to be fast-forwarded afterwards. No pull request lands in +this round unless a regression is proven; the candidate is `dev` as it stands. + +## Loop spec + +- Archetype: satisfy-spec, multi-cycle HOTL; one PABCD cycle per work-phase; wp1 is this + docs-only roadmap. +- Goal: candidate green on Cross-platform CI lane=all, regression review with no unfixed + blocker, `dev` pre-moved to 2.68.0, 2.67.0 and 2.67.0-preview.20260926 published and + verified, local `dev`/`main`/`preview` equal to origin. +- Non-goals: landing open contributor PRs; raising any file-size cap; direct pushes to + `dev`/`main`/`preview`; security write-ups in tracked directories; rewriting history. +- Verifier: job tables of the lane=all run at the candidate, push-event CI and Service lifecycle at + both promotion SHAs, `release.yml` outcome rows, npm dist-tags, `gh release view`, + `latest.json`, `git rev-parse` local vs origin. +- Stop condition: 2.67.0 published and verified, or a terminal outcome below. +- Terminal outcomes: DONE; UNSAFE if a proven regression cannot be fixed in-round (release + withheld); BLOCKED for a release gate that cannot pass after documented retries; NEEDS_HUMAN + for an owner decision (for example a security-sensitive fix). +- Resource bounds: no token or wall-clock budget set by the owner. Tool scope: gh (dispatch of + `ci.yml`, `dev-version-bump.yml`, `release.yml`; job reruns; PR creation; admin merges with + `--match-head-commit`), git in the native checkout and `/tmp` scratch worktrees, local bun + tests. Kimi subagents (read-only leaves) for architecture, regression review and audit. +- Memory artifact: this unit, the goalplan under + `.codexclaw/goalplans/release-opencodex-2-67-0-preview-stable-from-the/`, `030_done.md` at the end. + +## Phase map + +| wp | Doc | Change | Depends on | +|---|---|---|---| +| wp1 | 000 (this) | roadmap | — | +| wp2 | [010](010_wp2_candidate_ci_review.md) | candidate CI, regression review, fixes if proven | wp1 | +| wp3 | [020](020_wp3_release.md) | pre-move, promotion, publish, verify, local ff, close | wp2 | + +## Cross-cutting rules + +1. Exact-head evidence: a run counts only when its `headSha` is the SHA being judged and every + job concluded `success` at its latest attempt (`privacy gate` skipped on + `workflow_dispatch` by design). +2. Windows runner-stall signatures (`spawnSync ETIMEDOUT`, 480 s batch timeout with each file + passing alone, `EPERM` on temp cleanup) get one `gh run rerun --job`. A second identical + failure, or any non-Windows failure, is a defect. +3. A defect gets a focused fix PR to `dev` merged at a green exact head; the candidate moves to + the new `dev` tip and gets a new lane=all run. Review findings follow the same rule only + when confirmed by reading the code at the candidate or by a failing test. +4. Runners: competing queued or in-progress runs may be cancelled by hand one at a time after + reading workflow, branch and event. Never cancel runs on `main`, `preview`, the candidate + run, this round's PR runs, or `Release`. +5. `PV=2.67.0-preview.20260926` is fixed now (KST publish day) and reused by every later + command, even if the publish crosses midnight. diff --git a/devlog/_fin/260926_release_2670/010_wp2_candidate_ci_review.md b/devlog/_fin/260926_release_2670/010_wp2_candidate_ci_review.md new file mode 100644 index 0000000000..70ee7bde01 --- /dev/null +++ b/devlog/_fin/260926_release_2670/010_wp2_candidate_ci_review.md @@ -0,0 +1,118 @@ +# 010 — wp2: candidate CI and regression review + +## Candidate + +`CAND=08fd8a62844738c960e2da71681b9a064b2fede3` (`origin/dev` at round start). Cross-platform CI +lane=all run `36208751784` (`workflow_dispatch`, headSha = CAND) was dispatched at round start. +The previous full run on `f353aac859` (`36170156438`) passed; the only failure since v2.66.0 on +`dev` was `windows 7/9` on `ca74738bc5`, fixed by #5863. + +Acceptance: every job `success` at its latest attempt; aggregate `ci` success. Rerun and defect +rules are cross-cutting rules 2 and 3 in 000. + +## Regression review + +Four kimi read-only leaves review `v2.66.0..CAND`, split by surface: + +| Lane | Commits | +|---|---| +| standalone | #5761 worker embedding, `e830d8adee`, `f67993645e`, `3ed77e964f` | +| chat | #5844, #5843, #5845, `b603a5ce78`, #5863 | +| ops | #5856, #5840, #5841, #5842, #5756, #5758, #5790, `43345e3c0e` | +| features | #5850, #5870, #5872 | + +Each reports severity, file:line, failure scenario, evidence and a RELEASE-OK/BLOCK verdict. +Main synthesizes accept/rebut per finding (REVIEW-SYNTHESIS-01): a finding is accepted only when +confirmed at the candidate by reading the code or by a failing test. Accepted blockers become fix +PRs under rule 3; accepted non-blockers are listed as follow-ups in 030 and left on `dev`. + +## Local proof + +At CAND in a `/tmp` worktree: `bun run typecheck` plus the focused test files named by any +accepted finding. The lane=all run is the full-suite evidence; a full local `bun run test` is not +repeated (it is the same suite on three OSes in CI). + +## Exit + +Record run ID, job count, reruns, and the finding dispositions below this line, then close wp2. + +Sweep #5858 (merge `ca74738bc5`) is the union of twelve PRs, all already in the lanes above: +#5761 (standalone); #5844, #5843, #5845 (chat); #5856, #5840, #5841, #5842, #5756, #5758, #5790 (ops); +#5850 (features). Its integration commits `43345e3c0e` and `3ed77e964f` are in ops and standalone. +The remaining commits in the range (#5857 devlog, #5852 version pre-move) carry no runtime code. +Before wp2 exits, re-read the `dev` run list to confirm no new failure since this was written. + +## Status at wp1 close (01:36Z) + +Run `36208751784` in progress: skipped=1, success=20 of 37 jobs, no failure yet. Four kimi regression leaves +(standalone, chat, ops, features) dispatched in wp1's P as read-only discovery; their reports are +synthesized in wp2. + +## Regression review synthesis (REVIEW-SYNTHESIS-01) + +| Lane | Verdict | Findings and disposition | +|---|---|---| +| standalone | RELEASE-OK | none; compiled binary built and a policy worker ran inside it on macOS arm64 | +| chat | RELEASE-OK | orphan legacy `function` results now fail with 400 (ADR-0111, intended) — release note | +| ops | RELEASE-OK | Remote Workspace RPC v2 fails closed against v1 peers (documented in f19ef3dbfe) — release note; Kiro injected-`saveCredential` rollback skip is test-surface only — follow-up | +| features | RELEASE-BLOCK | F1-F3 below, all rebutted | + +- F1 hard-lock 429 without `Retry-After` when a blocking window has no future reset: rebutted. It is + the #5870 design (`260926_main_hard_lock_any_window/010_plan.md`), pinned by + `main-account-hard-lock-policy.test.ts`; a `Retry-After` at the 5h reset would promise an unlock + the weekly window may still refuse. +- F2 retained weekly block could persist: rebutted. `runMainAccountHardLockRecovery` + (`src/codex/auth-api/pool-mode-gate.ts`) force-refreshes WHAM every 60 s while blocked, and a + measured secondary window replaces the retained tuple. +- F3 `validateForwardAdmissionCredential` now below the Devin search dispatch: rebutted. The guard + protects bearer forwarding; the Devin path forwards no caller header, and `resolveApiAuth` already + admitted the request in `serve-options.ts` before `handleSearch`. Residual: no pin test for a + Devin-routed request with the admission secret as bearer — follow-up. + +An independent kimi audit confirmed all three rebuttals (NEAR-PASS; residuals are missing pin tests). + +## Candidate run 36208751784 + +`windows 2/9` failed once: `provider outbound GET transport > proxy mode reaches one real proxy` +timed out at 15 s after the fixture logged a local DNS failure (runner stall signature). One job rerun +per rule 2 once the run finishes. + +## Owner steering (2026-09-26): land #5866 before the candidate + +The owner asked to merge #5866 (owner-authored, "independent first-party switch for the Claude Code +CLI", 68 files) and then continue. Its PR CI passed at head `a79b8625` on base `ca74738bc5`, five +commits behind `dev`. Before merge: union worktree `/tmp/ocx-union-5866` (`dev` + #5866) passes +`bun run typecheck` and `bun run test:changed`, and a kimi regression review of the diff finds no +accepted blocker. Merge with `gh pr merge 5866 --squash --admin --match-head-commit a79b8625...`. +The candidate becomes the new `dev` tip; a new lane=all run on it replaces `36208751784` as the +candidate evidence. + +## Owner steering (2026-09-26): #5866 merged, then #5875 + +- #5866 merged as `03aa39340b` (squash, admin, match-head `a79b8625`). Union evidence before merge: + `bun run typecheck` exit 0 on `dev`+#5866; kimi review RELEASE-OK (default-off verified, no + credential exposure, union gates clean). The union `test:changed` run had 24 failures, all in + service ownership, native Grok/Codex toggle, and remote Linux sandbox files that #5866 does not + touch (this Mac's installed service and missing bubblewrap). The owner then asked that no further + local suites run; CI is the test evidence from here. +- Run `36208751784` (old candidate) was cancelled as superseded; lane=all `36209738124` started on + `03aa39340b`. +- The owner then asked to include #5875. It conflicted with #5872 in `QuotaSummaryBar.tsx`; resolved + in `/tmp/ocx-5875` by keeping #5875's `ref={publishStickyTop}` on the section and #5872's + `QuotaSummaryChips`, root and GUI `tsc` plus GUI lint exit 0, pushed as `90a556debf`. Merge + after its exact-head PR CI passes and a kimi review finds no blocker; the candidate then moves to + the new `dev` tip with a new lane=all run. + +- #5875 exact-head PR CI at `90a556debf`: every check pass (skips by path); kimi review RELEASE-OK (defaults-only + shadow-intercept users gain `gpt-6-luna`; a hand-written `sourceModels` list replaces defaults — release note). + Merged as `dac1d25f48`. Run `36209738124` on `03aa39340b` was cancelled as superseded after `windows 7/9` + failed five `cli-connect-readiness` cases on a 15 s spawnSync kill (`status: null`, cold-spawn warmup 17 s), + the stall signature; that shard passed on `08fd8a6284`. The new candidate run is its one retry. + +## Final candidate + +`CAND=dac1d25f48fad18420aa856631ff9ee9c1775b0f`, lane=all run `36210914271`. + +Result: run `36210914271` completed `success` at CAND, attempt 1: 39 jobs success, `privacy gate` skipped by +design, no reruns. `windows 7/9` passed, so the earlier `cli-connect-readiness` timeouts were the runner stall. +wp2 exits with no accepted blocker. diff --git a/devlog/_fin/260926_release_2670/020_wp3_release.md b/devlog/_fin/260926_release_2670/020_wp3_release.md new file mode 100644 index 0000000000..281e5d2a63 --- /dev/null +++ b/devlog/_fin/260926_release_2670/020_wp3_release.md @@ -0,0 +1,121 @@ +# 020 — wp3: pre-move, promotion, publish, verify + +Procedure follows `devlog/_fin/260925_release_2660_prs/060_wp7_release.md`; only values differ, +plus the pre-move now uses the workflow (see 1). + +| Ref | Commit | Version | +|---|---|---| +| `main` | `e70b3d86fb` | 2.66.0 (npm `latest`) | +| `preview` | `0c37e74002` | 2.66.0-preview.20260925 (npm `preview`) | +| `dev` | CAND | 2.67.0 | + +## 1. Dev pre-move to 2.68.0 + +`origin/main:.github/workflows/dev-version-bump.yml` is the hardened #5786 version (`4ebb4fb8d9`: +trusted checkout at `github.sha`, `dev` checked out as data without credentials), so the +workflow is used this time: + +```bash +gh workflow run dev-version-bump.yml --ref main -f intended-version=2.67.0 -f mode=pre-move +``` + +The PR it opens must change exactly the four version sources (`package.json`, +`desktop/src-tauri/tauri.conf.json`, `desktop/src-tauri/Cargo.toml`, the `opencodex-desktop` +entry of `desktop/src-tauri/Cargo.lock`) to 2.68.0. Merge with +`gh pr merge --squash --admin --match-head-commit ` after its exact-head checks pass. The +candidate SHA does not change. Fallback if the workflow fails: the manual steps from 2.66.0's 060 +audit amendment, with 2.67.0/2.68.0. + +## 2. Promotion PRs + +```bash +PV=2.67.0-preview.20260926 +git worktree add /tmp/ocx-rel-2670 "$CAND" && cd /tmp/ocx-rel-2670 +git switch -c codex/260926-release-preview-2.67.0 +git merge -s ours --no-edit origin/preview -m "release: promote the verified 2.67.0 preview tree to preview" +bun scripts/release-version-sources.ts sync "$PV" +git commit -am "release: prepare $PV version metadata" +bun scripts/release-version-sources.ts check "$PV" +git switch -c codex/260926-release-main-2.67.0 "$CAND" +git merge -s ours --no-edit origin/main -m "release: promote the verified 2.67.0 tree to main" +bun scripts/release-version-sources.ts check 2.67.0 +``` + +Checks: `git diff --stat $CAND codex/260926-release-main-2.67.0` empty; preview differs from CAND +only in the four version lines. Push with `--no-verify`, open both PRs from the template, merge +each with `gh pr merge --merge --admin --match-head-commit ` (merge commit, never squash). + +## 3. Release-branch CI + +At each promotion merge SHA: push-event `ci.yml` success and Service lifecycle success +(`package.json` changed since the previous tag). Failures follow 000 rules 2 and 3. + +## 4. Dispatch (preview first) + +```bash +gh workflow run release.yml --ref preview -f version="$PV" -f tag=preview -f expected-sha= -f dry-run=false +gh workflow run release.yml --ref main -f version=2.67.0 -f tag=latest -f expected-sha=
    -f dry-run=false +``` + +A job that fails after npm acknowledged publication is completed by re-dispatching with the same +version and expected SHA plus `resume-after-npm-publish=true`; a version is never republished. + +## 5. Verification + +```bash +curl -s https://registry.npmjs.org/@bitkyc08%2fopencodex | jq '."dist-tags"' +gh release view v2.67.0 --json assets,isPrerelease,targetCommitish +gh release view "v$PV" --json assets,isPrerelease,targetCommitish +curl -sL https://github.com/lidge-jun/opencodex/releases/latest/download/latest.json +``` + +Acceptance: npm `latest` = 2.67.0 and `preview` = `$PV`; both releases have 25 assets; +`latest.json` reports 2.67.0 with a signature for darwin-aarch64, darwin-x86_64, windows-x86_64, +linux-x86_64 and linux-x86_64-deb. A green run with registry verification `pending` is waited out. + +## 6. Local fast-forward and close + +`git fetch origin`; in the native checkout fast-forward `dev`, `main` and `preview` with +`--ff-only` (`git fetch origin main:main preview:preview` for branches not checked out); a +non-fast-forward is reported, never forced. Write `030_done.md`, move the unit to `devlog/_fin/`, +open a docs PR to `dev`, merge at green exact head, fast-forward local `dev` again. + +## Architect reflection (kimi, P phase) + +Confirmed against `origin/main` workflow blobs: `dev-version-bump.yml` is the #5786 version and must be +dispatched with `--ref main` (its guard refuses other refs). `release.yml` enforces in-workflow that +`expected-sha` is the branch head, a push-event `ci.yml` success and a Service lifecycle success exist at +that SHA, and `dev` already outranks the release version. Ordering is therefore strict: the pre-move PR +merges before any release dispatch, and a dispatch issued earlier fails at preflight. If the bump run +fails because `codex/dev-version-2.68.0` already exists with other content, read the run; do not retry +blindly. + +## Audit round 1 (kimi, NEAR-PASS) — dispositions + +1. MAJOR "#5858 not partitioned": rebutted with a mapping. The sweep's twelve PRs are exactly the + merges listed in 010's lanes; 010 now states the mapping explicitly. +2. Minor: stable dispatch runs only after the preview release run concludes `success` (ordering + note restored from 2.66.0's 060). +3. Minor: acceptance adds `isPrerelease` = true for `v$PV` and false for `v2.67.0`. +4. Minor: 010 re-verifies the `dev` failure list before wp2 exit. + +## wp3 P revalidation (2026-09-26T02:38Z) + +- `CAND=dac1d25f48fad18420aa856631ff9ee9c1775b0f` (green lane=all `36210914271`), not `08fd8a6284`; the + owner added #5866 and #5875 during wp2. +- `PV=2.67.0-preview.20260926` unchanged. `main` `e70b3d86fb`, `preview` `0c37e74002`. +- Pre-move PR #5895 (`codex/dev-version-2.68.0`, head `4c64fd4acc`, base `dac1d25f48`) was opened by the + workflow token, which does not trigger `pull_request` workflows; close/reopen by the maintainer started + its CI. Diff is exactly the four version sources, 2.67.0 → 2.68.0. + +## wp3 execution log + +- Pre-move: `dev-version-bump.yml` run `36210933641` success opened #5895 (head `4c64fd4acc`). Its CI did not + start until a maintainer close/reopen. `enforce-target` and `hygiene` failed with `unsponsored_surface` + (bot author on release-owned files) and the readiness gate held it in draft; every CI job including the + aggregate `ci` passed. Marked ready and merged with admin as `c56dd47a6f`; `dev` carries 2.68.0. +- Promotion: #5899 `preview` merge `9c6fb1ee8b8c06fd4ee68bd237099b6f9325802e` (tree = CAND + four version lines), + #5900 `main` merge `4bc92294aa23a7edba75805095808d892efa72e2` (tree = CAND). Both were drafted by the gate, + marked ready, merged with merge commits. +- Release-branch CI: preview CI `36213263882`, Service lifecycle `36213264006`; main CI `36213267338`, + Service lifecycle `36213267275`. diff --git a/devlog/_fin/260926_release_2670/030_done.md b/devlog/_fin/260926_release_2670/030_done.md new file mode 100644 index 0000000000..355b941bbc --- /dev/null +++ b/devlog/_fin/260926_release_2670/030_done.md @@ -0,0 +1,67 @@ +# 030 — done: 2.67.0 release round + +## Outcome + +2.67.0 shipped from `dev` candidate `dac1d25f48fad18420aa856631ff9ee9c1775b0f` as preview `2.67.0-preview.20260926` +and stable `2.67.0`. `dev` now carries 2.68.0. No regression was found that blocked the release. + +## What the candidate contains beyond v2.66.0 + +The 55 commits already on `dev` at round start (sweep #5858 with twelve PRs, #5863, #5870 98% lock on either +window, #5872 single-line quota strip), plus two owner-requested landings during the round: + +| PR | Merge commit | Notes | +|---|---|---| +| #5866 independent Claude Code CLI first-party switch | `03aa39340b` | green PR CI at `a79b8625`; union typecheck and kimi review clean | +| #5875 shadow-call intercept follows GPT-6 Luna, Models settings fold, pinned rail | `dac1d25f48` | conflict with #5872 resolved in `90a556debf`; exact-head PR CI green | +| #5895 dev pre-move to 2.68.0 | `c56dd47a6f` | workflow-opened; four version lines only | + +Promotions: #5899 `preview` `9c6fb1ee8b`, #5900 `main` `4bc92294aa` (merge commits from the candidate with +`-s ours`; `main` tree identical to the candidate, `preview` differs only in the four version lines). + +## Evidence + +- Candidate: Cross-platform CI lane=all run `36210914271` (`workflow_dispatch`, attempt 1) success at CAND, 39 jobs + success, `privacy gate` skipped by design. +- Release branches: push-event Cross-platform CI `36213263882` (preview) and `36213267338` (main) success; Service + lifecycle `36213264006` and `36213267275` success. +- Release runs: preview `36214728717` success; stable `36215532928` success. The stable run ended with npm registry + verification pending (publish acknowledged 03:57:54Z with provenance); see the registry note below. +- GitHub releases `v2.67.0` (prerelease false, target `4bc92294aa`) and `v2.67.0-preview.20260926` (prerelease + true, target `9c6fb1ee8b`) each carry 25 assets. `latest.json` reports 2.67.0 with signatures for darwin-aarch64, + darwin-x86_64, windows-x86_64, linux-x86_64 and linux-x86_64-deb. +- Regression review: four kimi lanes over `v2.66.0..08fd8a6284` plus reviews of #5866 and #5875; dispositions in 010. + +## Release-note items + +- Remote Workspace RPC v2: a 2.67.0 peer refuses v1 frames. Upgrade the Hub and Executors together. +- Chat Completions: an orphan legacy `function` result now fails with 400 instead of being dropped (ADR-0111). +- Main-account hard lock: either the 5h or the weekly window at 98% blocks; the 429 omits `Retry-After` when a + blocking window has no future reset. +- Shadow-call intercept: defaults now include `gpt-6-luna`; a hand-written `sourceModels` list still replaces + the defaults. + +## What did not go to plan + +- The first two lane=all runs were superseded when the owner added #5866 and then #5875; each was cancelled. + Their Windows failures (`windows 2/9` proxy fixture timeout, `windows 7/9` `cli-connect-readiness` spawn kills + with a 17 s cold-spawn warmup) did not recur on the final candidate. +- A PR opened by the workflow token does not start `pull_request` workflows. #5895 needed a close/reopen, and the + quality gate failed it with `unsponsored_surface` and held it (and both promotion PRs) in draft; they were marked + ready and merged with admin after every CI job passed. +- A local `test:changed` on the #5866 union produced 24 environment failures (installed service, no bubblewrap); + the owner then asked for no local suites, so CI carried the test evidence. +- PR-event CI on the already-merged promotion branches held macOS runners; it was cancelled so the push-event runs + on `main` and `preview` could start. + +## Follow-ups (not in this release) + +- Pin tests: a Devin-routed search presenting the admission secret as bearer; weekly unknown-retention release path. +- #5866 low findings: observed Claude intercept state is captured at listener start; `PUT /api/claude-code` has no + toggle-flight lock. +- Kiro forced-login rollback is skipped when a test injects `saveCredential` (test surface only). + +## Registry note + +A direct registry read at 04:05:51Z showed `latest` = 2.67.0 and `preview` = 2.67.0-preview.20260926 (package +`time.modified` 04:03:05Z), closing the stable run's pending verification. The run was not re-dispatched. diff --git a/devlog/_fin/260927_directive_marker_bridge/000_plan.md b/devlog/_fin/260927_directive_marker_bridge/000_plan.md new file mode 100644 index 0000000000..16322b406b --- /dev/null +++ b/devlog/_fin/260927_directive_marker_bridge/000_plan.md @@ -0,0 +1,57 @@ +# Codex App visualization references for models that cannot see private-use characters + +## Objective + +A routed model must be able to show a Codex App inline visualization. Today only models whose +provider preserves Basic Multilingual Plane private-use characters can, because the reference the +bundled Visualize skill teaches is `U+E200 visualize U+E202 {json} U+E201`. + +Observed on 2026-09-27 against the local proxy (2.68.0, `a1285fc648`): asked to list the code +points of `[A U+E200 B U+E202 C U+E201]`, `gpt-6-luna` and `xai/grok-4.7` return all six, while +`anthropic/claude-opus-5-5` and `cursor/claude-opus-5-5` return only `A B C`. OpenCodex's own +Anthropic request body still contains the three characters (checked by building the request in +process), so they are removed after the proxy. Claude therefore reads the skill template as +`visualize{"path":...}`, writes that plain text back, and the app shows it verbatim. + +## Constraints + +- Native OpenAI/ChatGPT passthrough stays byte-identical: it serializes `_rawBody`. +- Stored request history stays raw (`adapter-delivery.ts` hands `_rawBody` to `state.ts`). +- The citation filter semantics from #3150, #3843 and #6040 are unchanged. +- No new dependency, no Lab import on the request path, no file-size ratchet increase. + +## Work-phase map + +| Work-phase | Doc | Outcome | +|---|---|---| +| wp1 | this directory, 000-001 | Evidence and roadmap (docs only) | +| wp2 | [010](010_wp2_visualization_directive_normalization.md) | Parser-side normalization, tests, PR, CI, squash merge | +| wp3 | [020](020_wp3_rebuild_and_live_verify.md) | Local app rebuild from merged dev with a fresh sidecar, live Claude check | + +wp2 depends on the app evidence in [001](001_app_and_external_evidence.md); wp3 depends on wp2 being on `dev`. + +## Architect consultation (formal P) + +Architect: gpt-6-astra subagent `01a0e0ce-58fd-7fd3-8186-12e3752ec8f6` (Fermat), read-only. +Proposal D1-D14 received 2026-09-27. Main dispositions: + +| Id | Proposal | Disposition | +|---|---|---| +| D1 | New pure module `src/responses/visualization-directives.ts`, copy-on-change | Accepted | +| D2 | Hook at the parser's final `context` | Accepted; covers the replay expansion and the encrypted-agent reparse | +| D3 | Normalize conversation text only (system prompt, message strings, text parts), including fenced code | Accepted | +| D4 | Match only complete literal `U+E200visualize U+E202 … U+E201` spans | Accepted | +| D5 | Payload rules mirror the app's `f2` exactly | Accepted | +| D6 | Single template exception for `/.html` | Accepted | +| D7 | `codex-live-vis` for `type:"live"`, `mode="wide"` only for inline `mode:"wide"` | Accepted | +| D8 | Raw-body passthrough excluded | Accepted. Whether every raw-body route preserves the characters is not established; the observed failures are both on context-built adapters | +| D9 | No persistence change; history stays raw | Accepted | +| D10 | Deterministic output; Cursor checkpoint digest will differ once for affected prefixes | Accepted; the digest mismatch invalidates the old checkpoint, which is the safe direction. No dedicated Cursor test, recorded as residual | +| D11 | No output repair of bare `visualize{json}` | Accepted; revisit only if live checks show a model still emitting the bare form | +| D12 | Alternatives | Rejected: adapter hooks duplicate the rule per adapter and miss provider switches mid-thread; a request-prepare hook misses the encrypted-agent reparse and direct parser callers; changing the upstream skill text does not reach installed plugins or existing history | +| D13 | Sibling test file registered in both layout manifests | Accepted; replay-after-restart and Cursor checkpoint tests narrowed to raw-body and idempotence assertions | +| D14 | Live UI confirmation of the ASCII directive still pending | Tracked in wp3 | + +Limitations carried forward: raw-body passthrough routes are not normalized; replies already stored as bare `visualize{...}` are not repaired; a future template variant needs its own fixture; the live render of the ASCII form is confirmed only in wp3. + +Reflection: recorded in [010](010_wp2_visualization_directive_normalization.md#architect-reflection). diff --git a/devlog/_fin/260927_directive_marker_bridge/001_app_and_external_evidence.md b/devlog/_fin/260927_directive_marker_bridge/001_app_and_external_evidence.md new file mode 100644 index 0000000000..c23fabc6e4 --- /dev/null +++ b/devlog/_fin/260927_directive_marker_bridge/001_app_and_external_evidence.md @@ -0,0 +1,53 @@ +# Evidence: how the Codex App reads visualization references + +## App bundle (primary) + +Source: `/Applications/ChatGPT.app` (bundle `com.openai.codex`, 26.924.22138), `Contents/Resources/app.asar` +extracted read-only to `.tmp/asar/app`. Paths below are inside `webview/assets/`. + +- `app-shared-36eae88777f2.js`, function `f2`: every span matching + `/U+E200visualize U+E202([^U+E201]+)U+E201/g` outside markdown code tokens is rewritten to + `::codex-inline-vis{path="<abs>" title="…" mode="wide"}` before rendering. Payload handling: + - payload starting with `{` is `JSON.parse`d (failure keeps the span); otherwise it is `{path: payload}`; + - schema `{path: string|null, title?: string, type?: "inline"|"live", mode?: "wide", wide?: boolean}`; + - `path: null` becomes `::codex-live-vis{}` for `type:"live"` and is otherwise kept; + - the span is kept when the path has a `..` segment, a double quote or CR/LF, when its basename does + not match `/^[a-z0-9]+(?:-[a-z0-9]+)*\.html$/`, or when it is not absolute and is either JSON or not a bare basename; + - attribute is `path` for an absolute path and `file` for a bare basename; + - `title` is emitted only without quotes or CR/LF; `mode="wide"` only for inline with `mode:"wide"`. +- Same file, function `Err`: a message is a visualization when it contains the private-use form OR the + literal `::codex-inline-vis`, and parsing `f2(text)` yields a `codexDirective` named + `codex-inline-vis`. The plain ASCII directive is therefore the app's canonical form. +- Same file, `sj` (absolute path): `/…` but not `//…`, `/^[A-Za-z]:[\\/]/`, `/^\\\\[^\\]+\\[^\\]+/`, `/^\/\/[^/]+\/[^/]+/`. +- `app-primary-cca0c1a58f0f.js`, `i3e`: registers `renderElement` for the inline directive and reads + its `path` or `file` attribute; `o8e` attaches it whenever `renderInlineVisualizations` is set, which + `conversation-blocks-1768e325fd75.js` (`qg`) sets for assistant message blocks. +- `app-primary-cca0c1a58f0f.js`, `a3e`: while streaming, a trailing partial `U+E200visualize U+E202` is + hidden until it closes. + +## External sources + +| Claim | Source | Status | +|---|---|---| +| OpenAI documents the private-use citation grammar (`U+E200 cite U+E202 … U+E201`) | https://developers.openai.com/api/docs/guides/citation-formatting | primary | +| Claude Code strips BMP private-use characters (U+E000-U+F8FF) from tool input and output; supplementary PUA-A survives | https://github.com/anthropics/claude-code/issues/44525 (2026-04-07) | primary report | +| Same class reported earlier for U+E0A0 | https://github.com/anthropics/claude-code/issues/31849 (2026-03-07) | primary report | +| Streamed citation markers leak into third-party output | https://community.openai.com/t/streamed-web-search-citations-leaking-citation-markers-into-text-output/1390157 (2026-08-12) | lead | +| Codex desktop instructions use plain `::name{…}` directives (`::code-comment`, `::created-thread`) | Codex App system prompt in this session; community reproduction | primary (session) | + +Luna lanes: Plato, Ampere, Avicenna (`gpt-5.6-luna`). Aside exec session `boOE2Tp2gdGCg2Do`, +report under `~/.aside/u/0/artifacts/ocx-directive-research/`. + +## Conclusion + +Converting the private-use reference into `::codex-inline-vis{…}` before the model sees it gives a model +that cannot see private-use characters an ASCII instruction it can read and repeat. The bundle parses that +form directly (`Err`, `f2`); the live render is confirmed in wp3. + + +## Session observation + +A commentary message in this thread that contained `::codex-inline-vis{…}` reached the rollout only as a +`reasoning` summary, not as an assistant `message`, so the user saw plain text. Other commentary blocks +in the same thread show the same pattern. The render test therefore uses a final answer. How routed +Claude text becomes a reasoning item is outside this unit. diff --git a/devlog/_fin/260927_directive_marker_bridge/010_wp2_visualization_directive_normalization.md b/devlog/_fin/260927_directive_marker_bridge/010_wp2_visualization_directive_normalization.md new file mode 100644 index 0000000000..0e5550f22c --- /dev/null +++ b/devlog/_fin/260927_directive_marker_bridge/010_wp2_visualization_directive_normalization.md @@ -0,0 +1,109 @@ +# wp2: normalize visualization references in model-visible text + +## Scope + +IN: `src/responses/visualization-directives.ts` (new), `src/responses/parser.ts` (one call at the +return), `tests/responses/visualization-directives.test.ts` (new), `scripts/test-layout/layout.json`, +`tests/fixtures/test-layout-expected.json`, `structure/transports/responses.md`, +`docs-site/src/content/docs/guides/codex-integration.md` (new subsection after "Routed local tools"). +OUT: adapters, raw-body passthrough, response-side repair, persistence, citation filter. + +## Files + +### NEW `src/responses/visualization-directives.ts` + +Exports: + +- `normalizeVisualizationText(text: string): string` — returns `text` unchanged (same reference) + unless it contains the exact prefix `"\uE200visualize\uE202"` (no spaces). It produces the same + matches as the app's regex `/\uE200visualize\uE202([^\uE201]+)\uE201/g`, but with a linear scanner: + find the next prefix with `indexOf`; the payload runs to the next END (`indexOf` from the prefix end); + an empty payload resumes the prefix search one character later; no END after a prefix means no later + match exists, so scanning stops. Each match becomes `toAsciiDirective(payload)` or stays as is. + Unterminated spans are copied through. App-regex parity supersedes D4's opaque-payload rule: a + `visualize` span inside another keyword's payload is converted, as the app would convert it. + + Intentional differences from `f2`: code tokens are normalized too (the skill's example is fenced), + and the template exception below. Using the same regex means a `visualize` span that appears after + a malformed START of another keyword is converted exactly as the app would convert it. +- `normalizeVisualizationContext(context: OcxContext): OcxContext` — copy-on-change over + `systemPrompt`, string `content`, and `type:"text"` parts of every message role; returns the same + object when nothing changed. + +`toAsciiDirective(payload)` follows [001](001_app_and_external_evidence.md) rule for rule: every valid +`type:"live"` payload becomes `::codex-live-vis{…}` with no `mode`; `wide:true` is validated and +ignored; `mode="wide"` only for inline. Template exception: a JSON payload whose `path` is exactly +`<absolute-path>/<title>.html` skips path validation and always uses the `path` attribute (the app's +`sj` would pick `file` for it); schema, type, title and mode rules still apply. Both skill templates +have exact-output assertions. + +### MODIFY `src/responses/parser.ts` + +```diff ++import { normalizeVisualizationContext } from "./visualization-directives"; +@@ return { +- context, ++ context: normalizeVisualizationContext(context), +``` + +### NEW `tests/responses/visualization-directives.test.ts` + +Cases: absolute JSON path; title; wide; live with path and with null path; bare absolute path; bare +basename (`file=`); both skill templates; spans inside a fenced block; rejected payloads kept (invalid +JSON, relative JSON path, `..`, quote, bad basename, missing path); other keywords and unterminated +spans untouched; Windows drive and UNC paths; live null path with a title; malformed optional fields +(`title` number, `type` other, `mode` other, `wide` string); a decoded compaction summary carrying the +reference; 20,000 repeated unterminated prefixes normalized in under 200 ms; exact output for a +`visualize` span nested in a `cite` payload and for a malformed `visualize` prefix followed by a valid +one; idempotence; every +role and content shape through `parseRequest`, with image parts, tool-call arguments and reasoning +parts deep-equal to an unnormalized parse; a frozen input body (deep `Object.freeze`) parses without +throwing and `_rawBody` is the same object; an expanded continuation body (prior assistant output +with the private-use reference plus a new user turn) normalizes both turns; the Anthropic adapter +body contains the ASCII directive and no private-use character. + +### MODIFY layout manifests and `structure/transports/responses.md` + +Register the test under `responses`; add one paragraph to the transport doc naming the module and the +raw-body exclusion. + +### MODIFY `docs-site/src/content/docs/guides/codex-integration.md` + +New `### Inline visualizations with routed models` after `### Routed local tools`: the private-use +reference is rewritten to `::codex-inline-vis{…}` for context-built routes, raw-body passthrough routes +are unchanged, and replies already stored in the bare `visualize{…}` form stay as they are. Locale +copies gain nothing, so they do not contradict the English page. + +## Verification (PLAN-VERIFIER-REAL-01) + +| Command | Reads the target | +|---|---| +| `bun test tests/responses/visualization-directives.test.ts` | direct argument | +| `bun test tests/responses/citation-markers.test.ts tests/adapters/bridge.test.ts` | callers of the response path | +| `bun test tests/responses/responses-parser.test.ts tests/responses/responses-parser-agent-message.test.ts tests/responses/responses-parser-malformed-content.test.ts tests/responses/parser-content-audio.test.ts tests/responses/responses-state.test.ts` (run locally) | parser and replay consumers | +| `bun test tests/lab/core-lab-boundary.test.ts tests/ci-workflows/file-size-ratchet.test.ts tests/test-layout.test.ts tests/test-layout-tooling.test.ts` | Lab boundary, ratchet, layout manifests (source-reading guards `test:changed` cannot see) | +| `bun run test:changed` | import graph from `parser.ts`; local run is limited by the `~/.codex` checkout guard, the full suite is left to CI | +| `bun run typecheck`, `bun run structure:check`, `bun run privacy:scan` | whole tree | + +Activation scenarios: a rejected payload (C runs the "kept" cases and sees the private-use span unchanged); +a live null path (C sees `::codex-live-vis{}`). + +## Architect reflection + +Fermat returned MISALIGNED on the first plan revision with five gaps. Dispositions: + +1. Exact prefix — folded. Opaque unknown spans — superseded in the second reflection by app-regex + parity (the app converts a nested span, so the model should see the same thing); nested and + malformed-prefix fixtures pin the exact output. +2. Live selection, ignored `wide`, template exception scope — folded above. +3. Opaque-field, frozen-input and replay assertions — folded above. Cursor checkpoint rejection test — + rebutted: the Cursor builder compares a digest of the context it is given, and this change only + alters that context's text, so an old checkpoint mismatches and is discarded, which is the existing + safe path; recorded as residual instead of a new Cursor fixture. +4. Existing suites and source-reading guards named explicitly — folded above. +5. Universal claims removed; alternatives' reasons and limitations recorded in 000 and 001 — folded. + + +Second reflection (wp2 P): MISALIGNED on regex cost (quadratic on repeated unterminated prefixes, +measured 16/63/251 ms for 2k/4k/8k) and on the opaque-span claim. Both folded above: linear scanner with +the regex's semantics, parity recorded explicitly, adversarial and nested fixtures added. diff --git a/devlog/_fin/260927_directive_marker_bridge/020_wp3_rebuild_and_live_verify.md b/devlog/_fin/260927_directive_marker_bridge/020_wp3_rebuild_and_live_verify.md new file mode 100644 index 0000000000..3ac1294ebb --- /dev/null +++ b/devlog/_fin/260927_directive_marker_bridge/020_wp3_rebuild_and_live_verify.md @@ -0,0 +1,25 @@ +# wp3: rebuild the local app and verify with a Claude-routed model + +1. In the primary checkout, fast-forward `dev` to the merge commit. +2. Force a fresh standalone sidecar: `bun run build:standalone --target bun-darwin-arm64`. + `desktop/scripts/prepare-sidecar.ts` reuses `dist/standalone/*/ocx` when it exists, which shipped a + stale 2.61.0 proxy inside the 2.68.0 app on 2026-09-27. +3. `bun run build:gui`, then in `desktop/`: `bun run prepare-sidecar`, `bun run prepare-widget`, and + `bun run build:local` with the shared `.git/config` `core.bare` set to `false` only for the build + (Cargo reads it) and restored afterwards. +4. Extract `OpenCodex.app` from the DMG, verify `codesign --verify --deep --strict`, and check that + the bundled `ocx` contains `codex-inline-vis`. +5. Protocol check against the packaged sidecar once the user has reinstalled the app: confirm + `/healthz` reports the rebuilt version and that the running binary is the reinstalled bundle, then send a + Claude-routed request whose input carries the private-use reference and confirm the reply contains + `::codex-inline-vis{path="…"}` and no private-use character. +6. Render check in the Codex App: in this Claude-routed thread, write a real HTML fragment under the + thread's visualization directory and put `::codex-inline-vis{path="…"}` on its own line in a + **final answer** (a commentary block can be recorded as a reasoning summary, which the app does not + render as markdown directives; observed 2026-09-27). The user reports whether it renders. + Computer Use cannot inspect `com.openai.codex` (blocked by its safety policy), so rendering stays + "pending user confirmation" until that report. + +Shared `.git/config` note: Cargo (libgit2) ignores `config.worktree`, so the build flips `core.bare` for +the shortest possible window and restores it; do not run it while another task is using the checkout's +git config. diff --git a/devlog/_fin/260927_directive_marker_bridge/030_done.md b/devlog/_fin/260927_directive_marker_bridge/030_done.md new file mode 100644 index 0000000000..103964137d --- /dev/null +++ b/devlog/_fin/260927_directive_marker_bridge/030_done.md @@ -0,0 +1,35 @@ +# 030 — done: Codex App visualization references for routed models + +## Outcome + +Any routed model can now show a Codex App inline visualization. #6040 (`a1285fc648`) stopped the +citation filter from deleting non-citation directives; #6045 (`dc784d3e6f`) rewrites the private-use +`visualize` reference into the app's own `::codex-inline-vis{…}` directive in the text routed models +read, so a model whose provider drops private-use characters still reads and writes a form the app renders. + +## Evidence + +- App bundle 26.924.22138: `f2`, `Err` and `i3e` (see 001) render the ASCII directive directly. +- Local app rebuilt from `dc784d3e6f` with a forced standalone sidecar; the installed `ocx` reports + 2.68.0 and contains `codex-inline-vis`; `codesign --verify --deep --strict` passes. +- Live, through the installed proxy: `anthropic/claude-opus-5-5` and `cursor/claude-opus-5-5` answered a + private-use reference with `::codex-inline-vis{path="/tmp/demo-chart.html"}`; `gpt-6-luna` returned the + private-use form unchanged; `xai/grok-4.7` received and quoted the private-use form. +- Render: a Claude-routed final answer carrying `::codex-inline-vis{…}` rendered as an interactive widget in + Codex App; the widget reported `route: Claude, version: #6045, outcome: renders` back to the thread. +- Reviews: architect Fermat, auditor Lovelace, reviewer Archimedes (132,715 `f2` parity cases), and two + release regression lanes over #6040 and #6045 found no regression. + +## What did not go to plan + +- `desktop/scripts/prepare-sidecar.ts` reuses `dist/standalone/*/ocx` when it exists, so the first rebuilt + app shipped a stale 2.61.0 proxy. The standalone build has to be forced before `prepare-sidecar`. +- A commentary message is recorded as a reasoning summary on this route, which the app does not render + as markdown directives; the render check needed a final answer. +- #6045 merged by admin at the owner's instruction before its PR CI finished; the post-merge lane=all run + on `dc784d3e6f` is the CI evidence for the merged tree. + +## Follow-ups + +- Make `prepare-sidecar` rebuild when the source is newer than the cached standalone binary. +- Replies already stored in the bare `visualize{…}` form are not repaired. diff --git a/devlog/_fin/260927_release_2680/000_plan.md b/devlog/_fin/260927_release_2680/000_plan.md new file mode 100644 index 0000000000..8cfde2f80a --- /dev/null +++ b/devlog/_fin/260927_release_2680/000_plan.md @@ -0,0 +1,32 @@ +# 260927 release 2.68.0 — plan + +## Reader summary + +The owner asked (2026-09-27) for a main..dev regression review with astra reviewers, for the +Windows/Linux tray to gain the menu bar patches the macOS panel received, and for a full 2.68.0 +release. `origin/main` is v2.67.0 (`4bc92294aa`); `origin/dev` (`dc784d3e6f`) carries 86 commits +beyond it. Procedure follows [the 2.67.0 round](../../_fin/260926_release_2670/020_wp3_release.md); +only values differ. The owner asked for CI to be judged heuristically: a failure is a blocker only +when it reproduces or is tied to a change in the range. + +## Work-phase map + +| wp | Doc | Change | +|---|---|---| +| wp4 | [010](010_wp4_blockers_and_tray.md) | release blockers from the review, Windows tray parity | +| wp5 | [020](020_wp5_release.md) | candidate CI, pre-move to 2.69.0, promotion, publish, verify | + +## Review lanes (astra, read-only) + +| Lane | Range | Verdict | +|---|---|---| +| #6040 citation filter | `a1285fc648` | OK (700,168 comparisons) | +| #6045 visualization references | `dc784d3e6f` | OK | +| dev sanity + CI classification | whole range | OK; Devin test mock leak (test-only) | +| merge trains 1-5 | #5901..#5909 | OK | +| desktop, batches 6-8 | #5910..#5957 | BLOCK: native tray `switchFailed` cached | +| Kiro series | #5967..#6016 | BLOCK: first discovery failure never backs off | +| batches 9-10 | #5984..#6031 | BLOCK: Home-initiated Remote Link answers 503 | + +Owner disposition for the Remote Link finding (2026-09-27): keep 2.67.0 behaviour for Home-initiated +links only; Child-initiated links keep the new ownership proof. diff --git a/devlog/_fin/260927_release_2680/010_wp4_blockers_and_tray.md b/devlog/_fin/260927_release_2680/010_wp4_blockers_and_tray.md new file mode 100644 index 0000000000..cae311c70e --- /dev/null +++ b/devlog/_fin/260927_release_2680/010_wp4_blockers_and_tray.md @@ -0,0 +1,47 @@ +# 010 — wp4: release blockers and Windows tray parity + +## Fixes (one PR to dev) + +| Finding | Files | Change | Proof | +|---|---|---|---| +| Kiro discovery failure without a last good list retried on every request | `src/providers/kiro-model-catalog.ts` | `failedUntil` map carries the 60 s retry per account identity; cleared on success and by `clearKiroAccountModels` | new case in `tests/providers/kiro/kiro-model-catalog.test.ts` fails before (2 calls), passes after (1) | +| Native tray `switchFailed` cached and settling later switches | `desktop/src-tauri/src/native_tray.rs` | `cached()` drops the one-shot flag from the snapshot every refresh starts from | `cargo test --lib native_tray` | +| Home-initiated Child relays answer 503 | `src/client/link-relay.ts`, `src/client/runtime.ts` | explicit `HOME_INITIATED_LINK_TUNNEL` gate for a link-mode runtime without a sidecar; a relay with no gate still refuses | new case in `tests/clients/client-link-relay.test.ts`; the lane's repro passes | +| Devin preflight test leaks its adapter mock | `tests/responses/responses-grok-devin-preflight.test.ts` | restore the module in `afterAll` | 3-file run 82 pass (was 55/27) | + +## Windows/Linux tray parity (web tray `gui/src/pages/Tray.tsx`) + +| macOS patch | Web tray before | Change | +|---|---|---| +| #5920 windows that report data | present | none | +| #5922 provider marks, severity bars | missing | `ProviderIcon` in headings; `quotaSeverity` classes on bars (70/90) | +| #5931 switch the active account | missing | `switchState`/`exhausted` in `parseAccounts`, `accountSwitchRequest` routes, "Use this account" on hover/focus, pending and failure states | +| #5921 widget reload | not applicable (macOS widget) | none | + +The popup sends the switch with the dashboard session (`window.fetch` is session-wrapped by +`gui/src/api.ts`), not the desktop capability the native panel uses. + +## Verification + +`bun run typecheck`, GUI `tsc` and lint, `bun run structure:check`, `bun run privacy:scan`, the focused +test files above, `gui/tests/tray-data.test.ts`, and a Playwright capture of the web tray against the +running proxy. Full suite: PR CI. + +## Audit round 1 (Carver, astra) — FAIL, dispositions + +1. GUI build: `providerSources` `flatMap` inferred only `'codex'` — folded (`flatMap<TrayProviderSource>`; `tsc -p tsconfig.app.json` exit 0). +2. A Child-initiated link whose sidecar was deleted took the Home-initiated gate — folded. A join now + writes `link/child-initiated.json` with the link id; the runtime uses the Home gate only when neither + the sidecar nor a matching marker exists, and an unreadable marker counts as present. The auditor's + repro now answers 503 with no send for deleted and corrupt sidecars and for hub transport. + Residual: a 2.67.0 join made before this marker existed, with its sidecar later deleted, is treated as + Home-initiated, which is the 2.67.0 behaviour for every link. +3. Switch settled before the reload — folded: the row stays pending until the reload the switch started completes. +4. Unbounded PUT — folded: `createBoundedFetch(20 s)`; a timeout reports the switch failure. +Note (not folded): the web tray does not print `blockedReason`; a blocked account simply offers no "Use". +5. Round 2: pre-marker joins — partly folded. The runtime records the marker at start whenever the + sidecar is intact (`recordChildInitiatedLink`), so a pre-marker join that starts once on 2.68.0 is + protected from then on. Rebutted for the remaining case, a pre-marker join whose sidecar was deleted + before its first 2.68.0 start: that state is indistinguishable from a 2.67.0 Home-initiated link, and + failing it closed would break every existing Home-initiated link, which the owner chose to keep working. + It keeps exactly the 2.67.0 behaviour, the owner-accepted baseline. diff --git a/devlog/_fin/260927_release_2680/020_wp5_release.md b/devlog/_fin/260927_release_2680/020_wp5_release.md new file mode 100644 index 0000000000..f7e6966047 --- /dev/null +++ b/devlog/_fin/260927_release_2680/020_wp5_release.md @@ -0,0 +1,45 @@ +# 020 — wp5: release 2.68.0 + +Values for the 2.67.0 procedure: `CAND` = `origin/dev` after wp4 merges; `PV=2.68.0-preview.20260927`; +pre-move `dev-version-bump.yml --ref main -f intended-version=2.68.0 -f mode=pre-move` (dev → 2.69.0); +promotion branches `codex/260927-release-preview-2.68.0` and `codex/260927-release-main-2.68.0` built +with `git merge -s ours` and `scripts/release-version-sources.ts`; merge commits (never squash); push-event +CI and Service lifecycle at both promotion SHAs; `release.yml` preview first, then stable; verify npm +dist-tags, both GitHub releases' assets, and `latest.json` signatures; fast-forward local branches. + +Heuristic CI rule (owner): a failing job blocks only when it reproduces on rerun or its log points at a +change in main..dev. Runner-stall signatures get one job rerun. + +## wp5 P revalidation (2026-09-27) + +- `CAND=f764765c6453a718806d3465ea966015fa233123` (`origin/dev` after #6052). Lane=all run `36294278068` (workflow_dispatch) at CAND. +- `main` `4bc92294aa` (2.67.0, npm latest), `preview` `9c6fb1ee8b` (2.67.0-preview.20260926), `dev` 2.68.0. +- `PV=2.68.0-preview.20260927`. Pre-move: `gh workflow run dev-version-bump.yml --ref main -f intended-version=2.68.0 -f mode=pre-move` + (inputs verified on `origin/main`); its PR must change only the four version sources to 2.69.0. +- `release.yml` inputs verified: `version`, `tag`, `dry-run`, `resume-after-npm-publish`, `expected-sha`. + +## Audit (astra, NEAR-PASS) — folded + +- Order: the 2.69.0 pre-move PR merges before either release dispatch; `CAND` stays pinned before the bump. +- Both promotion branches start at `CAND` and merge their release branch with `-s ours`; `sync "$PV"` only on + preview, `check 2.68.0` on stable; promotion PRs merge with `--merge --match-head-commit`, never squash. +- The heuristic CI rule never waives release gates: push-event `ci.yml` and Service lifecycle must succeed at each + promotion merge SHA. +- Dispatch: `gh workflow run release.yml --ref preview -f version="$PV" -f tag=preview -f dry-run=false -f expected-sha=<sha>`; + stable only after the preview run succeeds, `--ref main -f version=2.68.0 -f tag=latest`. +- Verify: `npm view @bitkyc08/opencodex dist-tags --json`; `gh release view <tag>` 25 assets, non-draft, prerelease flags; + `latest.json` 2.68.0 with five signatures. + +## wp5 execution log + +- Pre-move: `dev-version-bump.yml` run `36294309818` success opened #6053 (head `7af8381049`, four version sources + 2.68.0 → 2.69.0 only). Merged by admin as `99d0a9400e` under the owner's heuristic CI rule (workflow-token PR CI + does not start; the diff is the same four lines as every pre-move). +- Promotion: branches built in `/tmp/ocx-rel-2680` from CAND; `release-version-sources.ts check` passed for both; + main tree equals CAND, preview differs only in four version lines. #6054 `preview` merged as `09081803c5`, + #6055 `main` merged as `93f4231e4b` (merge commits, `--match-head-commit`). +- Runner hygiene: PR-event runs on the merged promotion branches and the superseded CAND lane=all run + `36294278068` were cancelled one at a time so the push-event runs on `main` and `preview` could start. + CAND's tree equals #6052's exact head, whose PR CI passed (31 pass, 6 skipped). +- Release gates pending: main CI `36294473376`, main Service lifecycle `36294473362`; preview CI `36294469680`, + preview Service lifecycle `36294469713`. diff --git a/devlog/_fin/260927_release_2680/030_done.md b/devlog/_fin/260927_release_2680/030_done.md new file mode 100644 index 0000000000..ef25119e2a --- /dev/null +++ b/devlog/_fin/260927_release_2680/030_done.md @@ -0,0 +1,38 @@ +# 030 — done: 2.68.0 release round + +## Outcome + +2.68.0 shipped from candidate `f764765c6453a718806d3465ea966015fa233123` as preview `2.68.0-preview.20260927` and +stable `2.68.0`. `dev` carries 2.69.0 (#6053). The main..dev regression review found three release blockers; all +were fixed in #6052 before the candidate was cut, together with the Windows/Linux tray parity the owner asked for. + +## Evidence + +- Regression review: seven astra lanes over main..dev (86 commits); dispositions in [000](000_plan.md) and [010](010_wp4_blockers_and_tray.md). +- #6052 exact-head PR CI: 31 pass, 6 skipped; merged as `f764765c64`. Its tree is the candidate. +- Promotion: #6054 `preview` `09081803c5`, #6055 `main` `93f4231e4b` (merge commits). +- Release-branch CI: preview Cross-platform CI `36294469680` (23 success, 5 skipped) and Service lifecycle `36294469713`; + main Cross-platform CI `36294473376` (23 success, 5 skipped) and Service lifecycle `36294473362`. +- Release runs: preview `36295546215` success (14/14), stable `36296387672` success (14/14). +- GitHub releases: `v2.68.0` (25 assets, prerelease false, target `93f4231e4b`) and `v2.68.0-preview.20260927` + (25 assets, prerelease true, target `09081803c5`). `latest.json`: 2.68.0 with signatures for darwin-aarch64, + darwin-x86_64, windows-x86_64, linux-x86_64 and linux-x86_64-deb. +- npm: `preview` = `2.68.0-preview.20260927` after registry propagation; the stable publish was acknowledged with + provenance and `latest` is checked again after propagation (the preview took about ten minutes). + +## Release-note items + +- Codex App inline visualizations work with every routed model (#6040, #6045). +- Windows/Linux tray: provider marks, 70%/90% quota colors, and switching the active account (#6052). +- Remote Link: Home-initiated links keep 2.67.0 forwarding; Child-initiated links require the tunnel ownership proof. +- Kiro: account model discovery, quota metrics, device login, concurrency caps, 1M GPT-5.6 context windows; failed + discovery backs off. +- Items listed by the review lanes in [000](000_plan.md) (combo cooldowns, stall defaults, desktop title strip, and others). + +## What did not go to plan + +- The first candidate lane=all run failed `test 2/4` on a batch timeout whose files all passed alone (runner stall); + it and the second candidate run were superseded or cancelled to free macOS runners. The candidate's tree was + covered by #6052's exact-head CI and by the push-event CI on both promotion commits. +- The pre-move PR (#6053) was merged without PR CI under the owner's heuristic rule; workflow-token PRs do not start CI. +- A 2.67.0-era Child join whose sidecar was deleted before its first 2.68.0 start still forwards like 2.67.0 (owner decision). diff --git a/devlog/_fin/260928_anthropic_cooldown_recovery/000_decision.md b/devlog/_fin/260928_anthropic_cooldown_recovery/000_decision.md new file mode 100644 index 0000000000..eb75d9b5ec --- /dev/null +++ b/devlog/_fin/260928_anthropic_cooldown_recovery/000_decision.md @@ -0,0 +1,46 @@ +# Anthropic cooldown recovery ownership + +## Decision log + +- Purpose: let an authoritative Anthropic usage refresh release a stale reset-derived + cooldown without weakening explicit upstream backoff or clearing another account's state. +- Existing constraints: routing health is process-local, usage probes are asynchronous, and + credentials or a newer 429 can replace the state observed when a probe starts. +- Alternatives considered: clear every cooldown after any successful usage response; clear + only through the operator endpoint; or bind recovery to the observed refusal and credential. +- Decision: a probe captures the exact account credential generation and cooldown generation + before dispatch. Settlement requires the same live credential, the same reset-derived + cooldown, a fresh timestamp, and utilization below 100% for every window that the 429 marked + rejected. Account-level single-flight keys also include an active recovery generation, so a + forced post-429 refresh cannot join work dispatched before the refusal. Any later cooldown + mutation revokes publication ownership as well as settlement ownership. Partial, failed, + exhausted, stale, Retry-After, and default-backoff evidence does not recover anything. +- Why this option: a successful quota HTTP response alone says neither which credential it + measured nor whether a newer refusal arrived while it was in flight. Generation fences make + those ownership claims explicit while preserving the existing manual escape hatch. +- Impact and trade-off: recovered accounts re-enter routing immediately; uncertain evidence + remains fail-closed until expiry or `clear-cooldown`. The extra bookkeeping is process-local + and bounded to one generation plus one health record per account. Superseded probes return an + unavailable result instead of publishing quota that no longer describes the routing state. + +## Data flow + +1. A 429 records its source, rejected quota windows, and a monotonic cooldown generation. +2. A fresh usage probe captures that generation plus the stored credential generation. + When recovery is pending, both the provider-usage flight and the outer account-quota flight + are generation-scoped, so the probe dispatches after the claim instead of joining older work + that might describe pre-refusal state. +3. The usage response is parsed and checked for complete headroom evidence. +4. Publication and settlement both require the observed cooldown generation to remain current. + Settlement deletes only the still-matching reset-derived record. + +The CLI dispatches `openai` to `/api/codex-auth/accounts/clear-cooldown` and `anthropic` to +`/api/oauth/accounts/clear-cooldown`. Anthropic IDs and aliases are resolved through the OAuth +account list before the write; other providers remain rejected because they do not expose this +process-local cooldown owner. + +## Focused verification + +- `tests/providers/anthropic-cooldown-recovery.test.ts` +- `tests/cli/cli-account-pool-verbs.test.ts` +- `tests/adapters/anthropic/anthropic-ratelimit-headers.test.ts` diff --git a/devlog/_fin/260928_release_2_69_0/000_plan.md b/devlog/_fin/260928_release_2_69_0/000_plan.md new file mode 100644 index 0000000000..88274a6b1e --- /dev/null +++ b/devlog/_fin/260928_release_2_69_0/000_plan.md @@ -0,0 +1,17 @@ +# Release 2.69.0 + +Candidate: `870f39e75e` (dev after release train 4). Final Cross-platform CI run 36348371944 on this exact SHA passed (privacy gate skipped by its path condition). Version sources on the candidate read 2.69.0. + +## Steps + +1. Pre-move `dev` past the release: dispatch `dev-version-bump.yml` with `intended-version=2.69.0` from the default branch, verify the opened PR only changes version sources to 2.70.0, merge it through the maintainer dev integration path. +2. Preview: branch from the candidate, `git merge -s ours origin/preview`, run `bun scripts/release-version-sources.ts sync 2.69.0-preview.20260928`, check, open PR to `preview`, merge with a merge commit. Dispatch `release.yml` on `preview` with `version=2.69.0-preview.20260928`, `tag=preview`, `dry-run=false`, `expected-sha=<preview head>`. +3. Stable: branch from the candidate, `git merge -s ours origin/main`, check version sources for 2.69.0 (tree equals the candidate), open PR to `main`, merge with a merge commit. Dispatch `release.yml` on `main` with `version=2.69.0`, `tag=latest`, `dry-run=false`, `expected-sha=<main head>`. +4. Verify npm `latest=2.69.0` and `preview=2.69.0-preview.20260928`, both GitHub releases with the full asset set and correct prerelease flags, and `latest.json` reporting 2.69.0 with five platform signatures. +5. Land this record with the outcome on `dev`. + +## Guards + +Before each release dispatch, the merged promotion SHA itself must carry a successful push-event Cross-platform CI run and a successful Service lifecycle run (release.yml gates on both; the candidate's workflow_dispatch run 36348371944 does not satisfy them). Verify the preview merge tree differs from `870f39e75e` only in the four version sources, and the main merge tree equals it, then use each verified merge SHA as `expected-sha`. + +Do not weaken release preflight, exact-SHA, or CI gates. If a release run fails, read the failing job, fix through a PR to `dev` and re-promote, or use the workflow's documented resume input only when npm publication was acknowledged for the same commit. diff --git a/devlog/_fin/260928_release_2_69_0/090_outcome.md b/devlog/_fin/260928_release_2_69_0/090_outcome.md new file mode 100644 index 0000000000..1bc1087d2e --- /dev/null +++ b/devlog/_fin/260928_release_2_69_0/090_outcome.md @@ -0,0 +1,18 @@ +# Release 2.69.0 outcome + +Published 2026-09-28 (KST). npm `latest=2.69.0`, `preview=2.69.0-preview.20260928`. + +| Step | Evidence | +|---|---| +| Release tree | Candidate `870f39e75e` (release train 4; final Cross-platform CI run 36348371944 passed) plus #6140, which removed agent scratch files (`.agents/`, root `design-debt.md`, train 4 lane `_handoff.md`) and added a repo-hygiene guard. dev at `b3d445d8ec`. | +| dev pre-move | #6136 moved dev to 2.70.0 (`081b670892`) before any publish. | +| Superseded promotion | #6137 (preview) and #6138 (main) were merged from `870f39e75e` and then replaced before publication, when the owner asked for the scratch-file cleanup. Their push CI runs were cancelled; nothing was published from them. | +| Preview | #6141 merged as `24f55dcb3f` (tree = dev + four version sources at `2.69.0-preview.20260928`). Push CI 36355208544 and Service lifecycle 36355270913 (dispatched, since the merge touched no service path relative to the previous preview head) passed. Release run 36356404530 succeeded. GitHub release `v2.69.0-preview.20260928`: prerelease, 25 assets. | +| Stable | #6142 merged as `3cc34e1181` (tree = dev + four version sources at `2.69.0`). Push CI 36355212673 and Service lifecycle 36355272973 (dispatched) passed. Release run 36357595961 succeeded. GitHub release `v2.69.0`: not prerelease, 25 assets; `latest.json` reports 2.69.0 with five signed platforms. | +| npm | `2.69.0-preview.20260928` published 23:11Z; `2.69.0` published 23:27Z and visible on the registry at 23:33Z. | + +## Notes + +- The preview push CI skipped path-gated jobs (Windows shards, macOS control and others) because the merge differed from the previous preview head only in cleanup files. The full matrix ran and passed on `870f39e75e` (run 36348371944) and on the main promotion push (run 36355212673). +- An independent plan audit (gpt-6-sol) found that the release gate needs push-event CI and Service lifecycle on each promotion SHA; that was folded into [000_plan.md](000_plan.md) before publishing. + diff --git a/devlog/_fin/260929_dev_next_hardening_carry/010_plan.md b/devlog/_fin/260929_dev_next_hardening_carry/010_plan.md new file mode 100644 index 0000000000..66a999e206 --- /dev/null +++ b/devlog/_fin/260929_dev_next_hardening_carry/010_plan.md @@ -0,0 +1,77 @@ +# 260929 dev after 2.70.0: Claude contract hardening and PR carries + +Loop objective: harden the Sonnet 5.5 rollout and land a small, reviewed set of open PRs on `dev` +(owner request 2026-09-29, admin merge authorized). Base: `dev` `a118fcc64c` (2.71.0 line). +Research: two gpt-6-sol read-only leaves (hardening audit, PR/issue triage) and live probes. + +## Live evidence (2026-09-29, api.anthropic.com, OAuth, one field changed per request) + +| Model | temperature 0.2 | top_p 0.9 | top_k 5 | tool_choice any | thinking disabled | between_tools | +|---|---:|---:|---:|---:|---:|---:| +| claude-sonnet-5-5 | 400 | 400 | 400 | 400 | 400 | 200 (xhigh effort: 400) | +| claude-sonnet-5 | 400 | 400 | 400 | 200 | 200 | — | +| claude-fable-5-1 | 400 | 400 | 400 | 400 | 400 | 400 | +| claude-fable-5 | 400 | 400 | 400 | 200 | 400 | — | +| claude-opus-5-5 | 400 | 400 | 400 | 400 | 400 | 400 | +| claude-opus-5 | 400 | 400 | 400 | 200 | 200 | 400 | +| claude-opus-4-8, 4-7 | 400 | 400 | 400 | 200 | 200 | — | +| claude-opus-4-6, sonnet-4-6, haiku-4-5 | 200 | 200 | 200 | 200 | 200 | 400 (haiku) | + +`temperature: 1` (the default) returns 200 on Sonnet 5.5, Opus 5.5, Fable 5.1 and Opus 5. + +## Units + +| wp | Unit | Method | Why now | +|---|---|---|---| +| wp2 | Claude request-contract hardening | own PR | Sampling rejection covers every adaptive family, not only Sonnet 5.5; Fable 5.1 rejects forced tool choice; sidecars send `disabled` to Opus 5.5 / Fable, which reject it | +| wp3 | #6194 reject forged `ss` owner tuples (luvs01) | merge | approved, CLEAN, exact-head CI green; port-reclaim trust boundary | +| wp4 | #6195 preserve drift-heal ownership veto (luvs01) | merge | approved, CLEAN, CI green; foreign service-home write | +| wp5 | #6193 bound GLM checkpoint-envelope scanning (luvs01) | merge after disposing one CodeRabbit thread | approved, CLEAN, CI green | +| wp6 | #6089 Usage blank scroll (fflake33) | carry with Co-authored-by | maintainer-approved UI fix held in draft by the author checklist | + +Not landed: #6214 (removes Kiro preemptive rows; Kiro now publishes Opus 5.5 and the owner chose +preemptive rows, so it needs an owner decision), #6209 (draft, needs a real Windows run), #6119 (open +maintainer objection), Cursor/Devin effort-suffixed price lookup (logs record the base selector, so no +observed unpriced rows), Messages-native passthrough rewriting (caller-owned contract, +structure/data-planes/protocol-paths.md), dotted Bedrock ids (no direct Bedrock adapter reaches +`anthropic.ts`). + +## wp2 diff + +`src/adapters/anthropic-model-contract.ts`: +- `rejectsSamplingParameters`: Sonnet >= 5.0, Opus >= 4.7, every Fable. The adapter already drops + temperature/top_p whenever it sends thinking; this closes the no-reasoning path. +- `rejectsForcedToolChoice`: add Fable >= 5.1. +- `sidecarThinkingOff`: `between_tools` for Sonnet >= 5.5; omit the field for Opus 5.5 and Fable + (both reject `disabled` and `between_tools`); `disabled` otherwise. Sidecars spread the result. +Tests: extend `tests/adapters/anthropic/anthropic-sonnet-5-5-contract.test.ts` with the family table; +update any existing test that expects temperature on an adaptive family. Structure docs updated. +Verify: focused anthropic, web-search, vision tests; typecheck; structure; privacy; live re-probe. + +## wp3-wp5 + +Per PR: fetch head, merge onto current `dev` in a `/private/tmp` checkout, run the PR's focused tests +plus typecheck and the file-size ratchet on the combined tree, check open review threads, then +`gh pr merge --squash --admin` (author credit stays with the PR). + +## wp6 + +Cherry-pick #6089's commits onto a fresh branch from `dev`, add `Co-authored-by` for fflake33, run the +Usage tests, `bun run lint:gui` and `bun run build:gui`, open a PR carrying its screenshot link, admin +merge, then close #6089 with a pointer to the carry. + +## Audit (020, gpt-6-sol, NEAR-PASS) folded + +1. Sidecar budget for families that reject both `disabled` and `between_tools`: live probe at + max_tokens 1024 returned `end_turn` with 430-630 characters of text for Opus 5.5, Fable 5.1 and + Fable 5, both with thinking omitted and with `output_config.effort: "low"`. The sidecar sends + `output_config: {effort: "low"}` and no `thinking` for those families. +2. The sampling rule keys on the family-first parse; legacy `claude-3-7-sonnet` does not parse and + keeps its sampling fields, and `claude-opus-4-20250514` parses as Opus 4.0 (keeps them). +3. `tests/adapters/anthropic/anthropic-reasoning.test.ts` expects Fable 5 to keep `temperature`; it is + updated to the live contract. New wire cases cover each rejecting family and Fable 5.1 forced choice. +4. wp3/wp4 run the combined-tree checks on current `dev` before each admin merge. +5. wp5 disposes the open CodeRabbit thread (the old regex was already `/i`) before merging. +6. wp6 opens a template PR with the screenshot link, trailer + `Co-authored-by: Jian Gong <fflake33@icloud.com>`, waits for its exact-head CI, and adds typecheck, + structure and privacy checks to the GUI verification. diff --git a/devlog/_fin/260929_dev_next_hardening_carry/030_done_wp2.md b/devlog/_fin/260929_dev_next_hardening_carry/030_done_wp2.md new file mode 100644 index 0000000000..a885a3489f --- /dev/null +++ b/devlog/_fin/260929_dev_next_hardening_carry/030_done_wp2.md @@ -0,0 +1,20 @@ +# 030 Done (wp2): Claude request-contract hardening + +`src/adapters/anthropic-model-contract.ts` now encodes the family table measured live on 2026-09-29 +(010 plan, plus a follow-up probe of the 4.5/4.6 families): + +| Rule | Families | Before | After | +|---|---|---|---| +| Drop `temperature`/`top_p` | Opus 4.7+, Sonnet 5+, every Fable | kept unless thinking was sent; 400 "temperature is deprecated" | dropped; 200 | +| Keep one sampling field | Haiku 4.5, Sonnet 4.5/4.6, Opus 4.5/4.6 | both sent; 400 "cannot both be specified" | `top_p` dropped when both present; 200 | +| Forced `tool_choice` -> `auto` | adds Fable 5.1 (with Opus 5.5, Sonnet 5.5+) | `any`; 400 | `auto`; 200 | +| Sidecar thinking-off | Opus 5.5, Fable | `thinking: disabled`; 400 | `output_config.effort: low`, no `thinking`; 200 end_turn at 1,024 tokens | + +Adapter-built requests re-probed after the change: temperature+top_p, required tool choice and the +sidecar off switch return 200 on Opus 5.5, Fable 5.1, Fable 5, Opus 5, Sonnet 5, Sonnet 5.5, Opus 4.8; +Haiku 4.5 needed the combined-sampling rule and returns 200 with it. Unmeasured older ids +(`claude-3-7-sonnet`, `claude-opus-4-20250514`, Opus 4.1 which 404s on this account) keep both fields. + +Tests: `tests/adapters/anthropic/anthropic-sonnet-5-5-contract.test.ts` pins the table; +`anthropic-reasoning.test.ts` updated where it asserted the rejected wire shapes (Fable 5 temperature, +Sonnet 4.5 temperature+top_p). diff --git a/devlog/_fin/260929_dev_next_hardening_carry/090_outcome.md b/devlog/_fin/260929_dev_next_hardening_carry/090_outcome.md new file mode 100644 index 0000000000..c39c369c02 --- /dev/null +++ b/devlog/_fin/260929_dev_next_hardening_carry/090_outcome.md @@ -0,0 +1,24 @@ +# 090 Outcome: dev after 2.70.0 + +All six work-phases closed on 2026-09-29. `dev` moved from `a118fcc64c` to `2e35b2573d`. + +| wp | Unit | Landed | +|---|---|---| +| wp1 | roadmap, gpt-6-sol audit NEAR-PASS (folded) | this unit | +| wp2 | Claude request-contract hardening | #6217 `08a85175f3` | +| wp3 | forged `ss` owner tuples (luvs01) | #6194 `d0157e0f59` | +| wp4 | drift-heal ownership veto (luvs01) | #6195 `101fa61137` | +| wp5 | bounded GLM checkpoint scan (luvs01); CodeRabbit doc thread answered and resolved | #6193 `ab5807267e` | +| wp6 | Usage blank scroll, carried with `Co-authored-by: Jian Gong`; #6089 closed with a pointer | #6218 `2e35b2573d` | + +Verification on `2e35b2573d` in a `/private/tmp` checkout with isolated homes: typecheck; 337 pass / +1 skip / 0 fail across the contract, reasoning, tool-choice, web-search, usage-cost, port-reclaim, +catalog auto-refresh, GLM summary, metadata sync, file-size ratchet and test-layout files; GUI Usage +tests 44 pass; structure and privacy checks pass. #6194, #6195 and #6193 were first merged together +onto `08a85175f3` (typecheck, 102 pass / 1 skip); #6218 passed its full exact-head CI before merge. +#6217 and the three luvs01 PRs were admin-merged without new exact-head CI at the owner's request; +their approved heads had green CI. + +Held for an owner decision: #6214 removes the preemptive Kiro rows (Sonnet 5.5, Opus 5.5, GPT-6 +Sol/Luna); Kiro now publishes Opus 5.5 and the preemptive policy was the owner's choice. Also held: +#6209 (needs a real Windows run), #6119 (open maintainer objection). diff --git a/devlog/_fin/260929_release_2_71_0/000_plan.md b/devlog/_fin/260929_release_2_71_0/000_plan.md new file mode 100644 index 0000000000..15afb0c7a8 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/000_plan.md @@ -0,0 +1,70 @@ +# 000 Plan: integrate six reviewed PRs and release 2.71.0 + +Six open pull requests were reviewed as ready or nearly ready, and dev already carries eight +commits since v2.70.0. This unit lands the six PRs on dev through one integration PR, fixes the +small findings left on them, removes the Windows timeout that turned the dev tip red, and publishes +2.71.0 (stable and preview). Users get the Claude picker CA hardening, the requestPacing +concurrency cap in the GUI, the Cursor Private Inference installer link, the Windows /healthz +priority fix and the cross-home ownership proof; maintainers get a green dev tip. + +## Loop spec + +- Archetype: satisfy-spec, HOTL multi-cycle (cxc-loop), one integration lane. +- Trigger: owner request on 2026-09-29 to fix and merge the six PRs recommended by the Kimi + release triage, then prepare and deploy the next release; cross-platform CI only at the end; + Kimi subagents may be dispatched without limit. +- Goal: #6201 #6206 #6094 #5905 #6209 #6198 on dev with author credit, 2.71.0 and + 2.71.0-preview.20260929 on npm and GitHub. +- Non-goals: every other open PR (#6204 #6161 #6157 #6192 #6205 #6203 #6200 #6152 #6151 #6188 ...), + raising any file-size cap, weakening CI/release gates, updating the installed proxy/service/app on + this machine, touching real accounts or keychain. +- Verifier: local gates in 020 (typecheck, full test, lint:gui, build:gui, privacy:scan, + structure:check, skill:surface:check, focused per-PR tests) before the single push; then + exact-head Cross-platform CI + Service lifecycle on the integration PR head; then push-event CI + + Service lifecycle on each promotion SHA; then release.yml runs and npm/GitHub read-back. +- Stop condition: 090 outcome on dev with npm latest=2.71.0 and preview=2.71.0-preview.20260929. +- Memory artifact: this unit (000-040 plans, 090 outcome), goalplan + `.codexclaw/goalplans/release-opencodex-2-71-0-after-integrating-six-r`. +- Terminal outcomes: DONE as above; UNSAFE drops a PR that shows a security defect and continues; + BLOCKED when a secret/permission/CI gate fails twice for a non-code cause. +- Escalation: a required approval GitHub will not accept from an admin merge, or a release + workflow failure after npm acknowledged publication (use resume-after-npm-publish, never + republish). +- Resource bounds: tools = gh (owner token), git, bun; write scope = branch + codex/release-2-71-0, dev/preview/main via PRs, pr-assets; no token/time budget set by the user. + +## Baseline + +- v2.70.0 = 53834ff47b; dev = 37ad7e771b (version sources 2.71.0). +- dev tip Cross-platform CI run 36499924172 failed only `windows 8/9`: + `native main profile transactions > allows 32 profiles ...` timed out at 30s (35.0s). The test + normally runs 0.45s on Windows; dev dispatch history shows 2.7s, 3.6s, 6.9s, then 35s. Code + under test last changed 2026-09-09. Classified as runner I/O contention (Kimi flake report). + +## Source PRs (heads pinned 2026-09-29) + +| PR | head | author | review state | Kimi verdict | +|---|---|---|---|---| +| #6206 | 46bf1c9931 | Ingwannu | APPROVED (luvs01) | merge with doc fixes | +| #6201 | 986f9a0c99 | luvs01 | APPROVED (Ingwannu, security maintainer) at head | merge | +| #6209 | 8fb990dab0 | MeroZemory (Jio Kim) | draft, no human review | merge with doc fix | +| #6094 | de9880f75a | bradhallett | APPROVED at head | merge | +| #5905 | 2b3dacdde5 | halysondev | APPROVED at head | merge | +| #6198 | 8dbb264300 | luvs01 | CHANGES_REQUESTED on 3cff6018; all items fixed by later commits | merge with comment fix | + +Sequential 3-way application of each PR's merge-base..head diff onto dev in the order above is +conflict-free (tree 4d81bf2856). None of the 58 touched files is tracked in +tests/fixtures/file-size-baseline.json, and none crosses the 2000-line new-file threshold +(largest: src/cli/index.ts at 1979 after both #6209 and #6198). + +## Work-phase map (dependency order) + +1. wp0 — this roadmap (docs only). +2. wp1 — 010: integration branch, six author-preserving commits, four fix commits. +3. wp2 — 020: local regression gates and independent Kimi review of the integrated diff. +4. wp3 — 030: single push, integration PR, exact-head CI, admin merge, close source PRs. +5. wp4 — 040: dev pre-move, preview/main promotion, release.yml, verification, 090 outcome. + +## Architect consultation + +Recorded in 001_consultation.md. diff --git a/devlog/_fin/260929_release_2_71_0/001_consultation.md b/devlog/_fin/260929_release_2_71_0/001_consultation.md new file mode 100644 index 0000000000..3582993270 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/001_consultation.md @@ -0,0 +1,23 @@ +# 001 Architect consultation + +- Handle: Kimi architect `01a0eb33-e0ec-75e1-afe0-cd713995a00f` (third spawn; two earlier spawns + `01a0eb29-a74d...` and `01a0eb2e-74ad...` failed with provider 429 before producing output, so + they are transport failures, not task failures). +- Inputs besides the architect: Kimi per-PR reviewers for #6201 #6206 #6094 #5905 #6209 #6198 and a + flake investigator (results summarised in 000 and 010). + +| ID | Proposal | Main disposition | +|---|---|---| +| D1 | One squashed commit per PR, `--author` = PR author, Co-authored-by trailers; no cherry-pick of original commits | Accepted. Trailers copy every other identity found in the source PR's commits verbatim (human alt emails and bot identities), because `.github/scripts/pr-carry-attribution.cjs` matches exact names/emails and has no bot allowlist. The coordinator is the committer, so no extra coordinator trailer. | +| D1a | Carry gate only fires when a carry verb (reimplement/supersede/carry/rebase/adopts the design from) sits within 80 chars of a bare #N (pr-carry-attribution.cjs:15,24-37,307) | Accepted. Commit and PR text say "lands"/"from #N", never a carry verb next to a reference; trailers are present anyway. | +| D2 | Fold the review fixes into each PR's squash commit | Rejected. Folding would attribute coordinator edits to the original authors. The fixes are docs, a comment and a test budget, so separate commits cost nothing for bisect. | +| D3 | Integration PR body per template; gui/ in the union requires an embedded screenshot (pr-quality.cjs:211-215,315-321,579-584); unsponsored_surface cannot fire | Accepted. Verified: no path in the union matches RESTRICTED_PREFIXES/FILES in pr-sponsored-surface.cjs:24-56, and the author has push permission (exempt at :76). Screenshots go to pr-assets, linked by SHA. | +| D4 | Release after merge per the 2.70.0 procedure; pre-move touches only the four version sources (release-version-sources.ts:37-40) and must merge before release dispatch (release.yml:995) | Accepted; 040 orders pre-move after the integration merge so the candidate is final before any version movement. | +| R1-R4 | Loose commit text, bot identity mismatch, screenshot gate, author identity source | Folded into D1/D1a/D3; identities come from PR commit metadata (gh pr view --json commits). | + +Reflection: sent 000/001/010/020/030/040 revision 1 to the same architect (submission +01a0eb38-7d6f-71d3-a554-5b3078e6fd64). Result: ALIGNED. Identities in the 010 table match +`gh pr view N --json commits` for all six PRs; F1-F4 anchors verified on the PR heads; D2 rejection +creates no gate or attribution problem (the carry gate reads text, not diff authorship). Gaps and +dispositions: (1) 030 wording "every carried author" contains a carry verb — reworded to "every +source author"; (2) F4 locator comment in 010 is not in the source file — clarified as a locator. diff --git a/devlog/_fin/260929_release_2_71_0/010_integration.md b/devlog/_fin/260929_release_2_71_0/010_integration.md new file mode 100644 index 0000000000..a2dec42b28 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/010_integration.md @@ -0,0 +1,99 @@ +# 010 Integration (wp1): branch codex/release-2-71-0 + +Base: dev `37ad7e771b`. Each source diff is `git diff $(git merge-base canon/dev rt/pr-N) rt/pr-N` +applied with `git apply --3way --index`, then committed with the PR author as author. Order is the +conflict-free sequence proven in 000. + +## Land commits (author / trailers copied from each PR's commit metadata) + +| # | PR (head) | --author | Co-authored-by trailers | subject | +|---|---|---|---|---| +| 1 | #6206 (46bf1c9931) | Ingwannu <ingwannu@users.noreply.github.com> | — | docs(plan): define supported macOS quota admission (#6206) | +| 2 | #6201 (986f9a0c99) | luvs01 <27862058+luvs01@users.noreply.github.com> | Epinephrine <luvs01@hanmail.net>; Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> | fix(claude): validate picker CA scope and require the full minted CA profile (#6201) | +| 3 | #6209 (8fb990dab0) | Jio Kim <merozemory@gmail.com> | Claude Opus 5.5 <noreply@anthropic.com> | fix(windows): run the proxy at ABOVE_NORMAL priority so a saturated host cannot starve /healthz (#6209) | +| 4 | #6094 (de9880f75a) | Brad Hallett <53977268+bradhallett@users.noreply.github.com> | — | feat(gui): expose requestPacing.maxConcurrentRequests in provider settings (#6094) | +| 5 | #5905 (2b3dacdde5) | halysondev <halysoncesar2020@gmail.com> | codingbooo <9621077+codingbooo@users.noreply.github.com>; Claude Opus 5.5 <noreply@anthropic.com> | feat(cursor): surface the Private Inference local-mode installer for regular Cursor (#5905) | +| 6 | #6198 (8dbb264300) | luvs01 <27862058+luvs01@users.noreply.github.com> | Epinephrine <luvs01@hanmail.net> | fix(cli): prove cross-home ownership before deferring to a hinted port (#6198) | + +Body of each: "Lands #N at head <sha> on dev through the 2.71.0 integration branch." #6209's body +adds "Closes #6208." Acceptance: `git show --stat` of each commit equals the PR's merge-base..head +stat (same files, same +/- counts). + +## Fix commits (coordinator-authored) + +F1 `docs(plan): pin full source paths and redirect refusal in the macOS quota design` +MODIFY devlog/_plan/260928_macos_quota_gate/000_design.md: + +```diff +- the macOS limitation. `usage-policy.ts:26-48` controls two WHAM booleans, not the ++ the macOS limitation. `src/codex/desktop-compatibility/usage-policy.ts:26-48` controls two WHAM booleans, not the +- recovery is still unverified. `runtime-ownership.ts:15-27` binds exact PAC URL to ++ recovery is still unverified. `src/codex/desktop-compatibility/runtime-ownership.ts:15-27` binds exact PAC URL to +``` + +and after the sentence ending "even if another member of the closure is independently funded.": + +```diff ++ ++A 3xx response is terminal for admission: the reservation contract never fetches a ++`Location` and never treats a redirect destination as admitted. If redirects are ever ++supported, each resolved `Location` is a new immutable target that must pass the full ++funding classification, closure, generation, TLS and credential-attachment checks ++before dispatch. +``` + +Resolves CodeRabbit thread PRRT_kwDOS-0Gi86mz4CC (CWE-862 redirect refusal). Both paths exist on the +#6079 branch only; the doc already names #6079 as their source, so full paths are the accurate form. + +F2 `docs(cli): describe the Windows priority boost as best-effort` +MODIFY docs-site/src/content/docs/reference/cli.md — replace the paragraph #6209 added (it claims +the loop never waits past the ceilings, while the PR's own measurement shows p90 3.0s at 100% +saturation) with: + +```markdown +On Windows the proxy also raises its own process to ABOVE_NORMAL priority when it starts, which +reduces scheduling delays on a host saturated by other NORMAL-priority work (antivirus scans, +encoders, emulators) without guaranteeing the probe stays under these ceilings at extreme load. +The boost applies to the proxy process only — work it spawns still runs at NORMAL — and a +CPU-heavy proxy can itself delay NORMAL-priority applications. The change is best-effort; set +`OCX_DISABLE_PRIORITY_BOOST=1` in the proxy's environment to leave the priority unchanged. +``` + +F3 `chore(server): repair mangled punctuation in the proxy liveness comment` +MODIFY src/server/proxy-liveness.ts (~284): `probe uses ??"did not answer"` -> `probe uses — "did not answer"` +(od -c confirmed two literal '?' bytes). + +F4 `test(codex): give the 32-profile transaction case the bulk durable-IO budget` +MODIFY tests/codex-integration/native-profile-manager.test.ts: + +```diff +-import { INTERNAL_DEADLINE_MS, SPAWN_BUDGET_MS } from "../helpers/test-budget"; ++import { BULK_DURABLE_IO_BUDGET_MS, INTERNAL_DEADLINE_MS, SPAWN_BUDGET_MS } from "../helpers/test-budget"; +... +- }, 30_000); ++ }, BULK_DURABLE_IO_BUDGET_MS); +``` + +(Line 938, the closing line of the test "allows 32 profiles and rejects profile 33 without changing +the vault"; the import is line 21.) + +Justification per tests/helpers/test-budget.ts: the case performs 32 register transactions, each +several `atomicWriteFileAsync` calls that fsync (src/config/atomic-write.ts:181,245) and, on Windows, +harden the temp file; those writes are the assertion (32 accepted, 33rd refused, vault bytes +unchanged), so the wait is intrinsic, and the ablation still fails because no assertion depends on +the budget. Observed Windows durations: 0.45s (PR runs), 2.7/3.6/6.9s (dev dispatch), 35.0s (run +36499924172). Constant = 180s on win32, 90s elsewhere. No line is added (file-size safe). Sibling +30_000 budgets are left alone; they have not flaked. + +## Scope boundary + +IN: the six diffs, F1-F4. OUT: optional hardening noted by reviewers (response-size cap and +redirect policy for the Cursor manifest fetch, duplicate [3] TBS field strictness in picker-ca, +extra #6209 subprocess tests, localized copies of the cli.md priority note) — recorded as +follow-ups in 090, not built here. + +## Exit (checked in wp1 C) + +- `git log --format='%an <%ae>%n%(trailers:key=Co-authored-by)' canon/dev..HEAD` matches the table. +- `git diff canon/dev..HEAD --stat` = union of six PR stats + F1-F4 files. +- typecheck + the focused tests listed in 020 pass in the /private/tmp verification checkout. diff --git a/devlog/_fin/260929_release_2_71_0/011_wp1_execution.md b/devlog/_fin/260929_release_2_71_0/011_wp1_execution.md new file mode 100644 index 0000000000..cfa53f2b90 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/011_wp1_execution.md @@ -0,0 +1,58 @@ +# 011 wp1 execution script + +Revalidated at wp1 P (2026-09-29): dev 37ad7e771b, main 53834ff47b, preview aa3a8dda16 unchanged; +all six PR heads equal the 010 table. Previous D (wp0) conclusion: roadmap locked, audit PASS, +execute 010 as written. + +Run from the managed worktree on branch codex/release-2-71-0 (HEAD e69b48f71e = dev + roadmap, +plus this file once committed). The script lives at .tmp/rt2710/land.sh (gitignored): + +```sh +set -eu +git diff --quiet HEAD && git diff --cached --quiet +norm() { grep -E '^(diff --git|[-+])' | grep -vE '^(\+\+\+|---) '; } +land() { # n author subject trailers... + n=$1; author=$2; subject=$3; shift 3 + head=$(git rev-parse rt/pr-$n); base=$(git merge-base canon/dev rt/pr-$n) + git diff --binary "$base" "$head" > ".tmp/rt2710/pr-$n.patch" + git apply --3way --index ".tmp/rt2710/pr-$n.patch" + body="Lands #$n at head $head on dev through the 2.71.0 integration branch." + if [ "$n" = 6209 ]; then body="$body + +Closes #6208."; fi + trailers="" + for t in "$@"; do trailers="$trailers +Co-authored-by: $t"; done + msg="$subject + +$body" + if [ -n "$trailers" ]; then msg="$msg +$trailers"; fi + git commit -q --author="$author" -m "$msg" + pa=$(git diff HEAD^ HEAD | git patch-id --stable | cut -d' ' -f1) + pb=$(git patch-id --stable < ".tmp/rt2710/pr-$n.patch" | cut -d' ' -f1) + la=$(git diff --binary HEAD^ HEAD | norm | git hash-object --stdin) + lb=$(norm < ".tmp/rt2710/pr-$n.patch" | git hash-object --stdin) + [ "$pa" = "$pb" ] && [ "$la" = "$lb" ] || { echo "diff mismatch for #$n"; exit 1; } +} +# six land calls with the 010 table arguments, in order 6206 6201 6209 6094 5905 6198 +test -z "$(git diff --name-status 4d81bf2856 HEAD | grep -v 'devlog/_plan/260929_release_2_71_0/')" +``` + +D5-D8 (architect, wp1 proposal): fail-fast apply (set -e and clean-tree assert), equality of +each land commit's diff against the PR diff, a blank line before the trailer block (the trailers +string starts with a newline, so "body + newline + trailers" leaves one empty line), and the tree +check against 4d81bf2856 as a command. All accepted. The base assert is on the clean tree rather +than a fixed SHA because this file is committed first. + +D6 amended (main, wp1 P): byte equality of `git diff --binary` is the wrong check. A dry run in a +temporary index showed #6209, #5905 and #6198 produce different bytes only because an earlier +land commit already touched the same file (layout rosters, src/cli/index.ts), which shifts hunk +line numbers and context. The check is now `git patch-id --stable` equality plus equality of the +changed lines only (`diff --git` headers and +/- lines, file headers dropped). Dry run: all six +EQUAL on both measures; final tree 4d81bf2856. + +Then F1-F4 exactly as 010, each applied with apply_patch and committed by the coordinator. + +wp1 C (before wp2's full suite): `bun run typecheck` and the focused test files from 020 in +/private/tmp/rt2710-verify at the wp1 head. diff --git a/devlog/_fin/260929_release_2_71_0/020_verification.md b/devlog/_fin/260929_release_2_71_0/020_verification.md new file mode 100644 index 0000000000..24906de0d7 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/020_verification.md @@ -0,0 +1,53 @@ +# 020 Verification (wp2): local regression before the only push + +All commands run in a detached verification worktree at `/private/tmp/rt2710-verify` pinned to the +integration head. The coordinator's managed worktree lives under `~/.codex`, and +src/lib/test-home-guard.ts:232-237 refuses to delete any path inside the real Codex home from a test +process: a first full run inside the managed worktree at dev 37ad7e771b failed 1210 cases, every +sampled one with "refusing to remove a path inside the real Codex home". That run is discarded as +environment noise. Each result is recorded with exit code in the wp2 C attest and receipt. A +failure is classified by rerunning the same file in a second /private/tmp worktree at dev +37ad7e771b; identical failure there = pre-existing, otherwise it is ours. + +| Gate | Command | Reads the change because | +|---|---|---| +| deps | `bun install && (cd gui && bun install)` | lockfiles unchanged; ensures gui types | +| types | `bun run typecheck` | tsconfig includes src/ and tests/ (all six PRs' TS) | +| GUI types/lint | `bun run lint:gui` | gui/src (#6094, #5905 pages, i18n) | +| GUI build | `bun run build:gui` | bundles gui/src incl. new i18n keys | +| full suite | `bun run test` | tests/ incl. every new/changed test file below | +| layout | `bun test tests/test-layout.test.ts tests/test-layout-tooling.test.ts` | new test files registered in both rosters | +| structure | `bun run structure:check` | structure/*.md edits (#6201, #6209, #5905, #6198) | +| skill surface | `bun run skill:surface:check` | src/cli/capabilities.ts + skills/ocx (#5905) | +| privacy | `bun run privacy:scan` | devlog + docs + src text | +| GUI tests | `cd gui && bun test tests` | gui/tests/provider-settings-request-pacing.test.tsx, cursor-integration-page.test.tsx | + +Focused files (run first, then inside the full suite): + +- #6201: tests/claude-integration/claude-picker-ca.test.ts, claude-desktop-cli.test.ts +- #6209: tests/windows/windows-process-priority.test.ts (real-Windows case skips on macOS; covered by + CI windows shards), tests/cli/cli-start-*.test.ts +- #6094: gui/tests/provider-settings-request-pacing.test.tsx +- #5905: tests/providers/cursor/cursor-local-installer.test.ts, cursor-integration-status.test.ts, + gui/tests/cursor-integration-page.test.tsx +- #6198: tests/cli/sibling-home-client-sync.test.ts, tests/server/proxy-liveness-package-tree-fence.test.ts, + tests/cli/cli-dispatch.test.ts +- flake fix: tests/codex-integration/native-profile-manager.test.ts + +Activation checks for the conditional paths the fixes touch: + +- #6209 opt-out and non-win32 skip: covered by injected-platform unit cases in + windows-process-priority.test.ts (observable return value "skipped"). +- Budget change: the ablation still fails — the assertions (32 profiles, INVALID_REQUEST on 33, + vault bytes unchanged) do not depend on the budget; the budget only bounds a hang. + +Independent review: a Kimi reviewer reads `git diff canon/dev...HEAD` and the ten commit +messages; REVIEW-SYNTHESIS records accept/rebut before C. + +Exit: every gate exit 0, or a failure proven identical in the baseline log (named test, same +error) and unrelated to touched files. + +Baseline at dev 37ad7e771b in /private/tmp/rt2710-base (`bun run test`, 13m44s, load avg ~6 from +unrelated desktop processes): 9 failures, all timeouts in spawned-CLI cases — `CLI subcommand +help` x4 (40s), `ocx launcher graceful shutdown` x3 (20s), `ocx models` x2 (15s). None of these +files is touched by the six PRs. They are the classification reference for wp2. diff --git a/devlog/_fin/260929_release_2_71_0/021_wp2_execution.md b/devlog/_fin/260929_release_2_71_0/021_wp2_execution.md new file mode 100644 index 0000000000..e496cda3c3 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/021_wp2_execution.md @@ -0,0 +1,52 @@ +# 021 wp2 execution + +Previous D (wp1): integration head 441aeb179e built; typecheck and 16 focused files (321 tests) pass +in /private/tmp/rt2710-verify; the only skip is the real-Windows priority readback, which the +Windows CI shards run. Direction unchanged: run the full 020 gate set on that head. + +Head under test: 441aeb179e (verification worktree /private/tmp/rt2710-verify, detached). + +Gate script .tmp/rt2710/wp2-gates.sh (each gate's exit is recorded; the script does not stop at the +first failure so every gate reports): + +```sh +cd /private/tmp/rt2710-verify +test "$(git rev-parse HEAD)" = 441aeb179e173c7778afdbecf2d2ad1675146cd6 +run() { name=$1; shift; "$@" > /private/tmp/rt2710-gate-$name.log 2>&1; echo "$name exit=$?"; } +run typecheck bun run typecheck +run lint-gui bun run lint:gui +run build-gui bun run build:gui +run privacy bun run privacy:scan +run structure bun run structure:check +run skill-surface bun run skill:surface:check +run gui-tests sh -c 'cd gui && bun test tests' +run docs-build sh -c 'cd docs-site && bun install --frozen-lockfile && bun run build' +run full-test bun run test +git status --short # must be empty: no gate may leave tracked changes +``` + +Classification of any full-test failure: rerun the failing files alone with +`bun scripts/test.ts <file>` in /private/tmp/rt2710-verify and in /private/tmp/rt2710-base (dev +37ad7e771b). Pass alone at head = load flake (record); fail at head and pass at base = regression +(fix in a new commit, return to wp1 amendment); fail in both = pre-existing (record, not ours). +Base reference: 9 load-timeout failures (020). + +Independent review in parallel with the gates: a fresh Kimi reviewer reads +`git diff 37ad7e771b..441aeb179e` and `git log --format=%B` for the ten commits and reports +defects in the union (cross-PR interactions, i18n key collisions, capability counts, docs +contradictions). REVIEW-SYNTHESIS: each finding accepted (fix commit) or rebutted with reason. + +Exit: all gates exit 0 after classification, review findings dispositioned, working tree clean. + +D9 (architect, wp2): the union touches eight docs-site pages and no other gate builds docs-site +(deploy-docs.yml runs on main only), so an Astro build is added. Accepted. The architect confirmed +no path under desktop/, go/, app/ or native/ is touched, so the macOS/cargo helper suites are not +needed; root `bun run test` does not run gui/tests, so the separate gui gate stays. + +Audit (Kimi 01a0eb66, PASS). Dispositions: the deps gate ran when the verification worktree was +created (`bun install` root and gui at 441aeb179e, lockfiles unchanged vs dev). The "fail in both" +classification compares the named test and its error text, not only the file, because #6198 and +#6209 touch the spawned-CLI area where the base has load timeouts. Union finding 1 (F1 paths "do +not exist") is rebutted: both files exist on the #6079 head that the paragraph describes +(`git ls-tree rt/pr-6079`: src/codex/desktop-compatibility/usage-policy.ts and runtime-ownership.ts, +with the cited line ranges matching). diff --git a/devlog/_fin/260929_release_2_71_0/022_wp2_results.md b/devlog/_fin/260929_release_2_71_0/022_wp2_results.md new file mode 100644 index 0000000000..c8a774d164 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/022_wp2_results.md @@ -0,0 +1,45 @@ +# 022 wp2 results: local regression on the integration head + +Head 441aeb179e in /private/tmp/rt2710-verify; base dev 37ad7e771b in /private/tmp/rt2710-base. +Machine: macOS, 15 cores, load average 4-11 from unrelated desktop processes during the runs; a real +opencodex proxy listens on 127.0.0.1:10100. + +## Gates + +| Gate | Result | +|---|---| +| typecheck | exit 0 | +| lint:gui | exit 0 | +| build:gui | exit 0 | +| privacy:scan | exit 0 | +| structure:check | exit 0 | +| skill:surface:check | exit 0 | +| gui `bun test tests` | exit 0 | +| docs-site `bun run build` | exit 0 | +| focused files (wp1) | 321 tests, 0 fail, 1 skip (real-Windows readback) | +| full `bun run test` | exit 1 twice (196 and 116 failures); every failure classified below as pre-existing | + +## Classification of full-suite failures + +| Failing files | Evidence | Class | +|---|---|---| +| tests/claude-integration/{claude-picker-runtime, claude-picker-ca, claude-management-api, claude-models-discovery, claude-picker-recovery} | Pattern: one server-starting test hangs to its timeout, bun kills a dangling child, then the rest of that worker fails with `SpendLedgerOwnerError` (SPEND_LEDGER_OWNER_HOME_CONFLICT). The whole directory (64 files) at head: 1190 pass, 0 fail. The same directory at base: 7 fail with the identical hang + SpendLedgerOwnerError cascade in claude-models-discovery. The five files alone at head: 129/129 pass. | pre-existing flake (suite-level hang cascade), not introduced by the union | +| tests/service/shutdown-launcher.test.ts (3 cases, 20s) | Fails identically alone at head and alone at base; leaves orphan `--ocx-internal-launch-proof` proxies (cleaned up after each run). Also in the first base full run. | pre-existing, environment | +| tests/cli/cli-headless-parity.test.ts (2) | Passes alone at head (91 tests with shutdown-launcher: only the 3 launcher cases fail). | load flake | +| Base-only: CLI subcommand help x4, ocx models x2 (timeouts) | Did not recur at head. | load flake | + +No new file was added to the failing set by the union, and no failure reproduces at head in +isolation while passing at base. The union's own code paths have no new child-process spawns +(`git diff 37ad7e771b..441aeb179e -- src` adds only one bounded `fetch` in +src/integrations/cursor-local-installer.ts). + +Follow-up (outside this unit): the claude-integration hang cascade and the launcher orphan cases are +worth their own issue; they are reproducible on a loaded macOS host at dev. + +## Independent review + +Kimi reviewer 01a0eb66 reviewed the union diff (shared files between PRs, i18n, test-layout rosters, +skill surface, structure docs, F1-F4): PASS; one Low finding rebutted in 021. + +Cross-platform coverage (Linux, Windows, macOS shards) comes from the single CI run in wp3. + diff --git a/devlog/_fin/260929_release_2_71_0/030_land.md b/devlog/_fin/260929_release_2_71_0/030_land.md new file mode 100644 index 0000000000..47b4901222 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/030_land.md @@ -0,0 +1,22 @@ +# 030 Land (wp3): one push, exact-head CI, admin merge + +1. `git push -u origin codex/release-2-71-0` (the only push of integration content). +2. Upload GUI screenshots to `pr-assets` (the #6094 request-pacing panel and the #5905 Cursor + installer notice, taken from the source PR bodies' images) and link them by commit SHA. +3. `gh pr create --base dev --head codex/release-2-71-0` with the repository template (Summary, + Verification, Checklist), one bullet per source PR with its landing commit, `Closes #6208`, and + `Co-authored-by` lines for every source author (see 010 table). +4. Wait for exact-head results on the PR head: Cross-platform CI (all test shards, windows 1-9, + macOS, docker smoke, npm-global smokes), Service lifecycle when triggered, enforce-target, + hygiene, privacy gate, structure. Missing, pending, skipped-by-failure, cancelled or older-head + results are not a pass. Fix any failure with a new commit on the branch (new exact head). +5. `gh pr merge <n> --admin --merge --match-head-commit <head>` (owner-authorized maintainer + integration; record the decision in the PR). +6. Close each source PR with a comment naming its landing commit on dev; for #6209 note that + #6208 closes manually since dev is not the default branch. +7. Record dev tip SHA after merge for 040. + +Rollback: if a defect from the union surfaces after the merge and before promotion, fix forward +on dev with a new PR; if it cannot be fixed quickly, revert the offending land commit on dev +(each source PR is one commit, so `git revert <land sha>` isolates it) and re-cut the candidate. +Nothing is promoted until the candidate is green. diff --git a/devlog/_fin/260929_release_2_71_0/031_wp3_execution.md b/devlog/_fin/260929_release_2_71_0/031_wp3_execution.md new file mode 100644 index 0000000000..cd1fb7ed8e --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/031_wp3_execution.md @@ -0,0 +1,69 @@ +# 031 wp3 execution: push, PR, final CI, merge + +Previous D (wp2): all local gates green on 441aeb179e; full-suite failures proven pre-existing +(022). Direction unchanged: push once and let the cross-platform CI be the last gate. + +Branch head at push = 441aeb179e plus devlog-only commits (021 audit, 022 results, this file). +The devlog commits do not change code; CI runs on the pushed head, which is the head that merges. + +## 1. Screenshots to pr-assets (no working-tree change) + +```sh +IDX=$(mktemp /tmp/rt2710-assets.XXXXXX) +GIT_INDEX_FILE=$IDX git read-tree rt/pr-assets # rt/pr-assets = origin pr-assets ea1749154d +for f in pr-6094-provider-request-pacing.png:request-pacing.png pr-5905-2.bin:cursor-installer.png; do + src=${f%%:*}; dst=${f##*:} + blob=$(git hash-object -w .tmp/rt2710/shots/$src) + GIT_INDEX_FILE=$IDX git update-index --add --cacheinfo 100644,$blob,260929-release-2-71-0/$dst +done +tree=$(GIT_INDEX_FILE=$IDX git write-tree) +commit=$(git commit-tree $tree -p rt/pr-assets -m "assets: 2.71.0 integration screenshots (#6094, #5905)") +git push origin $commit:refs/heads/pr-assets # fast-forward only; fails if pr-assets moved +``` + +The images are the source PRs' own screenshots (#6094 by the author's run, #5905 second image); +the integrated code for those files is byte-identical to the PR heads (patch-id equality, 011). + +## 2. Push and open the PR + +`git push -u origin codex/release-2-71-0`, then `gh pr create --base dev --title +"chore(release): integrate six reviewed PRs for 2.71.0" --body-file .tmp/rt2710/pr-body.md`. +Body: template sections (Summary, Verification, Checklist); one bullet per source PR with author, +land commit and what it changes; the four fix commits; the two screenshots embedded; local +verification summary pointing to 022; `Closes #6208`; one `Co-authored-by` line per identity in the +010 table. + +## 3. Final CI (the only cross-platform run) + +Wait on the PR head SHA. Required: every check run on that SHA completes with success (skipped only +where the workflow's path filter or matrix helper skips by design, as on the source PRs). Cross- +platform CI must show all Linux test shards, windows 1-9, macOS, docker smoke, npm-global smokes; +enforce-target, hygiene, privacy gate, structure, react-doctor (runs on every PR, fails on +warnings) and Service lifecycle must pass. A failure is fixed with a new commit and +the wait restarts on the new head; a flake classified with evidence may be rerun once +(`gh run rerun --failed`). + +## 4. Merge and close + +Before merging: `scripts/ci/assert-mergeable-review.sh --maintainer-integration <n> lidge-jun/opencodex` +(MAINTAINERS.md change log 2026-09-06). It checks the actor against the dev roster and live +permissions and binds its snapshot to the current head and base; it is not CI proof. Then post a +PR comment recording the maintainer-integration decision (owner-authorized, 2026-09-29), the exact +head SHA and the CI run IDs that passed on it. + +`gh pr merge <n> --admin --merge --match-head-commit <head>` (merge commit keeps the six authored +commits). Then for each source PR: `gh pr comment <src> --body "Landed on dev in <land sha> through +#<n> (2.71.0 integration). Thanks!"` and `gh pr close <src>`. Close #6208 with a comment naming the +#6209 land commit (dev is not the default branch, so Closes does not fire). Record the dev merge SHA. + +D10 (architect, wp3): a) run the maintainer-integration helper and record the decision (above); +b) Service lifecycle must actually run and pass, because src/cli/index.ts is in its pull_request +paths (service-lifecycle.yml:15); c) `--merge` is allowed on dev (MAINTAINERS.md:204-205, no linear +history rule). ci.yml runs the full matrix on pull_request (ci.yml:6) and not on push to dev +(ci.yml:32-33), so the PR run is the only cross-platform evidence before promotion. Accepted. + +Audit (Kimi 01a0eb97): GO-WITH-FIXES, 1 blocker folded (react-doctor added to required checks), +`mktemp -u` replaced, line anchors corrected. The reviewer ran the pr-assets plumbing in a scratch +repo and confirmed it touches only refs/heads/pr-assets; confirmed the maintainer-integration helper +passes for the owner on a self-authored PR; confirmed the planned body satisfies pr-quality and the +carry gate (trailers use each author's git identity from the PR commits). diff --git a/devlog/_fin/260929_release_2_71_0/032_wp3_results.md b/devlog/_fin/260929_release_2_71_0/032_wp3_results.md new file mode 100644 index 0000000000..1affb33855 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/032_wp3_results.md @@ -0,0 +1,18 @@ +# 032 wp3 results: landed on dev + +| Step | Evidence | +|---|---| +| Screenshots | pr-assets 947e869c8b (`260929-release-2-71-0/request-pacing.png`, `cursor-installer.png`), fast-forward from ea1749154d | +| PR | #6224 `chore(release): integrate six reviewed PRs for 2.71.0`, head dcfbb2f708 | +| PR-event CI | Cross-platform CI 36525839698 success; Service lifecycle 36525839767 success; React Doctor 36525839694, PR hygiene, Enforce PR target branch success | +| Full-lane CI | Cross-platform CI workflow_dispatch 36526719414 (lane all) success: 39 jobs success, 1 skipped; includes windows 1/9-9/9 (the pull_request event skips the Windows matrix, ci.yml:835-839) and macos control | +| Head totals | 80 check-runs: 74 success, 6 skipped by design, 0 failed | +| Policy | `scripts/ci/assert-mergeable-review.sh --maintainer-integration 6224`: OK; decision recorded in PR comment 5884552184 | +| Merge | `gh pr merge 6224 --admin --merge --match-head-commit dcfbb2f708` -> dev d161c0e83e; all six land commits and four fixes are ancestors of dev | +| Source PRs | #6206 #6201 #6209 #6094 #5905 #6198 closed with a comment naming the land commit; issue #6208 closed with the #6209 commit | + +Note: windows 8/9 passed in 36526719414, including the 32-profile transaction case that failed +the dev tip in 36499924172, now under BULK_DURABLE_IO_BUDGET_MS. + +Candidate for 040: dev d161c0e83e (version sources 2.71.0). + diff --git a/devlog/_fin/260929_release_2_71_0/040_release.md b/devlog/_fin/260929_release_2_71_0/040_release.md new file mode 100644 index 0000000000..174aa21149 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/040_release.md @@ -0,0 +1,30 @@ +# 040 Release (wp4): 2.71.0 + +Same procedure as devlog/_fin/260929_sonnet_5_5_catalog/040_plan_release.md. + +Candidate = dev tip after the 030 merge (tree has version sources 2.71.0). The push-event CI run on +that merge commit is informative; the gating runs are on the promotion SHAs. + +1. Pre-move dev: `gh workflow run dev-version-bump.yml --ref main -f intended-version=2.71.0 -f mode=pre-move`; + the opened PR must change only the four version sources to 2.72.0 (package.json, + desktop/src-tauri/Cargo.toml, Cargo.lock, tauri.conf.json); merge it with --admin --squash + --match-head-commit (as #6213). +2. Preview: branch `codex/promote-preview-2-71-0` from the candidate, `git merge -s ours origin/preview`, + `bun scripts/release-version-sources.ts sync 2.71.0-preview.20260929`, commit, PR to preview, + merge commit. Wait for push-event Cross-platform CI and Service lifecycle success on the + preview head. Dispatch `gh workflow run release.yml --ref preview -f version=2.71.0-preview.20260929 + -f tag=preview -f dry-run=false -f expected-sha=<preview head>`. +3. Stable: branch `codex/promote-main-2-71-0` from the candidate, `git merge -s ours origin/main` + (tree equals candidate), PR to main, merge commit. Wait for push-event CI + Service lifecycle on + the main head. Dispatch release.yml on main with version=2.71.0, tag=latest, dry-run=false, + expected-sha=<main head>. +4. Verify: `npm view @bitkyc08/opencodex dist-tags --json` shows latest=2.71.0 and + preview=2.71.0-preview.20260929; `gh release view v2.71.0` (not prerelease, full asset set) and + `v2.71.0-preview.20260929` (prerelease); latest.json reports 2.71.0 with signed platforms. +5. Write 090_outcome.md with run IDs and SHAs, move the unit to devlog/_fin/, land on dev through a + docs PR. + +Guards: verify preview tree differs from candidate only in the four version sources and main tree +equals the candidate. A failing promotion run is fixed through dev and re-promoted. If a release +run fails after npm acknowledged publication, re-dispatch with resume-after-npm-publish=true and the +same expected-sha; never republish. diff --git a/devlog/_fin/260929_release_2_71_0/041_wp4_execution.md b/devlog/_fin/260929_release_2_71_0/041_wp4_execution.md new file mode 100644 index 0000000000..f1db2d5025 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/041_wp4_execution.md @@ -0,0 +1,59 @@ +# 041 wp4 execution: release 2.71.0 + +Previous D (wp3): #6224 merged, dev d161c0e83e; final CI green including windows 1-9. Direction +unchanged: release that candidate exactly as 040 (same shape as 2.70.0: #6211/#6212/#6213). + +Revalidated at wp4 P: dev d161c0e83e (package 2.71.0), preview aa3a8dda16 (2.70.0-preview.20260929), +main 53834ff47b (2.70.0); npm latest=2.70.0, preview=2.70.0-preview.20260929. Candidate C = +d161c0e83ea8a88027b2cef28c125ff3c1f29f8a. + +## Order and parallelism + +1. Pre-move dispatch (runs on GitHub, ~1 min): + `gh workflow run dev-version-bump.yml -R lidge-jun/opencodex --ref main -f intended-version=2.71.0 -f mode=pre-move` + -> PR `codex/dev-version-2.72.0`. Accept only if its files are exactly the four version sources and + each reads 2.72.0. Its PR CI runs while steps 2-3 proceed; merge it (`--admin --squash + --match-head-commit`, as #6213) once its Cross-platform CI and gates are green. It is merged + before any release dispatch; the stable dispatch strictly requires it (release-preflight.sh + assert-ahead), and doing it first for both keeps one order. +2. Preview branch (managed worktree, clean tree): + ```sh + git switch -c codex/promote-preview-2.71.0 d161c0e83ea8a88027b2cef28c125ff3c1f29f8a + git merge -s ours refs/remotes/canon/preview -m "Merge preview into 2.71.0-preview.20260929 promotion" + bun scripts/release-version-sources.ts sync 2.71.0-preview.20260929 + git commit -am "chore(release): 2.71.0-preview.20260929" + git diff --name-only d161c0e83e HEAD # must list exactly the four version sources + bun scripts/release-version-sources.ts # check mode: all sources 2.71.0-preview.20260929 + ``` + Push, PR to preview titled "release: 2.71.0-preview.20260929", body as #6211 (plus the two + pr-assets screenshots because the union touches gui/), merge with `--admin --merge + --match-head-commit`. +3. Main branch: + ```sh + git switch -c codex/promote-main-2.71.0 d161c0e83ea8a88027b2cef28c125ff3c1f29f8a + git merge -s ours refs/remotes/canon/main -m "Merge main into 2.71.0 promotion" + git diff --quiet d161c0e83e HEAD # tree equals the candidate + ``` + Push, PR to main titled "release: 2.71.0", body as #6212 plus screenshots, merge the same way. +4. Gate per promotion SHA (the merge commits on preview and main): push-event Cross-platform CI and + Service lifecycle both `success` (release.yml checks both). A red run is diagnosed; a flake with + evidence may be rerun once; a real defect is fixed on dev and re-promoted. +5. Dispatch, preview first: + `gh workflow run release.yml -R lidge-jun/opencodex --ref preview -f version=2.71.0-preview.20260929 -f tag=preview -f dry-run=false -f expected-sha=<preview merge sha>` + then after it succeeds: + `gh workflow run release.yml -R lidge-jun/opencodex --ref main -f version=2.71.0 -f tag=latest -f dry-run=false -f expected-sha=<main merge sha>` + On a failure after npm acknowledged publication: same inputs plus `-f resume-after-npm-publish=true`. +6. Verify: `npm view @bitkyc08/opencodex dist-tags --json` (latest=2.71.0, preview=2.71.0-preview.20260929), + `npm view @bitkyc08/opencodex@2.71.0 gitHead` = main merge sha, `gh release view v2.71.0` (not + prerelease, asset count as v2.70.0 = 25), `gh release view v2.71.0-preview.20260929` (prerelease), + latest.json from the v2.71.0 release reports 2.71.0 with signed platforms. +7. Outcome: 090_outcome.md, move the unit to devlog/_fin/260929_release_2_71_0, docs PR to dev from + a branch on the pre-moved dev tip; merge after its gates. + +Merging promotion PRs is owner-authorized in this session ("배포까지 완료"). enforce-target flags +promotion PRs as wrong base by design (2.70.0 precedent); that check is not required for preview/main. + +D11 (architect, wp4, ALIGNED): release.yml:847-867 accepts only a push-event ci.yml run on the exact +SHA (a PR run does not count); package.json is in both ci.yml push paths (:51) and +service-lifecycle.yml push paths (:19), so both gating runs fire on each promotion merge commit; +release-preflight.sh enforces dev strictly ahead, so the pre-move PR merges first. diff --git a/devlog/_fin/260929_release_2_71_0/090_outcome.md b/devlog/_fin/260929_release_2_71_0/090_outcome.md new file mode 100644 index 0000000000..f3bd2a59d2 --- /dev/null +++ b/devlog/_fin/260929_release_2_71_0/090_outcome.md @@ -0,0 +1,38 @@ +# 090 Outcome: 2.71.0 released + +Published 2026-09-29 (KST). GitHub releases `v2.71.0` and `v2.71.0-preview.20260929` exist with 25 +assets each; npm accepted both publications with signed provenance. + +Six reviewed PRs landed on dev with author credit, the Windows timeout that had turned the dev tip +red was fixed, and the whole set was released as 2.71.0 and 2.71.0-preview.20260929. + +| Step | Evidence | +|---|---| +| Integration | #6224 merged to dev as d161c0e83e (merge commit). Land commits: df951bc2ee #6206 (Ingwannu), 6d7af977a4 #6201 (luvs01), 21ccf35dbe #6209 (Jio Kim), e3fcf3e43c #6094 (Brad Hallett), 2b0ac07170 #5905 (halysondev), febc16d2a4 #6198 (luvs01); follow-ups 9155880950, 7b875559dd, 917db156de, 441aeb179e. Details in 032. | +| Local regression | 022: all gates exit 0; full-suite failures proven pre-existing at dev 37ad7e771b. | +| Final CI | #6224 head dcfbb2f708: Cross-platform CI pull_request 36525839698, workflow_dispatch lane=all 36526719414 (windows 1-9), Service lifecycle 36525839767, all success; 0 failed check-runs. | +| dev pre-move | Dev version bump run 36529404555 opened #6226; its PR checks were approval-gated (bot-authored, as #6213), so the version tests and typecheck ran locally on its head 316750815c (209 pass) before `--admin --squash`; dev 69ce312097 reads 2.72.0. | +| Preview | #6227 merged as fd346e2489 (tree = candidate + four version sources at 2.71.0-preview.20260929). Push Cross-platform CI 36529530341 and Service lifecycle 36529530333 success. Release run 36531705017 success; npm `preview=2.71.0-preview.20260929`, gitHead fd346e2489, provenance attestation present. | +| Stable | #6228 merged as 8a005dd98f (tree equal to the candidate). Push Cross-platform CI 36529537253 and Service lifecycle 36529537201 success. Release run 36533478070 success; npm publish acknowledged (`+ @bitkyc08/opencodex@2.71.0`, sigstore log 2995422874); GitHub release `v2.71.0` not prerelease, 25 assets; latest.json reports 2.71.0 with 5 signed platforms. | +| Source PRs | #6206 #6201 #6209 #6094 #5905 #6198 closed with landing comments; issue #6208 closed. | + +## Notes + +- The pull_request event skips the Windows matrix (ci.yml:835-839); a workflow_dispatch run on the + integration head covered windows 1-9 before merge. +- The promotion PRs' own pull_request CI runs (36529486815, 36529483354) were cancelled to free + runners after both promotions merged; the push runs on the same trees gate the release. +- The registry smoke step warned on both releases because npm took a few minutes to serve the new + versions; no republish was attempted. +- The installed proxy, service and desktop app on this machine were not updated. + +## What did not improve / open follow-ups + +- Local full `bun run test` on a loaded macOS host is not a clean signal: a claude-integration + server test can hang and cascade into `SpendLedgerOwnerError` for the rest of its worker, and + `tests/service/shutdown-launcher.test.ts` fails and leaves orphan proxies. Both reproduce at dev + 37ad7e771b; they deserve their own issue. +- Optional hardening noted by reviewers and not built here: a response-size cap and redirect policy + for the Cursor manifest fetch (#5905), strict single-[3] TBS parsing in picker-ca (#6201), extra + subprocess tests for #6209, localized copies of the cli.md priority note. + diff --git a/devlog/_fin/260929_sonnet_5_5_catalog/010_plan.md b/devlog/_fin/260929_sonnet_5_5_catalog/010_plan.md new file mode 100644 index 0000000000..9a8eb6c054 --- /dev/null +++ b/devlog/_fin/260929_sonnet_5_5_catalog/010_plan.md @@ -0,0 +1,123 @@ +# 260929 Claude Sonnet 5.5 catalog, pricing and adapter contract + +## Problem + +Anthropic released Claude Sonnet 5.5 (`claude-sonnet-5-5`) on 2026-09-28. Live Anthropic discovery on +the running proxy already lists `anthropic/claude-sonnet-5-5` (1M context) but with no +`max_output_tokens`, `supports_reasoning: false`, no effort ladder and no price, because nothing in +the static catalog knows the id. Every other provider that carries `claude-sonnet-5` has no row at all. + +The model also changes the request contract. Sonnet 5 accepts `thinking: {type: "disabled"}`, and the +Anthropic adapter sends exactly that for reasoning `none` on every Sonnet >= 5.0. Sonnet 5.5 rejects it +with a 400 and adds `thinking: {type: "between_tools"}` as its lowest setting. It also rejects forced +`tool_choice` (`any`/`tool`) like Opus 5.5, and non-default temperature/top_p/top_k. + +## Evidence (collected 2026-09-29) + +| Source | Surface | Facts | +|---|---|---| +| platform.claude.com/docs/en/about-claude/pricing | Aside exec | Sonnet 5.5: $2 in, $2.50 5m write, $4 1h write, $0.20 hit (standard 0.1x), $10 out; batch $1/$5; no fast mode; no long-context premium; `inference_geo: us` 1.1x | +| platform.claude.com/docs/en/models/overview | Aside exec | id and alias `claude-sonnet-5-5`; Bedrock `anthropic.claude-sonnet-5-5`; Google Cloud / Foundry `claude-sonnet-5-5`; 1M context; 128K output; adaptive thinking; default effort high | +| .../models/sonnet-5-5/whats-new-sonnet-5-5 and migration-guide | Aside exec | `thinking.type` accepts only `adaptive` and `between_tools`; `disabled` and `enabled` 400; `between_tools` 400 at xhigh/max; forced tool choice 400; non-default sampling params 400 | +| .../build-with-claude/effort | Aside exec | low, medium, high, xhigh, max; default high | +| .../build-with-claude/fast-mode | Aside exec, kimi | fast mode is Opus-only; no Sonnet 5.5 fast anywhere | +| models.dev/api.json | curl, kimi | anthropic, azure, vertex, kilo, openrouter, vercel, bedrock `global.anthropic.claude-sonnet-5-5`; all 1M / 128K, 2 / 10 / 0.2 / 2.5 | +| openrouter.ai/api/v1/models | curl | `anthropic/claude-sonnet-5.5` 1M / 128K, 2 / 10 / 0.2 / 2.5 | +| ai-gateway.vercel.sh/v1/models | curl | `anthropic/claude-sonnet-5.5` 1M / 128K, 2 / 10 / 0.2 / 2.5; no -fast | +| api.kilo.ai gateway models | kimi | `anthropic/claude-sonnet-5.5` 2 / 10 / 0.2 / 2.5 | +| api.venice.ai/api/v1/models | kimi | `claude-sonnet-5-5` 1M / 128K, 3.75 / 18.75 / 0.375 / 4.6875 | +| cursor.com/docs/models/claude-sonnet-5-5 | kimi | id `claude-sonnet-5-5`, 200K default / 1M max, thinking supported, 2 / 2.5 / 0.2 / 10; no fast; not yet in the live Cursor roster | +| github.blog changelog 2026-09-28, docs models-and-pricing | kimi | Copilot GA; $2 / $0.20 cached / $2.50 write / $10; 1M context; no API id published | +| docs.devin.ai/desktop/models, running proxy devin/* | kimi | no Sonnet 5.5 | +| kiro.dev/docs/models, changelog | kimi | no Sonnet 5.5 | +| api.commandcode.ai/provider/v1/models | kimi | no Sonnet 5.5 | +| opencode.ai/zen/v1/models, zen/go | kimi | no Sonnet 5.5 (go has no Claude at all) | +| api.opper.ai/v3/models | kimi | no Sonnet 5.5 | +| zenmux.ai/api/v1/models, Cloudflare catalog | kimi | no Sonnet 5.5 | +| installed Claude Code 2.1.283 binary | strings | `sonnet` alias still resolves to `claude-sonnet-5`; no `claude-sonnet-5-5` string | + +Raw captures: `.tmp/sonnet55/` (scratch, not committed). + +## Provider classification + +| Provider | Decision | Id | Numbers | +|---|---|---|---| +| anthropic, anthropic-apikey | ADD seed, context, snapshot row, overlays (verified) | `claude-sonnet-5-5` | 2 / 10 / 0.2 / 2.5, 1M / 128K | +| amazon-bedrock | ADD 6 snapshot rows; `global.` published, `anthropic.` from Anthropic docs, us/eu/jp/au preemptive | `*.anthropic.claude-sonnet-5-5` | base 2 / 10 / 0.2 / 2.5; regional 1.1x = 2.2 / 11 / 0.22 / 2.75 (same rule as Opus 5.5 rows) | +| openrouter, vercel-ai-gateway, kilo | ADD snapshot rows (published) | `anthropic/claude-sonnet-5.5` | 2 / 10 / 0.2 / 2.5 | +| venice | ADD snapshot row (published) | `claude-sonnet-5-5` | 3.75 / 18.75 / 0.375 / 4.6875 | +| github-copilot | ADD snapshot row (published price; hyphen id convention of its Sonnet 5 / Opus 5.5 rows) | `claude-sonnet-5-5` | 2 / 10 / 0.2 / 2.5 | +| cursor | ADD capability, effort tiers, picker family, price overlay (published price; roster shape preemptive) | `claude-sonnet-5-5` | 2 / 10 / 0.2 / 2.5; 1M window | +| devin, devin-cli | ADD preemptive seed, context and derived overlays | `claude-sonnet-5-5` | 1M; Anthropic list price | +| kiro | ADD preemptive model and 1M context (dot spelling like its other rows) | `claude-sonnet-5.5` | price via vendor fallback | +| opencode-zen | ADD preemptive snapshot row | `claude-sonnet-5-5` | 2 / 10 / 0.2 / 2.5 | +| zenmux | ADD preemptive snapshot row (its dot convention) | `anthropic/claude-sonnet-5.5` | 2 / 10 / 0.2 / 2.5 (its Sonnet 5 row shape) | +| cloudflare-ai-gateway | ADD preemptive snapshot row (its hyphen convention) | `anthropic/claude-sonnet-5-5` | 2 / 10 / 0.2 / 2.5 | +| Claude Desktop picker suggestions | ADD `claude-sonnet-5-5` (suggestion list only) | | | +| Command Code, Opper | NO static site: neither seeds `claude-sonnet-5` today; live discovery owns the roster and price resolves through the Anthropic vendor row | | | +| opencode-go | NO: carries no Claude model | | | +| any fast tier | NO: Anthropic has no Sonnet fast mode | | | +| Claude Code native tier map, web-search / vision sidecar defaults | UNCHANGED: Claude Code 2.1.283 still sends `claude-sonnet-5`; the sidecars send `thinking: disabled`, which Sonnet 5.5 rejects | | | + +## Diff-level plan (wp2) + +1. `src/adapters/anthropic.ts` + - `supportsExplicitThinkingDisable`: Sonnet 5.0 <= v < 5.5 only (5.5 rejects `disabled`). + - New `usesBetweenToolsFloor`: Sonnet >= 5.5. Reasoning `none` sends `thinking: {type: "between_tools"}` + with no effort (the API default high is inside the accepted low..high range) and drops temperature/top_p. + - `rejectsForcedToolChoice`: Opus 5.5 and Sonnet >= 5.5. + - `rejectsSamplingParameters`: Sonnet >= 5.5 drops temperature/top_p on every path. +2. `src/usage/expected-prices.ts`: `CLAUDE_SONNET_55` = 2 / 10 / 0.2 / 2.5; overlays anthropic, + anthropic-apikey (verified), cursor (verified, Cursor model page), devin and devin-cli + (verified-derived, preemptive). +3. `scripts/model-metadata.source.json` rows listed above, cloned from each provider's claude-sonnet-5 + row with the new id/name/cost; regenerate `src/generated/model-metadata.ts`. +4. `src/providers/registry/model-seeds.ts`: `claude-sonnet-5-5` before `claude-sonnet-5` in + `ANTHROPIC_MODELS`, 1M in the context map, provenance comment. +5. Devin: `entries-core.ts` seed list and `DEVIN_MODEL_CONTEXT_WINDOWS`. +6. Kiro: `KIRO_MODELS` + `KIRO_MODEL_CONTEXT_WINDOWS` (`claude-sonnet-5.5`). +7. Cursor: `CURSOR_CAPABILITIES["claude-sonnet-5-5"]` shaped like Opus 5.5 (flat effort ids, regular + FULL ladder, no fast/thinking variant until the live roster is measured); effort-map tier row; + `models-capabilities.ts` picker family regex. +8. `src/claude/intercept/model-bindings.ts` suggestion list. +9. `structure/providers/chat-compat.md`: Sonnet 5.5 thinking floor, forced tool choice and sampling. +10. Tests next to existing ones, within ratchet caps: adapter wire shape (none -> between_tools, forced + choice -> auto, sampling stripped, Sonnet 5 unchanged), usage-cost resolver across providers, + kiro/cursor rosters if pinned. + +## Verification + +- `bun run generate:model-metadata`, model-metadata sync test +- focused: tests/adapters/anthropic, tests/usage, tests/providers/cursor, tests/providers/kiro, + devin tests, registry parity; `bun run test:changed` +- `bun run typecheck`, file-size ratchet + test-layout, `bun run structure:check`, `bun run privacy:scan` +- Live: adapter-built requests against api.anthropic.com for Sonnet 5.5 with reasoning none, medium, + required tool choice; expect 200 after the fix (token read in memory, never printed). +- Fresh-process resolver probe for every provider row. + +## Bounds + +Write scope: files above plus this unit. PR to `dev`, exact-head CI, merge, then release via +`scripts/release.ts` (user asked to deploy on 2026-09-29). + + +## Plan after audit (020) + +- Item 5 is two files: `src/providers/registry/entries-core.ts` devin seed list and + `DEVIN_MODEL_CONTEXT_WINDOWS` in `src/adapters/devin/live-models.ts`. `DEVIN_STATIC_MODELS` stays + untouched; live discovery owns the roster. +- Item 6: no `KIRO_NATIVE_EFFORT_FIELDS` entry. Kiro's `claude-sonnet-5` uses emulated effort and only + Opus has a measured native field. +- Item 7: no `models-capabilities.ts` regex change, matching what Opus 5.5 actually shipped. The Cursor + row is regular-only until the id appears in the live GetUsableModels roster; that listing is the + trigger to re-shape it (thinking variant or flat effort ids, whichever the roster shows). +- Sidecars: add a shared `anthropicThinkingOff(modelId)` in the adapter and use it in the web-search + and vision sidecars, so a sidecar pointed at Sonnet 5.5 sends `between_tools` instead of a + rejected `disabled`. Defaults stay `claude-sonnet-5`. +- Tests: `kiro-adapter.test.ts` gets only the in-line roster element (3 lines of ratchet headroom); + new assertions go to sibling files registered in `scripts/test-layout/layout.json` and + `tests/fixtures/test-layout-expected.json`. Update the usage-cost surface loops, the cursor catalog + flat-wire case, the devin context pin and add anthropic wire-shape cases. +- Residuals recorded, not fixed here: dotted Bedrock ids do not parse in `claudeFamilyVersion` (all + families), and the Messages-native passthrough forwards caller `thinking`/sampling verbatim. + diff --git a/devlog/_fin/260929_sonnet_5_5_catalog/020_audit.md b/devlog/_fin/260929_sonnet_5_5_catalog/020_audit.md new file mode 100644 index 0000000000..bf132fd5b6 --- /dev/null +++ b/devlog/_fin/260929_sonnet_5_5_catalog/020_audit.md @@ -0,0 +1,20 @@ +# 020 Audit (wp1): Sonnet 5.5 plan + +Auditor: kimi/kimi-for-coding (max), read-only, 2026-09-29. Verdict: NEAR-PASS. The four adapter +predicates, the between_tools-without-effort choice and the provider classification were confirmed +against source. Dispositions: + +| # | Finding | Disposition | +|---|---|---| +| 1 | `claudeFamilyVersion` matches only `^` or `/` before `claude-`, so dotted Bedrock ids (`us.anthropic.claude-sonnet-5-5`) never parse | RESIDUAL. Pre-existing for every family including Opus 5.5; widening the boundary changes wire shape for all Bedrock-style Claude ids and needs its own live evidence. Recorded in 030 | +| 2 | Messages-native passthrough forwards caller `thinking` and sampling fields verbatim | RESIDUAL. The passthrough contract is caller-owned bytes; a Messages client addressing Sonnet 5.5 gets Anthropic's own 400. Recorded | +| 3 | Web-search and vision sidecars hard-code `thinking: disabled`; a user who points them at Sonnet 5.5 gets a 400 | RESIDUAL. Defaults stay `claude-sonnet-5`. Recorded as an override hazard | +| 4 | Kiro native effort field for `claude-sonnet-5.5` | DECIDED: no entry. Kiro's `claude-sonnet-5` uses emulated effort today and only Opus has a measured native field | +| 5 | Cursor local picker family regex was promised for Opus 5.5 and never landed | DECIDED: skip for Sonnet 5.5 too; the live bundle table owns that prediction | +| 6 | claude-cli and native Anthropic catalog spread `ANTHROPIC_MODELS` | ACCEPTED: user-visible roster addition, intended | +| 7 | Devin context map lives in `src/adapters/devin/live-models.ts`; `DEVIN_STATIC_MODELS` stays untouched | FOLDED into wp2 diff | +| 8 | Cursor row shape is regular-only while the docs say thinking is supported | FOLDED: follow-up trigger is the model appearing in the live GetUsableModels roster; shape then follows the measured ids | +| 9 | `kiro-adapter.test.ts` has 3 lines of ratchet headroom | FOLDED: only the in-line roster element there; any new assertion goes in a sibling file registered in both layout maps | +| 10 | Pinned tests: kiro roster, usage-cost surface loops, cursor catalog flat-wire case, devin context pin, anthropic reasoning wire cases | FOLDED into wp2 test list | +| 11 | No docs-site locale diff needed (no Sonnet fast tier) | ACCEPTED | + diff --git a/devlog/_fin/260929_sonnet_5_5_catalog/030_done.md b/devlog/_fin/260929_sonnet_5_5_catalog/030_done.md new file mode 100644 index 0000000000..cb86797172 --- /dev/null +++ b/devlog/_fin/260929_sonnet_5_5_catalog/030_done.md @@ -0,0 +1,39 @@ +# 030 Done (wp2): Sonnet 5.5 catalog, pricing and adapter contract + +## Outcome + +- Snapshot rows (15), regenerated `src/generated/model-metadata.ts`: anthropic `claude-sonnet-5-5`; + Bedrock `anthropic.`, `global.` (2 / 10 / 0.2 / 2.5) and `us.`, `eu.`, `jp.`, `au.` (1.1x: + 2.2 / 11 / 0.22 / 2.75); openrouter, vercel-ai-gateway, kilo `anthropic/claude-sonnet-5.5`; venice + `claude-sonnet-5-5` (3.75 / 18.75 / 0.375 / 4.6875); github-copilot `claude-sonnet-5-5`; preemptive + opencode-zen `claude-sonnet-5-5`, zenmux `anthropic/claude-sonnet-5.5`, cloudflare-ai-gateway + `anthropic/claude-sonnet-5-5`. All 1M / 128K. +- Seeds: `ANTHROPIC_MODELS` and context map (also reaches claude-cli and the native Anthropic catalog), + Devin seed and context (preemptive), Kiro `claude-sonnet-5.5` (preemptive), Cursor capability and + effort tier (regular-only flat ids until the live roster lists it), Claude Desktop picker suggestion. +- Price overlays: anthropic, anthropic-apikey, cursor (verified); devin, devin-cli (verified-derived, + preemptive). Venice, Kiro, Command Code and Opper resolve through the Anthropic vendor row, as Opus + 5.5 already does. +- Adapter: new `src/adapters/anthropic-model-contract.ts` owns the per-family rules. Sonnet 5.5 + reasoning `none` sends `between_tools`; forced tool choice degrades to `auto`; temperature/top_p + are dropped. The web-search and vision sidecars use the same module for their thinking-off field. + +## Live evidence (2026-09-29, api.anthropic.com, OAuth, adapter-built bodies) + +| Request | Before | After | +|---|---:|---:| +| `claude-sonnet-5-5`, reasoning none | 400 (`thinking.type.disabled` not supported) | 200 (`between_tools`) | +| `claude-sonnet-5-5`, temperature 0.2 | 400 (`temperature` is deprecated for this model) | 200 (stripped) | +| `claude-sonnet-5-5`, tool choice required | 400 (`any` not supported) | 200 (`auto`) | +| `claude-sonnet-5-5`, effort medium and max | 200 | 200 | +| `claude-sonnet-5`, reasoning none (control) | 200 (`disabled`) | 200 (unchanged) | + +## Residuals + +- Dotted Bedrock ids (`us.anthropic.claude-sonnet-5-5`) do not parse as a Claude family, so the rules + above do not apply on that id shape; true for every family today. +- Messages-native passthrough forwards the caller's `thinking` and sampling fields unchanged. +- Rows marked preemptive (Devin, Kiro, opencode-zen, ZenMux, Cloudflare, Bedrock regional) are inert + until each provider lists the model; Kiro and Devin show the id in their static pickers, so a call can + fail upstream until they ship it. +- The Cursor row is re-shaped once the live GetUsableModels roster lists `claude-sonnet-5-5`. diff --git a/devlog/_fin/260929_sonnet_5_5_catalog/040_plan_release.md b/devlog/_fin/260929_sonnet_5_5_catalog/040_plan_release.md new file mode 100644 index 0000000000..df0f0ad5e1 --- /dev/null +++ b/devlog/_fin/260929_sonnet_5_5_catalog/040_plan_release.md @@ -0,0 +1,26 @@ +# 040 Plan (wp3): release 2.70.0 + +Candidate: `c34c4d20db` (dev after #6210, Sonnet 5.5). Version sources on the candidate read 2.70.0. +Cross-platform CI on the candidate: run 36474965294 (workflow_dispatch). Same procedure as +`devlog/_fin/260928_release_2_69_0/000_plan.md`; the user asked to deploy on 2026-09-29. + +1. Pre-move `dev`: dispatch `dev-version-bump.yml` from `main` with `intended-version=2.70.0`; the + opened PR must change only version sources to 2.71.0; merge it through the maintainer dev path. +2. Preview: branch from the candidate, `git merge -s ours origin/preview`, + `bun scripts/release-version-sources.ts sync 2.70.0-preview.20260929`, PR to `preview`, merge + commit. Dispatch `release.yml` on `preview` with `version=2.70.0-preview.20260929`, `tag=preview`, + `dry-run=false`, `expected-sha=<preview head>`. +3. Stable: branch from the candidate, `git merge -s ours origin/main` (tree equals the candidate), + PR to `main`, merge commit. Dispatch `release.yml` on `main` with `version=2.70.0`, `tag=latest`, + `dry-run=false`, `expected-sha=<main head>`. +4. Verify npm `latest=2.70.0` and `preview=2.70.0-preview.20260929`, both GitHub releases with the + full asset set and prerelease flags, and `latest.json` at 2.70.0 with five signed platforms. +5. Land the outcome on `dev`. + +Guards: each promotion SHA needs a successful push-event Cross-platform CI and Service lifecycle run +before its release dispatch (release.yml gates on both). Verify the preview tree differs from the +candidate only in the four version sources and the main tree equals the candidate. Do not weaken +release preflight, exact-SHA or CI gates; a failing run is fixed through `dev` and re-promoted. + +If a release run fails after npm publication was acknowledged, re-dispatch with the same version and +expected-sha plus `resume-after-npm-publish: true`; never republish. diff --git a/devlog/_fin/260929_sonnet_5_5_catalog/090_outcome_release.md b/devlog/_fin/260929_sonnet_5_5_catalog/090_outcome_release.md new file mode 100644 index 0000000000..ac719e5f8b --- /dev/null +++ b/devlog/_fin/260929_sonnet_5_5_catalog/090_outcome_release.md @@ -0,0 +1,18 @@ +# 090 Outcome (wp3): release 2.70.0 + +Published 2026-09-29 (KST). npm `latest=2.70.0`, `preview=2.70.0-preview.20260929`. + +| Step | Evidence | +|---|---| +| Change | #6210 (Claude Sonnet 5.5 catalog, pricing and request contract) admin-merged to dev as `c34c4d20db` at the owner's request, before its PR CI finished. | +| dev pre-move | Dev version bump run 36475209662 opened #6213; merged as `034e787164` (four version sources 2.70.0 -> 2.71.0). | +| Preview | #6211 merged as `aa3a8dda16` (tree = candidate + four version sources at `2.70.0-preview.20260929`). Push Cross-platform CI 36475459900 and Service lifecycle 36475459622 passed. Release run 36480301229 succeeded; GitHub release `v2.70.0-preview.20260929` is a prerelease with 25 assets. | +| Stable | #6212 merged as `53834ff47b` (tree equal to the candidate). Push Cross-platform CI 36475469901 and Service lifecycle 36475469997 passed. Release run 36480501878 succeeded; GitHub release `v2.70.0` is not a prerelease, 25 assets; `latest.json` reports 2.70.0 with five signed platforms. | +| npm | publish jobs succeeded 21:04Z (preview) and 21:13Z (stable); both dist-tags confirmed on the registry. | + +## Notes + +- The PR CI for #6210 (run 36474873578) and a redundant dispatch on the candidate (36474965294) were + cancelled to free runners once the promotion push runs were queued; the main promotion tree equals + the candidate, so its push CI covers the same tree. +- The installed proxy, service and desktop app on this machine were not updated. diff --git a/devlog/_fin/260929_tokenlab_sponsor/000_roadmap.md b/devlog/_fin/260929_tokenlab_sponsor/000_roadmap.md new file mode 100644 index 0000000000..3d14bac63c --- /dev/null +++ b/devlog/_fin/260929_tokenlab_sponsor/000_roadmap.md @@ -0,0 +1,13 @@ +# TokenLab — preset merge and third Standard sponsor + +TokenLab (TOKENLAB AI INC.) signed up as a Standard sponsor on 2026-09-29 (DocuSign envelope sent +21:08 KST; USD 1,200 / 3 months, term starts at the npm release that ships the preset). Two units: + +| Doc | Work phase | Outcome | +|-----|-----------|---------| +| [010](./010_merge_6221.md) | wp1 | #6221 (fork `hedging8563`, ordinary API-key preset) merged into `dev` on green exact-head CI | +| [020](./020_sponsor_pr.md) | wp2 | Separate maintainer PR: sponsor field, dashboard sponsor surface, README row #3, docs, assets | + +The sponsor PR stays unmerged for owner review. The Responses-first proposal in TokenLab's asset pack +(`integration-notes.md`) is out of scope for both units: it changes the upstream protocol and needs +its own evidence and PR. diff --git a/devlog/_fin/260929_tokenlab_sponsor/010_merge_6221.md b/devlog/_fin/260929_tokenlab_sponsor/010_merge_6221.md new file mode 100644 index 0000000000..789fa1aa08 --- /dev/null +++ b/devlog/_fin/260929_tokenlab_sponsor/010_merge_6221.md @@ -0,0 +1,37 @@ +# 010 — Merge #6221 (TokenLab API-key preset) + +## State at plan time + +- Head repo `hedging8563/opencodex`, branch `codex/tokenlab-provider`, draft, `maintainerCanModify`. +- Branch updated onto `dev` with `gh pr update-branch` → head `6a9e571202`. +- Fork runs `Cross-platform CI` (36566988054) and `React Doctor` (36566988033) approved by the maintainer. +- Admission gap recorded in the PR body: routing/resale authorization evidence. The owner's + sponsorship agreement with the operator (§2.2: sourcing and delivery mode are TokenLab's + responsibility, not audited or guaranteed by opencodex) is the maintainer decision that closes it; + no upstream endorsement claim is added. + +## Steps + +0. Head moved to `fb565b7c48`: the `dev` union failed `docs-provider-preset-counts` (16 assertions, + localized 99/82 counts). Recounted the anchored lines; the test passes locally 18/18. +1. Wait for every required check on `6a9e571202` to complete successfully. A red, skipped or + older-head result blocks merge; fix forward only if the failure is in this diff. +2. Mark the PR ready (`gh pr ready 6221`). +3. Post a maintainer-integration comment: owner instruction, exact head SHA, check list, and the + admission decision above. +4. Squash-merge with the admin bypass (`gh pr merge 6221 --squash --admin`); the squash keeps the + PR author as commit author. +5. Verify `dev` contains the squash commit and `src/providers/registry/entries-extended.ts` carries + the `tokenlab` entry. + +## Not in this unit + +Sponsor field, README, logos — see 020. + +## Outcome (2026-09-29) + +Merged as `f6cddd7d69` at 12:51 UTC from head `fb565b7c48`: 24 checks passed, 7 skipped by path +conditions (Cross-platform CI 36567400620, React Doctor). CodeRabbit's assertive review found nothing +actionable. The readiness gate re-drafted the PR on `ready`; the maintainer ticked the last two boxes +after that review and recorded it on the PR. A full local `bun run test` cannot run from a worktree +under `~/.codex`: the harness refuses to delete temporary directories inside the real Codex home. diff --git a/devlog/_fin/260929_tokenlab_sponsor/020_sponsor_pr.md b/devlog/_fin/260929_tokenlab_sponsor/020_sponsor_pr.md new file mode 100644 index 0000000000..0cbee8371c --- /dev/null +++ b/devlog/_fin/260929_tokenlab_sponsor/020_sponsor_pr.md @@ -0,0 +1,63 @@ +# 020 — TokenLab third Standard sponsor PR + +Branch `codex/tokenlab-sponsor`, started on the #6221 head and rebased onto `dev` after 010 lands. +Pattern: #3914 (OrcaRouter, mechanism) and #3915 (PackyCode, second sponsor). Assets come from +TokenLab's corrected pack (`TokenLab-OpenCodex-assets-corrected.zip`), artwork unchanged. + +## Diff + +- `src/providers/registry/entries-extended.ts` — the `tokenlab` entry gains + `sponsor: { tier: "standard", url: "https://tokenlab.sh/?utm_source=opencodex&utm_medium=readme" }` + with a comment naming SPONSORS.md and the signing date. No routing, default or discovery change. +- `assets/sponsors/tokenlab-light.png`, `assets/sponsors/tokenlab-dark.png` — the pack's 500×125 + lockups, unchanged (`-light` = for light page backgrounds, as in the pack). Listed in + `package.json` `files` beside the OrcaRouter/PackyCode logos. +- `README.md` — third row of the `sponsors:standard` table, in signing order after PackyCode. + Logo in `<picture>` with a `prefers-color-scheme: dark` source so the dark lockup shows in + GitHub dark mode; the `<img>` fallback is the light lockup (npm). Text: "Thanks to TokenLab for + sponsoring this project!", the sponsor's English blurb verbatim, then the picker / + `ocx provider add tokenlab` line. +- `readme/README.{ko,ja,zh-CN,zh-TW,ru,fr,tr}.md` — the same row, translated like the existing + rows, image paths `../assets/...`. Keep the README drift gate green. +- `gui/public/provider-icons/tokenlab.svg` — the pack's symbol mark; `gui/src/provider-icons.ts` + maps `tokenlab` to it, sets the display name, and applies dark-mode treatment like other + monochrome marks. +- `gui/src/components/provider-workspace/ProviderSponsor.tsx` — replace the hardcoded + OrcaRouter/PackyCode ternaries with a small brand table so a sponsor is one row; add TokenLab. + i18n keys `pws.sponsor.tokenlabTitle` / `pws.sponsor.tokenlabDescription` in every + `gui/src/i18n/*.ts` locale (the closed catalog must stay exhaustive). +- `docs-site/src/content/docs/guides/providers.md` — sponsor paragraph after PackyCode, and the + base-URL table row if #6221 has not already added it. Translated locales must not contradict it. +- Tests — `tests/providers/sponsor-presets.test.ts` stays generic; the GUI sponsor overview test + covers the TokenLab card. Respect the file-size ratchet: move rather than grow capped files. + +## Verification + +## Audit folds (reviewer, NEAR-PASS) + +- `desktop/src-tauri/src/provider_icons.rs`: `("tokenlab", "tokenlab.svg")` in `ALIASES`, the same paint + arm as `MASKED_PROVIDER_ICONS`, and `svg!("tokenlab.svg")` (`gui/tests/provider-icons-native.test.ts`). +- `readme/i18n-manifest.json`: refresh all seven `sourceSha256` values to the new LF-normalized README hash + (`docs-readme-translation-parity.test.ts`). +- `<picture>` is gate-compatible: only `src=`/`href=` are asset-tracked, so each locale's `<img src>` must be + `../assets/sponsors/tokenlab-light.png`; include the dark `<source>` too. +- Icon provenance row in `gui/public/provider-icons/README.md`; `structure/dashboard-and-usage.md` owns the + `ProviderSponsor` description — update it for the brand table. +- Docs: extend the existing TokenLab section in `guides/providers.md` with the sponsor link; do not add a + second paragraph. +- `SPONSORS.md`: fix the stale "picker follows registry order" and "translated READMEs carry one linking + line" sentences. +- Branch: after #6221 squashes, move only the sponsor commits onto `dev` + (`git rebase --onto origin/dev fb565b7c48`). + +## Verification (commands) + +Focused: sponsor-presets, provider-registry-parity, tokenlab-provider, README drift/translation +gates, GUI sponsor tests, `bun run typecheck`, `bun run lint:gui`, `bun run privacy:scan`, +`bun run structure:check`, `bun run build:gui`. Then `bun run test` or the documented focused +exception. Screenshot of the dashboard sponsor card and picker uploaded to the `pr-assets` branch +and linked by SHA in the PR description, never committed to the PR branch. + +## Out of scope + +Responses-first adapter and the `X-TokenLab-Delivery-Policy` header (pack `integration-notes.md`). diff --git a/devlog/_fin/260930_desktop_external_links/010_plan.md b/devlog/_fin/260930_desktop_external_links/010_plan.md new file mode 100644 index 0000000000..74d1389b40 --- /dev/null +++ b/devlog/_fin/260930_desktop_external_links/010_plan.md @@ -0,0 +1,69 @@ +# 260930 desktop external links — plan + +## Conclusion + +Links the loopback dashboard asks to open in a new window (OAuth "didn't open?" fallback, +device-code verification links, `window.open`) now leave the desktop app through one Rust +handler that hands http/https URLs to the OS default browser. Nothing new is granted over IPC. + +## Problem + +User report: pressing a login button in the desktop app often does not open the default browser. + +Two read-only gpt-6.1-sol lanes and direct reads of the pinned crates (tauri 2.11.6, wry 0.55.1, +tauri-plugin-opener 2.5.3) agree: + +1. The server-side launch (`/api/oauth/login` -> `openUrl`) is intact for browser flows, but its + result is discarded, and device-code flows (Copilot, Kimi, Nous, Meta Muse, Kiro, Codex device) + never launch server-side by design. Both depend on the GUI's `target="_blank"` link. +2. `tauri_plugin_opener::init()` injects a click listener that `preventDefault()`s every `_blank` + click and invokes `plugin:opener|open_url`. The dashboard is the remote origin + `http://127.0.0.1:*`; its capabilities (`dashboard-titlebar.json`, `dashboard-zoom.json`) grant + no opener permission, so the IPC is denied after the click was already cancelled: nothing opens. +3. No webview installs `on_new_window`. Without it wry drops `window.open` on WebView2 + (`SetHandled(true)`) and WebKitGTK (no `create` handler); WKWebView only reaches the browser + because its navigation policy sees the URL first. + +## Options considered + +- A. Grant `opener:allow-open-url` to the remote dashboard origin. Fixes anchors only, leaves + `window.open` broken on Windows/Linux, and widens the IPC surface of a remote origin. +- B (chosen). Disable the plugin's JS interceptor (`open_js_links_on_click(false)`) and install an + `on_new_window` handler on the main window and the tray popup that opens http/https in the default + browser and returns `NewWindowResponse::Deny`. One decision point in Rust, no new grant, covers + both anchors and `window.open` on all three platforms. + +## Diff-level plan + +- `desktop/src-tauri/src/window.rs`: extract `opens_in_default_browser(&Url)` (http/https only) + and `open_in_default_browser`; reuse it in `navigation_allowed`; add + `open_new_windows_in_default_browser<R>()` returning the handler; unit test for the scheme filter. +- `desktop/src-tauri/src/lib.rs`: build the opener plugin with `open_js_links_on_click(false)`; + add `.on_new_window(...)` to the main window builder. +- `desktop/src-tauri/src/popup.rs`: add `.on_new_window(...)` to the tray popup builder, and open + external http/https URLs in `popup_navigation_allowed` before refusing them. Audit round 1 (FAIL) + found that WKWebView consults this policy before it would create a window for a `_blank` link, so + a refusal without opening kept the popup's external links dead on macOS; round 2 passed. + +Bundled pages (`desktop/ui`) contain no `_blank` anchors or opener calls, so disabling the +interceptor removes nothing they relied on. `mailto:`/`tel:` were never reachable from the +dashboard (same denied IPC) and stay out of the external-open filter. + +## Verification + +Same steps as the CI `desktop` job: placeholder sidecar and resource files (gitignored), then +`cargo fmt --check`, `cargo clippy --all-targets -D warnings`, `cargo test` for +`desktop/src-tauri`. Exact-head PR CI is the merge gate. A packaged-app click test is not +available locally; the behavior claim rests on the pinned wry/opener sources cited above. + +## Outcome + +Implemented as planned plus the audit fold in `popup.rs`. Local proof on macOS arm64: +`cargo fmt --check`, `cargo clippy --all-targets -D warnings` and `cargo test` (190 passed, +including `only_web_addresses_are_handed_to_the_default_browser`) exit 0; the desktop, +release-contract and repo-hygiene Bun suites (24 files, 334 pass, 2 platform skips), +`privacy:scan` and `structure:check` pass. Windows and Linux behavior is source-reviewed +against wry 0.55.1 and covered by the hosted desktop CI job, not by a packaged click test. + +What this does not change: server-side `openUrl` still discards its launch result, and +device-code logins still do not auto-open a browser; both remain separate follow-ups. diff --git a/devlog/_fin/260930_release_2_72_0/000_plan.md b/devlog/_fin/260930_release_2_72_0/000_plan.md new file mode 100644 index 0000000000..149e4c57f1 --- /dev/null +++ b/devlog/_fin/260930_release_2_72_0/000_plan.md @@ -0,0 +1,17 @@ +# 2.72.0 — TokenLab release + +Owner request (2026-09-30): merge the TokenLab sponsor PR, verify for regressions, release, check +TokenLab's payment in the WORKS inbox through Aside, and have Aside email Vincent. + +Previous unit: `devlog/_fin/260929_tokenlab_sponsor/` merged #6221 (preset, `f6cddd7d69`) and opened +#6240 (sponsor placement + CLI pinning, head `21cddd35c9`, 32/32 PR checks green). Release shape is +2.71.0 (`devlog/_fin/260929_release_2_71_0/040_release.md`). + +| Doc | Work phase | Outcome | +|-----|-----------|---------| +| [010](./010_merge_6240.md) | wp2 | #6240 merged on full-lane CI (incl. windows 1–9) at its exact head | +| [020](./020_release.md) | wp3 | npm latest 2.72.0, preview 2.72.0-preview.20260930, GitHub releases, gitHead | +| [030](./030_payment_and_email.md) | wp4 | Payment/DocuSign status from WORKS; Vincent emailed by Aside exec; outcome recorded | + +Release content since v2.71.0: #6221 (TokenLab preset), #6240 (sponsor placement, CLI sponsor +pinning), devlog-only commits. diff --git a/devlog/_fin/260930_release_2_72_0/010_merge_6240.md b/devlog/_fin/260930_release_2_72_0/010_merge_6240.md new file mode 100644 index 0000000000..ec33545400 --- /dev/null +++ b/devlog/_fin/260930_release_2_72_0/010_merge_6240.md @@ -0,0 +1,14 @@ +# 010 — Regression gate and merge #6240 (wp2) + +1. Dispatch full-lane Cross-platform CI on the PR head: + `gh workflow run ci.yml -R lidge-jun/opencodex --ref codex/tokenlab-sponsor -f lane=all` + (the pull_request event skips windows 1–9 and macOS control; 2.71.0 used the same dispatch). + Required: every job success, windows 1/9–9/9 present, on the final head `2b9295fea6` (the run on `21cddd35c9` was cancelled when the final sponsor copy landed; see 030). +2. Local regression scope: the lanes' focused tests plus `test:changed`; the full local suite is + not runnable from a worktree under `~/.codex` (test home guard), so CI is the full-suite proof. +3. `scripts/ci/assert-mergeable-review.sh --maintainer-integration 6240 lidge-jun/opencodex`, a PR + comment recording owner authorization, exact head and run IDs: PR-event Cross-platform CI, + Service lifecycle (triggered by `desktop/**` and `package.json`), and the `lane=all` dispatch. +4. `gh pr merge 6240 --admin --squash --match-head-commit <head>`. C is that exact squash commit, + fixed before anything else lands on `dev`; assert `git show C:package.json` reads 2.72.0. + The `260929_tokenlab_sponsor` plan docs on the branch land inside the squash. diff --git a/devlog/_fin/260930_release_2_72_0/011_wp2_execution.md b/devlog/_fin/260930_release_2_72_0/011_wp2_execution.md new file mode 100644 index 0000000000..6105dfe65d --- /dev/null +++ b/devlog/_fin/260930_release_2_72_0/011_wp2_execution.md @@ -0,0 +1,14 @@ +# 011 wp2 execution: regression gate and merge + +| Step | Evidence | +|---|---| +| Final copy | #6240 took TokenLab's final blurbs and referral link at `2b9295fea6` (030 findings); the `lane=all` run on `21cddd35c9` was cancelled | +| PR-event CI on `2b9295fea6` | Cross-platform CI 36591071773, Service lifecycle 36591071699, React Doctor 36591071756: success; 32 pass, 5 path-skipped | +| Full-lane gate | Cross-platform CI `lane=all` 36591083341: attempt 1 failed windows 7/9 (`cli-connect-readiness` installed-root probe, exit null at 18 s) and windows 8/9 (`main quota policy at native admission`, 32 s cases); neither loads a changed module (`init`, `provider-runtime` are lazy CLI imports). One rerun (attempt 2): both shards and `ci` success | +| Review | Codex P2 and CodeRabbit sponsor-first-run fixed; CodeRabbit wording suggestion declined (verbatim sponsor copy, adapter chip visible, Responses-first pending) | +| Policy | `assert-mergeable-review.sh --maintainer-integration 6240`: OK; decision comment 5894282491 | +| Merge | `gh pr merge 6240 --admin --squash --match-head-commit 2b9295fea6` → dev `1cd9d25517` = C; `package.json` 2.72.0, version-sources check 2.72.0 passes | +| Pre-move | #6243 (four version sources 2.72.0 → 2.73.0), `maintainer-sponsored` after review, merged → dev `73289d46ae` (2.73.0) | + +Local regression scope: focused sponsor/registry/README/GUI suites and `test:changed`; a full local run +is not possible from a worktree under `~/.codex` (test home guard), so CI above is the full-suite proof. diff --git a/devlog/_fin/260930_release_2_72_0/020_release.md b/devlog/_fin/260930_release_2_72_0/020_release.md new file mode 100644 index 0000000000..98da49e340 --- /dev/null +++ b/devlog/_fin/260930_release_2_72_0/020_release.md @@ -0,0 +1,25 @@ +# 020 — Release 2.72.0 (wp3) + +Same procedure as `devlog/_fin/260929_release_2_71_0/041_wp4_execution.md`, with C = the #6240 +squash commit from 010 (never a later `dev` tip, which would carry 2.73.0). + +1. Pre-move: `gh workflow run dev-version-bump.yml -R lidge-jun/opencodex --ref main + -f intended-version=2.72.0 -f mode=pre-move` → PR moving exactly the four version sources to + 2.73.0; merge `--admin --squash --match-head-commit` after its CI. +2. Preview: branch `codex/promote-preview-2.72.0` from C, `git merge -s ours origin/preview`, + `bun scripts/release-version-sources.ts sync 2.72.0-preview.20260930`, commit; diff vs C must be + exactly the four version sources, and `bun scripts/release-version-sources.ts` check mode passes. + PR to preview with the #6240 pr-assets screenshots, `gh pr merge --admin --merge --match-head-commit`. +3. Main: branch `codex/promote-main-2.72.0` from C, `git merge -s ours origin/main`, tree equals C. + (`git diff --quiet C HEAD`). PR to main with the screenshots, merged the same way. enforce-target + flags promotion PRs as wrong base by design; it is not required on preview/main. +4. Gate each promotion SHA: push-event Cross-platform CI and Service lifecycle `success`. +5. Dispatch preview then stable: + `gh workflow run release.yml --ref preview -f version=2.72.0-preview.20260930 -f tag=preview + -f dry-run=false -f expected-sha=<preview sha>`, then `--ref main -f version=2.72.0 -f tag=latest + -f dry-run=false -f expected-sha=<main sha>`. After an npm-acknowledged failure, resume with + `-f resume-after-npm-publish=true`; never republish. +6. Verify dist-tags, `npm view @bitkyc08/opencodex@2.72.0 gitHead` = main sha, `gh release view v2.72.0` + (not prerelease, 25 assets as v2.71.0), preview release prerelease, latest.json 2.72.0 signed. + +The installed proxy/app on this machine is not updated (same as 2.70.0/2.71.0). diff --git a/devlog/_fin/260930_release_2_72_0/021_wp3_execution.md b/devlog/_fin/260930_release_2_72_0/021_wp3_execution.md new file mode 100644 index 0000000000..a268eb8e9e --- /dev/null +++ b/devlog/_fin/260930_release_2_72_0/021_wp3_execution.md @@ -0,0 +1,29 @@ +# 021 wp3 execution: promotion and release + +| Step | Evidence | +|---|---| +| Candidate | C = dev `1cd9d25517` (#6240 squash), version sources 2.72.0 | +| Pre-move | #6243 → dev `73289d46ae` (2.73.0), `maintainer-sponsored` after review | +| Preview promotion | `codex/promote-preview-2.72.0`: `-s ours` merge of origin/preview + sync to 2.72.0-preview.20260930 (`14ccfe1a1a`); diff vs C = four version sources; PR #6245 merged (merge commit) → preview `4f9e3f0afb` | +| Main promotion | `codex/promote-main-2.72.0`: `-s ours` merge of origin/main (`77cb00512f`), tree equals C; PR #6246 merged → main `5ab6d52b2a` | +| Push-event gates | preview: Cross-platform CI 36597831993, Service lifecycle 36597831950; main: Cross-platform CI 36597841262, Service lifecycle 36597841450 | + +Dispatches (after both gates of a SHA succeed), preview first: + +```sh +gh workflow run release.yml -R lidge-jun/opencodex --ref preview -f version=2.72.0-preview.20260930 -f tag=preview -f dry-run=false -f expected-sha=4f9e3f0afbbcf54a2b0421db8e962ec3d5682d5e +gh workflow run release.yml -R lidge-jun/opencodex --ref main -f version=2.72.0 -f tag=latest -f dry-run=false -f expected-sha=5ab6d52b2a4da722d398e4ab50a6c621ac3ce087 +``` + +## Results + +| Check | Evidence | +|---|---| +| Preview gates | Cross-platform CI 36597831993 success (push), Service lifecycle 36597831950 success | +| Main gates | Service lifecycle 36597841450 success; Cross-platform CI 36597841262 attempt 1 failed only `test 2/4` (batch 10/48 hit the 120 s process bound; the attribution sweep reported every file passing alone, "the timeout lives in multi-file process state"); one rerun, attempt 2 success | +| Preview release | release.yml 36602348988 success; npm `preview` = 2.72.0-preview.20260930, gitHead `4f9e3f0afb`, bins `ocx`/`opencodex` intact; GitHub release prerelease, 25 assets | +| Stable release | release.yml 36603799783 success; npm `latest` = 2.72.0 (published 17:45 UTC, visible ~10 min later, as with 2.71.0), gitHead `5ab6d52b2a`, bins intact; GitHub release v2.72.0 not prerelease, 25 assets; latest.json 2.72.0 signed for darwin-aarch64, darwin-x86_64, linux-x86_64, linux-x86_64-deb, windows-x86_64 | + +npm printed `"bin[...]" script name bin/ocx.mjs was invalid and removed` during both publishes; 2.70.0 and +2.71.0 printed the same, and the registry metadata keeps both bins (the `./` prefix is normalized). +The installed proxy and desktop app on this machine were not updated. diff --git a/devlog/_fin/260930_release_2_72_0/030_payment_and_email.md b/devlog/_fin/260930_release_2_72_0/030_payment_and_email.md new file mode 100644 index 0000000000..c76199ee79 --- /dev/null +++ b/devlog/_fin/260930_release_2_72_0/030_payment_and_email.md @@ -0,0 +1,29 @@ +# 030 — Payment check and sponsor email (wp4) + +1. Aside exec, read-only, WORKS inbox: TokenLab messages since 2026-09-29 18:00 KST (payment + confirmation, transaction hash, amount), DocuSign completion status. Where a transaction hash is + given, confirm it read-only on the public chain explorer against the recipient addresses in the + agreement (1,200 USDT, TRC-20 or ERC-20). +2. Only after npm `latest` reads 2.72.0 (the term starts at that release), Aside exec replies in the TokenLab thread from the maintainer mailbox, politely: payment received (only if confirmed), #6221/#6240 merged, released in + opencodex 2.72.0 (npm `@bitkyc08/opencodex`, release link), the 3-month term starts on that + release date per the agreement, README/picker placement live, the Responses-first proposal will be + evaluated separately. No attachments, no other recipients. +3. Record 090_outcome.md; move this unit and `260929_tokenlab_sponsor` to `devlog/_fin/` through a + docs PR to dev. + +Wallet addresses and transaction hashes stay out of the repository; the outcome records only that +payment was confirmed and when. + +## Findings (2026-09-30, before release) + +- WORKS inbox (Aside exec, read-only): Vincent wrote on 2026-09-29 23:08 KST that he signed the + agreement and paid 1,200 USDT on TRC-20, with a transaction hash, and sent final sponsor copy: a + longer English blurb, a Chinese blurb, and `https://tokenlab.sh/r/OPENCODEX` as the README and + picker link. A 23:13 message offers a USD 20 API-credit code for integration testing. +- On-chain (Tronscan, read-only): the hash is a confirmed, successful transfer on the official + USDT contract of exactly 1,200.000000 USDT to the agreement's TRC-20 address, at 2026-09-29 + 13:53 UTC; not flagged as risky. +- DocuSign: the only DocuSign mail in WORKS is the 21:08 sender-verification notice. No completion notice reached the maintainer mailbox; status notices go to the envelope sender's address, which this + check did not cover. Signature completion stays unverified here. +- Consequence for 010: #6240 takes the final copy and referral link (`2b9295fea6`) before the + regression gate; the earlier `lane=all` run on `21cddd35c9` was cancelled. diff --git a/devlog/_fin/260930_release_2_72_0/090_outcome.md b/devlog/_fin/260930_release_2_72_0/090_outcome.md new file mode 100644 index 0000000000..e8aedc57c7 --- /dev/null +++ b/devlog/_fin/260930_release_2_72_0/090_outcome.md @@ -0,0 +1,23 @@ +# 090 Outcome — 2.72.0 TokenLab release + +Closed 2026-09-30 (KST). + +| Criterion | Result | +|---|---| +| #6240 merged after full-lane CI on its exact head | Merged `1cd9d25517` from head `2b9295fea6`; lane=all 36591083341 success on attempt 2 (windows 7/9 and 8/9 reran once; neither loads a changed module) | +| npm and GitHub releases | `latest` 2.72.0 (gitHead `5ab6d52b2a`, main via #6246), `preview` 2.72.0-preview.20260930 (gitHead `4f9e3f0afb`, preview via #6245); release.yml 36603799783 / 36602348988; v2.72.0 has 25 assets and a signed latest.json | +| Payment and sponsor email | TokenLab's 1,200 USDT payment confirmed on-chain (2026-09-29 13:53 UTC). Reply sent from the maintainer mailbox through Aside exec at 2026-09-30 02:59 KST, confirming receipt, the 2.72.0 release, the placements and the term start | + +Release content since 2.71.0: #6221 (TokenLab preset, by @hedging8563), #6240 (sponsor placement, +CLI sponsor pinning, final sponsor copy and referral link). #6243 moved `dev` to 2.73.0 after the +candidate was pinned and is not part of 2.72.0. + +Per the agreement, the three-month sponsorship term starts with the 2.72.0 npm release +(2026-09-29 17:45 UTC, 2026-09-30 KST). + +Open items, outside this unit: +- DocuSign completion is not confirmed from the maintainer mailbox; TokenLab reports signing. Envelope + status goes to the sender account. +- TokenLab's Responses-first preset proposal (`X-TokenLab-Delivery-Policy`) needs its own PR and evidence. +- TokenLab offered a USD 20 API credit for integration testing; redeeming it is the maintainer's choice. +- The installed proxy and desktop app on this machine were not updated. diff --git a/devlog/_plan/260920_desktop_app_stabilization/030_icons_and_widget.md b/devlog/_plan/260920_desktop_app_stabilization/030_icons_and_widget.md new file mode 100644 index 0000000000..6ea275bebe --- /dev/null +++ b/devlog/_plan/260920_desktop_app_stabilization/030_icons_and_widget.md @@ -0,0 +1,114 @@ +# wp4 — one vector source for the icons, and a verdict on the widget + +## Why the icon set needed a source + +`desktop/src-tauri/icons/` carried eighteen raster files and no vector. Every size was an +independent artifact: nothing tied `Square107x107Logo.png` to `icon.png`, nothing could tell +whether one of them had been hand-edited, and adding a platform size meant drawing it again. The +`.icns` and `.ico` containers hid the problem further, because a wrong member inside them is not +visible in a diff at all. + +The fix is a single `icon.svg` plus `desktop/scripts/generate-icons.ts`, exposed as +`bun run icons` and `bun run icons:check`. Fifteen PNGs render through `rsvg-convert`, the +`.icns` is assembled by `iconutil` from its ten members, and the `.ico` is written directly with +six PNG-embedded entries (16, 32, 48, 64, 128, 256). `--check` regenerates into a temporary +directory and compares byte for byte, so a hand-edited PNG fails instead of silently disagreeing +with the source. + +## The geometry was measured, not redrawn + +A redrawn mark would have been a different icon wearing the same name. The shape in `icon.png` +was measured instead: it spans 58..453 on both axes, the stroke is 48 wide, and the outer corner +turns at radius 135. A centred stroke therefore sits at `x=82 y=82 w=348 h=348` with +`stroke-width=48`, and the corner radius was swept to find the closest match. `rx=127` reproduces +the original to within **430 of 262144 pixels at 512×512 — 0.164%**, which is antialiasing along +the curve rather than a changed silhouette. + +The mark stays pure black on transparency. Both macOS and Windows composite it over their own +backgrounds, so a baked background would appear as a card on one of the two. + +## The widget question + +`OpenCodexWidget.appex` is bundled, and the acceptance note requires a verdict either way rather +than an absence. + +**The extension registers, and that part is settled.** `pluginkit` lists it from the installed +application with the parent bundle resolved and no disabled or ignored marker: + +``` +com.opencodex.desktop.widget(2.61.0) + SDK = com.apple.widgetkit-extension + Parent Bundle = /Applications/OpenCodex.app + Parent Name = OpenCodex + Platform = macOS +``` + +That record is structurally identical to a system widget queried the same way, so the earlier +working hypothesis — that ad-hoc signing keeps the extension from being adopted at all — is wrong +and is recorded here as wrong. Registration is not the obstacle. + +**And it does not appear in the gallery.** The gallery was opened on this machine and checked: +OpenCodex is not among the offered widgets. No `OpenCodexWidget` process has ever run here +either, so nothing has asked the extension for a timeline. Registration and adoption are two +different things, and only the first of them holds. + +**What the signing state actually costs.** The host bundle carries the linker-signed placeholder: + +``` +host app Identifier = opencodex_desktop-b89067d97e1c189c + flags = 0x20002(adhoc,linker-signed) + Info.plist = not bound + Sealed Resources = none +appex Identifier = com.opencodex.desktop.widget + flags = 0x2(adhoc) +``` + +The host's `CFBundleIdentifier` is `com.opencodex.desktop`, but its *signed* identity is the +placeholder, its `Info.plist` is not bound into the signature, and it seals no resources. Locally +that is tolerated because the machine built the bundle itself. A distributed copy has no sealed +host for the system to validate the extension's containment against, and nothing binds the +declared identifier to the signed one. + +**The verdict, then:** the extension is registered and the gallery does not offer it. The host +bundle is the thing that fails a requirement — its signed identity is not the identity it +declares, and it seals nothing — so nothing downstream can establish that this extension belongs +to `com.opencodex.desktop`. Until the release pipeline signs the host with a Developer ID +identity, the widget ships but cannot be added. That is the finding; it is not worked around here, +and no part of the icon work depends on it. + +## What the icon check does and does not cover + +`bun run icons:check` compares all seventeen generated artifacts — fifteen PNGs, the `.ico` and +the `.icns` — byte for byte against a fresh render. It needs `rsvg-convert` and `iconutil`, and +when `iconutil` is missing it now says the `.icns` was not compared and fails, rather than +reporting a pass over a file it never looked at. + +That check does not run in CI, and claiming otherwise would be the easy lie here. The renderer is +not pinned, so two machines with different librsvg builds produce different bytes with nothing +wrong; asserting byte identity on a hosted runner would be asserting the runner's renderer +version. What CI runs instead is `tests/ci-workflows/build-desktop-icon-set.test.ts`, which needs +no renderer at all and reads its expectations out of the generator: every declared size committed +at exactly that size, the `.ico` directory carrying exactly the packed sizes with each payload a +real PNG of its declared dimension, the `.icns` walking cleanly end to end with one image member +per declared entry, and nothing hand-added beside the generated set. It was driven red on a +resized raster and on a stray file before being trusted. + +So the split is: shape is enforced everywhere, byte identity is enforced wherever the toolchain +exists. + +## Files + +- `desktop/src-tauri/icons/icon.svg` — new, the single source. +- `desktop/scripts/generate-icons.ts` — new, renderer and `--check` verifier. +- `desktop/package.json` — `icons` and `icons:check` scripts. +- Seventeen regenerated raster artifacts under `desktop/src-tauri/icons/`. +- `tests/ci-workflows/build-desktop-icon-set.test.ts` — new, the renderer-free structural guard, + registered in `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`. + +## Acceptance + +`bun run icons:check` passes on the committed tree over all seventeen artifacts, +`build-desktop-icon-set.test.ts` passes and has been shown to fail on a wrong-sized raster and on +a stray file, `bun run build:local` produces a bundle whose `Contents/Resources/icon.icns` is the +generated one, and the widget verdict is an observation of the gallery rather than an inference +from registration. diff --git a/devlog/_plan/260920_desktop_app_stabilization/040_widget_never_offered.md b/devlog/_plan/260920_desktop_app_stabilization/040_widget_never_offered.md new file mode 100644 index 0000000000..062fbfa4d2 --- /dev/null +++ b/devlog/_plan/260920_desktop_app_stabilization/040_widget_never_offered.md @@ -0,0 +1,215 @@ +# wp5 — the widget registered, and offered nothing + +## The symptom, and why it was not a signing problem + +The extension installs, `pluginkit` lists it beside the system widgets, and the gallery does not +show it. The obvious reading was signing: a locally built host is ad-hoc signed, so of course the +system will not adopt its extension. That reading was wrong, and following it would have produced +a signing change that fixed nothing, because the released build is signed and notarized and the +widget is missing there too. + +The actual defect is in the binary. `app/Package.swift` forced the executable's entry point: + +```swift +.unsafeFlags(["-Xlinker", "-e", "-Xlinker", "_NSExtensionMain"]), +``` + +and `app/Sources/OpenCodexWidget/main.swift` held nothing but a comment explaining that the entry +was handled by that flag. So `OpenCodexWidgetBundle` — which `Views.swift` defines correctly, +with a display name, a description and three supported families — was never referenced by +anything, and no code ever handed it to the extension host. + +Read off the shipped bundle: + +``` +LC_MAIN entryoff -> _NSExtensionMain +nm: SnapshotProvider present, OpenCodexWidgetBundle absent +Info.plist: NSExtensionPointIdentifier = com.apple.widgetkit-extension + NSExtensionPrincipalClass = (absent) +``` + +That combination is exactly consistent with the symptom. `pluginkit` registers from the +Info.plist, which is complete, so registration succeeds. `NSExtensionMain` then looks for an +`NSExtensionPrincipalClass`, which a SwiftUI widget does not declare because Xcode's `@main` on +the `WidgetBundle` is what connects it instead. Nothing errors. The gallery simply has no +configuration to offer. + +## The fix, and the wrong turn on the way to it + +The first attempt was to delete the linker override and call the bundle from `main.swift`. That +made the bundle's symbols appear in the binary and did not work either — it replaced a silent +failure with a loud one. Every launch died: + +``` +EXC_BREAKPOINT (SIGTRAP) + ExtensionFoundation closure #1 in ... _EXRunningExtension._shared + ExtensionFoundation MainActor.assumeIsolated + ExtensionFoundation _EXExtension.bootstrap(with:) + WidgetKit + OpenCodexWidget main +chronod: [com.opencodex.desktop::com.opencodex.desktop.widget] query failed - will try lazy + reload later +``` + +Seventeen crash reports accumulated in `~/Library/Logs/DiagnosticReports` while the gallery stayed +empty, because `chronod` asks the extension for its descriptors and the extension never survives +long enough to answer. + +**The extension needs both halves of what Xcode does, and each is useless alone.** `@main` on the +`WidgetBundle` is what keeps it in the binary; `-e _NSExtensionMain` is what makes the process +start as an extension rather than as a program. The original code had the second without the +first, this branch briefly had the first without the second, and only both together produce a +widget the system will talk to. With both in place the crash reports stop at zero and `chronod` +processes the extension normally. + +`tests/clients/desktop-widget-entry.test.ts` asserts both, plus that no `main.swift` has come back +to compete with `@main`, and that the bundle carries a widget with a display name rather than an +empty body — the same failure by a third route. + +The deployment target moved to macOS 14 at the same time, which drops the per-declaration +`@available(macOS 14, *)` guards and puts the binary's `minos` at 14.0, matching every working +widget on the machine this was measured on. + +## The sandbox is not optional + +While narrowing this down, the extension was rebuilt without `com.apple.security.app-sandbox` to +test whether the sandbox was implicated. It is required, and the system says so plainly: + +``` +pkd: Ignoring mis-configured plugin at [.../OpenCodexWidget.appex]: plug-ins must be sandboxed +``` + +An unsandboxed extension is not rejected at launch — it is never registered at all, so it vanishes +from `pluginkit` entirely. That also settles the snapshot path: the host writes into +`~/Library/Containers/com.opencodex.desktop.widget/Data/...` precisely because the extension reads +its own container, and that arrangement has to stay. + +## What the public record says about this failure + +The `_EXRunningExtension` crash is not unique to this repository, and finding the precedent +changed how much of the fix is guesswork. A forensic report on macOS 26.5 with Swift 6.3.2 +describes the same trap from the same cause — a widget extension assembled from a SwiftPM +`.executableTarget` and wrapped into an `.appex` by hand — and records that neither Info.plist +shape avoids it, because SwiftPM has no app-extension target and therefore never applies the +entry-point setup Xcode's WidgetKit template provides. That project's resolution was to stop +using SwiftPM for the extension and build a real Xcode app-extension target instead. + +Two other projects keep SwiftPM and supply the missing pieces by hand, which is the route taken +here: the linker entry (`-Xlinker -e -Xlinker _NSExtensionMain`) and the compiler's +extension-only mode (`-application-extension`, which is what Xcode spells +`APPLICATION_EXTENSION_API_ONLY`). Both are now set, and the extension launches and answers +`chronod` without a crash report. + +Three things the same record settles that were open questions here: + +- **Ad-hoc signing does not prevent gallery appearance.** Developer ID and notarization matter for + Gatekeeper, not for gallery mechanics. The containing app does have to be launched once after + installation, which is what makes the first-run behaviour in this branch load-bearing for more + than the menu bar. +- **App Groups do not work under ad-hoc signing**, and the documented fallback is exactly what + this repository already does — the host writes into the extension's own container. +- **`CFBundleVersion` must match between host and extension** or WidgetKit rejects timeline + reloads. Verified on the installed bundle: both read 2.61.0. + +If the gallery still refuses this extension after the entry point and the extension-only build, +the remaining known cause is the Xcode app-extension target itself, and that is a larger change +than this unit: it means adding an Xcode project for the widget and building it with +`xcodebuild` rather than `swift build`. + +## The signing defect underneath it + +Fixing the entry point does not make a *released* widget adoptable on someone else's machine, +because the release pipeline would not sign it. + +`.github/workflows/release.yml` ran `build-widget.sh` with no `env:` block. `MACOS_SIGN_IDENTITY` +was set one step later, on the Tauri build, which never reads it. So the script took its +`codesign --force --sign -` branch, and the bundler does not re-sign anything under `PlugIns/` — +its nested-code walker handles `.framework`, `.xpc` and `.app`, not `.appex`. + +**This has not harmed a release yet, and the reason matters.** No release has ever published a +macOS application: the last three carry no desktop assets at all, and the signing secrets did not +exist until after the most recent one was cut. `MACOS_SIGN_IDENTITY` reads a secret that was not +there, so the real-signing branch has never executed and the Developer ID path in the Tauri step +has never executed either. The bug is a mine rather than a crater — the next release is the first +one that would step on it. Saying otherwise would be inventing a history this repository does not +have. + +**Signing one path is also not enough.** A bundler that did not place a file does not sign it, and +picking binaries by file extension misses the ones that have none. The durable form of the check +is to find Mach-O files by their magic bytes and require every one of them to carry the release +identity, rather than naming the paths that are expected to exist. + +Three changes: + +- The certificate is imported into a temporary keychain in a step **before** the widget build, and + the keychain is deleted in an `always()` step so it cannot outlive a failed job. +- The widget build receives `MACOS_SIGN_IDENTITY`, and `build-widget.sh` now signs with + `--options runtime` as well as `--timestamp`, both of which notarization requires. +- A step after the widget build asserts the result rather than printing it: strict verification, + the configured team identifier, the runtime flag, and a secure timestamp. Without a configured + team it says so and skips, so a fork's build still works and still cannot pretend to be signed. + +This half cannot be proven here. It needs maintainer-held credentials, and the proof is a +notarized artifact installed on a machine that did not build it, launched once, with the gallery +then checked. That is recorded as the outstanding verification rather than claimed. + +## The menu bar had the same shape of problem + +Start at Login was purely opt-in. Nothing enabled it on first run, so an install left the user +with a menu bar item only for as long as the app happened to be running — and a menu bar app that +is not running has no menu bar item. After a reboot the app was simply absent. + +`first_run::apply_start_at_login_default` enables it once per installation, keyed on a marker in +the app config directory, and runs before `tray::install` so the tray checkbox reads the state it +leaves behind. The marker is written before the login item is touched and is never removed, so a +user who turns the setting off keeps it off. Writing afterwards would let a failed enable retry +every launch and eventually flip the setting back under someone who had deliberately disabled it. + +The marker distinguishes a fresh install from a user who opted out, but it cannot distinguish +either from an install that predates the marker. The desktop shell and the widget both landed the +same day this was written and no release tag contains them, so there is no such population; if +that changes, this needs a migration rather than a marker. + +## What was verified here + +Rebuilt, installed to `/Applications`, and launched: + +``` +LC_MAIN entryoff 5656 -> _main (was _NSExtensionMain) +nm: _$s15OpenCodexWidget0abC6BundleV4bodyQrvpQOMQ present +pluginkit: com.opencodex.desktop.widget re-registered, parent bundle resolved +~/Library/Application Support/com.opencodex.desktop/start-at-login-claimed written +~/Library/LaunchAgents/OpenCodex.plist created +``` + +So the entry point is connected and the login item is registered, both on a real install rather +than in a test double. + +## Verdict: it appears + +The gallery was opened on this machine after the fix and OpenCodex is in it, between OKX and +PASS, with all three declared families rendering real data rather than placeholders: + +``` +com.opencodex.desktop::com.opencodex.desktop.widget:OpenCodexWidget:systemSmall +com.opencodex.desktop::com.opencodex.desktop.widget:OpenCodexWidget:systemMedium +com.opencodex.desktop::com.opencodex.desktop.widget:OpenCodexWidget:systemLarge + "OpenCodex — Proxy status, today's usage, and quota at a glance." +``` + +Small shows the token count for the day, medium adds requests, cost and the account quota rows, +large adds the 24-hour per-model timeline. The list icon is the mark generated from `icon.svg`. + +That settles the whole question the acceptance note left open, and it settles it the right way +round: the extension was never rejected by signing or by the sandbox. It had no widget in it, and +then it had one that could not start. Both are fixed, and the fix is a SwiftPM configuration +rather than the Xcode app-extension target the public record recommends — so the cheaper route +does work, provided all three of `@main`, the `_NSExtensionMain` entry and +`-application-extension` are present. + +## Acceptance + +The entry-point half is closed by the gallery observation above. The signing half closes when a +release build's extension reports the team identifier, the runtime flag and a timestamp, and a +clean install on a machine that did not build it offers the widget. That second half needs a real +release and is recorded as outstanding. diff --git a/devlog/_plan/260920_desktop_app_stabilization/050_landing.md b/devlog/_plan/260920_desktop_app_stabilization/050_landing.md new file mode 100644 index 0000000000..4e6135b9d7 --- /dev/null +++ b/devlog/_plan/260920_desktop_app_stabilization/050_landing.md @@ -0,0 +1,66 @@ +# wp5 — landing the stack + +## Shape + +Four pull requests, each based on the one below it, all ultimately targeting `dev`: + +| PR | branch | what it carries | +|---|---|---| +| #5327 | `codex/260920-app-stabilization` | release profile, stale-dist report, `build:local`, the lockfile and test-layout repairs | +| #5328 | `codex/260920-claude-desktop-mode-visibility` | the first-party reachability message | +| #5329 | `codex/260920-app-icons` | one SVG source, the generator, the renderer-free CI guard | +| #5339 | `codex/260920-widget-entry` | the widget entry point, the login-item default, release signing | + +They merge bottom-up. After each one lands, the next is retargeted to `dev` and its exact head is +read again, because a squash merge rewrites the parent and the child's base disappears. + +## Two repairs in here are not ours + +`dev` was already red when this stack was cut, in two independent places, and both were fixed +here because every branch cut from `dev` inherits them. + +`tests/providers/stepfun-provider.test.ts` landed with no entry in either inventory and no regex +seed that resolves its name, so the membership oracle failed on `dev` and on everything branched +from it. Registering it under `providers` restores the gate for everyone. + +`macos widget + bundle` failed with *A public key has been found, but no private key*. The job is +an unsigned build by design, so the key is correctly absent — but the committed config sets +`bundle.createUpdaterArtifacts` and `plugins.updater.pubkey`, so `tauri build` writes the updater +archive and then refuses to finish. That half is #5338's, which turns the artifact off for that one +invocation; this stack does not duplicate it. + +Fixing the build revealed the rest of the job, which had never run. Its first assertion looked for +`Contents/MacOS/OpenCodex` — `productName` — while the bundle carries `opencodex-desktop`, the +crate name. That half landed separately as #5351, and better than the version written here: it +reads `CFBundleExecutable` out of the bundle instead of restating the name, so the check follows +the config rather than drifting from it. This stack's copy was dropped in favour of it. + +What remains here is the assertion with no equivalent: that the WidgetBundle is actually linked +into the extension. The appex builds, signs and registers identically with the bundle dropped by +the linker, so nothing else in this job would have noticed the defect that shipped. + +Three of this stack's incidental repairs turned out to be running in parallel with the +maintainer's own: the StepFun layout registration (#5335), the widget job's updater override +(#5338), and this executable assertion (#5351). Each was dropped here once the other landed. The +pattern is worth noting for the next batch — a repair found while passing through is worth +checking against open pull requests before it is written. + +## What closes this + +Each merge reads the exact head's check runs rather than a rollup, distinguishes a job the event +requested from one it skipped, and treats a missing, skipped, or cancelled job as not a pass. The +last merge is followed by reading `dev`'s own push run, because five of the eight defects found in +this unit were invisible until two changes met. + +## Deliberately not changed here + +Review asked for the public macOS install guidance to move with the release path, since +`README.md`, `guides/desktop-app.md` and `guides/macos-menu-bar.md` all tell the reader the app is +ad-hoc signed and not notarized, while this stack makes a real release refuse to run without a +Developer ID and the full notarization credential set. + +Those pages are accurate today and will stop being accurate at the next release, not at this +merge. No release has ever published a macOS application, so rewriting them now would describe an +artifact nobody can download and would leave the Gatekeeper walkthrough — still correct for a +locally built app — reading as though it were obsolete. The pages move with the first notarized +artifact, which is also when someone can check the instructions against a real download. diff --git a/devlog/_plan/260920_round2_followups/010_r1_paginated_history_guard.md b/devlog/_plan/260920_round2_followups/010_r1_paginated_history_guard.md new file mode 100644 index 0000000000..a3db495799 --- /dev/null +++ b/devlog/_plan/260920_round2_followups/010_r1_paginated_history_guard.md @@ -0,0 +1,91 @@ +# R1 — the paginated-history guard, from activation and from recovery + +Scope: #5321 (activation) and #4812 (recovery). Branch `codex/260920-r1-paginated-history-guard`. + +## What the guard was actually protecting + +`preflightCodexHistoryInjection` returns `history_paginated_openai_requires_native_writer` when a +provider-table transition finds a thread row that is both `model_provider = 'openai'` and +`history_mode = 'paginated'`. The reasoning is sound. The transition takes the root +`openai_base_url` out, a paginated row cannot be relabeled, and Codex builds its provider map as +`merge_configured_model_providers(built_in_model_providers(openai_base_url), model_providers)`, so +without that root line the built-in `openai` entry is `api.openai.com`. The conversation would +resume outside the proxy. + +What made it a lockout is that 2.60.0 classified it alongside "something is wrong with this +store". `src/codex/inject.ts` refuses every reason that is not exactly `HISTORY_RELABEL_STANDS_DOWN`, +so nothing was written at all: no config, no profile, no `model_catalog_json`, integration +disabled. Before 2.60.0 the same home returned the plain stand-down, and the routing and catalog +half landed while the relabel stood down. + +## The state that was already in the tree + +The injector already builds the safe state for one routing form. `keepRootOverrideAlongsideTable` +keeps the marker-owned root override beside the provider table for client compaction, for exactly +this reason, and passes `resumeHistory: false` so the relabel never runs. Authless was excluded +deliberately — its point is `requires_openai_auth = false` — on the assumption that it could +always forward-tag resume history instead. On a paginated home that assumption is false, and the +refusal is where that showed up. + +So the fix is not a new mechanism. `src/codex/inject/paginated-openai-compat.ts` selects the +existing one from the preflight verdict rather than from the routing form: when the reason is the +paginated-openai code and the target can own a root key, retain the override, downgrade the reason +to the stand-down constant, and let the transition complete. The paginated row is never read or +written; it simply keeps resolving to this proxy. + +Two cases cannot reach that state, and both are honest outcomes rather than traps: + +- An admission-token form cannot use the root key at all, because Codex's built-in `openai` entry + carries no `x-opencodex-api-key` header. It keeps the refusal, and the message now names + `unauthenticatedLoopbackListener` and `syncResumeHistory` instead of "do not retry". +- A root line the user owns is left alone. The conversation follows the destination they chose, + which is the guarantee the injector already makes everywhere else about a line it does not own, + and the journal correctly records the line as not ours. + +## Where it had to live + +`src/codex/inject.ts` was at 984 of its 987-line ratchet cap, so the decision could not be +inlined. The new module costs the injector one import and one net line; the file now sits at +exactly 987. The refusal code became an exported constant in `src/codex/history-provider.ts` +because the same literal in two files is how the stand-down pair drifted the first time. + +## #4812, checked rather than assumed + +The recovery half is already closed on `dev`: `resolveRestoreHistoryDisposition` stands down on +`HISTORY_RELABEL_STANDS_DOWN` and removal retains the provider table. The new code cannot reach +restore at all — it is only set under `providerTableMode`, and restore preflights with +`providerTableMode = false`, whose row predicate is `model_provider = 'opencodex'`. + +Two things were still wrong on that side. `ocx restore --remove-codex-provider-table` existed but +appeared in no usage or help text, so the escape hatch was reachable only by reading the parser; +it is now in the command registry and top-level usage, bound by a test that reads the flag out of +`dispatch.ts` rather than restating it. And the public guide in all eight locales still said +restore and removal refuse on paginated history and that such a home cannot be uninstalled, which +has not been true since 2026-09-17. + +## Verification + +Static review plus exact-head hosted CI. Per the lane constraints, NOT RUN locally: `bun test`, +any individual test file, `bun run typecheck`, any build, any install, live `ocx`, service +restart, and credential or configuration changes. + +Regression coverage added: + +- `tests/codex-integration/history-paginated-openai-compat.test.ts` — the resolver itself: root + override retained and placed before the first table, CRLF preserved, a user-owned line left + untouched and not claimed, the admission-token refusal naming both remedies as keys that are + asserted to exist in `src/types/config.ts`, every other reason passing through unchanged, and a + source-oracle check that the refusal code is defined once. +- `tests/codex-integration/codex-inject-integration.test.ts` — the end-to-end regression, rewritten + from "refuses" to the full transition: config carries both the table and the marker-owned root + override, the rollout bytes and the thread row are unchanged, and `ocx restore` afterwards takes + the retained override back out. That last assertion is the one that keeps this from trading + #5321 for a new #4812. +- `tests/cli/cli-restore-back.test.ts` — the removal flag is discoverable in both help surfaces. + +## Not in this lane + +The other hard-refusal reasons on the recovery side still have no named repair command: a missing +state database with pending manifest entries, and a backup manifest that is unreadable, foreign, +or schema-invalid. Those are a different failure family from the guard and are left open rather +than folded in here. diff --git a/devlog/_plan/260920_round2_followups/020_r2_desktop_ci.md b/devlog/_plan/260920_round2_followups/020_r2_desktop_ci.md new file mode 100644 index 0000000000..47c658ef83 --- /dev/null +++ b/devlog/_plan/260920_round2_followups/020_r2_desktop_ci.md @@ -0,0 +1,54 @@ +# R2 — the `macos widget + bundle` failure on `dev` + +Status: the job had two independent defects stacked on top of each other. The second was +invisible until the first was fixed, because it lived in a step that had never once executed. + +## First layer: the verification build demanded the release key + +`bundle.createUpdaterArtifacts` is on and the updater public key is committed, so `tauri build` +concluded it had to emit a signed update artifact and stopped with `A public key has been found, +but no private key`. On macOS this bites even with `--bundles app`, because the macOS updater +artifact is derived from the `.app` itself. + +#5338 scoped the opt-out to the verification build with a `--config` override and left +`tauri.conf.json` alone, so release signing stays in `release.yml` where the secret lives. +`BundleConfig` carries `deny_unknown_fields`, so a misspelled override key fails the build +rather than silently reverting to signing — the override cannot rot into a no-op. + +## Second layer: the Verify step asserted a filename that never existed + +With the build green the Verify step ran for the first time and failed on its first line, +`test -x "$app/Contents/MacOS/OpenCodex"`, printing nothing because `test` is silent. + +Tauri renames the main binary only when `mainBinaryName` is set (`tauri-cli` +`src/interface/mod.rs`, with `rename_app` in `src/interface/rust/desktop.rs` a no-op +otherwise). This config does not set it, so the bundled executable keeps the Cargo bin name +`opencodex-desktop`. The job log had said so all along: `Built application at: +.../target/release/opencodex-desktop`. + +The fix reads `CFBundleExecutable` from the bundle's own `Info.plist`. `tauri-bundler` +`create_info_plist` writes that key from the same `main_binary_name()` that +`copy_binaries_to_bundle` uses for the filename, so the plist and the file on disk cannot +disagree. An empty value is rejected so a missing key cannot pass by testing the `MacOS` +directory. + +The other three assertions were checked against the same source and were already correct: +`Settings::copy_binaries` strips the `-<target>` suffix so the sidecar lands as +`Contents/MacOS/ocx`, and `copy_custom_files_to_bundle` resolves `bundle.macOS.files` +relative to `Contents` and errors when the source is missing, so the appex is present with its +executable bit intact. + +## What this leaves open + +CI no longer exercises updater bundling at all. A regression there surfaces only during a +release. Two release-time backstops contain it — `collect-release-assets.ts` throws when the +macOS `app.tar.gz` is missing, and `updater-manifest.ts --require-all` refuses a partially +signed `latest.json` — and both are covered by `tests/ci-workflows/release-desktop-scripts.test.ts`. +What nothing covers is `tauri.conf.json` itself: no test reads it, so flipping +`createUpdaterArtifacts` off or mangling the `plugins.updater` block stays green everywhere +until a release runs. A static contract test over that file is the cheap follow-up. + +The macOS updater filename is also restated by hand in four places — +`collect-release-assets.ts`, `updater-manifest.ts`, the `release.yml` matrix, and +`structure/desktop-shell.md` — with nothing deriving one from another. The same contract test +should tie them together. diff --git a/devlog/_plan/260920_round2_followups/030_lane_r3.md b/devlog/_plan/260920_round2_followups/030_lane_r3.md new file mode 100644 index 0000000000..d38a851a40 --- /dev/null +++ b/devlog/_plan/260920_round2_followups/030_lane_r3.md @@ -0,0 +1,90 @@ +# R3 — the roster and login remainders + +Status: implemented, awaiting review. Scope was #5292 and the two #5261 remainders. + +## #5292 was already closed before this lane opened + +The plan's table says `gui/src/pages/Logs.tsx` restates the recovery-kind union with nine of +thirteen members. That was true when the table was written and stopped being true two hours +earlier: `555f0cacdf` (#5300, 18:41) replaced the copy with the durable roster, and the plan +commit landed at 20:48 from a snapshot taken before it. + +Current `dev` already has all of it. `Logs.tsx` imports `AttemptRecoveryKind` from +`src/usage/telemetry-contract.ts` and its label map closes with +`satisfies Record<AttemptRecoveryKind, string>`, so a fourteenth kind is a typecheck failure +there rather than an "Unknown recovery reason". All ten catalogs carry all thirteen labels plus +the fallback, and `tests/usage/request-outcome-agreement.test.ts` holds both: the label map has +to cover every member of `ATTEMPT_RECOVERY_KIND_ROSTER`, and every key it names has to exist in +every catalog. Verified by reading the tree, not by rerunning the suite. + +Nothing was changed for it. The row is stale, not open. + +## #5261, remainder one: the two CLI logins that discarded the launch + +`src/oauth/login-cli.ts` called `void openUrl(...)` in both `handleOAuthLogin` and +`handleKeyLogin`. Each printed a URL, said it was opening a browser, and asked a question that +assumes it opened — indistinguishable from a login that is working. + +The part that made this more than a missing `console.warn`: `OAuthController.onAuth` returns +`void` and every one of the thirteen provider call sites invokes it as `ctrl.onAuth?.(...)` and +moves on. The launcher's answer therefore arrives after the flow has continued, and on a +callback-server provider `#waitForCallback` has already called `onManualCodeInput` by then. A +warning written at that moment lands on the line the user is typing on. + +Making `onAuth` awaitable would mean changing the controller contract and all thirteen call +sites, which is a much larger change than the defect deserves. Instead the launch reports itself +when it settles, and the two things that could collide with it wait on that report: the +manual-code prompt awaits it before asking, and the key login awaits it before it constructs a +reader at all. A polling provider that never prompts is still told before the login claims to +have worked. + +`BROWSER_LAUNCH_FAILED_HINT` in `src/cli/account-auth.ts` kept its ChatGPT-specific second line +and now derives its first from `BROWSER_LAUNCH_FAILED_NOTICE`, so the sentence has one home +across all three logins. + +The handlers took an optional deps object. The contract worth holding is an order, and an order +is only observable from something that records both events; spawning a launcher and attaching to +stdin to find that out would test the operating system. Production passes none of them. + +## #5261, remainder two: the roster that kept last-good rows silently + +`useCodexAccountPool` kept its rows after a failed read and also kept reporting `ready`. Keeping +the rows is right — blanking a populated pool because one 30s poll missed is its own defect — but +the surface then could not tell a list the server had just confirmed from one that predated a +failure. The reported shape: add an account, the read that would bring it over fails, and the +older accounts are on screen with the new one absent. + +`refreshFailed` sits beside `loadState` rather than inside it, for the same reason `refreshing` +already does. `loadState` answers what the surface can draw and a warm failure does not change +that answer; folding it in would mean either flashing the cold skeleton over good data or saying +nothing. A cold failure still replaces the surface with the error it already had, and the banner +only renders when rows survived, so an empty cold failure is never annotated instead of explained. + +## Verification + +Static review and hosted CI at the exact head. The lane ran no local suite, no individual test, +no typecheck, no build, no install, no `ocx`, and changed no credential or configuration — +recorded as NOT RUN. + +Checked by reading rather than running, because the ratchets are what a merge breaks: + +- No file this lane touches appears in `tests/fixtures/file-size-baseline.json`. The ten i18n + catalogs are in its `exempt` list. +- `tests/oauth/oauth-login-cli-browser-launch.test.ts` is registered in both + `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`. The gui + suite has no layout guard. +- The one new i18n key is in all ten catalogs, which `gui/tests/locale-parity.test.ts` and + `gui/tests/claude-desktop-locale.test.ts` both require. +- `CodexAccountLoadState` gained no member. `CodexAccountPoolController` gained one, and the + source-oracle roster in `gui/tests/codex-account-pool-controller.test.ts` names it. +- `CodexAccountPoolLoadStates` stopped restating the load-state union and derives it. + +## The GUI screenshot gate + +`enforce-target` requires a screenshot for a PR that touches `gui`. Producing one needs +`bun run build:gui` and a running proxy, both of which this lane is forbidden to do, so the pull +request says so and offers what can be checked instead: the rendered markup is asserted against a +mounted DOM in `gui/tests/codex-account-pool-stale-refresh.test.tsx` — the banner appears with +surviving rows, carries the catalog string, does not appear on a successful refresh, and does not +replace the cold error — and the new class reuses the existing `.pwi-auth-state` block with the +`--amber` pair already used elsewhere in the theme. diff --git a/devlog/_plan/260920_round2_followups/040_r4_retry_rework.md b/devlog/_plan/260920_round2_followups/040_r4_retry_rework.md new file mode 100644 index 0000000000..6022b10626 --- /dev/null +++ b/devlog/_plan/260920_round2_followups/040_r4_retry_rework.md @@ -0,0 +1,53 @@ +# R4 — #4942 and #4989 as one ambiguous-resend gate + +Status: OPEN. Branch `codex/260920-r4-retry-rework`, cut from `origin/dev` at `d6d87440b7`. + +## Why the two pull requests are one change + +#4942 (FredAmartey) replays a native Responses send whose connection died before any response +head, behind a per-provider opt-in. #4989 (lidge-jun) replaces a native Responses SSE stream that +died after the head while the body had carried only control events. Written apart they read as two +features. Against the stage table #5266 landed in `src/lib/request-failure-model.ts` they are one +row: a stage whose `stageCommitment` is `nothing-observed`, with a cause whose +`causeEvidence` is `unknown`. `resendPermission` answers `refused-ambiguous` for both, and +the module already names the only thing that may override it — "a narrowly scoped, explicitly +opted-in recovery that a maintainer reasoned about and bounded". + +Two overrides is one too many. #4942 spreads `replayResets: 2` into every dispatch leg of the +request and #4989 takes `Math.min(1, remaining)` of the transient budget at the stream boundary, +so one logical request that reset before the head and again after it would buy a replacement on +each. The rework gives the override a single per-request allowance and makes both stages claim +from it. + +## Shape + +- `src/lib/request-resend-gate.ts` — the one gate. Pure table lookup for the stages the caller + already observed something at, plus the operator override for the ambiguous row. It never + restates the table: stage, cause, permission and send class all come from + `request-failure-model.ts`, and the cause comes from the `AttemptRecoveryKind` that will be + recorded, so the reason in the log and the send it authorised cannot disagree. +- `src/lib/request-execution-budget.ts` — the allowance lives on the shared send ledger, which is + what a combo child inherits through `deriveRequestExecutionBudget`. Parent and child therefore + cannot each hold one. +- `src/server/responses/reset-replay.ts` — the provider opt-in and the body judgment from #4942, + plus the per-request authority both call sites use. +- `src/lib/upstream-retry.ts` — the pre-header claim, as a callback rather than a number. +- `src/server/responses/combo-stream-preflight.ts` — the preflight reports the stage it observed + instead of a boolean, so the gate rather than the preflight decides. + +## Stage classification at the stream boundary + +#4989 gated on `responseCreated && !outputCommitted && !terminal`. That is `protocol-prelude`. +A read error before any parsed event is `headers-only`, which the table gives the same +commitment and therefore the same answer; the rework admits it rather than refusing a row the +table permits. Everything else the preflight can see is `semantic-output` or `terminal`, and +those refuse regardless of cause. + +## In scope from the remainders + +#4191 and #5180 only to the extent the resend gate reaches them. Recorded in 050. + +## Verification + +Static review plus exact-head hosted CI. Local suites, individual tests, typecheck, build, +install and live `ocx` execution are NOT RUN by lane policy. diff --git a/devlog/_plan/260920_round2_followups/050_lane_r5.md b/devlog/_plan/260920_round2_followups/050_lane_r5.md new file mode 100644 index 0000000000..7343408d0c --- /dev/null +++ b/devlog/_plan/260920_round2_followups/050_lane_r5.md @@ -0,0 +1,191 @@ +# Lane R5 — the four telemetry pull requests as derived consumers of the recorder + +Status: OPEN. Branch `codex/260920-r5-telemetry-derived`, rebased onto `dev` after `origin/dev` +advanced mid-lane. One branch, ordered commits, one pull request to `dev`. + +Lane C deferred #2366, #3748, #3983 and #5063 "as implemented", because each adds a parallel store +or a second emission path. [030_lane_c2.md](../260920_meaning_preservation_batch/030_lane_c2.md) +then specified the derived form for each. This lane builds those four forms. It adds no store: the +durable shapes stay `PersistedUsageAttempt` and `PersistedUsageEntry`, and every projection reads +them. + +## What landed, per pull request + +### #2366 (chilung-cgu) — durable failure attribution, in the landed vocabulary + +`failureStage` and `failureCause` now ride the attempt that ended a request and the logical row, +both closed roster members. `FailureSide` and the seven-member `FailureStage` are not here: two +attribution vocabularies for one question is the class that blocked 2.60.0. The PR's widening of +`transportPhase` and `terminalSource` to arbitrary strings is not here either; those validators +stay closed, and `terminalStatus` — which was a plain `string` — joined them, because it is now a +grouping-key slot and it is assembled from an upstream frame. + +The derivation reads only closed values. `errorCode` and `upstreamError` are excluded on purpose: +both carry upstream text, so a classification keyed on them is a different answer per provider and +per locale, and a key built from them cannot promise it carries no content. That exclusion is what +lets the pair be a Prometheus label and a fingerprint component with no masking pass. + +It runs at `addFinalRequestLog`, the one seam every request passes exactly once, and before the +attempt snapshot so the disk row and the live attempt carry the same pair. `addRequestLog` rebuilds +the persisted row field by field, so the pair is written there explicitly — a field omitted at that +line reaches `/api/logs` and never reaches `usage.jsonl`. + +**The resend verdict is not stored.** `/api/logs` computes `resendPermission` at read time for the +row and each attempt. The tables that decide it live in this build; a row written months ago must +not assert a permission the current tables refuse. + +**Known limit, recorded rather than hidden.** Only the attempt that ends a request, plus the one +sealed by a key-account rotation, carry attribution. The other intermediate finalizers — +`policy-fallback.ts` and five sites in `core-combo.ts` — still reach the ledger unattributed. Each +has different evidence in scope and a branch verified by static review alone should not add six new +classification call sites at once. The logical row is attributed in every case, which is what the +projection and the exporter read. + +**Second known limit.** A ciphertext or reasoning-parameter recovery that SUCCEEDED, followed by an +unrelated 400 on the same attempt, still reads as that recovery's cause. The rule is narrowed to +the last recorded kind on the matching status, and the proper fix — clearing recovery evidence on +success in `core-opaque-recovery.ts` — belongs in the recovery path, not the derivation. + +### #3748 (yansigit) — a failure grouping, not a second ledger + +`src/telemetry/` and its SQLite store are not built. Failed rows are grouped by a versioned +fingerprint over a fixed-arity tuple of closed roster members, folded during a scan of +`usage.jsonl` through the existing `scanUsageLedgerCooperatively`. The projection holds a count and +two timestamps per group; delete a ledger row and it leaves the grouping on the next rebuild. + +The free-text `signature` and its regex masking are replaced by construction rather than by a +better regex: an expression can only assert it removed what it matched, while a tuple whose every +slot comes from a frozen list has nothing to remove. Absent facts are explicit nulls in fixed +positions, because omitting them would let `[a, null, b]` and `[a, b]` collide. + +**A deliberate divergence from 030_lane_c2.md, flagged for the coordinator.** That document says +"No provider". The lane brief for R5 says the fingerprint is over "closed cause + provider + model +class". The brief is the later and more direct instruction, so `providerClass` is in the tuple — +resolved against the provider registry so it is a registry id or `null`, never the alias a user +typed. Model class is NOT in the tuple: no closed model-class vocabulary exists in this repository +and inventing one is the union-exhaustive hazard this batch exists to avoid. Removing +`providerClass` is one slot and a version bump if the coordinator prefers the C2 shape. + +The mutable `monitoring/dispatched/fixed/ignored` status and its notes are absent. They are +operator state; they cannot be reconstructed from immutable request rows, so presenting them as a +derived ledger would be a claim this projection cannot make. + +The reader is `GET /api/usage?failures=1` rather than a new route: it answers a different question +from the usage summary and costs a scan, so it is opt-in and no new CLI-parity surface appears. + +### #3983 (yansigit) — five counts on the attempt, no second emission path + +`emitDebugLine` writes the in-process ring AND stderr, and stderr is redirected to the service log +under launchd and systemd, so the PR's per-event lines would give an installed service a durable +per-event history beside the ledger. Its per-payload HMAC used a process-global random key, making +every repeated prompt fragment, tool name and error message correlatable for the process lifetime. + +Instead the attempt carries adapter events, relayed frames, semantic bytes, side effects and +terminal frames. Adapter events are counted at the existing adapter-parse seam; relayed frames +after a SUCCESSFUL `controller.enqueue`. Counting both at the reader would make them equal by +construction and erase the loss signal. The recorder is bound to the request's translator budget +and reaches the current attempt through a callback, so a mid-request attempt rotation credits the +live attempt rather than one already finalized. The debug ring now FORMATS one line per finalized +attempt from those counts, through `appendDebugLogLine` and never `emitDebugLine`. + +Adversarial review caught the case this design gets wrong on its own: a non-streaming turn delivers +one body and calls no per-frame recorder, so every buffered response would have persisted adapter +events with zero relayed ones — the loss signal, raised on every buffered request. The buffered +seam now records its delivery from the body it built. + +`run-turn-execution.ts` is untouched. Its accounting distinguishes adapters that report their own +physical sends, and the PR's unconditional pre-count would double-charge them. + +### #5063 (Vocllum) — retention with a revision contract + +`usageLedgerMaxBytes` is unset by default and unset means unlimited. When set, an append that +crosses it publishes the newest whole rows byte for byte through the shared atomic writer. + +The defect this closes: #5063 captured a size, copied a suffix and renamed over whatever was there, +so a row appended in between was silently dropped; its own concurrency test performed two +sequential calls and said it could not test concurrency. Two things close it. The append is +synchronous and the compaction runs inside the same call stack, so no in-process append can +interleave, and a second server on the same home cannot append at all — it is refused by the +existing ledger-owner lease, which is why the hook is installed after ownership. And +`validateBeforeRename` re-opens the target immediately before the rename and refuses unless +identity, size and revision metadata are byte-for-byte what was copied. A focused test drives that +exact window through an injected hook. + +Rows are copied and never parsed, which is what keeps a field a newer build wrote intact through a +compaction. The writer gained a streaming form so the retained span is not held in memory, and that +form fsyncs the temp before the rename and the parent directory after it. + +The invalidation half was missing from the original entirely. A compaction now discards the +2,000-entry Logs ring, the retained usage aggregate and failure projection, and the request-history +index — otherwise `/api/logs` keeps serving rows the ledger no longer has. + +**This does not close #5063.** The Usage-page control it also asks for is not here: this lane may +not build or run the GUI, so it cannot produce the screenshot the gate requires, and shipping an +unverifiable control is worse than shipping the policy it would set. The limit is settable in +`config.json` today and the configuration reference says so. Remaining scope: the dashboard +control, its management route, and the ten catalog strings. + +## The GUI screenshot gate + +This branch changes `gui/src/pages/Logs.tsx` and the ten locale catalogs, so `missing_ui_screenshot` +fires. It fires on changed paths under `gui/`, not on words in a description, and this lane may not +run `bun run build:gui`. A maintainer comment or the `gui-screenshot-waived` label is the documented +resolution. + +The evidence to judge it without the screenshot: the catalog edits are purely additive (+29 lines, +0 removed, in each of ten files, all exempt from the file-size ratchet), every new key exists in all +ten catalogs, and three `satisfies` clauses make a missing label a typecheck failure rather than a +silent fallback. The visible change is three rows added to the Logs detail dialog for a failed +request — the cause, the stage it reached and the resend verdict — and a named cause where the +attempt table previously led with a bare wire code. + +## Pre-existing defect found and deliberately not fixed here + +`src/config/atomic-write.ts` scrubs a failed temp through `effective.write(tmp, "")`, but the default +writer opens with `"wx"`, so that fallback always fails with `EEXIST` on an existing temp. It only +matters when `truncate` has also failed, and the temp is owner-only. It predates this branch and +affects every atomic config write, including secret-bearing ones, so fixing it is a change to a +security-adjacent path that belongs in its own lane rather than inside a telemetry branch. + +## The file-size ratchet caught this branch once + +`src/server/request-log.ts` carries the whole request-logging surface and was 1,962 lines against +the repository's 2,000-line seed threshold. The attribution wiring pushed it to 2,015, and +`file-size ratchet: repository` reported `NEW_OVERSIZED` on the first exact-head run. The remedy is +the one AGENTS.md gives — a move, never a number — so the two places a stage and cause are decided +and written moved to `src/server/request-log-failure-attribution.ts`, leaving the file at 1,979. + +Worth recording for the next lane that touches this file: 21 lines of headroom is not much, and +the cap only ever moves down. + +## Verification + +Static source review plus exact-head hosted CI, and three adversarial reviews at high effort +covering typecheck hazards, repository gates, and runtime correctness and privacy. Their findings +are in the branch: the transport-evidence precedence, the 402 mapping, `transport-unsent` no longer +being the fall-through, the parent-directory fsync, the buffered delivery accounting, the rosters +read instead of restated in two tests, and the invariant split into INV-RESEND-01 and +INV-ATTRIBUTION-01 so each binds exactly one test. + +NOT RUN on this branch, by instruction: `bun run test`, any individual `bun test` file, +`bun run typecheck`, `bun run build:gui`, `bun run lint:gui`, `bun install`, +`bun run structure:check`, `bun run privacy:scan`, and any live `ocx` execution. None of these may +be recorded as passing. + +Checked statically: + +- no file this branch touches is at or over its file-size ratchet cap; `src/server/index.ts` sits at + 884 against 893, and the ten catalogs are exempt; +- `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json` agree key for key, + and each new test's regex seed resolves to the domain it is registered to — `failure-attribution` + is named to avoid the `request-` seed that would have placed it in `usage`; +- `src/usage/telemetry-contract.ts` still has no imports, `src/usage/request-outcome.ts` still reaches + nothing but it, and `gui/src/pages/Logs.tsx` still never names `src/usage/log`; +- no test or document restates a source constant: the rosters, the fingerprint version and the label + keys are read from the modules that declare them. + +## Issues + +#2366, #3748 and #3983 are addressed by these derived forms; the coordinator decides closure. #5063 +is partially addressed and must not be closed — its dashboard control is named above as remaining +scope. diff --git a/devlog/_plan/260920_round2_followups/050_r4_remainders.md b/devlog/_plan/260920_round2_followups/050_r4_remainders.md new file mode 100644 index 0000000000..4c826a9d17 --- /dev/null +++ b/devlog/_plan/260920_round2_followups/050_r4_remainders.md @@ -0,0 +1,59 @@ +# R4 — what the remainders reach, and what they do not + +The lane brief put #4191 and #5180 in R4 "as far as the rework reaches". This records where that +line actually fell, with the evidence, so the next lane starts from a finding rather than a +re-investigation. + +## #4191 — reached: nothing. Found: a wrong stage in the shared vocabulary + +The resend gate does not consult the WebSocket projection. The post-header path excludes a +`isCodexWsUpstreamResponse` body on purpose: the WS transport settles its own ambiguous +failures and marks them non-replayable, and a second reader of one exchange is a defect, not a +recovery. So the durable threading the round-2 plan names is untouched here. + +The investigation did surface a real defect in the projection itself. +`classifyCodexWsFailure` in `src/server/responses/codex-ws-wire.ts` returns +`after-response-started` — which `CODEX_WS_FAILURE_PROJECTION` maps to `semantic-output` — +as soon as `relayedEvents > 0`. But `src/server/responses/codex-ws-exchange.ts` increments +`relayedEvents` for every non-metadata Responses event, and `response.created` is one: +`controlFrame` is set only when the metadata channel consumes the frame, not for lifecycle +events. `src/lib/request-failure-model.ts` puts `response.created` in `protocol-prelude` +and requires an output-bearing event for `semantic-output`. A WS failure carrying only a +created event therefore projects as committed output today. + +It is left here rather than fixed because the fix needs a counter the classifier does not have, +and `CodexWsStageRecord` is derived from `CodexWsFailureStage` by `Omit`, so adding one +lands in a persisted record whose read-back whitelist in `src/usage/log.ts` would reject every +row written before it. Adding the counter and `Omit`-ing it from the durable twin avoids that, +but the output-bearing predicate lives in `combo-stream-preflight.ts` and restating it in the +exchange is the class of duplication this round already paid for three times. It belongs with +the lane that threads `failureStage` / `failureCause` into the record, where both halves can +be written once. + +The SSE fallback the issue asks for stays out regardless. After `ws.send()` returns, a +fallback is a second physical send on another transport, which is a transport decision with its +own duplicate-inference policy — not a retry-gate change. + +## #5180 — reached: nothing. The symptom is upstream of this gate + +The reported failure is a key-auth `openai-chat` provider answering a bare 429. Traced on +current `dev`: `rateLimitRetryPolicyFor` returns null for every provider except the +OpenCode Go destination, so the same-target wait never runs; key rotation needs a pool of at +least two; and `fetchWithResetRetry` returns the first received HTTP response without +consulting its status. One send, 429 returned, which is exactly what the reporter saw. +`Retry-After` is forwarded to the client — synthesized as `2` for a bare retryable 429 by +`src/lib/retry-after.ts` — but the proxy never waits on it itself. + +None of that is an ambiguous-resend question: a received 429 is `headers-only` with cause +`rate-limit`, which the stage table already answers `permitted` and funds from the +`transient` class. It needs no grant and no override. What it needs is a policy default and a +process-wide cooldown that a single-key provider can write, and `keyCooldowns` cannot be +reused unchanged because both its identity and its write path require a multi-key pool. + +One adjacent accounting gap is worth recording for whoever takes it. On the generic adapter +path, `prepareAdapterExchange` passes `attempts` and `onSendsConsumed` to its retry helper +only when `transientRetryOn5xx` is configured. An unconfigured provider's initial send is +therefore recorded in the attempt log but never charged to the request-wide send counter. It is +bounded today — without `replaySafe` the reset helper makes exactly one send — so it is an +under-count rather than an amplification, and widening it without a suite to run is not a change +worth making blind. diff --git a/devlog/_plan/260920_round2_followups/060_r6_usage_models_table.md b/devlog/_plan/260920_round2_followups/060_r6_usage_models_table.md new file mode 100644 index 0000000000..e3bd25cd13 --- /dev/null +++ b/devlog/_plan/260920_round2_followups/060_r6_usage_models_table.md @@ -0,0 +1,95 @@ +# R6 — the Usage models table + +Status: OPEN until the pull request lands on `dev`. + +Four defects a user hit on the dashboard Usage tab, all in the models table. Three are layout; the +first is a reading the table gets wrong. + +## The hit rate was withheld from every provider that reports partial cache detail + +On the reporter's dashboard `gpt-5.6-sol` shows 2.8B cache hits and a hit rate of `—`. +`gpt-6-astra`, `k3[1m]`, `gemini-3.8-flash`, `gpt-5.6-luna` and `grok-4.6` show the same thing. The +reporter read it as the zero in the cache-writes column suppressing the rate. + +It is not the writes column. The summary is right and the dashboard was throwing its answer away. + +`calculateCacheHitRate` in `src/usage/summary.ts` averages cache reads over +`cacheObservedInputTokens` — the input tokens whose cache detail was actually reported — and +returns `null` when nothing was observed. That denominator is the #4546 contract recorded in +`structure/gui-and-management-api.md`: a synthesized zero and an unreported detail must not be +averaged as cache misses, or a pool that discarded every warm prefix reports a plausible hit rate. +A provider that reports reads and never reports writes is observed, and it has a rate. + +The dashboard then required that denominator to cover the row's **entire** input before it would +show the number: + +```tsx +model.cacheObservedInputTokens >= model.inputTokens ? model.cacheHitRate : null +``` + +One request in the row with no cache detail — a locally answered turn, an unreported usage record, +a row written by an older proxy — puts the denominator below `inputTokens` and blanks the column. +For a busy model that is every row, which is why six models with billions of measured hits all read +`—`. The gate arrived with the cache columns in #5268 and was never the server's rule. + +The fix drops the gate. The cell renders whatever the summary supplied, because the summary already +refused to supply a number it could not justify, and the coverage becomes a tooltip instead of a +reason to hide the value: `usage.cacheHitRate.partial` names the measured and total input tokens on +a partially observed row, `usage.cacheHitRate.unmeasured` explains the em dash on a row where +nothing reported cache detail. That row — no basis at all — is now the only one that shows `—`. + +The coverage sentence is carried twice: a `title` for a pointer, and an `sr-only` span so it is not +mouse-only. A `td` is not focusable and a `title` never reaches a keyboard or a touch screen, and a +cell whose whole point is to explain a number should not explain it to one input device. + +No server change. The denominator, the provenance split and the `null` are all correct as they +stand, and the structure doc that owns the contract stays accurate. + +## Column order + +`Model, Provider, Share, Tokens, API list-price`, then the per-request detail: +`Requests, Measured, Input tokens, Output tokens, Cache hits, Cache writes, Hit rate`. Identity +first, then the three figures a reader compares models on, then the evidence behind them. The +previous order buried share and price behind five cache columns. + +## Sideways scroll and pinned identity columns + +`.tbl` is `width: 100%`, so twelve columns divided the shell between them until eight-digit token +totals folded onto a second line. The models table is now `width: max-content; min-width: 100%` and +the shell scrolls sideways — `.tbl-wrap` was already `overflow-x: auto`, so nothing else had to +move. Model and provider are `position: sticky` at fixed widths so a row stays identifiable while +its numbers scroll; both offsets are one `var(--space-3)` step negative, the same trick the sticky +header plays with `top`, so a stuck cell repaints the scrollport padding it slides over. Under +720px the pinning stands down, because at that width two pinned columns cost more reading room than +scrolling the whole table does. + +Every selector is doubled as `.tbl.usage-models-tbl`. This file is `@import`ed from the top of +`styles.css`, so the whole of `styles.css` cascades after it, and a single class ties +`.tbl { width: 100% }` on specificity and loses on source order — the sizing contract reads as +applied and does nothing. The rules that already lived in this file buy the same margin with a +`.usw-section` prefix. The source-oracle case asserts the doubled form, because the single-class +version is the failure that looks correct. + +## The exclusion caption + +`(56 requests excluded)` shared a line with the amount and folded mid-phrase. It is a block now, so +the amount is the first line and the caption is the second. + +## Verification + +GUI change, so the screenshot gate applies and this lane cannot satisfy it: builds are not +permitted here, so no dashboard was rendered to photograph. The evidence offered instead is the +column order and cell layout written out above, the regression assertions below, and hosted CI. + +- `gui/tests/usage-custom-range.test.tsx` — the partially observed row now asserts `90%` where it + asserted `—`, with both tooltips, and the header sequence asserts the new order. +- `gui/tests/usage-layout.test.ts` — new source-oracle case binding the scroll, the pinned columns + and the block caption, so removing the stylesheet rules fails rather than degrading silently. +- Adversarial static review by a second agent, since nothing here may be executed: it reproduced + the cascade defect above independently and hand-evaluated the rendered cell arrays for all three + fixture rows against the new JSX. +- NOT RUN: `bun run test`, `bun test` on any single file, `bun run typecheck`, `bun run lint:gui`, + `bun run build:gui`, `bun install`, and any `ocx` execution. Hosted CI at the exact head is the + only execution evidence for this lane; GUI lint, typecheck and `gui` tests all run in the + `gates` job of `Cross-platform CI`, which a branch push does not trigger and the pull request + does. diff --git a/devlog/_plan/260920_round2_followups/070_widget_signing.md b/devlog/_plan/260920_round2_followups/070_widget_signing.md new file mode 100644 index 0000000000..9fbf7ec005 --- /dev/null +++ b/devlog/_plan/260920_round2_followups/070_widget_signing.md @@ -0,0 +1,70 @@ +# Widget extension signing + +Status: OPEN until the pull request lands on `dev`. + +The macOS app would have installed with no widget, and nothing in the build would have said so. + +## What was wrong + +`macOS.files` in `desktop/src-tauri/tauri.conf.json` puts `PlugIns/OpenCodexWidget.appex` into the +bundle. The Tauri bundler copies it and never signs it: `copy_custom_files_to_bundle` in +tauri-bundler 2.5.0 writes the file and does not add it to `sign_paths`, which only ever holds +`Contents/MacOS`, `Contents/Frameworks` and the `.app` itself. There is no `--deep` anywhere in +that path. Whatever signature `build-widget.sh` leaves is therefore the signature that ships. + +`build-widget.sh` left an ad-hoc one. Its signing branch keys on `MACOS_SIGN_IDENTITY`, and the +release workflow set that variable only on the `Build desktop bundles` step — the step *after* the +widget was built. `Build WidgetKit extension` carried no `env:` block at all, so the script always +took its `codesign --force --sign -` fallback, with `--timestamp=none` and no hardened runtime. + +macOS does not register an extension signed that way, and notarization rejects any Mach-O in a +bundle that lacks the hardened runtime. + +## What had not happened yet + +No release has shipped a macOS app. v2.58, v2.59 and v2.60 all carry zero desktop assets, and the +`APPLE_*` secrets were added to the repository hours after the last release ran. The signed branch +of this script has never executed. This is a defect found before its first victim, not one being +recovered from — the next release is where it would have landed. + +## The fix + +The script resolves its signing identity before the Swift build, so a release that holds Developer +ID material and somehow has no identity fails in a second instead of after a universal build, and +never leaves a half-built unsigned appex behind. `WIDGET_SIGN_REQUIRED=1` makes that refusal the +behaviour whenever the workflow holds a certificate; the ad-hoc branch stays for local builds. + +Signing walks every Mach-O the bundle actually contains, chosen by magic bytes rather than by name. +Today that set is one file. A suffix filter is the thing that fails silently when that stops being +true: a helper tool or an embedded dylib carries no extension to match, stays unsigned, and the +submission comes back "The binary is not signed with a valid Developer ID certificate" while the +containing bundle looks perfectly signed. Every signature now carries `--options runtime`, and the +script re-reads its own result and fails if the runtime flag is missing. + +The workflow imports the certificate into a temporary keychain before the widget is built, because +codesign resolves an identity through the keychain search list and Tauri does not build its own +keychain until the bundling step. Tauri re-adds itself to the same search list, so the two do not +collide, and a cleanup step deletes the keychain on any outcome. + +## Verification + +Run locally on macOS 27 with Xcode 27.0 and a real Developer ID in the keychain. + +- Signed path: `flags=0x10000(runtime)`, `Authority=Developer ID Application`, `TeamIdentifier` + set, secure timestamp present, `com.apple.security.app-sandbox` preserved, and + `codesign --verify --deep --strict` clean. `CFBundleShortVersionString` and `CFBundleVersion` + both resolve to the Tauri version. +- Ad-hoc path with no identity: `flags=0x10002(adhoc,runtime)` — the hardened runtime is now on + the local build too, so the two paths differ only in who signed. +- `WIDGET_SIGN_REQUIRED=1` with no identity: refuses in under a second, before the build. +- Bundle simulation: an `.app` holding the signed appex under `Contents/PlugIns`, signed the way + Tauri signs — inner executables, then the bundle, no `--deep` — keeps the nested Developer ID + signature, runtime flag, team identifier and entitlements intact, and + `codesign --verify --deep --strict` reports `--validated:...OpenCodexWidget.appex`. +- The same simulation over an appex left unsigned fails outer signing with + `In subcomponent: .../OpenCodexWidget.appex`. +- `tests/ci-workflows/release-desktop-scripts.test.ts` — 11 pass, binding the workflow wiring, the + magic-byte sweep, the hardened runtime and its self-check, and the refusal. + +End-to-end notarization of a full OpenCodex `.app` was not run; that needs a complete `tauri build` +and the release workflow is where it belongs. diff --git a/devlog/_plan/260920_round2_followups/090_closeout.md b/devlog/_plan/260920_round2_followups/090_closeout.md new file mode 100644 index 0000000000..7d491dff65 --- /dev/null +++ b/devlog/_plan/260920_round2_followups/090_closeout.md @@ -0,0 +1,66 @@ +# Round 2 closeout + +Status: CLOSED. Every R lane landed on `dev` and the branch is green again. This file records +what landed, the two incidents the round produced, and the rule the maintainer approved because +of them. + +## What landed + +| Lane | Pull request | Subject | +| --- | --- | --- | +| R1 | #5331 | Complete the provider-table transition on a paginated OpenAI home | +| R2 | #5338, #5351 | Keep the verification build out of updater signing, then assert the executable the bundle declares | +| R3 | #5332 | Make a failed browser launch and a failed account refresh visible (#5261) | +| R4 | #5342 | Rework #4942 and #4989 into one ambiguous-resend gate with one grant per request | +| R5 | #5347 | Derive the four telemetry pull requests from the landed recorder | +| R6 | #5333, #5345, #5353 | Usage table readability, WidgetKit Developer ID signing, keychain step location | + +#5342 is the one to notice. An earlier lane had ruled that #4942 and #4989 must not each buy an +independent replacement send for one logical request, and that they therefore belonged in a +single reworked change rather than two. That disposition closed as an implementation rather than +as a note. + +## Incident one: a default flip that no test could see + +#5271 removed a hostname test that decided the `developer` wire role. Deleting the inference was +right — a gateway proxying OpenAI accepts the role and the hostname cannot say so. The +replacement default was wrong in the other direction: forwarding to every destination assumed +each one accepts a standard role until an operator marks it. + +Three lane dispatches died on `400 role 'developer' is not allowed` within four seconds of +starting. Nothing in this repository saw it first, because every test in the tree was written +against the new default and passed. What broke was outside the tree. + +#5334 made the key tri-state with the unset state on the safe side, and then three more landings +were needed because three suites still asserted the forwarded role and the first sweep missed +them: the Lab conformance vector in `src/lab/` (#5341), a suite whose messages come from a +helper rather than a literal (#5344), and a suite about documents that reads the role only to +locate the turn (#5346). Searching for a string is not how you find what asserts a default; the +reliable question is which tests call the adapter at all. + +## Incident two: a verification step that had never run + +The `macos widget + bundle` job failed on `tauri build` because the updater public key is +committed and the private key is not in CI. #5338 scoped the opt-out to the verification build. +With that green, the Verify step ran for the first time and failed on its first line, silently, +because `test` prints nothing: it asserted `Contents/MacOS/OpenCodex` while Tauri keeps the Cargo +bin name unless `mainBinaryName` is set. #5351 reads `CFBundleExecutable` from the bundle instead. + +The same shape appeared once more at the end. #5345's test located a workflow step by name, #5339 +renamed that step while the branch was open, and the rename survived the merge while the assertion +did not. #5353 locates the steps by what they run. + +## The rule the maintainer approved + +A change that flips an existing default is a separate approval item before merge. Tests in the +tree are written against the new default and pass; what breaks is the set of real destinations +outside it, which exact-head CI cannot reach. Two instances landed on the same day — the 1 MiB +queue budget in #5182 and the role default in #5271 — and only the second was caught by a human +noticing that dispatch had stopped working. + +## Still open + +#5261 keeps two remainders: generic OAuth and key login still discard the launch result, and the +dashboard roster keeps last-good rows after a failed refresh. #4191 wants the SSE fallback and +#5180 the shared cooldown, both transport and routing changes. #5292 records the Logs page union +restatement. #2366, #3748, #3983 and #5063 remain deferred with reasons recorded on each. diff --git a/devlog/_plan/260920_round2_followups/r6-usage-table/r6-01-column-order.png b/devlog/_plan/260920_round2_followups/r6-usage-table/r6-01-column-order.png new file mode 100644 index 0000000000..eae536967f Binary files /dev/null and b/devlog/_plan/260920_round2_followups/r6-usage-table/r6-01-column-order.png differ diff --git a/devlog/_plan/260920_round2_followups/r6-usage-table/r6-02-pinned-scroll.png b/devlog/_plan/260920_round2_followups/r6-usage-table/r6-02-pinned-scroll.png new file mode 100644 index 0000000000..dc835c86e0 Binary files /dev/null and b/devlog/_plan/260920_round2_followups/r6-usage-table/r6-02-pinned-scroll.png differ diff --git a/devlog/_plan/260920_round2_followups/r6-usage-table/r6-03-narrow-fallback.png b/devlog/_plan/260920_round2_followups/r6-usage-table/r6-03-narrow-fallback.png new file mode 100644 index 0000000000..623f653e1a Binary files /dev/null and b/devlog/_plan/260920_round2_followups/r6-usage-table/r6-03-narrow-fallback.png differ diff --git a/devlog/_plan/260921_app_runtime_ownership/000_charter.md b/devlog/_plan/260921_app_runtime_ownership/000_charter.md new file mode 100644 index 0000000000..6e4b1fc88f --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/000_charter.md @@ -0,0 +1,56 @@ +# One runtime, one owner + +## What was asked + +Three things, in the user's words: + +1. Launching the app should stop the npm-installed runtime safely and bring up the app's own + runtime instead — on every platform. +2. Whatever permissions the app needs should be requested up front at first launch, the way + Karabiner does, rather than failing later. +3. Cmd+Q should leave the app in the menu bar and keep it running, not end the process. + +These are not three separate features. They are three faces of one question the codebase has never +answered: **who owns the running proxy, and how does ownership change hands.** + +## Why the current code cannot answer it + +The desktop shell decides ownership with a single boolean set once at startup. +`desktop/src-tauri/src/sidecar.rs` waits up to two seconds for anything to answer `/healthz` on the +discovered port; if something does, it returns `None` and the app is a guest, and if nothing does it +spawns the bundled sidecar and the app is the owner. `AppState::spawned_by_us` carries that answer +for the rest of the process lifetime. + +Every one of the user's three asks breaks on that boolean. + +- **Takeover has no representation at all.** There is no path from guest to owner. An existing npm + runtime is joined, never replaced, and nothing asks the user which they want. +- **Quit is `CommandChild.kill()`**, which is SIGKILL on Unix (`desktop/src-tauri/src/lib.rs`). The + CLI's own stop path restores client configuration, drains in-flight requests and clears state + files; the app's quit path does not wait for any of it. Cmd+Q reaching that code is not a + cosmetic problem — it is the destructive path firing on a keystroke the user expects to mean + "hide". +- **Permissions are never requested.** `first_run.rs` enables Start at Login once per install and + swallows every failure, which is the opposite of asking up front. + +## The gap this unit has to close + +Core already knows how to answer the ownership question. `src/server/proxy-liveness.ts` resolves a +live proxy from the pid record plus the runtime-port record, requires the `/healthz` body to +identify as opencodex, and carries back the version and the listener role. `src/service/state.ts` +records which launcher installed the service. The Rust side reimplemented a weaker version of the +same question — `discovery.rs` reads `runtime-port.json`, falls back to 10100 and then starts with +`--port 10100`, so a user on a custom `config.port` gets a different port than the one they +configured. + +So the work is mostly connection, not invention: give the shell the identity, readiness, graceful +stop and restart meaning that already exist in core, and add the one thing core does not have — +an explicit handover between two installations. + +## Status + +Interview open. The charter is recorded; the diff-level plan is not written yet because the +takeover semantics, the platform scope and the permission surface are still open questions with +the user. Evidence gathered so far is in `010_coexistence_findings.md` and +`020_windows_linux_findings.md`. + diff --git a/devlog/_plan/260921_app_runtime_ownership/010_coexistence_findings.md b/devlog/_plan/260921_app_runtime_ownership/010_coexistence_findings.md new file mode 100644 index 0000000000..fae8bbe1ea --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/010_coexistence_findings.md @@ -0,0 +1,73 @@ +# Coexistence with an existing npm installation + +External review of the desktop shell against an existing npm install, recorded here as claims plus +what this tree actually says. Findings are labelled **confirmed** when read directly out of the +source at `a499746395`, and **unverified** when the reasoning is sound but the behaviour was not +reproduced. + +## What happens today when an npm user launches the app + +| existing state | what the code does | consequence | +| --- | --- | --- | +| npm server already running | joins it, does not start the bundled one | the app ships a newer engine and dashboard than the one in use | +| npm server stopped, custom port | no runtime-port record, so 10100 is chosen | starts on a port the user did not configure | +| npm service autostarts at login | the app also enables its own Start at Login | two owners race at next login | +| terminal-only `OPENCODEX_HOME` | the app's environment lacks it, so a different home is read | looks like accounts disappeared | +| app started the server, then Quit | `kill()` on the child | in-flight requests and config restoration are cut off | +| server fails to start | the window and tray are created after `ensure_proxy` returns | nothing on screen explains the failure | + +## Confirmed in this tree + +- **Version split is invisible.** `sidecar.rs` accepts any successful `/healthz` and `lib.rs` then + navigates to that server's `/#/usage`. Nothing compares the app's version, the engine's version + or the dashboard's build. A user on 2.60.0 who installs a newer app keeps using 2.60.0 and has no + way to see it. +- **Port and home are guessed separately from core.** `discovery.rs` reads only + `runtime-port.json` and falls back to `DEFAULT_PORT = 10100`; `sidecar.rs` then passes + `--port 10100` explicitly rather than letting the CLI resolve `config.port`. `config_directory` + expands `~` itself instead of using the CLI's resolution. +- **Quit bypasses graceful stop.** `AppState::shutdown_child` calls `CommandChild.kill()`; there is + no `RunEvent::ExitRequested` handler, so Cmd+Q reaches it directly. +- **Ownership is a startup boolean.** `spawned_by_us` is set from whether a child was spawned, not + from whether the process now answering the port is that child. A slow-starting npm service that + wins the port after the spawn attempt would be recorded as app-owned. +- **Start at Login is enabled unconditionally on first run.** `first_run.rs` does not look for an + existing service, and it is not gated to macOS. +- **The updater does not coordinate a stop.** `updater.rs` installs and restarts with no drain of + an owned server first. + +## Confirmed, and already solved one layer down + +`src/server/proxy-liveness.ts` resolves liveness from the pid record and the runtime-port record, +requires the `/healthz` body to identify as opencodex, and returns the reported `version` and +`role`. It also carries deliberately tuned probe budgets — `START_OWNERSHIP_LIVENESS` exists +because a single unanswered 750ms probe was enough to start a duplicate proxy on Windows. The Rust +shell reimplemented the weaker form of this question and inherited the bug the comment describes. + +`src/service/state.ts` `stableLauncherEntry()` prefers the **recorded** `launcherPath` over a +fresh `PATH` walk, for a documented reason. The consequence for this unit is direct: a repair +driven from the app keeps pointing the service at the npm launcher. + +## Unverified + +- Whether the local management client can be diverted by system proxy settings. `ProxyClient` sets + a timeout and a user agent and does not disable reqwest's system-proxy default. No token exposure + was observed; the concern is that a management token rides a client that has not been forced + direct. +- Whether the 20 × 150ms start wait plus per-request timeouts produces a user-visible hang. The + arithmetic is real — `proxy.rs` sets a 4s timeout and `sidecar.rs` loops 20 times — but no + measurement was taken. +- Whether incremental local builds actually ship a stale engine. `prepare-sidecar.ts` reuses an + existing `dist/standalone/<target>/ocx` without checking that it came from the current source, + so a new app with an old engine is possible; it was not reproduced. + +## Recommended ordering from the review + +1. An identity-checked connection contract: pid, version, role and config home, custom port kept, + no connection to a foreign listener. +2. A visible startup and recovery surface: the app opens even when the server does not. +3. Coordinated stop for quit, stop and update: drain only what the app owns, never an external one. +4. Bundle consistency: app, engine and dashboard from one source, or an explicit build refusal. +5. Service coexistence, takeover and removal: one owner after login, and uninstall never touches an + existing npm install. + diff --git a/devlog/_plan/260921_app_runtime_ownership/020_windows_linux_findings.md b/devlog/_plan/260921_app_runtime_ownership/020_windows_linux_findings.md new file mode 100644 index 0000000000..509c70cf83 --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/020_windows_linux_findings.md @@ -0,0 +1,93 @@ +# Windows and Linux release readiness + +Second external review, covering what stands between this tree and shipping the desktop app on +Windows and Linux. Every claim below was re-read against `a499746395` before being recorded. + +## Confirmed by reading the tree + +**The Windows release job runs bash syntax under PowerShell.** +`.github/workflows/release.yml:347` — `Rename release assets` uses backslash line continuations +and `"$RELEASE_VERSION"` expansion, and carries no `shell: bash`. The workflow sets no top-level +`defaults.run.shell` either; only two other steps (lines 117 and 146) opt in explicitly. A Windows +runner defaults to PowerShell, so this step does not mean on Windows what it means elsewhere. + +**Checksums record a path the verifier cannot resolve.** +`release.yml:159` writes `sha256sum "dist/ocx-<version>-<target>.tar.gz" > dist/....sha256`, so +the checksum file contains the path `dist/ocx-...`. `release.yml:472` then verifies with +`cd dist/release && shasum -a 256 -c ./*.sha256`, which resolves that recorded path relative to +`dist/release` — a directory that has no `dist/` inside it. + +**Publishing does not depend on packaging.** +`release.yml:483` — `publish` declares `needs: validate-dispatch` only. npm publication and the +GitHub release can proceed while desktop packaging is failing, which is how a version becomes +public with no app attached. + +**`ocx.exe` is not recognised as an opencodex process.** +`src/config/process-state.ts` `isOcxCommandLine` matches +`(?:ocx|opencodex)(?:\.cmd)?` — no `.exe`. Meanwhile `scripts/build-standalone.ts:37` and +`desktop/scripts/prepare-sidecar.ts:39` both emit `ocx.exe` on Windows targets, and the sidecar is +copied as `ocx-<triple>.exe`. This predicate feeds pid identity, so the shipped Windows binary is +the one shape the identity check does not know. + +**The Windows app origin is not in the navigation allowlist.** +`desktop/src-tauri/src/window.rs` permits the `tauri` scheme and `http://127.0.0.1:<port>`, and +sends everything else to the external browser. Tauri serves the local app over +`http://tauri.localhost` on Windows, which lands in the external-browser branch. The policy +mismatch is confirmed; what the WebView2 first navigation actually does was not reproduced. + +**The service path filter does not cover the service directory.** +`.github/workflows/service-lifecycle.yml:7` and `release.yml:636` both key on `src/service.ts`. +The implementation is `src/service/**` — eleven files. A change to `launchd.ts` or +`windows-scheduler.ts` alone does not trigger the lifecycle workflow. + +**`desktop shell` does not exercise a real sidecar.** +`.github/workflows/ci.yml:1253` creates the sidecar with `: > "desktop/src-tauri/binaries/ocx-<triple>"` +and `chmod +x`, then runs fmt, clippy and cargo test. That is a useful Rust check and it is not +evidence that the bundled binary runs. + +**The Windows suite is out of the push gate by design.** +`ci.yml` gates `platform-windows` on `workflow_dispatch`, with a comment saying Windows +re-enters the gate once its tracked failures are fixed. So the review's observation is right, but +this is a recorded decision rather than an oversight. It still means a green push tells you nothing +about the Windows app. + +**Standalone binaries target modern x64 only.** +`scripts/build-standalone.ts` builds `bun-windows-x64` and `bun-linux-x64` with no baseline +variant. A CPU without the newer instruction set would fail as an immediate sidecar exit, which the +shell currently reports as a generic health failure. + +**Start-up failures are indistinguishable and can be slow.** +`proxy.rs` sets a 4s per-request timeout; `sidecar.rs` polls 20 times with 150ms sleeps and +discards the spawn event stream into `_events`. A failure mode where every probe times out is +arithmetically over a minute, with no exit code and no diagnostic surfaced. + +## Confirmed shape, consequence not reproduced + +- Stop treats an HTTP 200 with parseable JSON as success without reading `success: false`, and + does not wait for the backend's post-response drain before killing the child. +- Linux inherits the macOS menu-bar assumption: the window is created hidden and close always + hides, which on a desktop without a working tray leaves a running process with no way back in. +- Tray capability differs per platform — a title is macOS-only — so usage shown as tray title has + no Windows or Linux equivalent. +- `.deb` and AppImage are both shipped while the updater manifest is AppImage-shaped, and the + update code does not branch on install format. +- Rust reads `HOME` before `USERPROFILE`; Node's `homedir()` prefers `USERPROFILE` on Windows. + Under Git Bash the two can differ, which presents to a user as missing accounts. +- Windows code signing for the installer and executables is separate from the updater's minisign + key, and no Authenticode configuration was found. + +## Ordering the review proposes + +1. Release pipeline: the Windows shell, the checksum paths, and no publication before packaging. +2. Instance identity and config home: recognise `.exe`, one home for both sides, prove the + connected process is the spawned child. +3. A graceful shutdown coordinator shared by stop, quit and update. +4. Per-OS first run, window and tray behaviour. +5. Update target and install format separation, app versus CLI. +6. An installed-artifact gate: first run, coexistence, stop, update and uninstall from the real + MSI, deb and AppImage. + +The closing judgement is the one worth keeping: the Swift episode was not about Swift. Compiling, +bundling and registering each failed to prove running. The same gap is still open on Windows and +Linux. + diff --git a/devlog/_plan/260921_app_runtime_ownership/030_contradictions.md b/devlog/_plan/260921_app_runtime_ownership/030_contradictions.md new file mode 100644 index 0000000000..06c58d1d22 --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/030_contradictions.md @@ -0,0 +1,78 @@ +# Contradiction round 1 + +Three read-only lenses were run against the charter and the user's answers. They returned 22 +contradictions, 17 of them high. Recorded here so the plan has to answer them rather than +rediscover them. + +## The premise that did not survive + +**The verification hosts were miscounted, and that was my error.** The host I took for a Mac is +in fact the Windows machine, and the Linux one failed to resolve because I used the wrong short +name. With the right name and a permitted account all three platforms are reachable; see +`040_verification_hosts.md`. The contradiction that survives is narrower: the Linux box has no +`ocx` installed, so the npm side of the coexistence scenario does not exist there yet. + +**The sync button's silence is not a permission problem.** There are two different sync buttons and +they behave differently. The dashboard's model sync (`gui/src/pages/use-dashboard-data.ts:779`) +posts to `/api/sync` and renders both a success and a failure toast +(`dashboard-overview-sections.tsx:243`, backend at +`src/server/management/config-routes.ts:700`). The Integrations client sync +(`gui/src/pages/Integrations.tsx:73`) posts to `/api/machine/sync`, **ignores the status and the +body entirely**, and only clears a busy flag — so it cannot report anything, ever, no matter what +the server says. Neither path calls an OS elevation API. Elevating the app would not change either. + +## Ownership cannot be expressed yet + +- Ownership is the process-local `spawned_by_us` boolean; persisted service state has no consent + field and no desktop-owner field, so "asked once" and "permanent owner" cannot both be enforced + across an app restart. +- Disabling the npm service's autostart does not survive `ocx service repair` or `ocx update`: a + disabled registration still counts as installed, repair re-enables and restarts it, and the + recorded `launcherPath` still names the npm launcher. +- The app's Start at Login and the service's autostart are independent switches with no + mutual-exclusion invariant, so both can fire at the next login and race for the port. +- The app cannot prove the process answering the port is the child it spawned: any successful + health response after `spawn()` yields `Some(child)` and therefore app ownership. +- A plain `POST /api/stop` cannot perform the promised graceful takeover of a *managed* runtime. + The endpoint deliberately refuses launchd/systemd self-unload and the Windows respawn case unless + a receipt-backed `ocx stop` owns the teardown, so the app either stalls on 409 or bypasses the + drain and client-restore contract. +- Cmd+Q reaches `shutdown_child()` with no `ExitRequested` interception, and the updater's + `app.restart()` takes the same hard-kill path. + +## Elevation is the wrong tool + +Everything this app owns is per-user: the app spawns its sidecar as the current user, macOS uses +`~/Library/LaunchAgents`, Linux uses `systemctl --user`, and the Windows task is registered +`InteractiveToken` with `LeastPrivilege`. Windows already has a *conditional* elevation +fallback that only crosses UAC after an access-denied create — and which explicitly fails when a +*different* administrator supplies the credentials, because that account cannot read the staged +payload. An unconditional up-front prompt would therefore be both unnecessary and, for a standard +account, misleading. + +Separately, and worth fixing regardless: `ProxyClient` sends the admin token to `127.0.0.1` +without `no_proxy()`, and the pinned reqwest enables system proxies by default. + +## Evidence CI does not provide + +- Nothing installs an MSI, a deb or an AppImage anywhere in the repository; no `msiexec`, no + `dpkg -i`, no AppImage execution. +- The service-lifecycle workflow installs a service *from a source checkout* — it never models an + npm-installed runtime being handed to an installed app. +- That workflow's path filter names `src/service.ts` and omits both `src/service/**` and + `desktop/**`, so the ownership implementation can land green without any lifecycle evidence. +- `platform-windows` runs only on `workflow_dispatch`, by recorded decision. +- The Windows release lane builds an MSI and then runs POSIX syntax under PowerShell. +- `publish` depends only on `validate-dispatch`, so a version can go public while packaging fails. + +## Open assumptions + +1. Linux has a host (Ubuntu 24.04 GNOME) but no npm `ocx` on it, so the coexistence scenario has to be + staged there before it can be exercised. +2. `.deb` and AppImage are one "Linux" in the charter but two update contracts — the manifest + names AppImage only. +3. "Cmd+Q keeps it in the menu bar" has no literal equivalent on Windows or Linux; the portable + statement is that closing the window and quitting the window are different actions, and only the + explicit tray Quit ends the runtime. +4. Which of the two sync buttons the user pressed is not yet known. + diff --git a/devlog/_plan/260921_app_runtime_ownership/040_verification_hosts.md b/devlog/_plan/260921_app_runtime_ownership/040_verification_hosts.md new file mode 100644 index 0000000000..b2bb4b8db1 --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/040_verification_hosts.md @@ -0,0 +1,43 @@ +# Verification hosts + +Three machines cover the three platforms, all reachable over a private mesh. They are described +here by role only — the concrete names, addresses and accounts are operator detail and live in +scratch, not in this directory. + +| platform | what it is | npm-installed ocx already present | +| --- | --- | --- | +| macOS | the development machine, macOS 27, with the signed app installed | yes, a global install on `PATH` | +| Windows | Windows 11 25H2, reached over a POSIX shell layer | yes, both the launcher and its `.cmd` form | +| Linux | Ubuntu 24.04 LTS with a live GNOME session | no — only `npm` and `node` | + +## Why each one matters + +**The Windows box is the coexistence case, not a spare runner.** It already carries an +npm-installed `ocx` on `PATH`, which is exactly the situation the takeover has to handle. It is +also where the `isOcxCommandLine` gap becomes real: the npm launcher there is `ocx.cmd`, which +the predicate *does* match, while the app's bundled sidecar is `ocx.exe`, which it does not. Both +shapes exist on the same machine, so the predicate can be shown to be wrong rather than argued +about. + +**The Linux box has a real graphical session**, so the tray question can be answered rather than +assumed. It runs stock GNOME — both the Wayland and Xorg sessions are installed, and there is an +active seat — and stock GNOME ships **no tray** without an AppIndicator extension. That is +precisely the configuration the Windows/Linux review warned about: a window created hidden plus a +close handler that always hides leaves a running process with no way back in. `systemctl --user` +is running and FUSE is available, so the systemd user unit and the AppImage path are both testable +there. + +That box has no `ocx` yet, so the npm side of the coexistence scenario has to be staged before +the handover can be exercised there. + +## A note on what belongs here + +The first draft of this file named the mesh hostnames, an address and the SSH accounts that are +and are not permitted, and it was committed locally before being caught. `devlog/` is a public +directory in a public repository, so that was operator detail heading for publication. It has been +removed from the working tree and from history — nothing was pushed. + +Worth recording for its own sake: `bun run privacy:scan` passed on that draft. The scan covers +credentials and account identifiers, not mesh topology or login names, so passing it is not +evidence that a file is safe to publish. + diff --git a/devlog/_plan/260921_app_runtime_ownership/050_webview_dialogs.md b/devlog/_plan/260921_app_runtime_ownership/050_webview_dialogs.md new file mode 100644 index 0000000000..25916ecbb2 --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/050_webview_dialogs.md @@ -0,0 +1,58 @@ +# The app's webview has no JavaScript dialogs + +The reported symptom was the sidebar's proxy refresh orb: press it, nothing happens, no popup. It +is not that button, and it is not a permission. + +## What the button does + +The refresh orb beside the red power orb is `dash.codexRestart` — "Codex 모델 목록 새로고침" — and +its handler opens with a consent gate: + + if (!confirm(t("dash.codexRestartConfirm"))) return null; + +Every outcome after that is delivered by `alert()`: success, nothing-running, partial, HTTP +failure, unreachable, timeout, malformed. The confirm is deliberate and documented — stopping an +app-server can interrupt a Codex turn that is running right now, so the click is where the user +gives that consent. + +## Why nothing happens + +The app embeds wry 0.55.1 under Tauri 2.11.6. Its `WryWebViewUIDelegate` implements exactly three +`WKUIDelegate` methods: the file open panel, the media capture permission request, and window +creation for a navigation action. A search of the whole crate for +`runJavaScriptAlertPanel`, `runJavaScriptConfirmPanel` or `runJavaScriptTextInputPanel` returns +nothing. + +WKWebView does not display a JavaScript dialog when its UI delegate does not implement the matching +panel method. So inside the app `confirm()` returns `false` without ever drawing anything, and +`alert()` draws nothing at all. The handler takes its early return and the click is swallowed. +In a browser the same dashboard works, which is why this reads as "the app is broken" rather than +"the dashboard is broken". + +## It is a class, not a button + +13 `confirm` gates and 8 `alert` reports across the dashboard are inoperative inside the app. +Among them: + +- the sidebar's red power orb — `dash.stopConfirm` — so **stopping the proxy from the app does + nothing either**; +- removing a provider key, removing an account, removing a routing profile, deleting a custom + model, hiding a model, switching provider account mode; +- uninstalling the tray helper from the startup page; +- the memory observability confirmation; +- every result message the Codex refresh would have shown. + +Every one of these fails the same way: the user clicks, is silently declined, and sees nothing. +The destructive ones fail safe — nothing is destroyed — but the user cannot tell a refusal from a +no-op, and the two non-destructive ones (stop, refresh) simply never run. + +## What this means for the unit + +This is a third answer to "who owns the runtime", from an unexpected direction. The app is supposed +to become the owner, and the two controls that act on the runtime from inside the app — stop and +refresh — are both gated behind a dialog the app cannot draw. Any takeover consent prompt written +as `confirm()` would be auto-declined the same way. + +So the consent surface has to be real UI rather than a platform dialog, or the shell has to supply +the delegate methods. That choice belongs in the plan, not here. + diff --git a/devlog/_plan/260921_app_runtime_ownership/060_contradictions_round2.md b/devlog/_plan/260921_app_runtime_ownership/060_contradictions_round2.md new file mode 100644 index 0000000000..29015ede25 --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/060_contradictions_round2.md @@ -0,0 +1,62 @@ +# Contradiction round 2 + +Run after the four decisions were made: ask-once permanent takeover, keep the npm registration and +record an owner, per-user by default with elevation only at the point of failure, all three +platforms. Two lenses, 11 contradictions, 6 high. The weakest dimension going in was success +criteria, and that is where most of them landed. + +## Nothing here is observable yet + +- **"Ask once, then own permanently" has no durable state.** Ownership is recomputed each launch + from whether this process spawned a child; service state has no owner field and no consent field. + A restart can silently demote the app back to guest and no test would see it. +- **"Keep the registration, supersede it" has no marker either.** State records a launcher path and + a backend, and repair still prefers the recorded launcher. There is nothing to write the decision + into and nothing to assert against. +- **The quit criterion is currently inverted on macOS and undefined elsewhere.** Tray Quit calls + `app.exit`, `RunEvent::Exit` calls `shutdown_child`, and that kills the child. Windows and + Linux have no Cmd+Q equivalent named anywhere, so they could be called compliant without proving + the runtime survived their equivalent gesture. +- **No check observes an installed app taking over a real runtime.** Desktop CI builds against a + zero-byte sidecar; lifecycle CI installs a service from a source checkout and never stages an npm + install to hand over. + +## The dialog defect reaches further than one button + +- `stop-proxy.ts` treats *every* fetch exception as acceptance, so a failed stop and a successful + one are already indistinguishable before the missing alert. +- `window.prompt()` is used for alias editing on the provider and model pages. wry implements no + text input panel either, so those edits cannot be made in the app at all. +- Account, key, model, routing and tray-uninstall changes are all gated the same way. AGENTS.md + requires identity-affecting actions to sit behind an explicit gate; inside the app that gate + cannot be passed, so the action fails safe but also fails silently. + +## Two things that make this cheaper than it looks + +- **The fix already exists in the tree.** `OAuthTosWarningModal` and `ConsequenceDialog` are + in-page `<dialog>` components with real consent flows. The dashboard does not need a platform + dialog; it needs to stop using one. +- **The shell is already detectable.** `gui/src/lib/desktop-shell.ts` exists and is used today + only to reroute external links, so there is a seam to branch on if a branch is wanted rather than + a straight replacement. + +## Why CI could never have caught it + +The existing GUI tests encode browser dialogs as available. `codex-stale-banner-dom.test.tsx` +stubs `confirm()` to true and `alert()` to a no-op; `memory-observability-card.test.tsx` forces +confirmation; `app-stop.test.ts` asserts that `alert()` *exists*. Each of those is reasonable on +its own and together they make the desktop failure invisible. A regression test for this has to +assert the absence of platform dialogs, not stub them in. + +## Open assumptions carried forward + +1. The Linux box has no npm `ocx`, so the coexistence scenario has to be staged there before it + can be exercised. +2. `.deb` and AppImage are one "Linux" in the charter but two update contracts; the manifest + names AppImage only. +3. "Cmd+Q keeps it in the menu bar" has no literal equivalent on Windows or Linux. The portable + statement is that closing a window and quitting the app are different actions, and only the + explicit tray Quit ends the runtime. +4. Whether the takeover consent becomes an in-page dialog or the shell gains the delegate methods + is a plan decision, not an interview one. + diff --git a/devlog/_plan/260921_app_runtime_ownership/070_decisions.md b/devlog/_plan/260921_app_runtime_ownership/070_decisions.md new file mode 100644 index 0000000000..10201892f1 --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/070_decisions.md @@ -0,0 +1,76 @@ +# Decisions + +Settled by the maintainer, then by an automated decider round in which each fork was given to an +independent reader with the evidence and the trade-offs and asked to choose one option and own its +cost. Each entry records the choice and the cost that was accepted with it, because the cost is the +part a later reader will want. + +## Fixed by the maintainer + +| | decision | +| --- | --- | +| Takeover | Ask once on first discovery of an existing npm runtime. On approval the app is the permanent owner. | +| The npm install | The user's service registration is kept, never deleted. A durable owner marker supersedes it and repair and update must respect it. | +| Elevation | Per-user by default. Elevate only at the point a per-user operation actually fails, which is what the Windows path already does. | +| Platforms | macOS, Windows and Linux, with a verification machine for each. | +| Quit | Cmd+Q leaves the app in the menu bar with the runtime alive. | + +## D1 — the consent and feedback surface + +**Every platform dialog leaves the dashboard.** `confirm`, `alert` and `prompt` are removed +from `gui/src` and replaced with the in-page dialog and feedback components already in the tree, +with a source guard so they cannot come back. + +The decider checked the other two platforms rather than assuming: wry leaves WebView2's script +dialog setting untouched and WebView2 enables script dialogs by default, and WebKitGTK shows +dialogs through its default handler. So implementing the macOS delegate would repair one platform +and leave the product's consent UI platform-dependent. **Cost accepted:** the macOS shell still +cannot draw an accidental future platform dialog, so repository code has to keep enforcing the ban +statically. + +## D2 — what ends the runtime + +**Window close and the OS quit gesture both hide to the tray, on all three platforms. Only the +explicit tray Quit ends the app**, and that path drains an app-owned runtime before exiting. + +Observable per platform: close and Cmd+Q on macOS, close and Alt+F4 on Windows, and the window +manager's close on Linux all leave the window hidden with both pids alive and the window +reopenable from the tray; tray Quit drains in-flight work and then ends both. **Cost accepted:** +on a Linux desktop with no tray this strands the user — which is D6. + +## D3 — where ownership lives + +**Both records, with the shared service state authoritative.** The service install state gains an +owner, an install id and a consent generation, written compare-and-swap and preserved by every +writer; the app keeps its own install identity so a reinstalled app can tell its own prior consent +from another installation's. + +The decider rejected the single-record option for a specific reason: an install id stored only in +the shared record gives the app no independent value to compare against, so a reinstalled app +cannot tell whose consent it inherited. **Cost accepted:** two records mean mismatch and orphan +recovery, and losing app-local state can force explicit re-consent, because the two writes cannot +be one atomic act. + +## D4 — how an existing managed runtime is stopped + +**The app shells out to its own bundled `ocx stop`.** The receipt-backed teardown, the drain, the +Windows respawn verification and the client-configuration restore then run exactly as they do from +a terminal, and the shell reads the exit code and the output. + +The alternatives were disqualified by the same fact: launchd and systemd can terminate the request +handler during self-unload, and the Windows respawn window can only be verified after that process +exits, so an in-process management endpoint cannot own its own teardown. **Cost accepted:** the +takeover path now depends on spawning a CLI and surfacing a human-readable result rather than a +structured one. + +## D5 — how the shell resolves the port, the home and liveness + +**It stops resolving them.** The shell asks the bundled CLI through a machine-readable resolve +command, with a strict timeout, and if the binary is slow or missing it opens a local recovery UI +and refuses to guess a home, a port or a liveness verdict. + +The reason is in the comments of the code it would otherwise duplicate: the tuned probe budgets in +the liveness path exist because small divergence produced duplicate proxies, twice. **Cost +accepted:** every launch pays one bounded process start, and startup now depends explicitly on the +bundled binary being executable — which is also why D7's recovery window has to exist first. + diff --git a/devlog/_plan/260921_app_runtime_ownership/080_decisions_round2.md b/devlog/_plan/260921_app_runtime_ownership/080_decisions_round2.md new file mode 100644 index 0000000000..e5f504c7aa --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/080_decisions_round2.md @@ -0,0 +1,65 @@ +# Decisions, round 2 + +## D6 — Linux without a tray + +**Detect real tray availability and branch.** Where there is no usable tray the window is shown on +launch, close really closes and quits through the graceful drain, and hide-to-tray is simply not +used. Where there is a tray, D2 applies unchanged. + +The decider found why construction success is not enough: the pinned Linux backend creates an +`AppIndicator` and returns success without checking for a StatusNotifier watcher, so +`tray::install` succeeding proves nothing about whether an icon is reachable. **Cost accepted:** +Linux behaviour becomes session-dependent, so both modes have to be supported and verified, and D2 +gains an explicit no-tray exception. + +## D7 — the startup surface + +**The window is created and shown first, always.** Resolve, liveness, takeover consent, start, +permission registration and the Start at Login decision all run inside it as named states under one +overall deadline, with a retry, the child's exit code and a copyable diagnostic. A launch that came +from login autostart starts hidden; that is the only difference. + +The retry surface already exists in `desktop/ui` — it is just created hidden and never promoted +into a real state machine. **Cost accepted:** an ordinary manual launch now shows a window even +when everything succeeds immediately. + +## D8 — the Linux update contract + +**Both formats update in place.** The pinned updater already branches between AppImage and +`.deb`, detects dpkg ownership, validates the payload and installs through package-manager +elevation, and Tauri exposes the bundle type embedded at packaging time, so the app can select the +right manifest entry rather than guess. The current mismatch is that both artifacts are collected +but only the AppImage is published as a Linux updater target. + +**Cost accepted:** Linux release and verification become a two-format matrix, and a `.deb` update +asks for package-manager authorization — which is consistent with the elevation rule, because the +prompt comes only after the user chooses Install. + +## D9 — what gates a desktop release + +**Fix the three pipeline defects, and add one installed-artifact smoke gate** that runs on a +machine per platform: install the real artifact, launch it, prove which runtime it connected to, +exercise takeover, quit, and confirm the runtime survived or drained as specified. Publication +waits for packaging and for that smoke. + +The release contract becomes package, then install-smoke on all three platforms, then publish and +attach, with a missing platform result blocking publication. **Cost accepted:** publication now +depends on three stateful GUI machines, each run needs strict rollback and cleanup, and AppImage +update behaviour, full uninstall coverage and the zero-sidecar PR job stay follow-up. + +## The observable contract this produces + +Every decision above was required to state what a test or a screenshot must show. Collected: + +- A staged npm runtime on a non-default port is drained, its registration is still present + afterwards, the desktop install id is recorded as owner with exactly one consent-generation + increment, and `/healthz` reports the bundled sidecar's pid and version on the preserved home + and port. +- Window close and the platform quit gesture each leave both pids alive with the window reopenable + from the tray; tray Quit during an in-flight request lets that request finish and then ends both. +- A second launch does not ask for consent again. +- On a Linux session with no usable tray: the dashboard appears on first launch, no tray icon is + claimed, and closing the window drains and exits rather than hiding. +- An older AppImage updates without elevation and keeps its path; an older dpkg install asks for + authorization only after Install is chosen, and cancelling leaves the old version in place. + diff --git a/devlog/_plan/260921_app_runtime_ownership/090_lanes.md b/devlog/_plan/260921_app_runtime_ownership/090_lanes.md new file mode 100644 index 0000000000..a4ff13602f --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/090_lanes.md @@ -0,0 +1,85 @@ +# Lanes + +Nine decisions, split into work that can proceed in parallel. Each lane is one branch, ordered +commits, one pull request to `dev`. No native stacks, no child PR chains. + +Lane order matters in two places only. **A** publishes the CLI resolve and stop contracts that **B** +consumes, and **C** must land before **B** wires the stop shell-out, because every service-state +writer has to become preserve-and-swap before a second writer exists at all (see R4 in +`100_resolutions.md`). Everything else is independent. + +## A — the CLI contract the shell will consume (D5, D4) + +A machine-readable resolve that returns the config home, the effective port and the liveness +verdict, and a stop invocation the shell can drive and read. Both are thin surfaces over +`src/config/paths.ts`, `src/server/proxy-liveness.ts` and the existing receipt-backed stop in +`src/cli/` — the point is to expose what already exists, not to reimplement it. + +Owns: the new CLI verb and its schema, and the contract tests. Must not change the meaning of the +existing stop path. + +## B — the shell: startup, quit, tray, consent plumbing (D7, D2, D6) + +The window is created and shown first and startup runs inside it as named states with one deadline, +a retry, the child's exit code and a copyable diagnostic; login autostart starts hidden. +`ExitRequested` is intercepted so close and the quit gesture hide, and only tray Quit drains and +exits. Linux detects real tray availability and, where there is none, shows the window and lets +close mean close. + +Owns: `desktop/src-tauri/src/` and `desktop/ui/`. Consumes A's contracts. Blocked on A only for +the resolve and stop call sites; the quit and tray work can start immediately. + +## C — durable ownership (D3) + +The service install state gains an owner, an install id and a consent generation, written +compare-and-swap and preserved by every writer; repair and update learn to respect it; the app +keeps its own install identity beside it. + +Owns: `src/service/` and `src/update/`. This is the lane with the widest reader list, so it +lands early and alone. + +## D — the dashboard consent surface (D1) + +`confirm`, `alert` and `prompt` leave `gui/src` entirely, replaced with the in-page dialog +and feedback components already in the tree, with a source guard so they cannot return, and with +tests that assert the absence of platform dialogs rather than stubbing them in. + +Owns: `gui/`. Independent of every other lane. This is also the lane that unblocks the takeover +consent prompt, since a `confirm`-based prompt would be auto-declined. + +## E — the release pipeline (D9, part one) + +The Windows shell override, the checksum path, and the dependency graph so publication cannot +precede packaging. Plus the service path filter that names one file while the implementation is +eleven, and the `.exe` the process predicate does not recognise. + +Owns: `.github/workflows/` and `src/config/process-state.ts`. Touches release automation, so it +carries the explicit security review the repository requires. + +## F — the installed-artifact gate (D9, part two) and the Linux update contract (D8) + +The smoke that installs the real artifact on each platform, launches it, proves which runtime it +connected to, exercises takeover and quit, and reports. Plus publishing both AppImage and `.deb` +as distinct updater targets and selecting the right one from the bundle type. + +Owns: `desktop/scripts/` and the new workflow. Registering self-hosted runners is a maintainer +action outside the diff; the lane delivers the workflow and the drivers. + +## What every lane owes + +- A focused regression test near the existing tests for that subsystem, driven red once. +- Any new test file registered in **both** `scripts/test-layout/layout.json` and + `tests/fixtures/test-layout-expected.json`. +- No new line in a file already at its size cap; move the case to a sibling file instead. +- Exact-head CI read at the SHA, with skipped and cancelled jobs named rather than counted green. +- English in every public artifact, and no host names, addresses, accounts or absolute user paths + anywhere in the tree. + +## Who is running each lane + +C, B and D run on one model and A, E and F on another, deliberately split so a systematic blind +spot in either does not cover all six. The split as dispatched is not the one that was intended: +A, E and F went out on a third model by a dispatch error on my part. By the time it was caught, +all three had substantial work in flight — a dozen modified files between them and two commits on +F — so they were left alone rather than restarted. It is recorded here because a later reader +comparing lane quality should know the split was not what the plan says. diff --git a/devlog/_plan/260921_app_runtime_ownership/100_resolutions.md b/devlog/_plan/260921_app_runtime_ownership/100_resolutions.md new file mode 100644 index 0000000000..6014d97fa5 --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/100_resolutions.md @@ -0,0 +1,75 @@ +# Resolutions + +A final scan over the decided set returned sixteen items. Most were the unit's own premise restated +— "the code does not do this yet" is not a contradiction between decisions. Six were real, and each +is resolved here so no lane has to guess. + +## R1 — no tray and login autostart (D6 against D7) + +D6 shows the window where there is no usable tray; D7 starts hidden when the launch came from login +autostart. A no-tray login launch satisfies both rules and they disagree. + +**Resolved: tray availability wins over launch origin.** With no usable tray there is nowhere to +hide, so the window is shown even on a login launch. The hidden start is a property of *having a +place to be hidden in*, not of how the process was started. + +## R2 — update restart against tray-only quit (D2 against D8) + +D2 says only the tray Quit ends the app. An update installs and restarts. + +**Resolved: an update restart is a coordinated restart, not a quit.** It runs the same graceful +drain as tray Quit, then comes back. What D2 forbids is an *uncoordinated* exit — the current +`app.restart()` straight into the hard kill — not the existence of a restart. The exit path must +be able to tell a coordinated restart from a user quit gesture, which is already in D2's blast +radius. + +## R3 — AppImage update verification (D8 against D9) + +D8 makes both Linux formats update in place. D9 accepted deferring AppImage update behaviour as +follow-up. Those cannot both hold. + +**Resolved: D8 wins and D9's deferral is withdrawn.** If both formats carry an update contract, +the gate has to exercise both, so update verification for AppImage and `.deb` moves into lane F's +scope rather than after it. A gate that cannot see one of the two promised paths is not a gate. + +## R4 — ownership writes against the external stopper (D3 against D4) + +D3 wants compare-and-swap ownership fields. D4 has the app drive an external `ocx stop`, and the +service-state writers today reconstruct the whole record and overwrite it, so a concurrent repair, +update or stop would drop the ownership fields entirely. + +**Resolved, and it fixes the lane order.** Lane C lands **before** lane B wires the stop shell-out. +C's scope explicitly includes converting every writer in `orchestration.ts`, `launchd.ts`, +`systemd.ts`, `windows-ops.ts`, `windows-scheduler.ts` and `repair.ts` from +reconstruct-and-replace to preserve-and-swap, with a revision check, before any new writer exists. +A preserved field is not optional politeness here; it is the only thing that makes consent durable. + +## R5 — the dialog guard must ban the call, not the word (D1) + +A lexical ban on `confirm`, `alert` and `prompt` would reject legitimate code: an admin-token +helper, a `confirm()` method on a session object, and an executable sample string that contains +the word. + +**Resolved: the guard matches the global call form**, not the identifier. `window.confirm(` and a +bare `confirm(` at call position are banned; a method call on a receiver, a property name and a +string literal are not. The guard has to be driven red against a real global call and green against +each of those three legitimate shapes before it counts. + +## R6 — two constraints every lane inherits + +**Security review.** Lane E and lane F change GitHub Actions and release automation, which the +repository requires to have explicit security review. That is a gate on those lanes landing, not a +thing to discover at merge time. + +**The size ratchet.** `gui/src/pages/Models.tsx` has one line of headroom against its cap, and +lane D has to touch it. Additive dialog code there fails CI for that branch and for every branch cut +from `dev` afterwards. Lane D extracts to a sibling file and registers it in both test-layout maps +rather than adding a line. + +## Remaining open assumptions + +1. Registering self-hosted runners for the installed-artifact gate is a maintainer action outside + any diff; lane F delivers the workflow and the drivers and stops there. +2. The Linux verification machine has no `ocx` installed, so the npm side of the coexistence + scenario has to be staged before the handover can be exercised there. + diff --git a/devlog/_plan/260921_app_runtime_ownership/110_reaudit.md b/devlog/_plan/260921_app_runtime_ownership/110_reaudit.md new file mode 100644 index 0000000000..3e6097d6bf --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/110_reaudit.md @@ -0,0 +1,124 @@ +# External re-audit of the lane branches + +Two independent reviews read the pushed lane branches at fixed SHAs and reported on the same day +the lanes were opened. Both agree the direction is right and both refuse to call it shippable. The +distinction they draw is the one worth keeping: **"better than before" and "safe in the failure +path" are not the same verdict.** + +What they credit as genuinely fixed: the Windows packaging shell, the checksum path, packaging +before publication, `.exe` process identity, the window being created before the runtime starts, +the removal of the direct `child.kill()`, the StatusNotifier probe, and the replacement of the +platform dialogs. Those are not re-listed as defects. + +## P0 — the installed gate can destroy a real installation + +`desktop/scripts/installed-gate.ts`. The preflight detects an existing service, an existing +state file in the default home, or a running app, and refuses to verify. But refusal only sets a +flag; the `finally` block then runs its cleanup unconditionally, killing processes matching the +app name and attempting a service and artifact uninstall — **including on Windows, where it can +reach the MSI removal path without the test ever having installed anything.** + +So the very situation that makes the gate refuse is the situation in which it acts. A `return` +inside `try` does not help: `finally` still runs. The preflight has to complete outside the +block that owns destructive cleanup, or the cleanup has to be limited to the exact pids, service +registrations and install results this run recorded for itself. + +Completion condition: **a run that refuses because it found an existing app, service or state +makes zero mutating calls, cleanup included.** + +## The update path still does not drain first + +The pinned updater's Windows install implementation ends in `process::exit(0)`. The lane calls +`download_and_install()` and only then asks the exit coordinator to restart, so on Windows that +second call is not reached. Removing the direct kill was real progress; it did not put the Windows +in-app update on the coordinated path. + +The order has to be: download and verify the signature, re-confirm who owns the current runtime, +drain and confirm the child actually exited, **then** install. A failed drain must refuse the +install rather than proceed. + +## A failed drain is still recorded as drained + +The exit state machine logs a drain failure and then calls the same completion path, so both a +successful and a failed drain end in the exiting or restarting branch. For a user pressing Quit +that is a defensible trade — better to leave a runtime than to refuse to close. **For a coordinated +restart it is not the same judgement.** A failed stop followed by a restart means the new app +re-attaches to the old runtime while the user believes they are on the new version. + +`DrainFailed` and `OwnershipUnknown` need to be states the restart path refuses, separately from +what the quit path tolerates. + +## Ownership is computed outside the lock it is written under + +The writer resolves ownership **before** taking the lock, then takes the lock, reads the current +record, and preserves the ownership it read earlier. A revocation that lands in between is +overwritten by the stale value. The revision check does not prevent this: the read is fresh and the +value being written is not. + +Two more in the same file: the state file is written in place rather than written and renamed, so +an interrupted write leaves half a document; and the lock is reclaimed on mtime alone, with no +holder identity, so a slow writer can delete a lock another process now owns. + +A third, and it is a different question from the CAS: **the record API takes an owner and an install +id, which cannot express "is the generation the user consented to still the current one".** An +internal retry that succeeds against a newer record has silently applied the consent to a different +subject. + +## Not stopping is not the same as safe to replace + +When ownership is unknown the update path skips stopping the runtime and skips refreshing the +service — but still proceeds to replace the package. If the live process is running out of the +files being replaced, that is a file lock on Windows and a mixed on-disk version elsewhere. + +Three decisions have to be separated: may the package be replaced, may the runtime be stopped, may +the service be restored. Unknown should block the first, not only the second and third. + +## The old CLI on the user's machine is not retrofitted + +The shipped 2.60.0 launcher calls the old `stop` before replacing the package whenever a service +or runtime record exists, and it knows nothing about an ownership field. So a user who takes +ownership in the app and then runs `ocx update` from the npm install on their `PATH` gets the +old teardown first. The protection added here is the new CLI's protection; it cannot reach backward. + +Taking permanent ownership therefore has to check the managing CLI's compatibility first, and +either upgrade it with consent or withhold the takeover and say why. + +## Smaller, each concrete + +- The resolve verb uses the default probe budget, one attempt at 750ms, and reports a timeout as + `not-found`. The start-ownership path uses 1500ms three times for exactly this reason. Alive, + absent-proven and unknown need to be three answers, and unknown must not authorise a new runtime. +- The resolve verb passes through the CLI root's automatic shim restore, so a read-only lookup made + to populate a consent screen can cause a repair side effect first. +- `if (await deps.handleStop())` still reads a now-object return as a boolean, so a failed stop + prints the downtime warning. +- The dashboard's stop client maps every fetch exception to accepted. Accepted, rejected and unknown + are different, and unknown needs a follow-up read rather than an assumption. +- A consent dialog can outlive its subject: the target can change or the surface unmount while it is + open, and the request is then sent against the captured closure. +- The dialog guard skips template literals wholesale, so `${window.confirm("...")}` inside one is + a real call it does not see. +- Attaching to a different proxy does not reset the ownership flag, so a retry that lands on a + foreign runtime can still send an owner-only stop to it. +- Tray availability is recorded from the host probe before `tray::install` runs, and an install + failure only logs. Host present, icon registered and currently reachable are three different + facts. +- The startup deadline does not wrap the registration that runs before the resolver, and the + existing-proxy budget is counted from process start, so registration can consume it. +- The Windows app origin is still not in the navigation allowlist. +- The local management client needs redirects refused and instance identity confirmed before the + token is sent, not only the system proxy disabled. +- Publication still precedes checksum, signature and manifest validation, because that validation + lives in the attach step that depends on publish. Packaging-before-publish closed a narrower gap + than the one that remains. +- The gate driver and the ownership record disagree on schema, install the wrong package spec, and + re-hardcode the macOS executable name that was removed once already. Its second-launch check + observes single-instance behaviour rather than a real relaunch, and the deb update is only + verified through cancellation, never through a successful install. + +## The three sentences both reports converge on + +**An unknown result is not turned into an absence or a success. The subject the user approved is +confirmed to be the subject being changed. An update begins only after the correct runtime is +confirmed stopped.** + diff --git a/devlog/_plan/260921_app_runtime_ownership/120_gate_runbook.md b/devlog/_plan/260921_app_runtime_ownership/120_gate_runbook.md new file mode 100644 index 0000000000..ef2d50eb13 --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/120_gate_runbook.md @@ -0,0 +1,90 @@ +# Operating the installed-artifact gate + +The gate from D9 part two lives in `.github/workflows/desktop-installed-gate.yml` with its +drivers under `desktop/scripts/`. It installs the real desktop artifact on one GUI machine +per platform, drives the ownership contract from `080_decisions_round2.md`, and uploads a +JSON report per job. This page is the operator procedure for its first live run. Registering +the runners is a maintainer action; nothing here is automated yet. + +## Runners + +One self-hosted runner per platform, each with a live GUI session (the gate drives real +windows and tray menus): + +| label | machine needs | +| --- | --- | +| `opencodex-gate-macos` | macOS with a desktop session; the app's tray automation uses System Events, so the runner account needs Accessibility permission for `osascript` | +| `opencodex-gate-windows` | Windows with an interactive session; PowerShell and `msiexec` (system), Git Bash for the workflow shell | +| `opencodex-gate-linux` | A desktop session with a working tray (an AppIndicator/StatusNotifier extension on GNOME), `systemctl --user`, and non-interactive dpkg rights for install/remove (`sudo -n dpkg -i/-r`) | + +Every runner also needs `gh` (artifact download) and `npm`/`node` (the gate stages the npm +runtime itself). Bun comes from the workflow's own setup action. + +Two protections are part of the design, not optional hardening: + +- Restrict each runner group so only this workflow can land on these machines. +- Add required reviewers to the `opencodex-desktop-gate` environment. Every dispatch then + waits for a maintainer approval. The jobs check out the protected `dev` branch for the + driver code — never the dispatched ref — so an approval is a review of inputs, not of + smuggled code. + +## GUI hooks + +OS automation cannot reach everything the contract needs: the in-page consent dialog, the +Windows and Linux tray, and the deb update's elevation prompt. The operator installs audited +executable files in a hooks directory on each runner and sets the repository or organization +variable `OPENCODEX_GATE_HOOKS_DIR` to that directory. Dispatch inputs then select hooks by +file name only: + +| input | the hook answers | +| --- | --- | +| `consent-hook` | the takeover consent prompt (accept) | +| `tray-click-hook` | left-clicks the tray icon | +| `tray-quit-hook` | opens the tray menu and chooses Quit | +| `tray-check-hook` | chooses Check for Updates (Linux update phases) | +| `tray-install-hook` | chooses the enabled Install update item (Linux update phases) | +| `elevate-accept-hook` | answers the deb update's authorization prompt, driving the accept path | + +macOS has built-in defaults for the tray actions; Windows and Linux have none on purpose — +without a hook, the phase that needs it fails with a diagnostic rather than guessing. Hook +files run directly, never through a shell, and the workflow accepts names, never command +text. + +## Running it + +The gate is `workflow_dispatch` only. Inputs: + +- `version` (required): the release whose artifacts are verified, e.g. `2.62.0`. The release + must already exist with its desktop assets and updater signatures attached. +- `from-version` (required): an older release, strictly lower by semver. It stages the npm + runtime that the app takes over and, on Linux, is the version the update phases start + from. +- the hook names above, as needed per runner. + +Artifacts come from the GitHub release itself, so the sequence is: publish (or draft) the +release, then dispatch the gate against it. Wiring publication to wait for a green gate is +lane E's release.yml surface and is tracked there. + +A run that finds the machine dirty refuses before touching anything: an existing service +registration, a default-home state file, a running app, or a dormant installed package all +fail `preflight-isolation`, and a refused run makes zero mutating calls. Clean the machine +or use another one; do not retry until the probe goes green. + +## Reading the report + +Each job uploads `installed-gate-report-<job>` (also on failure). The JSON lists one entry +per phase with `status`, `detail` and `evidence`, in contract order: + +`preflight-isolation`, `runner-readiness`, `stage-npm-runtime`, `install-artifact`, +`launch-and-take-over`, `runtime-identity`, `close-gesture`, `quit-gesture`, +`relaunch-consent`, `tray-quit-drains`, `update-verify` (Linux only), `cleanup`. + +The first failing phase stops verification; cleanup always runs and its own failure fails +the report. `ok` is true only when every phase ran and passed, so a report that crashed +midway is red even if everything recorded is green. When a phase fails, its `evidence` +carries the observed state (healthz bodies, ownership records, elevation sightings, digests) +needed to tell a product defect apart from a runner problem. + +Until the takeover and consent lanes land, the takeover and gesture phases fail against +current behavior — that is the gate doing its job, and the report names which contract item +failed. diff --git a/devlog/_plan/260921_app_runtime_ownership/120_install_verification.md b/devlog/_plan/260921_app_runtime_ownership/120_install_verification.md new file mode 100644 index 0000000000..e70c17b4ad --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/120_install_verification.md @@ -0,0 +1,84 @@ +# 120 — Installed-artifact verification on a real Linux desktop + +First run of the D6/D8 surface against an actual GNOME desktop session rather than a +unit test. The machine is described by role only: a GNOME 24.04 workstation on an X11 +session, with a user-level npm install of the proxy already listening on the default +port, and no desktop package installed before this run. + +The tree under test is `dev` after lanes C, A, E, F and the runtime-ownership follow-up +landed. Lane B (desktop shell) and lane D (consent surface) were **not** in the tree, so +everything below is the pre-B baseline, not a verdict on them. + +## The documented local build produces nothing on Linux + +`desktop/README.md` tells a contributor to run `bun run build:local`. On Linux that asks +for `appimage,deb` in that order. AppImage bundling fails: + + Bundling OpenCodex_2.61.0_amd64.AppImage (...) + failed to bundle project `failed to run linuxdeploy` + Error failed to bundle project `failed to run linuxdeploy` + +The failure is fatal for the whole invocation, and because AppImage is requested first, +the deb is never attempted. The bundle directory is empty afterwards. A contributor +following the README gets no installable artifact and an error that names a tool they +did not invoke. Installing `libfuse2t64` and setting `APPIMAGE_EXTRACT_AND_RUN=1` did not +change the outcome, and linuxdeploy's own diagnostics are swallowed by the bundler. + +Requesting the deb on its own succeeds in 43 seconds and produces +`OpenCodex_2.61.0_amd64.deb`, which installs cleanly through `dpkg -i` and registers +`open-codex 2.61.0` with the desktop-file and icon triggers. + +Two things follow. The local path should order the Linux bundles so that a failure in the +optional format cannot destroy the installable one, and it should surface the bundler's +stderr instead of a bare "failed to run" line. This is separate from D8: the release +workflow builds the AppImage on its own runner image and is not known to be affected. + +## No tray host means no visible application at all + +The session has no `StatusNotifierWatcher` on the session bus — stock GNOME with no +AppIndicator extension, which is the exact configuration D6 was written for. The +installed app was launched from that session's environment. + +The process starts and stays alive. No window is mapped: an X client enumeration lists +the shell's own windows and the user's browser, and nothing belonging to the app. There +is no tray icon either, because there is nothing hosting one. The application is running +and completely unreachable — the user has no surface to click and no way to know it +started. That is the failure D6 describes, now observed rather than argued. + +The only line the process wrote was an updater probe failure: + + updater check failed: Could not fetch a valid release JSON from the remote + +which is accurate for a tree whose release channel has not published a manifest yet, but +it is also the only feedback a first-run user would get if they had a way to see it. + +## Ownership was not taken, and nothing was disturbed + +The pre-existing user-level runtime kept the port for the entire run: `/healthz` reported +the same pid and version before, during and after. The desktop app wrote no install-state +record into the config home. Stopping the app left the original runtime healthy and +untouched. + +That is the correct outcome for this tree — the takeover path and its consent prompt are +lane B and lane D work — and it establishes the baseline those lanes have to change. + +## Windows is blocked on code signing, not on this batch + +The Windows verification machine runs with Smart App Control enabled and code-integrity +enforcement active. A local build fails when cargo executes its first unsigned build +script, with the OS reporting that an application-control policy blocked the file. + +This is not a toolchain gap: the build tools and the Rust MSVC toolchain install fine. +It means a machine in that configuration cannot build the shell locally, and — because +the project does not sign Windows artifacts yet — probably cannot run an installer +produced anywhere else either. Windows verification therefore depends on either an +unprotected machine or on wiring Authenticode signing, and the choice belongs to the +maintainer rather than to a lane. + +## Status + +- Linux deb: built and installed. NOT VERIFIED beyond installation, because the + behaviour under test lives in lanes that have not landed. +- Linux AppImage: NOT BUILT (bundler failure above). +- Windows: NOT BUILT (blocked by application-control policy). +- Local suites, typecheck and builds of the repository itself: NOT RUN, per the batch rule. diff --git a/devlog/_plan/260921_app_runtime_ownership/130_linux_surface_findings.md b/devlog/_plan/260921_app_runtime_ownership/130_linux_surface_findings.md new file mode 100644 index 0000000000..9d15f8bb24 --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/130_linux_surface_findings.md @@ -0,0 +1,72 @@ +# 130 — What the installed Linux build actually does + +Follow-up to 120, after the desktop shell landed. Same machine described by role: a GNOME +workstation on an X11 session with no tray host, and a user-level npm install of the proxy already +holding the default port. + +## The no-tray case is fixed + +Before the shell landed, the installed app ran with no window and no tray icon — alive and +unreachable. With the shell in the tree, the same machine shows a real window: an X client +enumeration lists an `OpenCodex` window at 1100x720 alongside the session's own windows. That is +D6 doing what it was written to do, now observed rather than argued. + +The runtime already on the port was left alone throughout: `/healthz` reported the same pid and +version before, during and after every run, and no install-state record was written. Takeover is +gated on consent, so that is the expected shape for this tree. + +## The startup surface never runs on Linux + +The window renders, and then nothing happens. The headline stays on the markup's default, the phase +checklist stays empty, and no terminal state is ever reached. The page's JavaScript does not execute +at all. + +Narrowing it took four builds, and the order matters because three plausible causes were eliminated +by measurement rather than by reading: + +1. **The asset is served correctly.** A probe that fetches the script from the page sees + `status=200`, `content-type: text/javascript`, 5941 bytes. Not a missing asset, not a MIME + refusal. +2. **Inline script runs when the policy is removed.** With the configured `csp` deleted, an inline + probe paints immediately, and the page's own script runs to completion: the checklist renders, + the registration phase completes, and the resolve phase becomes active. +3. **Widening the policy does not help.** Naming the asset-protocol scheme and host in `script-src` + changed nothing. +4. **Neither does `'unsafe-inline'`.** This is the informative one. `'unsafe-inline'` is ignored when + a nonce or a hash appears in the same directive, so the policy the webview enforces is not the + policy in the configuration file — the directive is being rewritten into a form that admits + neither the page's script nor an inline one. + +The dashboard is unaffected because it loads from the proxy's loopback origin and carries that +origin's own headers. Only the embedded bootstrap page is dead, which is why the product looks fine +until the moment it has to explain itself — and a startup surface that cannot report is exactly the +failure class this unit exists to close. + +Raised as its own issue with the evidence chain, and handed to the desktop lane. The fix has to +admit the script legitimately rather than remove the policy, so it is a design decision about how +the embedded page is served, not a widening of sources. + +## Two defects fixed on the way + +Both were found by looking at the screen and then confirmed in source, and both landed. + +**The failure block ignored its own `hidden` attribute.** An id rule with `display: grid` outranks +the user agent's `[hidden] { display: none }`, so the Retry button and an empty read-only diagnostic +box were painted during every normal start, under a headline that still said the runtime was +starting. That is precisely the screen a user reads as a dead application with one button. Removing +it is visible in the before/after captures from the same machine. + +**The page had no deadline of its own.** `invoke` returns a promise that neither settles nor rejects +when the command never answers, so the page could sit on its first handshake forever while the +shell's own deadline ran somewhere the user could not see. The handshake is now bounded and a +timeout is reported through the existing failure path. + +## Status + +- Linux deb: built, installed, launched, and inspected on a real session. +- Window presence with no tray host: VERIFIED. +- Existing runtime left undisturbed: VERIFIED. +- Startup surface reporting on Linux: FAILS — open issue, not closed by this unit. +- Windows: the verification machine required disabling its application-control policy before the + toolchain could build at all; that is recorded separately. +- Repository suites, typecheck and builds of the repository itself: NOT RUN, per the batch rule. diff --git a/devlog/_plan/260921_app_runtime_ownership/140_tray_popup_polish.md b/devlog/_plan/260921_app_runtime_ownership/140_tray_popup_polish.md new file mode 100644 index 0000000000..b93ad351bc --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/140_tray_popup_polish.md @@ -0,0 +1,78 @@ +# 140 — Tray usage popup: rustfmt, React Doctor, and a native glass surface + +The tray usage popup (#5452, carrying #5436 by JayYun98) is functionally complete and +running on macOS, but three things keep it from landing and from looking like the +WidgetKit widget it sits next to. + +## What is actually wrong + +**`desktop shell` is red on `cargo fmt --check`.** The conflict resolution left a +`matches()` body past the width limit and a double blank line before +`set_visibility`. Clippy and the Rust tests never ran because the format step gates +them. + +**React Doctor reports nine blocking findings** at `blocking: warning`. The action is +configured with `comment: false`, `review-comments: false`, `commit-status: false`, so +the findings exist only in the run's job summary. Reproduced locally with the +repository's own pinned scan, `react-doctor@0.9.11 --scope changed --base origin/dev`: + +| Rule | Location | +|---|---| +| `no-barrel-import` | `Tray.tsx:2` — `../i18n` re-exports from `./shared` | +| `no-set-state-after-await-in-effect` | `Tray.tsx:26` | +| `js-set-map-lookups` ×5 | `Tray.tsx:87` ×2, `Tray.tsx:153`, `tray-data.ts:80`, `:81` | +| `prefer-module-scope-pure-function` | `Tray.tsx:123` | +| `no-array-index-as-key` | `Tray.tsx:164` | + +**The popup is an opaque `#202022` rectangle.** The widget beside it uses the system +material, rounded numerals, and `.secondary` labels; the popup uses flat hex fills and +hairline dividers everywhere. They do not read as the same product. + +## Delivery + +One branch, `codex/260921-tray-usage-popup`, one PR to `dev` (#5452), ordered commits. + +### Native surface — `desktop/src-tauri/` + +Tauri 2.11.6 exposes `WebviewWindowBuilder::effects(WindowEffectsConfig)`, so the +vibrancy needs no extra dependency. It does need two things the tree does not have +yet: the `macos-private-api` Cargo feature on `tauri` and `app.macOSPrivateApi` in +`tauri.conf.json`. Both are required because `transparent` on macOS is a private-API +surface, confirmed from `tauri-2.11.6/src/lib.rs`. The cost is real and worth naming: +it forecloses Mac App Store submission. This app ships as a Developer ID DMG, so the +door it closes is one we are not using. + +Transparency and effects are applied on macOS and Windows only. Linux keeps the opaque +surface, because blur there belongs to the compositor and `window-vibrancy` documents +it as unsupported. + +The page has to know which surface it got, or its CSS would punch a hole in an opaque +window on Linux. A `cfg`-derived constant drives both the builder and the +initialization script, so the two cannot disagree. + +### Page — `gui/src/pages/` + +Fix all nine findings at the root rather than suppressing them. The +`no-set-state-after-await-in-effect` case is the only one that needs judgment: the +effect already guards every write with `active()`, so the fix is to make the guard +legible rather than to add one. + +Restyle to the widget's vocabulary: the system material behind a translucent panel, +rounded tabular numerals for the figures, secondary-tone labels, and dividers only +where a section genuinely changes subject. + +## Acceptance + +- `cargo fmt --check` clean; `desktop shell` green. +- The pinned React Doctor scan reports zero issues on the changed scope. +- Every job the pull_request event requested is green at the exact head, including + the aggregate `ci`. +- A screenshot of the glass popup in the PR body, since the description mentions gui. +- After landing: close #5436 as superseded with credit; the `Co-authored-by` trailer + for JayYun98 stays on the branch. + +## Not run + +Local `bun run test`, `test:changed`, `typecheck`, `build`, and `bun install` are out +of scope for this batch by standing instruction. `cargo fmt` and `cargo check` on the +desktop crate are run, under the local-build authorization given for the desktop app. diff --git a/devlog/_plan/260921_app_runtime_ownership/150_closeout.md b/devlog/_plan/260921_app_runtime_ownership/150_closeout.md new file mode 100644 index 0000000000..657dfc326c --- /dev/null +++ b/devlog/_plan/260921_app_runtime_ownership/150_closeout.md @@ -0,0 +1,96 @@ +# 150 — Closeout + +Every lane in this unit is on `dev`, and `dev` is green at `71d02e3619` with the aggregate +`ci` check passing. This records what landed, and the two findings worth carrying forward. + +## What landed + +| Change | Commit | +|---|---| +| Lane A — CLI resolve and stop contracts (#5383) | `c2a4b1`-era, see 090 | +| Lane B — desktop shell (#5384) | see 090 | +| Lane C — ownership state (#5386, #5400, #5406) | `2fb2dfb947` and follow-ups | +| Lane D — dashboard consent surface (#5387) | see 090 | +| Lane E — release pipeline (#5388, #5405) | `34ddb4d5fd` and follow-up | +| Lane F — installed gate and Linux updates (#5391) | see 090 | +| Bootstrap surface as one page the policy can name (#5445) | `1e233a4bd1` | +| Tray usage popup, glass surface, widget vocabulary (#5452) | `8f94a6fee9` | +| Tray left click reaches the popup (#5462) | `f2ebc5a8d6` | +| Startup surface cannot wait forever (#5451) | `71d02e3619` | + +`#5436` by JayYun98 was carried rather than merged and is closed as superseded, with the +`Co-authored-by` trailer on the branch so the attribution survives the squash. Issue `#5416` +is closed by `#5445`. + +## The popup surface, and the constant that holds it together + +The popup uses the native material on macOS (active HUD window, 12-point radius) and Acrylic on +Windows, both through Tauri's own effects builder. Linux stays opaque because blur there belongs +to the compositor. + +That asymmetry is the whole design problem. A transparent stylesheet on an opaque window does not +degrade gracefully — it paints a hole where the panel should be. So the platform verdict is a +single `cfg` constant, `VIBRANT_SURFACE` in `desktop/src-tauri/src/popup.rs`, and it drives both +the transparent native builder and the `data-tray-vibrancy` attribute the page selects on. Neither +side restates the other. + +Nothing in either toolchain connects a Rust constant to a CSS attribute selector, so +`tests/gui/gui-tray-vibrancy-surface.test.ts` reads `popup.rs` and `tray.css` together and fails +if they drift. Transparent windows on macOS also require the `macos-private-api` feature and +`app.macOSPrivateApi`; that forecloses Mac App Store submission, which this Developer ID DMG +channel does not use. + +## The defect static review could not see + +The popup shipped in `#5452` with a left-click handler that could never run to a visible effect on +macOS or Windows. + +`tray-icon` calls `NSStatusItem.setMenu` whenever a menu is attached. AppKit then pops that menu +on mouse-down, before the crate's own click handler — the one that reads `menu_on_left_click` — +is reached. `show_menu_on_left_click(false)` sets an ivar that never gets consulted. The menu item +that opens the popup was Linux-only, so on the two platforms where the icon click *is* the +interaction, there was no way in at all. + +Every reading of the code says it works. The handler exists, the event fires, and the wrong +surface simply appears on top of the right one. It took building the bundle and clicking the icon. +The fix makes the menu item unconditional and anchors it on the tray icon's rect; +`show_menu_on_left_click(false)` stays because it does what it says on Windows. + +Two smaller things fell out of the same round. `cargo fmt --check` had been failing, and it gates +clippy and the Rust tests, so neither had run on the popup since it landed on its branch — a +clippy error was waiting behind it. And React Doctor is configured with no comment, no review +comment and no commit status, so its nine blocking findings existed only inside a job summary +nobody opens. + +## Carried forward + +**The bundle can ship a stale app.** `bundle/macos/OpenCodex.app.tar.gz` is not refreshed by +`build:local`, so a directory holding a fresh DMG can hold a day-old archive beside it. Local +verification has to take the `.app` out of the DMG. The same class already bit the sidecar: +`prepare-sidecar` builds the standalone binary only when the file is missing. + +**A freshly compiled standalone binary is killed on macOS** until it is re-signed with +`codesign --force -s -`; from the parent that looks like "exit no exit code". + +**The installed gate refuses a symlinked prefix.** Running the bundle from a temporary directory +is rejected because that path resolves through a symlink, which is correct and worth knowing +before blaming the build. + +**Windows installed-bundle verification is still blocked.** The verification machine has no +interactive login session, so `link.exe` dies with `0xc0000142`. That needs credentials. + +**`macos 1/2` sits close to its budget.** `platform-macos` allows 20 minutes and recent runs took +8, 13 and 14; one run crossed the line and GitHub reported the expiry as a cancellation, which +reads like infrastructure noise and is not. Rerun that job rather than widening the limit — +`gh run rerun --failed` does not act on a cancelled job, so it needs `--job`. + +**`privacy:scan` never runs on the commits that add devlog content.** The scan lives in the +`gates` job, and `gates` is gated on the `ci` paths filter, whose allowlist does not include +`devlog/**`. A devlog-only change therefore skips it and the aggregate check still goes green. + +That is the one change class where the scan matters most. `AGENTS.md` says reading `devlog/` is +"what makes a public devlog safe rather than merely visible", and this pull request — which adds +sixteen devlog files to a public repository — was proven only by a hand sweep for addresses, mesh +names, accounts and absolute user paths. The fix is not to add `devlog/**` to `ci`, which would +start the cross-platform suite for a prose edit; it is to give the privacy scan its own trigger, +the way `docs-site/**` already has its own build gate. Raised separately. diff --git a/devlog/_plan/260921_app_runtime_ownership/tray-usage-popup-glass.png b/devlog/_plan/260921_app_runtime_ownership/tray-usage-popup-glass.png new file mode 100644 index 0000000000..74cbcc9cc2 Binary files /dev/null and b/devlog/_plan/260921_app_runtime_ownership/tray-usage-popup-glass.png differ diff --git a/devlog/_plan/260921_app_runtime_ownership/tray-usage-popup.png b/devlog/_plan/260921_app_runtime_ownership/tray-usage-popup.png new file mode 100644 index 0000000000..1b1a5f22f3 Binary files /dev/null and b/devlog/_plan/260921_app_runtime_ownership/tray-usage-popup.png differ diff --git a/devlog/_plan/260922_native_tray_release/000_plan.md b/devlog/_plan/260922_native_tray_release/000_plan.md new file mode 100644 index 0000000000..59dfa81d9d --- /dev/null +++ b/devlog/_plan/260922_native_tray_release/000_plan.md @@ -0,0 +1,50 @@ +# Native macOS tray and release verification + +The macOS usage popup currently embeds a web page whose clipping and scrolling disagree with its native window. Replace that popup with an AppKit popover hosting SwiftUI, keeping one application and the existing runtime owner. Preserve the dashboard and Windows/Linux popup. Review and test the changes since v2.59.0 before publishing stable and preview releases. + +## Loop contract + +- Archetype: satisfy-spec, sequential full PABCD cycles. +- Trigger: user-reproduced blocked scrolling and square outer corners over other windows; explicit request for native macOS UI and both release channels. +- Goal: native popup parity, verified regressions, signed installed app, stable main and preview publications. +- Non-goals: replacing the full dashboard, adding another app/runtime owner, changing accounts/configuration, weakening signing or protected branches. +- Verifiers: native model executable, Rust focused/full tests, GUI build and relevant tests, root typecheck/full test, privacy/structure gates, real installed app interaction, exact-head hosted CI and release receipts. Each implementation phase records actual command availability and target coverage before use. +- Stop: all work phases and recorded acceptance criteria complete, with exact release and installed artifact identities. +- Artifacts: this numbered unit; private/raw test results in `.tmp/native-tray-design/` and `.codexclaw/evidence/`; only sanitized evidence committed. +- Outcomes: DONE requires real proof; report missing external permissions/access or unsafe publication honestly; never reinterpret skipped/cancelled checks as success. +- Escalation: concrete credential/access blockers or incompatible requested behavior; new defects append a planned cycle without shrinking verification. +- Resources: no user-imposed token/cost or wall-clock cap; commands run as managed background processes with bounded polls. One release workflow at a time. +- Authority: user authorized implementation, verification and BOTH main stable and preview deployment. At 09:40 the user superseded Sol parallelism: closed all children, main implements and reviews directly, one PABCD cycle at a time. No further delegation; independent-agent consultation/review is therefore NOT RUN, not claimed. + +## Dependency order + +| Cycle | Design | Delivered outcome | +|---|---|---| +| wp0 | this roadmap and `001_evidence.md` | reviewed docs-only roadmap; no production patch | +| wp1 | `010_native_presentation.md` | versioned native DTO/model and SwiftUI content, model regression tests | +| wp2 | `020_shell_integration.md` | same-process AppKit popover, Rust transport/events, build and signing integration | +| wp3 | `030_regression.md` | v2.59.0-to-final diff audit, concrete repairs, local/native/hosted test and UI evidence | +| wp4 | `040_release.md` | dev integration, main/preview promotion, published artifacts and local install verified | + +Source layout: `app/Sources/NativeTray/` owns native presentation; `desktop/src-tauri/src/native_tray*.rs` owns transport and bridge; existing `proxy.rs` owns authenticated runtime requests; `tray.rs` remains the one icon/menu owner. Structure authorities are `structure/desktop-shell.md` and `structure/gui-and-management-api.md`. No new server endpoint. + +## Review decisions + +A1: link a Swift static library into the existing Rust process. Reject a second Swift app/helper because it duplicates lifecycle, icon and signing ownership. A scratch Rust-to-Swift link probe passed on this host. +A2: use the pinned Tauri `with_inner_tray_icon` / tray-icon `ns_status_item()` seam to anchor `NSPopover` to the real `NSStatusBarButton`; no transparent auxiliary window. +A3: retain Rust `ProxyClient` as the only authenticated network owner; Swift receives a versioned, whitelisted display DTO and emits a small callback event enum. Reject raw config/token delivery to Swift and independent Swift networking. +A4: keep Windows/Linux web popup and all main-window/runtime/update paths. macOS dispatch changes only the usage-popup branch. +A5: AppKit owns popup geometry, material, dismissal and outer corners. SwiftUI owns a bounded ScrollView and content; no CSS/native double backgrounds. +A6: user no-delegation steering overrides architect/reviewer dispatch requirements. Main performs explicit plan and adversarial code review in separate passes, recording the lack of independent reviewer. + +## Cycle record + +P wp0: requirements and source evidence collected, scratch FFI link passed, full phase map written. No production source changed. + +B wp0: roadmap locked after main-direct review; all four implementation/delivery decade docs populated. The latest user instruction supersedes prior parallel-agent plans. Next cycle implements only native presentation foundations. + +C/D wp0: reviewed the roadmap as an executable sequence; staged-document whitespace check passed. This cycle fixes no runtime behavior and makes no regression claim. The main risk remains real native integration, so wp1 must prove SwiftUI/framework linking and model semantics before wp2 can activate it. Continue with wp1 as written; no scope/criterion reduction. + +C/D wp1: native presentation and ABI foundation compiled for macOS 13, SwiftUI/Charts/AppKit linked into Rust, 26 model assertions passed, and a real native fixture reached provider 40 by scrolling while retaining header/footer. Apple Liquid Glass requirement is implemented with the system view on supported SDK/OS; actual anchored/background composition remains wp2. Evidence and limitations: `011_presentation_verification.md`. Continue wp2; no full-app or release claim yet. + +Latest user steering: main remains the implementer; Sol read-only verifiers are now required for each PABCD design/implementation verification. This supersedes the earlier blanket no-delegation instruction for verification only. Current wp2 has a Sol verifier; future cycles retain that role. Earlier wp0/wp1 remain honestly recorded as main-reviewed under the instruction in force then; Sol will review their inherited design/code before wp2 closes. diff --git a/devlog/_plan/260922_native_tray_release/001_evidence.md b/devlog/_plan/260922_native_tray_release/001_evidence.md new file mode 100644 index 0000000000..5d7086f942 --- /dev/null +++ b/devlog/_plan/260922_native_tray_release/001_evidence.md @@ -0,0 +1,18 @@ +# Baseline and design evidence + +- v2.59.0: `134c92a01b120162f00c7275189cc47858720379`. +- origin/main: `7c625fc9755c9824653ab944190e243091a2c85c`, package version 2.60.0. +- origin/preview: `84c4f014c8da51ff50c3e8b64f2d82b9ee3792da`, package version 2.60.0. +- origin/dev / starting HEAD: `e4ceeb38da74f0c727bd9a0228856f150908dd19`, package version 2.61.0. Re-read refs before integration/release. +- `gui/src/pages/Tray.tsx`: reads companion settings, today/30d usage, timeline, config and account-level quota projections; hidden views abort polling. `tray-data.ts` owns filtering, missing-vs-zero, masking and quota normalization. +- `desktop/src-tauri/src/popup.rs`: current transparent webview + HUD effect with radius 12; `gui/src/pages/tray.css` clips body overflow without a scroll container. Native wheel input leaves viewport unchanged. +- `desktop/src-tauri/src/proxy.rs`: identity-bound credential transport, redirects and system proxy disabled; keep this owner. +- Pinned Tauri 2.11.6 exposes `with_inner_tray_icon`; pinned tray-icon 0.24.2 exposes retained `NSStatusItem` on macOS. It must be used on the main thread. +- Apple NSPopover owns anchored positioning and transient dismissal: https://developer.apple.com/documentation/appkit/nspopover . NSHostingController embeds SwiftUI inside AppKit: https://developer.apple.com/documentation/swiftui/nshostingcontroller . MenuBarExtra is an alternative SwiftUI application scene, not needed for an existing Tauri-owned application. +- Scratch probe: `swiftc -parse-as-library -emit-library -static -module-name NativeProbe -target arm64-apple-macos13.0 …` then `rustc -L native=… -l static=NativeProbe -C link-arg=-L/usr/lib/swift -C link-arg=-Wl,-rpath,/usr/lib/swift …`; calling the exported Swift function returned 42, exit 0. `swift-autolink-extract` is not installed, so the design does not depend on it; Darwin linker autolinking worked. +- Earlier local app rebuild reproduced invalid embedded-CLI signing (resolve killed with exit 137), fixed by signing the inner CLI then the bundle. Packaging verification must execute resolve as well as validate bundle signature. No credentials or account identifiers belong in committed evidence. +- Child design/data/release agents were interrupted and closed on user instruction; shutdown confirmed. No delivered report is used as review evidence. +- Apple Liquid Glass steering: https://developer.apple.com/documentation/appkit/nsglasseffectview and its contentView page document the native effect and required content placement, macOS 26+. Installed macOS 27 SDK confirms availability; NSPopover.hasFullSizeContent (macOS 14+) clips full-size content to its native outline. No private view hierarchy manipulation or simulated CSS material. +- User requested WidgetKit/Xcode integration history review. `38a5ab9fc4` originally contained native `MenuBarUI/PopoverPanel.swift`: it records NSPopover from an accessory process failing key eligibility on macOS 27 and uses a key-capable nonactivating NSPanel with NSGlassEffectView. `2ff7f3385d` removed MenuBarUI when adopting the Tauri desktop stack. This is relevant executable precedent, not a reason to restore the old app/process owner. The initial wp1 standalone NSPopover preview produced no accessible window; the NSWindow-hosted content did render, consistent with the old finding. +- `7fead8d10b` requires WidgetBundle @main AND _NSExtensionMain AND application-extension compiler flag. `9a39eea40d` signs all nested Mach-O members inside-out with runtime options and the app team's Developer ID. Preserve those WidgetKit contracts unchanged. +- Active toolchain: Xcode 27.0 (27A266a), selected `/Applications/Xcode.app/Contents/Developer`. `xcodebuild -list -json` in app resolves NativeTray, NativeTrayTests, MenuBarCoreTests and OpenCodexWidget schemes. diff --git a/devlog/_plan/260922_native_tray_release/002_plan_review.md b/devlog/_plan/260922_native_tray_release/002_plan_review.md new file mode 100644 index 0000000000..fd345c5d2f --- /dev/null +++ b/devlog/_plan/260922_native_tray_release/002_plan_review.md @@ -0,0 +1,15 @@ +# Roadmap review and dispositions + +Main direct review, required by the user's no-delegation steering; this is not an independent reviewer sign-off. + +1. Existing Swift package has macOS 14 minimum for WidgetKit; linking that unchanged as the desktop implementation would accidentally raise the desktop minimum (currently 13). Accepted: the Cargo build compiles the native source set directly with deployment target 13, while SwiftPM tests remain on the available host. Compile availability is an explicit wp1 check. +2. Tray icon must not be duplicated by a new MenuBarExtra scene. Accepted: use the existing NSStatusItem via the pinned public inner-tray seam, all AppKit calls on main thread. Closing the usage popover cannot stop a proxy or exit the app. +3. Raw config/account responses could over-broaden the Swift interface and make testing ambiguous. Accepted: Rust projects an explicit display DTO; no config/credential persistence or network client in Swift. +4. A native implementation cannot claim parity using only provider aggregate quotas. Accepted: enumerate account roster paths, active OpenAI selection and missing data semantics; include mixed partial failure and hidden/model filters in fixtures. +5. A model test does not prove window corner/scroll behavior. Accepted: wp2/w3 explicitly require installed native interactions over both desktop and another window, scroll to footer and back, Escape/outside-click/reopen. +6. Release versions cannot be selected from the user's 2.59.0 reference: current main/preview are 2.60.0. Accepted: preserve v2.59.0 as regression baseline and read live version line before selecting publication versions. +7. Static library feasibility: Rust calls the compiled Swift export successfully. Do not depend on missing swift-autolink-extract. Native view/framework linking remains a wp1 executable check. + +No production source has changed. File ownership and dependency order are explicit. New-file method bodies remain implementation detail, bounded by the stated ABI/model contracts and fixtures; any change of contract or file scope must amend the corresponding phase design first. + +VERDICT: PASS — sequential roadmap ready; independent architect/reviewer consultation NOT RUN under explicit user instruction. diff --git a/devlog/_plan/260922_native_tray_release/010_native_presentation.md b/devlog/_plan/260922_native_tray_release/010_native_presentation.md new file mode 100644 index 0000000000..e12a83b6c7 --- /dev/null +++ b/devlog/_plan/260922_native_tray_release/010_native_presentation.md @@ -0,0 +1,29 @@ +# wp1 — Native presentation foundations + +Depends on wp0. This cycle delivers native models/views with a narrow ABI, not application activation. + +## File changes + +- NEW `app/Sources/NativeTray/Models.swift`: `NativeTraySnapshot: Decodable` with `schemaVersion=1`, loading/refreshing flag, bounded section errors, optional updatedAt, display settings, today/thirtyDay `Totals`, model rows, timeline series, provider/account/window rows. All numeric values optional; sanitize nonfinite/negative values, clamp percentages only for bar fill, preserve actual percentages for labels; missing is never zero. Unknown schema is rejected by the update boundary. Pure Foundation formatting of tokens/reset dates. +- NEW `app/Sources/NativeTray/UsageView.swift`: SwiftUI ScrollView in fixed-width bounded native content, Today/30d sections, cached input ratio, output/cost/requests/coverage, model counts/tokens, per-account quotas with reset times, accessible missing/error state, Refresh and Dashboard/Settings actions. Use semantic system fonts/colors, no painted outer background/corner mask. Swift Charts consumes timeline ids/times and honors the actual line/stackedBar setting. Split a chart/account view sibling if cohesion/size warrants. +- NEW `app/Sources/NativeTray/Popover.swift`: main-thread controller and `@_cdecl` ABI declarations: show/toggle with borrowed status-item pointer and callback `(Int32)->Void`; hide; update with borrowed UTF-8 JSON copied during the call; visible query. `NSPopover.behavior=.transient`, `NSHostingController`, anchor to status item button, height bounded by the screen visible frame, `.onExitCommand` closes. One controller per app, no timer/network/runtime ownership in Swift. +- MODIFY `app/Package.swift`: add a static NativeTray library target/product and an executable NativeTrayTests target/product following the existing executable-test convention. Existing widget target stays intact; Rust's direct Swift build targets macOS 13 independently of the widget's package minimum. +- NEW `app/Sources/NativeTrayTests/main.swift`: fixture-based schema/number/missing-vs-zero/reset/duplicate-account identity checks, decode fixture identical to Rust wire contract. No real API/Keychain/network use. + +## Contract and flow + +Rust creates display DTO -> serde_json encodes -> FFI copied Data -> JSONDecoder typed snapshot -> SwiftUI render. `schemaVersion` exists at all four stages. Callback events: 1 opened/refresh, 2 closed, 3 dashboard, 4 settings; producer Swift controller, C integer serialization, Rust exhaustive match with unknown ignored, consumers refresh cancellation/main-window navigation. Unsupported values never become stop/update actions. + +## Verification and acceptance + +Run the actual Swift executable model tests once created; compile all NativeTray sources for macOS 13 with `swiftc` to prove availability. Fixture cases: missing usage but present quotas; measuredRequests=0; pricedRequests=0; percentages >100; reset timestamps seconds vs milliseconds; unsupported schema; Unicode labels; many accounts. UI interaction waits for wp2's installed native host. Update this plan before deviating from ABI or DTO shape. Do not claim a standalone test proves the actual app path. + +## P revalidation and direct audit (wp1) + +Previous D: roadmap fixes no runtime behavior; prove native availability and model semantics before integration. Source unchanged since roadmap. Clarified wire keys: `schemaVersion`, `refreshing`, `errors`, `updatedAt`, `settings`, `today`, `month`, `models`, `chart`, `providers`; display settings use showToday/show30Days/showChart/showModels/showAccounts/showCost/chartStyle. No credentials in this DTO. Chart/account subviews may live in `UsageSections.swift` to keep modules focused. Persistent native header/footer surround the bounded ScrollView, so Refresh and Dashboard stay reachable even with many accounts. AppKit screen sizing is enforced by the controller. + +Direct audit: main-thread-only borrowed pointers must be copied/used synchronously; event callback must never outlive the app singleton. Reject unknown schema without replacing a valid snapshot; publish a fixed human error. Swift has no proxy client, filesystem state or new timer. Unit fixtures cover this data contract; installed UI remains a wp2 criterion. Independent consultation is NOT RUN under the user's no-subagent instruction. VERDICT: PASS. + +B source recheck: companion settings have only line/stackedBar chart styles, not area/stacked. The implementation and plan now use the canonical setting. `show30Days` is a display DTO field fixed true (the existing API always shows this section), not a new persisted setting. Partial price coverage retains an asterisk and explicit tooltip. + +User steering: Apple Liquid Glass is required. Add `app/Sources/NativeTray/Surface.swift` as an AppKit host controller: on macOS 26+ with a supported SDK, put the SwiftUI hosting view in `NSGlassEffectView.contentView` with regular glass. The native popover owns the sole outer clipping shape (`hasFullSizeContent` on the glass branch), with no inner rounded background. Older OS/SDK retains the native NSPopover material; no deployment-target increase. `NSGlassEffectView` availability was verified in Apple's documentation and the installed macOS 27 SDK. Respect reduced transparency through the system component. This is native API use, not CSS blur. Direct design re-audit accepts this amendment; wp2 must verify the glass branch on the actual installed app above both desktop and other windows. diff --git a/devlog/_plan/260922_native_tray_release/011_presentation_verification.md b/devlog/_plan/260922_native_tray_release/011_presentation_verification.md new file mode 100644 index 0000000000..55225952b6 --- /dev/null +++ b/devlog/_plan/260922_native_tray_release/011_presentation_verification.md @@ -0,0 +1,31 @@ +# wp1 verification + +## Automated and build proof + +- `swift run --package-path app NativeTrayTests`: PASS, 26 assertions. Fixture scope: schema/malformed response, unknown vs measured zero, absent usage with available account quota, pricing coverage, Unicode, 80-provider roster, reset units/date bounds and >100% quota bar. +- `swiftc -parse-as-library -emit-library -static -module-name NativeTray -target arm64-apple-macos13.0 app/Sources/NativeTray/*.swift ...`: exit 0, including the runtime-gated Apple Liquid Glass source. This proves the desktop deployment minimum is not raised to the WidgetKit package minimum. +- Rust link probe referencing the actual NativeTray static archive with SwiftUI/Charts/AppKit: exit 0. No swift-autolink-extract dependency. +- `bun run structure:check`: PASS after documenting the library owner. + +## Render and adversarial pass + +Used a temporary native NSWindow/NSHostingController host with synthetic usage and 40 synthetic account providers, at 420 x 660 content points. This is native view proof, not a claim that the installed application already uses it. + +| Scenario | Observation | Evidence | +|---|---|---| +| Populated content | Native totals, chart, models and provider rows render; no webview or raster substitute | `evidence/native-glass-top.png` | +| Long list | Native scroll reaches Example Provider 40; scrollbar value 1; fixed header/footer remain visible | `evidence/native-glass-bottom.png` | +| Missing values | Synthetic month omits input/output/cost; UI shows em dashes | top capture + model assertions | +| Partial price coverage | Today displays $2.50* for 30 priced out of 40 requests | top capture + model assertions | +| CJK/model boundaries | Unicode survives decoding; no credentials or real user account data in fixtures | 26-assertion executable | +| Apple glass | Native host includes NSGlassEffectView on this macOS 27 build; final anchored popover/background composition remains a wp2 acceptance item | Surface.swift + captures | + +A test-host sizing issue was found: assigning NSHostingController resets a window to its fitting size. The test host now sets content size after assigning its controller; the production NSPopover likewise sets contentSize after constructing its controller. The first test-host screenshot was rejected; only corrected captures are evidence. + +Main performed functional and visual review separately; independent reviewers were not used because the user explicitly forbade subagents. The actual menu icon, live transport, outside-click/Escape lifecycle and glass above background windows still require wp2 integration proof. No production-app completion claim is made in this cycle. + +Teardown: temporary NativeTrayPreview process terminated; no proxy/server or user configuration was created by this fixture. Temporary source/build artifacts are retained in ignored scratch space for reproducibility. + +D direction: native presentation foundation is ready. Continue wp2 with the prewritten bridge/transport/packaging plan, including a release-builder check that the published macOS artifact was built with Liquid Glass-capable SDK (not merely runtime availability on this host). + +Capture integrity: CUA supplied JPEG bytes despite the scratch `.png` name. Re-encoded those unchanged pixels to actual PNG before committing; verified PNG signature and 420x692 dimensions (420x660 content plus native titlebar). This correction does not synthesize or edit UI content. Functional pass checks FE-A11Y-POLISH-01 (persistent actions and accessible controls); visual pass checks bounded content and native-system styling. Both passes are main-owned under no-delegation. Headless browser from the preceding CSS investigation was also stopped. diff --git a/devlog/_plan/260922_native_tray_release/020_shell_integration.md b/devlog/_plan/260922_native_tray_release/020_shell_integration.md new file mode 100644 index 0000000000..fc7aabca5f --- /dev/null +++ b/devlog/_plan/260922_native_tray_release/020_shell_integration.md @@ -0,0 +1,47 @@ +# wp2 — Integrate the native popup + +Depends on wp1. Preserve the running proxy, menu ownership and main dashboard. + +## File changes and before/after + +- NEW `desktop/src-tauri/src/native_tray.rs`: macOS-only C ABI owner, one AppHandle binding and managed refresh task/generation. Before: popup show always creates a webview. After: macOS show/toggle obtains existing main tray's NSStatusItem on main thread and invokes Swift; Swift callbacks are marshalled through Tauri main-thread/async APIs. No additional NSApplication, icon, service, account refresh or runtime process. +- NEW `desktop/src-tauri/src/native_tray_data.rs`: GET-only collection using the existing bound ProxyClient. Before: JS fetches config/settings/usage/timeline/account roster. After: Rust selects the same roster endpoints, whitelists display fields and masks emails, filters selected models/hidden providers, emits the wp1 DTO. Preserve partial-section results and fixed human error text; raw server errors/config/credentials never cross the ABI or logs. OpenAI active selection read remains separate and is not inferred if unavailable. +- MODIFY `desktop/src-tauri/src/proxy.rs`: expose the existing identity-checked GET method only `pub(crate)` for the native collector; all auth/redirect/proxy/identity code unchanged. +- MODIFY `desktop/src-tauri/src/lib.rs`: register macOS bridge/data modules and managed native refresh state. Existing startup/exit/update sequence unchanged. +- MODIFY `desktop/src-tauri/src/popup.rs`: macOS show/toggle/hide forward to native adapter; Windows/Linux keep webview implementation. Separate web popup implementation if needed to avoid macOS dead-code warnings and preserve current Rust tests. +- MODIFY `desktop/src-tauri/build.rs`: on macOS compile the native Swift source set into OUT_DIR static archive for Cargo target architecture, deployment macOS 13; emit rerun-if-changed and framework/runtime link search/rpath. Fail build if compiler fails. Non-macOS must not invoke Swift. Prefer direct swiftc (verified), no external dependency/download. Existing tauri_build call remains. +- MODIFY `desktop/scripts/build-local.ts` only if needed: ensure ad-hoc integrity signing of inner CLI and native bundle in local output, then package verified output. Published signing requirements are never relaxed. Add a regression test if behavior changes. +- MODIFY `structure/desktop-shell.md`, `structure/gui-and-management-api.md`, `docs-site/src/content/docs/guides/desktop-app.md`: document native macOS popup and preserved other-platform route, transport/lifecycle ownership and scroll behavior. +- NEW/UPDATE focused tests under `tests/clients/` and both layout manifests if a new root-suite test file is added; use actual bridge/build fixture behavior and DTO tests rather than only string checks. + +## Reachable activation cases + +Open/show/toggle repeatedly through real tray; close by outside click and Escape; scroll dozens of providers to Dashboard footer; refresh while loading; close during request; reopen gets fresh data without old generation overwrite. Existing runtime absent/binding changed returns honest unavailable, never spawns through popup. Every GET is identity-bound; stale results discarded on close/runtime change. Verify only one polling task while visible, none while hidden. Dashboard/settings navigate the existing main window. Both desktop wallpaper and another window behind popup show one native outline, no square web underlay. + +## Acceptance + +Swift/Rust tests, cargo fmt/clippy/test, existing desktop runtime/ownership tests, GUI build for remaining platforms, macOS app build, signature validation AND embedded `ocx resolve --json`, installed real UI screenshots and scroll evidence. No full suite claimed here; full release gates belong to wp3. Keep source code, output hashes and actual source revision linked. + +## P revalidation after wp1 + +Previous D: native display and ABI compile/link and long-list native view proof passed; installed-app, transport and anchored glass remain this cycle. Source now has `NativeTrayHostingController` and `NativeTraySnapshot` from wp1; ABI export names are ocx_native_tray_show/hide/visible/update. This cycle adds `native_tray_snapshot.rs` (pure projection and tests) and `native_tray_accounts.rs` (account/window projection) beside `native_tray_data.rs` (async collector) to avoid a monolithic data module. The callback enum is fixed: 1 open/refresh, 2 close/cancel, 3 Dashboard, 4 Settings. + +Keep `popup.rs` as the web implementation module selected on non-macOS (and tests); select `native_tray.rs` as the `popup` module on macOS via cfg/path, retaining the caller's existing show/toggle/hide signatures. Existing Rust web geometry tests can be compiled under a test-only `web_popup` alias, without activating the web popup on macOS. The global bridge stores only the current Tauri AppHandle; task/cache/generation state stays managed by that app. Close aborts the one task; JoinSet owns bounded concurrent GETs and cancels them on drop. Publication checks visibility, generation and current runtime binding. + +Apple Liquid Glass is an explicit acceptance criterion, not an inferred material name. Verify build artifacts reference NSGlassEffectView; if the release builder's SDK cannot compile the branch, select a supported toolchain in its existing macOS job rather than shipping a silent fallback to supported OS users. Release signing policy remains unchanged. Direct plan audit: PASS; credential/data-boundary review is retained in ignored scratch. + +## History-driven design amendment during integration + +A2/A5 amended with user-requested historical evidence: use a key-capable nonactivating `NSPanel`, anchored in screen points from the existing NSStatusBarButton, rather than NSPopover. The former native companion in commit 38a5ab9fc4 documented the macOS27 accessory-popover keyboard failure, and the wp1 transient test host was not accessible. NEW `app/Sources/NativeTray/Panel.swift` adapts only that native presentation mechanism: transparent borderless host, system NSGlassEffectView (regular, radius16) or NSVisualEffectView.popover fallback, screen clamp, key focus, Escape/outside-click dismissal, idempotent monitor teardown. SwiftUI remains the content; existing Rust process/tray remains the owner. MODIFY Popover.swift and Surface.swift accordingly; the stable C ABI is unchanged. No new helper process or NSApplication delegate. Direct amended-design audit PASS: the new panel removes the failed key route and makes the single native material own its actual outer corners; real panel keyboard/background proof remains required before C. The originally proposed NSPopover is not claimed shipped. + +Sol review round1 was a setup FAIL because its review skill reads the staged snapshot and the current implementation was not staged. Main accepted this blocker and staged the exact wp2 change set; the same verifier was resumed. No code-level PASS was inferred from that empty review. Cargo release tests on the source-identical scratch tree passed 91/91 before installed-app QA. + +Packaging amendment: NEW desktop/src-tauri/Entitlements.plist and MODIFY tauri.conf.json to supply the Bun runtime's minimal JIT entitlement; local helper explicitly requests ad-hoc nested/bundle signing, keeping published Developer ID/updater signing untouched. The installed CLI itself must execute successfully, not merely pass codesign. Add contract coverage in release-desktop-scripts.test.ts, and the existing widget CI job runs NativeTrayTests plus bundled CLI resolve against an isolated home. Build.rs rejects release archives without NSGlassEffectView, so an old build SDK cannot silently omit the requested effect. Detailed security reasoning and controlled entitlement probe remain in ignored scratch. + +Accessibility integration amendment: MODIFY macOS `menu.rs` to expose View > Show Usage (Cmd+Shift+U), routed to the same native panel and existing status-item anchor. The status-extra menu is not exposed by the available computer-use AX snapshot; keyboard/app-menu access also gives users a discoverable route without pointer precision on a crowded menu bar. This is a normal product action, not an external test command or arbitrary IPC hook. Preserve the Quit gesture and all standard edit actions. + +Sol round2 FAIL dispositions: accept high partial-result loss from the global timeout; replace it with per-section bounded concurrent reads and incremental snapshot publication. Account fanout retains completed rows and marks unfinished providers unavailable when its deadline expires. Add a deterministic stalled-quota/available-usage test. Accept focus observation: remove forced application activation and recheck panel keyboard behavior. For the medium JIT scope finding, Tauri's current bundle entitlement setting is shared across packaged executables; retain the minimal single-key exception, document that actual scope, and assert exact final dictionaries/runtime flags on the host, sidecar and widget. Detailed risk decision remains scratch. Do not call the security observation resolved until Sol reviews the final assertions. + +Final signed-artifact assertion addition: NEW `desktop/scripts/verify-macos-runtime.sh` reads the generated app's actual CFBundleExecutable, verifies its bundle seal, checks hardened-runtime flags and exact entitlement dictionaries for app/ocx/widget, requires the native glass class reference, and executes isolated-home read-only `ocx resolve`. The existing macOS bundle CI invokes it. The host and ocx intentionally share Tauri's single minimal JIT entitlement configuration; the widget retains only its existing sandbox entitlement. This is an explicit shared-bundler scope decision, not a claim of sidecar-only permissions. + +Parity recheck found that the existing web chart labels series by its full id, while the first native projection used only model name. Correct that projection and explicitly group native LineMark by series id so identical model names from different providers never connect into one line. Keep the legend outside the plot's fixed height (native grid inside the existing scroll view), so many configured series cannot consume the whole plot. Add a two-provider/same-model projection case; no wire keys or privilege changes. diff --git a/devlog/_plan/260922_native_tray_release/021_integration_verification.md b/devlog/_plan/260922_native_tray_release/021_integration_verification.md new file mode 100644 index 0000000000..b2b61e0894 --- /dev/null +++ b/devlog/_plan/260922_native_tray_release/021_integration_verification.md @@ -0,0 +1,32 @@ +# Native tray integration verification + +The macOS tray now opens an AppKit panel containing SwiftUI. The existing Tauri process owns the status item and runtime attachment. One native Liquid Glass surface owns the outline on supported systems; older systems use native popover material. Windows/Linux retain their web popup. + +## Observed evidence + +- The user confirmed the installed app works on the actual display on 2026-09-22. This is human acceptance of the reported interaction/appearance issue, not an automated compositor measurement. +- The installed app loaded real usage/accounts, scrolled to the bottom (AX scroll value 1), dismissed with Escape and reopened. Private account captures remain ignored scratch, not public evidence. +- The production panel classes rendered the synthetic fixture in light/dark mode; screenshots below contain no real account data. The latest light image includes the final chart legend. Native chart and legend remain readable above a bounded scroll region with fixed header/footer. +- Native fixture exercised toggle-close, toggle-reopen and dismissal on key loss. Runtime material was NSGlassEffectView. The original probe measured external focus preservation before its own activation and key-window state afterwards, so that log does not prove both simultaneously on first opening. +- Xcode 27 built the NativeTray scheme and WidgetKit extension. Widget history retained @main, _NSExtensionMain and application-extension compilation together. The existing widget is independently packaged. +- Rust release tests: 95 passed, including a real HTTP stalled-quota/available-usage case, deadline retention, missing-vs-zero projection and same-model/different-provider series. Swift native model assertions: 26 passed; MenuBarCore: 118 passed. Focused desktop release/widget/CLI contracts: 40 passed. +- Cargo clippy with warnings denied passed before the final chart-only projection change. Docs site built 497 pages; privacy and structure checks passed. +- Signed bundle verification checks hardened runtime and exact JIT-only host/CLI entitlements, sandbox-only widget entitlement, deep signature, final dyld NSGlassEffectView binding and isolated-home bundled CLI resolve. No runtime service takeover is needed to open the tray. + +![Native panel, light appearance with synthetic data](evidence/native-panel-light.png) +![Native panel, dark appearance with synthetic data](evidence/native-panel-dark.png) +![Native scroll reaches the final synthetic provider](evidence/native-panel-bottom.png) + +The dark and bottom images precede the final chart-only legend adjustment; panel geometry, material and scroll implementation are identical. All images show the native window capture, not a full-screen composite. Xcode 27 logs a nonfatal Swift Charts custom-UnitPoint warning even with standard axis anchors; inspected labels are aligned, and this record does not claim the warning was fixed. + +## Independent review and limits + +Sol identified loss of partial results under an overall timeout, forced foreground activation, and the scope of the JIT entitlement. Main implemented independent bounded section collection, removed forced activation and added final-artifact entitlement assertions. The host and bundled CLI deliberately share Tauri's minimal JIT entitlement; the widget remains sandbox-only. Detailed pre-publication security review stays in ignored scratch. + +The native integration does not certify release readiness. Full prepush reached 28,463 passes, 36 skips and 21 failures. A source-identical baseline reproduced the release fixture failure and six other named failures, then exceeded the suite's 900-second limit with remote-workspace/account-pool tests still running. No failed, timed-out or absent gate is counted as green. The next PABCD cycle owns the full v2.59.0-to-candidate review, those failures and release gates. + +The failed NSPopover direction was replaced with the key-capable NSPanel mechanism from 38a5ab9fc4; the old companion process was not restored. Evidence contradicting the current approach would be a reproducible focus, scrolling, rounded-outline or partial-result failure in the installed native panel. User acceptance and targeted checks support this integration; full-suite and cross-platform release proof remain outstanding. + +## Final wp2 check + +Final C receipt executed Rust95, cargo fmt/clippy with warnings denied, the signed-bundle verifier, structure check and diff whitespace checks: exit 0. Sol's final staged-delta review reported no remaining native code findings, blocking_issues=0, VERDICT: PASS. The chart-final app was installed with a preserved previous-app backup; its native Show Usage menu loaded real data and ten chart series. The running proxy was not replaced or restarted. This closes wp2 only; wp3 and both releases remain open. diff --git a/devlog/_plan/260922_native_tray_release/030_regression.md b/devlog/_plan/260922_native_tray_release/030_regression.md new file mode 100644 index 0000000000..d865a0e274 --- /dev/null +++ b/devlog/_plan/260922_native_tray_release/030_regression.md @@ -0,0 +1,70 @@ +# wp3 — Regression review from 2.59.0 + +Depends on wp2. Audit actual v2.59.0/main/preview-to-final changes sequentially, with main implementing and reviewing as the user requested. + +## File/change map + +- READ full `git diff --name-status v2.59.0..<candidate>` and commit log, plus origin/main and origin/preview deltas. Classify each changed owned subsystem into runtime/adapters/provider catalog, ownership/install/update, GUI/desktop/widget, release/CI/docs. Record exact baseline/candidate SHAs and every covered scope in a numbered verification document in this unit. +- MODIFY only a proven regression's owning source, focused regression test and structure doc; add an explicit plan amendment describing trigger, before/after, test and file-size/layout constraints before each repair. No blanket cleanup and no speculative changes. Unreleased security analysis stays in `.tmp/` until shipped. +- MODIFY this unit's `031_verification.md` with sanitized command receipts, failed-case resolution, review limitations and final frozen SHA. Exact-head checks are re-run only when a later delta invalidates their scope. + +## Gates + +Root `bun run typecheck`, `bun run test`, `bun run privacy:scan`, `bun run structure:check`; GUI `bun test tests`, `bun run lint`, `bun run build`; native executable tests; cargo fmt/clippy/test; widget/local app build and installed smoke. Read scripts/config first and record what each command observes; no vacuous command is a pass. + +Hosted PR and merged-dev gates must include all expected jobs and platform legs at the recorded SHA, event, run and attempt. The final suite is the repository's actual defined suite, including indirect/source-oracle tests. Baseline comparison distinguishes preexisting environment failures from regressions with reproduced evidence; do not waive a named failed gate in its own report. + +UI matrix: desktop vs window behind, light/dark appearance, long scroll bottom and back, smaller work area, open/close/reopen, error/empty/loading, unavailable quotas and partial usage, settings hide sections/models/providers, refresh once, exit/update/service ownership. Non-macOS UI remains covered by its build and hosted platform tests; report limits where no interactive host is available. + +Completion means no known unresolved regression in the audited/tested scope; it is not a mathematical claim that no possible bug exists. + +## wp3 P revalidation + +Previous D concluded: the native shell is integrated, signed locally, independently reviewed and accepted by the user; full-suite failures remain release blockers. Continue that direction. Satisfy-spec loop; trigger is the requested regression audit and dual-channel release. Main alone edits implementation. Sol supplies read-only architecture/audit/verification. No new external account action, live proxy restart, credential mutation or dependency upgrade is required for this cycle. GitHub read/PR/CI access is within the authorized release; publishing remains wp4. No numerical token/cost or wall-clock budget was imposed; managed commands retain their existing finite test deadlines. Record in this unit and ignored diagnostic scratch; success means all named gates pass with no known unresolved regression in the audited scope. Missing access or a new unsafe release condition is reported; timeouts remain failures. + +Fresh baseline: v2.59.0=134c92a01b; initial dev=e4ceeb38da; candidate native commit=48822a5452. Fetch advanced origin/dev to39143fddf4 (Google permission enum PR #5243, two files). Integrate this reviewed delta in B before final tests and bind the audit inventory to that resulting candidate. Main/preview refs are rechecked before promotion. The since-v2.59.0 inventory has 1,208 changed paths, including source/runtime, native/web UI, tests, docs and release tooling; inventory generation is not claimed as review. + +### Proven failure repairs proposed + +- R1 MODIFY tests/ci-workflows/ci-workflows.test.ts: the publication-shell fixture omits GITHUB_SHA although the extracted production script expands it with nounset. Supply a deterministic synthetic commit in the fixture environment; keep all ten acknowledged-publication/recovery assertions. No production publishing change. Respect the file-size ratchet by changing the existing environment line. +- R2 MODIFY tests/clients/remote-workspace-command-runner.test.ts: two argv tests substitute process.execPath for bubblewrap. Local Bun is a hardlink (nlink2), while production correctly requires a private executable (nlink1). Give these tests their own executable fixture outside the writable workspace. Keep production hardlink/symlink rejection and the existing adversarial tests intact. Exercise both argv construction and later toolchain substitution. +- R3 MODIFY tests/codex-integration/codex-shim-destroyed-probe.test.ts: the fixture installs a real shim with a five-second observation window inside a five-second test. Use the existing observation-duration test seam, as codex-shim.test.ts already does, reset after each case. Preserve the actual FIFO replacement and one-second child process deadline. Do not shorten production safety deadlines. +- R4 MODIFY tests/claude-integration/claude-models-discovery.test.ts: an isolated reproduction shows native-main admission is blocked with foreign-ownership before and after waiting for startup. startServer reads the real host's default service state through its explicit path resolver; a running local service therefore suppresses the mock entitlement fetch. Inject the existing StartServerDeps ownership inspection seam with an explicitly owned fixture result for these discovery-contract tests. Separate startup-ownership tests continue to use actual hostile ownership inputs; do not bypass any production check. +- R5 MODIFY scripts/test.ts plus its owning test-runner coverage/docs: service-state authority lives at the process-start default home shared by Bun parallel workers. A focused four-file run reproduces one file reading another file's authority, while isolated baseline cases pass. Put service-ownership-state, service-sqlite-home, service.test and native-grok-toggle into the existing isolated full-suite lanes, each retaining all assertions and a fresh process/home. Inspect any additional failures before extending isolation. Verify the generated lane roster and actual complete suite, including hosted platform/shard behavior. +- R6 MODIFY gui/src/pages/tray.css and relevant docs: non-macOS vibrant web popup still clips body overflow without a bounded inner scroll container. Constrain html/body/root to the viewport and make the page itself scroll within that height; retain opaque Linux and Acrylic Windows surface. Verify a synthetic long list reaches the footer with vibrancy both on/off, capture browser output, and run GUI tests/lint/build. Do not alter accepted macOS native geometry. + +The two baseline hung files (remote-workspace-server and account-pool-management-api) passed together in isolation: 42/42 in one second. This narrows the issue to suite interaction/load; it is not a waiver of the full-suite deadline. The final complete run must settle successfully. + +### Coverage and verification + +Review changed code by ownership groups: desktop/app/widget and web UI; runtime adapters/transport/routing; Codex/provider/account/integration; service/install/update/security/release; usage/config/remaining modules. Compare each group with its relevant tests and docs, record concrete defects and limits, and amend this plan before repair. Credential/security reasoning stays scratch until published. For any new public field, trace producer/serialization/consumer; no such production field is currently proposed. The test changes are executable fixtures, not production enforcement. A runner can bypass scripts/test.ts by using bare Bun; isolation is only guaranteed by that defined full-suite command and its CI lanes, not a security boundary. + +Actual executed verifiers at P: full prepush exit1 (21 failures); baseline full suite exit124 (900s); focused six-failure set exit1; service four-file set exit1; previously hung pair exit0 (42 tests). Each directly names or discovers the changed tests. Native wp2 receipt remains valid for unchanged native code. Full prepush, GUI tests/lint/build, privacy, structure and exact-head hosted CI must pass after repairs. scripts/test.ts discovers ./tests and SERIAL_FULL_SUITE_FILES; source-oracle tests are included. Architecture docs synchronize fixture/isolation and non-macOS scrolling semantics. No passing check is repeated without a source or evidence-binding reason. + +### Architecture consultation and dispositions + +Sol architect Bacon (01a0c737-0fff-7fd1-bf7f-36364d012af6) proposed WP3-D01 candidate freeze/inventory, D02 fixture repairs, D03 isolated authority lanes, D04 non-macOS scroll repair and D05 sequential gates. Main accepts D01-D05. Reflection on this plan aligned R1-R5 and required two clarifications: R3 sets the observation seam BEFORE withInstalledShim (the helper installs before its callback), resetting in afterEach; R6 names its verification and docs below. These clarify execution rather than change module responsibilities or interfaces. + +R6 exact owners: MODIFY gui/src/pages/tray.css, structure/desktop-shell.md and structure/gui-and-management-api.md. Runtime layout regression is an executable browser probe in .tmp/native-tray-design/web-tray-scroll-check.mjs against the actual CSS and a synthetic long provider list; record measured clientHeight/scrollHeight, positive scrollTop and visible footer for vibrancy on/off at 440x520 and 440x720 in 031_verification.md. Persist observed screenshots at evidence/web-tray-vibrant-bottom.png and evidence/web-tray-opaque-bottom.png. This is render-grounded regression evidence, not a new permanent source-string assertion or browser dependency. Existing gui/tests/tray-data.test.ts retains data-contract coverage; full GUI tests/lint/build remain required. + +Architect reflection after those clarifications: ALIGNED. D01-D05 form a bounded evidence-based sequence; production guards are preserved, fixture isolation is distinct from product fixes, and independent A may begin. No unresolved architecture blocker remains. + +Verifier compatibility note: installed agbrowse has no `script` subcommand; its help output was not counted as a pass. The named browser probe runs with `node .tmp/native-tray-design/web-tray-scroll-check.mjs` and uses the installed evaluate/resize/snapshot/screenshot commands via execFileSync. Its baseline result is exit1: page clientHeight=scrollHeight=2911, scrollTop=0 and footer outside the viewport in the four bounded-page scenarios. This proves absence of the proposed inner scrolling region; opaque mode's existing document scroll is not claimed broken by that assertion. + +### A round1 synthesis + +Independent Sol reviewer Cicero (01a0c740-3316-7812-be52-67ddb5e7034f) returned GO-WITH-FIXES with two concrete blockers. Both are accepted; neither is waived. + +R5 expands to every defined complete-suite path. SERIAL_FULL_SUITE_FILES in scripts/test.ts remains the single roster. MODIFY scripts/ci/run-bun-test-batches.sh to read that roster with the selected Bun, keep the existing sorted/sharded ownership, and split each selected batch into its ordinary group and singleton roster entries. Every selected file still runs exactly once in a primary process, with the same failure/timeout/crash disposition; attribution cannot repair a failed result. Invalid/failed manifest reads fail closed. MODIFY the macos-control Test step in .github/workflows/ci.yml to call the existing complete-suite wrapper with --parallel=1 --timeout60000; the ordinary set stays one unsharded process, while the explicit isolation exceptions each get their own process/home. This changes process topology, not assertion coverage. Keep the normal macOS manifest consumer. MODIFY tests/ci-workflows/ci-crash-disposition.test.ts to prove singleton ownership plus failure preservation, and adjust ci-workflows.test.ts/ci-bun-crash-classifier.test.ts to the changed control invocation. Existing test-runner and macos-serial-lanes tests verify the other consumers. Update structure/ops/docs-and-release.md for the topology. No new external dependency, permission, secret or workflow trigger is added. + +R6 final evidence command is node .tmp/native-tray-design/web-tray-scroll-check.mjs --capture. Its oracle additionally requires the document scrollingElement scrollHeight <= clientHeight, zero outer scroll offset, positive inner scroll offset and a visible footer, with the page height bounded by the viewport. The captured images must be opened and observed before C closes. The initial no-capture run was only the baseline diagnostic, not the final evidence receipt. + +The same architect reflected on D03's hosted-runner amendment: ALIGNED. Sorted shard membership, first-failure disposition and singleton coverage are preserved; implementation proof remains C. The final A reviewer receives this revised plan, not the earlier local-only roster proposal. + +R5 local execution detail: replace the batch script's single Bash4-only mapfile statement with an equivalent NUL-delimited Bash3-compatible read loop. This preserves file discovery and enables the existing fake-toolchain behavioral harness on macOS as well as Linux (Windows still uses real hosted Git Bash). Extend its skip condition accordingly; no new shell dependency is installed. This lets the singleton/no-retry contract be executed locally rather than asserted from source only. + +A round2: the same reviewer returned PASS, no remaining blocker, after the hosted isolation and browser-oracle amendments. Main proceeds to B with all six repairs and the independent since-v2.59.0 code-review groups still active. + +B review found a further R5 integration issue: macos-control previously measured 50m39s under a 75-minute job budget, but the wrapper defaults to a 15-minute main-process bound. Accept this finding. Add a validated OCX_TEST_MAIN_TIMEOUT_MS override (integer 60,000–3,600,000; default remains900,000), consumed by resolveBunTestPlan; set3,600,000 only on the macos-control Test step. Existing singleton bounds and the75-minute job backstop remain. Pin valid/invalid/default parsing and the control env in existing runner/workflow tests. Environment chain: workflow env→process.env→validated lane.timeoutMs→runTestLane watchdog; no public runtime configuration changes. + +R1-R6 root suite result before this timeout-only adjustment: 28,731 passed, zero failed across the parallel set and all isolated lanes. Prepush returned0; React Doctor additionally reported two test-only findings, tracked for repair rather than ignored. Other since-v2.59.0 review findings are under main verification; security-sensitive working plans remain ignored scratch as required by AGENTS.md. No release-readiness claim is made. diff --git a/devlog/_plan/260922_native_tray_release/031_verification.md b/devlog/_plan/260922_native_tray_release/031_verification.md new file mode 100644 index 0000000000..bc8eb102a3 --- /dev/null +++ b/devlog/_plan/260922_native_tray_release/031_verification.md @@ -0,0 +1,140 @@ +# Regression candidate verification + +The native tray integration has user acceptance and the evidence recorded in +[021](021_integration_verification.md). This candidate also addresses findings from the +since-v2.59.0 source review and preserves the existing dashboard and runtime ownership. +Detailed security working notes remain outside the tracked tree. + +## Verification boundary + +The user explicitly prohibited all further local tests and requested a no-verify push on +2026-09-22. The active local GUI suite was terminated (exit143); no result from that interrupted +run is counted as passing. Subsequent verification belongs to hosted CI and read-only review. +The candidate is not release-ready until its exact commit has the required hosted results. + +Before that instruction, the original21 failures were resolved and a complete root run passed +28,731 tests. Later focused evidence includes Rust101, Swift121, the Bun updater's12 scenarios, +model migration45, cache/routing41 and release-resume20 tests. Those results are historical, +scoped evidence; they are not presented as a complete final-candidate suite. GUI harness findings +were repaired and React Doctor subsequently reported no issues. The latest dependency audit +reported no high-severity failure. Hosted CI must judge the final committed tree. + +Native and web screenshots in evidence/ contain synthetic data. The web scroll checks observed +440px content in500x433 and500x633 Chrome viewports (requested outer window sizes were440x520 +and440x720), positive inner scrolling, reachable footer and no outer document overflow. They do +not claim Windows desktop compositor coverage. No new local visual checks run after the prohibition. + +## Remaining delivery + +The exact-head PR checks, remaining read-only review, merged-dev checks, stable/main and preview +publication, registry tags/assets and final installation evidence remain outstanding. No release +or universal no-regression claim is made by this checkpoint. Follow the user-directed hosted-only +verification path and retain every failed, missing, cancelled or timed-out job as unresolved. + +## Hosted follow-up at 9233d4f3a3 + +PR #5490 targets dev. Its Cross-platform CI run 35692642447 completed the macOS +widget/bundle and desktop-shell jobs successfully. Service-lifecycle checks passed on Linux, +macOS and Windows. The full workflow is not green: Linux shard 2 failed two sandbox-fixture +cases because the temporary executable inherited a writable ancestor. Manual all-platform +run 35692726962 also exposed Windows failures, including retention publication while its +source reader remained open. Fixes and unrun regression cases follow in the next commit; +that commit requires fresh hosted evidence. No local tests were run for these repairs. + +The Swift optional-filter decoder and malformed-receipt fixtures received a read-only Sol +PASS. The settings-unavailable account section intentionally retains independently fetched +account limits, matching the web/default behavior, while reporting the unavailable settings. +Other baseline and security reviews remain separate from these two findings. + +The next hosted candidate also narrows the Windows scheduler fixture's synthetic system path, +awaits server/child cleanup in vision and outbound-proxy cases, and waits for the real native-main +startup gate before asserting discovery rows. Proxy fixture phase diagnostics preserve a bounded +failure if transport rather than teardown remains stuck. The Linux fixture owns a disposable +executable beside the trusted interpreter instead of changing shared-file permissions. + +Replacement ordering and CA startup repairs have source-review follow-ups; their new regression +cases are committed for hosted execution only. Final read-only reviews and exact-head hosted +results remain required before integration. + +## Final source-review follow-up + +The remaining baseline reviewer completed 63/63 assigned files. Its last three findings were +corrected and received a read-only source/security PASS, with regression cases committed but not +run locally. An additional Windows shutdown audit found two test files whose production-server +cleanup was not awaited; both now await release before deleting their directories. + +At 4a38eb4bb9, service lifecycle passed on all three platforms, and manual Windows shards 4, 5 +and 6 passed. This is intermediate evidence only: the subsequent source-review fixes require +fresh hosted checks. Source review does not establish runtime success or a universal absence +of regressions. + +Hosted ec7ad275f2 exposed a test-injection error: the new ZCode failed-stat case overrode +`store.io` while the planner consumes `input.io`. The fixture now injects through the consumed +seam and retains the preview refusal, actual-write refusal and no-write assertions. Sol reviewed +the correction. A later Windows shard exposed the existing 100ms timing margin in the stalled +400-body case. Its helper now waits for the real bounded reader's timeout before releasing the +upstream suffix; retry rules and product timeouts are unchanged. Sol confirmed the call ordering. +The helper is committed with the test. All of these checks remain unrun locally. + +The older macOS control run 35692726962 timed out after stopping in the first structure-SSOT +test. Its synchronous Git child is the source-based inference; the log does not identify the +child PID. The file now runs through the existing shared singleton roster, retaining every +assertion and existing time limit. Separately, the bridge-stall test's fixed six-second outer +ceiling pre-empted its CI-scaled 30/45-second inner watchdog. Its outer ceiling now retains the +same two-second cleanup margin on each platform; the product's one-second stall setting and +all terminal/cancellation assertions are unchanged. Sol reviewed both adjustments. + +Obsolete failed manual runs 35694628931 and 35695558779 were cancelled after their failure logs +were captured, to release runner capacity. Their partial successful jobs remain historical +diagnostics only. Cancelled workflows do not count as passing verification. + +Windows shard 8 at d7d2838341 exposed two management-auth teardown failures. The test drained +ACL reaps before draining native-main startup releases, allowing the latter to finish work that +registered a later reap. Teardown now drains native-main releases, all config-directory hardening, +then ACL child reaps before deleting the temporary home. Sol reviewed this ordering; removal +retry budgets and all management-auth assertions remain unchanged. + +At b5529c5bb2, Windows shards 1, 3, 4, 5, 6 and 8 passed, and both macOS shards plus the +widget/bundle job passed in the manual run. PR macOS shard 1 independently wedged after the +injection-write-lock zero-byte case and reached its job timeout; source review identifies the +next case's synchronous child spawn/reap boundary as the likely blocked point. That file joins +the existing fresh-process roster without changing assertions or deadlines. + +The manual Windows run found two further fixture lifetime failures. Five native-profile crash +phases shared one 90-second test; they now run as five independently bounded cases, preserving +every transaction/recovery assertion, with TERM/SIGKILL/reap bounds on switch-child cleanup. +A passthrough-cancellation fixture left its second pull pending forever despite request abort, +then deleted its accounting home before late cancellation finalized. Its fetch-shaped helper +now settles the pending pull on abort, and the case waits for the 499 cancellation log before +teardown. Sol reviewed both corrections; ownership enforcement is unchanged. No local tests ran. + +At 6b919f8dea, a stale-status CLI fixture inferred the human process's health verdict from +separate JSON invocations. The human process could legitimately see an intervening refusal +failure while both other probes reported stale. A preload observer now delegates to the real +probe, records that same process's boolean on stderr, and returns it unchanged. The formatter +assertion runs only when its own observed verdict is true; missing output still fails. Sol +reviewed the observation and import ordering. Product status behavior is unchanged. + +## macOS control process boundary amendment +The older b5529c5bb2 control again stalled in a different synchronous subprocess test after the +structure case was isolated. It stopped after assert-mergeable-review/malformed_reviews, reported +a killed dangling process at the per-test ceiling, then emitted no result for twenty minutes. +Keeping one indefinitely growing Bun isolate pool was not yielding reliable completion evidence. + +The control now enumerates the entire 1/1 test list through the existing bounded batch runner: +at most twelve files per fresh process, one worker, 300-second process bound, unchanged 60-second +per-test ceiling and 75-minute job cap. Dedicated storage/API families and declared serial files +remain singleton primary processes. Every selected file runs once; primary failures remain red +even if diagnostic attribution is clean. This preserves assertions and file membership but no +longer claims whole-suite shared-process contamination coverage. Behavioral fixtures check exact +membership, argument shape, special-family ownership, invalid input and failure disposition. +Sol architecture, behavioral and explicit workflow/dependency security reviews accepted the change. +The new cases are unrun locally; hosted execution is still required. + +The new macOS full-membership control passed at 9df4499dd2 (manual run 35704045906). +The same head's PR Linux run exposed one native Anthropic reject-path fixture race: it freed +an ephemeral upstream port before starting the proxy, allowing reuse/self-targeting instead of +a connection refusal. The fixture now rejects only its exact synthetic upstream origin through +the fetch boundary while retaining real HTTP ingress, an exact-one-upstream-call assertion, +502/api_error/message assertions, and global restoration. Sol accepted the change; it is unrun +locally and requires the next hosted candidate. diff --git a/devlog/_plan/260922_native_tray_release/040_release.md b/devlog/_plan/260922_native_tray_release/040_release.md new file mode 100644 index 0000000000..7ebfd64abc --- /dev/null +++ b/devlog/_plan/260922_native_tray_release/040_release.md @@ -0,0 +1,121 @@ +# wp4 — Stable and preview delivery + +Depends on wp3. User explicitly selected both main and preview publication. + +## Changes and operations + +- MODIFY `.github/PULL_REQUEST_TEMPLATE.md` sections in the actual PR body only: problem/result, exact-head validation, native screenshot, checklist and maintainer integration decision. Ordinary PR to dev, one branch with ordered commits; no native stack. Push/merge are authorized by the release request, subject to actual checks/review policy. +- READ latest `MAINTAINERS.md`, `scripts/release.ts`, release/dev-version-bump workflows, npm/GitHub published versions and branch rules before selecting versions. Record exact version matrix and promotion SHA. +- If dev does not outrank the intended release, use the repository's dev-version-bump PR flow before publication. MODIFY only version-bearing files selected by that canonical flow, no ad-hoc drift. +- Promote dev through PRs to main/preview following required review/branch policy. No direct protected-branch push or force push. Keep objections and security review separate; do not fabricate independent approval. +- Execute the canonical release command/workflow with exact expected SHA, branch, version and dist-tag. Stable and preview runs are serialized. A failed or pending workflow is not published success; reconcile before retrying. +- VERIFY GitHub release/tag and artifact inventory/checksums/signatures/updater manifest, npm versions/dist-tags/gitHead and required exact-head CI. Use hosted installed-artifact validation when runners exist. The authorized local app update preserves backup and runtime ownership, but local interaction, CLI and health probes remain NOT RUN under the latest instruction. Record artifact identity without claiming local execution proof. +- MODIFY this unit's `041_release_receipts.md`, then archive the unit to `devlog/_fin/` only once all cycles are terminal. + +## Acceptance and rollback + +Both channels have reachable verified artifacts at their recorded commits; the authorized app update is completed and local interaction remains explicitly unverified under the no-local-tests instruction. Keep the prior application backup and prior published version/digest so local rollback is reversible. Never republish the same version to repair a bad artifact; use repository release policy. If a protected promotion requires an independent maintainer action not available to this session, stop that publication step with the exact blocker while completing all independent preparation; no bypass inferred from beta status. + +## Publication observation during wp3 + +On 2026-09-22 the official npm registry reports version2.60.0 exists with gitHead7c625fc9755c9824653ab944190e243091a2c85c, matching origin/main and the published GitHub v2.60.0 release. However the live npm tags are latest=2.59.0 and preview=2.55.0-preview.20260914. This was re-read with the explicit official registry and prefer-online; no dist-tag mutation was performed. Both requested channel deliveries must verify the actual final registry tags in addition to GitHub assets and version existence. Do not republish2.60.0 or silently count it as the current latest tag. + +## Executable wp4 plan — 2026-09-22 +The native tray and regression candidate is verified at 3d64bd3040b2da7953962da3c05be14f31991e56. +This phase integrates that exact candidate and publishes independently derived preview and stable +artifacts. Release notes and receipts distinguish source review, hosted execution and installation. + +Loop: satisfy-spec, C4 release operations; trigger: explicit both-channel delivery and subsequent +dev admin-merge authorization. Goal: both channels published with verified artifacts and the +authorized app update. Non-goals: no local tests, typechecks, builds or QA probes; no live runtime +restart, credential changes or protection-rule mutation. Main executes, Sol reviews read-only. +No user-imposed time/token budget. Memory artifact: this unit and its release receipts, plus +ignored operational receipts in .tmp/native-tray-design. Success ends only after both channels +are verified; pending or partial publication remains unfinished. Escalate only actual unavailable +promotion authority, signing credentials or installation access, after completing independent work. + +### Dependency order and file map +1. D1: Re-read PR #5490 head, all exact-head hosted results, automated reviews and maintainer + objections. Explicit owner-authorized admin merge targets dev only. It is not an independent + approval. Source/security review is recorded separately. Keep this plan amendment uncommitted + until the delivery metadata commit; the remote PR head remains the verified candidate. +2. D2: Fetch the resulting dev merge SHA and freeze its immutable 2.61.0 RC branch/tree before + changing the dev version. Confirm the merge contains the reviewed changes without unexpected + product differences. Keep the source candidate pinned if dev advances. +3. D3: Dispatch dev-version-bump.yml from main with intended-version=2.61.0, mode=pre-move; + review and merge its package-only PR after hosted checks. Confirm dev is 2.62.0. +4. D4-D6: Prepare preview through an ordinary promotion PR based on current preview plus the + frozen RC. Use actual KST publication date in 2.61.0-preview.YYYYMMDD, adding an unused ordinal + when necessary. MODIFY package.json, desktop/src-tauri/tauri.conf.json, Cargo.toml and Cargo.lock + consistently; apart from release metadata, retain the frozen product tree. Observe final + promotion-SHA push CI/service success. Dispatch release.yml dry-run then publication, serialized, + tag=preview and exact expected-sha. Verify embedded desktop version in hosted artifacts and + npm version/dist-tag/gitHead/integrity/provenance, tag/release target, asset set/checksums and + updater signatures. Preview uses its tag-specific manifest; stable updater discovery is unchanged. +5. D7-D9: Prepare main independently from the same RC, never from preview or post-bump dev. + All four version authorities remain 2.61.0. Repeat final-SHA push CI/service, canonical dry-run + and publication with tag=latest. Verify both dist-tags, all assets and the stable latest manifest. + Use hosted installed-artifact validation where configured. Local update is authorized but local + execution checks remain NOT RUN under the latest prohibition; preserve the prior app backup, + user configuration and running proxy. Do not represent installation alone as interaction proof. +6. D10: If publication is partial, inspect the actual registry/tag/release state. Resume only an + acknowledged npm publication at the identical version/SHA using the canonical source-bound + resume path. Missing/mismatched provenance or signing evidence refuses completion. +7. D11: Promotions retain current MAINTAINERS.md rules. Record any explicitly authorized owner + override as an override, never an independent approving review. Do not weaken rulesets. +8. UPDATE 041_release_receipts.md with exact SHAs, versions, URLs and observable limitations; + archive the unit only after both channels and acceptance criteria are terminal. + +### Reachable verification and failures +- GitHub gh run view/watch reads exact head/status/jobs: the wp3 PR, all-platform and service runs + completed successfully; receipt command exit0 was observed at the clean candidate. These are + read-only hosted-result queries, not local tests. +- New release/promotion runs are NOT RUN yet. Their workflow definitions read checkout/version, + expected-sha, CI/service history, signing inputs, generated bundle bytes and registry state. + Actual triggers are workflow_dispatch on the selected protected branch with a full expected-sha. +- Branch movement must fail the dispatch identity check; existing consumed versions must fail + fresh publication; missing signing inputs or invalid assets must fail before publish; registry + source mismatch must fail resume. Do not activate destructive failures against public versions. +- No local verification command is implied by this plan. Windows/macOS skipped jobs, cancellation, + old-commit runs and review comments are not substituted for current execution evidence. + +### Architect consultation +Architect: Newton (01a0c744-11eb-7dd2-9af9-37d2b735b545), read-only Sol. Proposal D1-D11 accepted. +Main amendment to D9: latest no-local-tests instruction excludes local probes; use hosted artifact +checks and report local execution unverified. Promotion authority remains explicit per D11. +Same-architect reflection and independent audit are recorded before execution. + +### D0 — integration-review correction before D1 +GitHub Codex review at the verified head reported comment4070117466: a committed gateway write +followed by unreadable first-party settings returns before persisting the gateway mode and apply +fingerprint. Accepted for correction in this integration phase, without invalidating the prior +wp3 evidence at its recorded head. MODIFY the CLI apply path and both management paths to persist +committed gateway bookkeeping before reporting cleanup failure, while keeping the failed cleanup +visible and preserving any separate bookkeeping warning. Add focused cases to the existing +Claude Desktop first-party suite: start from first-party, make settings unreadable, apply gateway, +observe failure/partial cleanup and persisted gateway mode/fingerprint, and confirm a subsequent +default apply selects gateway. Include API, native-toggle and CLI paths as applicable. Update the +owning Desktop contract. Require Sol read-only review and fresh exact-head hosted CI before merge; +the previous head's green result does not certify this correction. No local tests are allowed. + +Architect reflection disposition: D0-D5 and D7-D11 aligned. D6 amendment accepted: preview +verification explicitly requires GitHub prerelease=true and npm latest unchanged from the +recorded pre-preview value. Main records that before stable publication changes latest. +Final same-architect reflection: Newton returned ALIGNED for D0-D11 after the D6 amendment; +no remaining architecture gap. Independent A audit follows. + +Latest owner steering: Latest owner instruction ci 걍 무시하고 머지해 executed: PR5490 admin squash merged to dev6c2f7676dcedba21bdbacf4fb84a7b2c286d1ee6 at2026-09-22T09:24:09Z. Priorcandidate3d64 hadPR/allplatform/serviceSUCCESS. D1 now precedesD0 by explicituseroverride; no fabricated B order. D0reviewfinding gatewaypartialbookkeeping remains narrowfollowup beforefreezeRC/publish. Bothpreview+stabledeployment remainsauthorized. No-local-tests andno-verify unchanged. + +Independent A audit: Volta NEAR-PASS. Both text gaps are folded: D0 explicitly requires credential-boundary security review under MAINTAINERS.md in addition to ordinary source review; the original local-verification wording above now matches the latest no-local-execution restriction. New promotion drafts #5510/#5511 are provisional and will receive the corrected RC. No release has been published. + +### D0b — ordinary macOS shard process boundaries +The post-merge dev run35710172686 hit its 20-minute limit in macOS shard1. Its last recorded +passing cases were in catalog-full-picker-order around09:30:51; no further test output appeared +before cancellation around09:48:13. The precise subsequent blocked import/cleanup boundary is +not visible in the log. The same membership passed under the bounded full-control batches. +Replace ordinary macOS shard monolithic processes and their separate serial loop with that +shared batch runner: sorted all-file1/2 and2/2, maximum12files, parallel1,300-second batch bound, +60-second per-test ceiling, existing20-minute job cap. Preserve singleton families, fail-red +attribution and all tests. The actual workflow harness must prove complete/disjoint membership, +exact-path collision handling and failure disposition. Sol source/security review precedes the +owner-authorized immediate dev merge; local checks remain NOT RUN. diff --git a/devlog/_plan/260922_native_tray_release/evidence/native-glass-bottom.png b/devlog/_plan/260922_native_tray_release/evidence/native-glass-bottom.png new file mode 100644 index 0000000000..239147da12 Binary files /dev/null and b/devlog/_plan/260922_native_tray_release/evidence/native-glass-bottom.png differ diff --git a/devlog/_plan/260922_native_tray_release/evidence/native-glass-top.png b/devlog/_plan/260922_native_tray_release/evidence/native-glass-top.png new file mode 100644 index 0000000000..5f704c8473 Binary files /dev/null and b/devlog/_plan/260922_native_tray_release/evidence/native-glass-top.png differ diff --git a/devlog/_plan/260922_native_tray_release/evidence/native-panel-bottom.png b/devlog/_plan/260922_native_tray_release/evidence/native-panel-bottom.png new file mode 100644 index 0000000000..d301ddbbab Binary files /dev/null and b/devlog/_plan/260922_native_tray_release/evidence/native-panel-bottom.png differ diff --git a/devlog/_plan/260922_native_tray_release/evidence/native-panel-dark.png b/devlog/_plan/260922_native_tray_release/evidence/native-panel-dark.png new file mode 100644 index 0000000000..c6cc27d8dd Binary files /dev/null and b/devlog/_plan/260922_native_tray_release/evidence/native-panel-dark.png differ diff --git a/devlog/_plan/260922_native_tray_release/evidence/native-panel-light.png b/devlog/_plan/260922_native_tray_release/evidence/native-panel-light.png new file mode 100644 index 0000000000..a9d5c40632 Binary files /dev/null and b/devlog/_plan/260922_native_tray_release/evidence/native-panel-light.png differ diff --git a/devlog/_plan/260922_native_tray_release/evidence/web-tray-opaque-bottom.png b/devlog/_plan/260922_native_tray_release/evidence/web-tray-opaque-bottom.png new file mode 100644 index 0000000000..aff3279624 Binary files /dev/null and b/devlog/_plan/260922_native_tray_release/evidence/web-tray-opaque-bottom.png differ diff --git a/devlog/_plan/260922_native_tray_release/evidence/web-tray-vibrant-bottom.png b/devlog/_plan/260922_native_tray_release/evidence/web-tray-vibrant-bottom.png new file mode 100644 index 0000000000..d8a90e3c65 Binary files /dev/null and b/devlog/_plan/260922_native_tray_release/evidence/web-tray-vibrant-bottom.png differ diff --git a/devlog/_plan/260923_anthropic_fast_speed/010_plan.md b/devlog/_plan/260923_anthropic_fast_speed/010_plan.md new file mode 100644 index 0000000000..1ccfcddb55 --- /dev/null +++ b/devlog/_plan/260923_anthropic_fast_speed/010_plan.md @@ -0,0 +1,46 @@ +# Anthropic fast mode (`speed: "fast"`) as a native FastWire — plan (wp1) + +## Loop spec (HOTL wp1) + +- Scope: user asked to verify Claude fast mode over the Anthropic OAuth token, research pricing in the Claude docs via Aside, check the other Claude OAuth models, bind it natively the way the xAI Grok OAuth fast lane was bound, and open a PR. Unlimited gpt-6-sol subagents granted. +- Write scope: branch `codex/anthropic-fast-speed` in this worktree. No merge, release, service restart, or live config edit. +- Certification: focused local tests plus exact-head hosted CI on the PR. + +## Evidence + +- Live probe matrix: [020_probe-evidence.md](020_probe-evidence.md). +- Official contract (Aside, 2026-09-23): platform.claude.com/docs/en/build-with-claude/fast-mode, /about-claude/pricing, code.claude.com/docs/en/fast-mode. Request `speed: "fast"` + `anthropic-beta: fast-mode-2026-02-01`; echo `usage.speed` ("fast" | "standard"); supported models exactly `claude-opus-5-5`, `claude-opus-5`, `claude-opus-4-8`; Opus 4.6 silently runs standard; fast price is 2x standard input/output with cache multipliers applied to fast input (Opus 5.5 8/40, Opus 5 and 4.8 10/50). Fast has its own rate-limit pool; 429 on fast exhaustion, 529 on capacity. The API does not fall back; Claude Code retries a rejected fast request at standard speed. Subscription fast (Pro/Max/Team/Enterprise) draws on usage credits. + +## Decisions + +- D1 Wire: `FAST_WIRE_ADAPTERS["anthropic-speed"]` = {"anthropic"}. `service-tier` stays OpenAI-only. +- D2 Registry: `anthropic` (OAuth) and `anthropic-apikey` declare `fastWire: {kind:"anthropic-speed", canonicalToWire:{priority:"fast"}, foreignCallerTiers:"drop", betas:["fast-mode-2026-02-01"]}` and `modelSupportsServiceTier` for the three documented ids. No provider-wide `supportsServiceTier`; Opus 4.6/4.7, Sonnet, Haiku, Fable and future ids stay unclassified. OAuth is included because the probe shows the OAuth lane accepts the field and gates only on account entitlement (usage credits / org enablement), which is the documented Claude Code subscription path. +- D3 Adapter request: on a `set` decision whose value is the declared wire value, emit `body.speed` and add the declared betas to the single `anthropic-beta` header per D3a. Adapter owns `tierLog` with wireKind `anthropic-speed`. +- D4 Adapter response: observe `usage.speed` from `message_start` / `message_delta` (stream) and the buffered body. "fast" confirms, "standard" downgrades (`response-declined`), absent leaves `assumed`. +- D5 Refusal downgrade (final, after reflection 030 and audit 040). Recognition is narrow: status 400 or 429 whose Anthropic error message names fast mode or the `speed` parameter (probe strings: `Usage credits are required for fast mode.`, `Fast mode is not enabled for your organization`, and "does not support the `speed` parameter"), or a 429 carrying `anthropic-fast-input-tokens-remaining: 0` or `anthropic-fast-output-tokens-remaining: 0`. Generic 429/529 keep today's path. The body is read from a bounded clone, so the original refusal survives if the resend is not admitted. + - Scope is the main adapter recovery loop in `src/server/responses/adapter-dispatch.ts`: one arm after the 401 arms and before the same-target 429 wait and key/OAuth rotation, guarded once per request. Before touching the response it reserves the resend with `reserveCredentialHop("repair", "<provider>|<model>|anthropic-fast-downgrade", countedExternally)` exactly as the generic OAuth 429 arm does (`countedExternally` true only on a helper-reported transient-policy leg); a refused reservation leaves the original refusal untouched and falls through. The permit rides `sendBudgetState.pendingHopPermit`, is confirmed with `permit.use()` in `rebuildAndRefetch`'s `onDispatch`, and is released on any pre-send failure. It replaces `parsed.options.tierDecision` with `drop`, marks `parsed.options.tierObservation.upstreamDeclinedFast = true`, invalidates the same-target cache, and calls `rebuildAndRefetch("anthropic-fast-downgrade")`. The refused response never reaches rotation or cooldown. + - No process memo. The decision lives on the request, so every later build of the same request (refetches, tool continuations, sidecar iterations that reuse the parsed options) stays standard, and credential rotation or token refresh cannot lose it. Each new turn pays one refused round trip, as Claude Code does. + - `createAdapterTierMetadata` reports `downgraded` / `response-declined` when the observation carries `upstreamDeclinedFast`, instead of `wire-unavailable`. + - The resend is a visible, paced, attempt-logged send with its recovery kind and a real request-budget charge. A dispatched resend also charges the root workflow once; a refused pre-send reservation does not. Tests pin workflow exhaustion and that a spent request budget returns the original refusal. + - New recovery kind `anthropic-fast-downgrade`: roster, cause `parameter-rejected`, status-confirmed 400 mapping, metrics class `fast_downgrade` (additive, via a kind override so `effort_downgrade` keeps meaning reasoning effort), dashboard log label in all ten locales. + - A first fast send that is refused inside a continuation or sidecar owner (not the main dispatch) keeps today's handling (residual). +- D3a Header merge: provider header overrides are merged case-insensitively into a single `anthropic-beta` with deduped tokens; OAuth betas are preserved; the fast beta is appended after overrides whenever `speed` is emitted, so `speed` is never sent without it. +- D6 Pricing: `PRIORITY_PRICING_RULES` gains 2x rules with `requiresResponseConfirmation` for the three models on `anthropic` and `anthropic-apikey`. Unconfirmed or downgraded turns keep standard price. +- D7 Picker: `--fast` rows follow the existing eligibility predicate; no new listing code. +- D8 Docs/SoT: structure owners (providers-and-adapters, transports/responses, gui-and-management-api cost note) and the stale comment in `src/server/claude-messages.ts`. + +## Acceptance criteria + +- C1 `fastPolicyForModel` is `eligible` for the three ids on both registry entries; Opus 4.6, Sonnet 5 and Haiku stay unclassified; `fastWire: null` still disables. +- C2 Adapter emits speed + beta only on a set decision; drop/default emits neither; header merge keeps the OAuth betas. +- C3 Stream and buffered echo map to confirmed / downgraded / assumed. +- C4 In the main dispatch loop a recognized fast refusal is replaced by exactly one visible standard resend (recovery kind recorded), never on a standard send, never twice, never for a generic 429/529; later builds of the same request stay standard; the outcome is downgraded/response-declined and priced 1x. +- C5 Confirmed fast turns price at 2x; standard echo or fallback stays 1x. +- C6 Existing pins that flip are rewritten deliberately (fastwire-policy anthropic-speed wire-unavailable, registry roster of explicit FastWire entries). +- C7 Focused tests, typecheck, ratchet/layout, structure:check, privacy:scan pass; PR open with template; exact-head CI inspected. + +## Residuals + +- Each new turn on an account without fast entitlement pays one refused round trip (about 400 ms) before the standard resend. Operators disable with `fastMode: false` or by not selecting `--fast`. +- A refused first fast send inside a continuation or sidecar owner, Claude Messages native passthrough (caller auth, raw caller `speed`), and the Anthropic web-search provider sidecar keep today's handling. +- None of the user's six OAuth accounts can currently run fast (four lack usage credits, two orgs have it disabled), so a confirmed `usage.speed: "fast"` echo is proven from the docs, not from a live 200. diff --git a/devlog/_plan/260923_anthropic_fast_speed/020_probe-evidence.md b/devlog/_plan/260923_anthropic_fast_speed/020_probe-evidence.md new file mode 100644 index 0000000000..ab4324516c --- /dev/null +++ b/devlog/_plan/260923_anthropic_fast_speed/020_probe-evidence.md @@ -0,0 +1,42 @@ +# Live probe evidence — Anthropic OAuth `speed: "fast"` (2026-09-23 KST) + +Mechanics: direct POST https://api.anthropic.com/v1/messages with the active ocx Anthropic OAuth credential and ocx's own OAuth fingerprint (`ANTHROPIC_OAUTH_BETA`, Claude Code system block, `CLAUDE_CODE_HEADERS`), `max_tokens: 256`, prompt "Reply with exactly: OK". Script kept in scratch (`.tmp/claude-fast/probe.ts`); no token printed or stored. + +## Beta header + +| Request | Result | +|---|---| +| claude-opus-5-5, speed fast, no fast beta | 400 `speed: Extra inputs are not permitted` | +| claude-opus-5-5, speed fast, `fast-mode-2026-02-01` | 429 `rate_limit_error: Usage credits are required for fast mode.` | +| claude-opus-5-5, speed `turbo` + beta | 400 `speed: Input should be 'standard' or 'fast'` | +| claude-opus-5-5, speed `standard` + beta | 200, `usage.speed: "standard"` | + +## Model matrix (active account, standard control vs speed fast + beta) + +| Model | Standard | Fast | +|---|---|---| +| claude-opus-5-5 | 200 OK | 429 usage credits required | +| claude-opus-5 | 200 OK | 429 usage credits required | +| claude-opus-4-8 | 200 OK | 429 usage credits required | +| claude-opus-4-7 | 200 OK | 400 does not support the `speed` parameter | +| claude-opus-4-6 | 200 OK | 200, `usage.speed: "standard"` (silent downgrade, documented) | +| claude-fable-5-1 | 200 OK | 400 does not support `speed` | +| claude-fable-5 | 200 OK | 400 does not support `speed` | +| claude-sonnet-5 | 200 OK | 400 does not support `speed` | +| claude-sonnet-4-6 | 200 OK | 400 does not support `speed` | +| claude-haiku-4-5 | 200 OK | 400 does not support `speed` | + +Every configured Claude model works on the OAuth lane at standard speed. Standard responses carry `usage.service_tier: "standard"` and no `usage.speed`. + +## Accounts (claude-opus-5-5, speed fast + beta) + +| Pool slot | Result | +|---|---| +| 1 (active) – 4 | 429 `Usage credits are required for fast mode.` | +| 5, 6 | 400 `Fast mode is not enabled for your organization. An organization admin must enable this feature.` | + +No account currently serves a fast turn: the feature is on, but fast draws on usage credits (extra usage), which none of the four personal accounts has funded, and the two org accounts have it disabled by admin. + +## Streaming + +claude-opus-4-6 fast stream: the echo is in `message_start.message.usage.speed` ("standard"); `message_delta.usage` carries no speed. diff --git a/devlog/_plan/260923_anthropic_fast_speed/030_reflection.md b/devlog/_plan/260923_anthropic_fast_speed/030_reflection.md new file mode 100644 index 0000000000..a7af3f1004 --- /dev/null +++ b/devlog/_plan/260923_anthropic_fast_speed/030_reflection.md @@ -0,0 +1,37 @@ +# Reflection: Anthropic Fast D5 dispatch fallback + +**Verdict: FAIL as written; NEAR-PASS after the concrete changes below.** D1-D4/D6 are consistent with the new probe: all three models accepted the OAuth Fast field but this account lacked entitlement (`devlog/_plan/260923_anthropic_fast_speed/020_probe-evidence.md:9-20,31-42`). The plan's pricing rule for both provider IDs is reasonable as a *confirmed-speed list-price estimate*, but D5's claim that `oauthDispatch` wraps every non-forward Anthropic send, and its proposed hidden second send there, do not hold (`010_plan.md:16-23`). This is a source audit, not a runtime test. + +## Where `oauthDispatch` runs + +| Path | Finding | +| --- | --- | +| Ordinary Responses adapter initial and refetch | **YES**, for `fetchResponse` and generic adapters, via `providerFetch(...dispatchOverride:oauthDispatch(request))`. `src/server/responses/adapter-dispatch.ts:288-329,440-502`. Anthropic currently uses the generic `buildRequest`/fetch path (`src/adapters/anthropic.ts:942-950,1125-1128`). | +| Adapter continuation | **YES**, both adapter-owned and generic fetch branches. `src/server/responses/adapter-continuation.ts:197-240`. | +| Responses sidecar execution, model iteration of image/video and web-search loops | **YES** for the routed model's `fetchForRequest`, which is recreated per iteration with `oauthDispatch(request,iterParsed)`. `src/server/responses/sidecar-execution.ts:332-361,409-423`; `src/images/loop.ts:575-621`; `src/web-search/loop.ts:467-519`. | +| Responses passthrough dispatch | **YES** at its HTTP send/recovery sites, e.g. `src/server/responses/passthrough-dispatch.ts:859-884,1227-1247,1358-1380,1665-1681`. But that path is the `passthrough` adapter branch (`src/server/responses/core.ts:114-125`), not a final `anthropic` adapter eligible for `anthropic-speed`; an override to an OpenAI adapter should fail the Anthropic wire compatibility check (`src/providers/fastwire.ts:204-233`). | +| Routed `/responses/compact` | **YES indirectly**: it builds a synthetic internal Responses request and calls `handleResponses`, which prepares this transport. `src/server/responses/compact.ts:1377-1404`; `src/server/responses/core.ts:101-105`. Native `/responses/compact` is **NO** (direct `providerFetch` with no override), but its gate is restricted to canonical OpenAI backends, not Anthropic. `src/server/responses/compact.ts:760-785,1011-1026`; `src/providers/openai-tiers-destination.ts:58-71`. | +| Native Chat Completions | **NO**: separate `providerFetch` override, restricted to key/local `openai-chat`, so not an eligible Anthropic Messages Fast route. `src/server/chat-native.ts:151-159,319-373`. Other Chat requests translated to Responses can enter the ordinary adapter path. | +| Claude Messages native passthrough | **NO**: direct `fetchWithHeaderDeadline` and caller-auth headers, bypassing the Anthropic adapter and this override. The `--fast` selector currently blocks this shortcut and reaches translation/adapter dispatch, but a caller's *raw* `speed:"fast"` request can still take native passthrough; D5 does not cover it. `src/server/claude-messages.ts:409-450,743-747,782-788`. `/v1/messages/count_tokens` also uses the direct path (`:1262`). Decide explicitly whether native caller-auth fallback is outside scope; do not say every Anthropic send is covered. | +| Anthropic web-search *provider sidecar* | **NO**: separate `runAnthropicWebSearch` uses its own `fetchWithResetRetry` and direct `fetch`, with a body that never requests `speed`. `src/web-search/loop.ts:731-732`; `src/web-search/anthropic-executor.ts:160-218`. It needs no Fast fallback unless Fast is added to that sidecar. | +| `runTurn` adapters | **NO**: they receive `providerFetch` without `oauthDispatch` (`src/server/responses/run-turn-execution.ts:150-184`); Anthropic is not a `runTurn` adapter. `src/adapters/anthropic.ts:942-950`. | +| Forward-auth routes | **NO by design**: `oauthDispatch` returns undefined for `authMode:"forward"`. `src/server/responses/request-transport.ts:403-405`. | + +`prepareResponsesTransport` is composed before the passthrough/sidecar/runTurn/adapter branches (`src/server/responses/core.ts:101-164`), so most *routed Anthropic adapter* requests are covered. That is narrower than every non-forward Anthropic HTTP send. + +## D5 correctness blockers + +1. **Physical-send budget and logs — FAIL.** The outer `fetchWithResetRetry`/`fetchWithTransientRetry` invokes the dispatch callback once and reports that as one send, including the shared request/workflow counters (`src/lib/upstream-retry.ts:619-626,708-726`; `src/server/responses/request-send-budget.ts:49-64`). If `oauthDispatch` calls `sendWithConnectionPolicy` twice before returning, the fallback is invisible to those counters. The outer caller also calls `noteRoutedAttemptSend` once (`src/server/responses/adapter-dispatch.ts:317-329`; sidecar loops `src/images/loop.ts:603-621`, `src/web-search/loop.ts:499-519`), so OAuth attempt `sendCount` remains one. For key auth, `commitKeyAttemptSend()` at `request-transport.ts:422` would have to run again to count the second send (`:250-262`; `src/server/request-log.ts:1779-1825,1858-1865`), but that still leaves the request/workflow budget uncharged. A hidden fallback can exceed the bounded physical-send contract and conceal an extra charge. The budget state is constructed **after** transport preparation (`src/server/responses/core.ts:101-115`), which is another sign that a minimal hook inside this closure is the wrong owner. +2. **Pacing and deadlines — FAIL/unspecified.** `providerFetch` acquires one pacing slot before calling the override (`src/server/responses/fetch-helpers.ts:240-263,288-300`); a second direct `sendWithConnectionPolicy` inside it does not wait for another slot. `fetchWithHeaderTimeout` arms one abort timer around the whole override (`:362-397`): the standard send inherits only time remaining after the Fast refusal. The image/web-search loops also have their own iteration header deadlines (`src/images/loop.ts:556-560`; `src/web-search/loop.ts:445-453,555-561`). Put the retry at those owners so pacing and the chosen per-leg/cumulative deadline policy are explicit; do not reset the sidecar's deliberately cumulative rotation deadline accidentally. +3. **Same-target cache and telemetry — FAIL if only `init` is stripped.** Adapter refetch reuses `sameTargetRequest` by `parsed` reference and `transportToken`; after a standard retry made from a temporary `init`, that cache still contains Fast body/header and an observer reporting Fast (`src/server/responses/adapter-dispatch.ts:277-283,403-429`). Image/web-search iteration caches retain the same request too (`src/images/loop.ts:575-589`; `src/web-search/loop.ts:445-486`). If the standard response later triggers 429/401/413 or another recovery, the next physical send can silently reintroduce Fast, defeat the one-shot promise, and misprice the attempt. Rebuild/replace the cached request or maintain an explicit settled standard request, update `iterParsed`/`parsed` where a rebuild reads it, and invalidate the same-target token when changing tier (`src/server/responses/request-transport.ts:112-121`). The live `tierLog.outcome` attached before dispatch must record the fallback as downgraded/`response-declined`; simply calling `createAdapterTierMetadata` on a `drop` decision would classify the route as wire-unavailable, and leaving the Fast observer untouched could price a standard turn at 2x (`src/providers/fastwire.ts:306-341,367-421`; `src/server/request-log.ts:744-767`; `src/usage/cost.ts:413-434,545-552`). Account rotations must still bind the new credential, as `oauthDispatch` currently does at `:447-475`. +4. **Refusal classification — NEAR-PASS only if narrow.** A generic 429 or 529 on a Fast request is not necessarily a *Fast-pool* refusal. D5 currently says to skip key-failure and Anthropic quota-header observation for any such status (`010_plan.md:20`), but `oauthDispatch` presently records failures and account quota headers before returning (`src/server/responses/request-transport.ts:427-445`), and downstream 429 handling can wait, cool/rotate the account (`src/server/responses/adapter-dispatch.ts:641-670,739-765`; sidecars `src/server/responses/sidecar-execution.ts:210-255`). Use bounded inspection of the 400/429/529 error body and/or documented Fast headers; suppress account effects only for an identified Fast-specific denial. `recordKeyAttemptFailure` is specifically a key-attempt status/usage-preservation hook, not the cooldown mechanism; skipping it for a replaced refusal may also discard usage if that refusal reports any (`src/server/request-log.ts:1748-1763`). The probe gives real patterns: credits-required 429 and org-disabled fast 400 (`020_probe-evidence.md:9-12,31-38`). Keep ordinary quota/overload responses in the existing failure/rotation path. Cancel or bounded-drain the refused response body before the replacement, without logging it. If standard fallback itself fails, return and record that response normally. + +## Safer diff shape + +Use one shared **pure** helper to classify an Anthropic Fast-specific refusal and transform an already serialized `AdapterRequest` into a fresh standard-speed request. Parse the final JSON `init.body`/`wireRequest.body` only after verifying it is a bounded, replayable string object; remove its top-level `speed`, remove only the exact `fast-mode-2026-02-01` token from a case-insensitive `anthropic-beta` header, preserve OAuth betas, provider headers, recovery flags, URL and auth binding. This is preferable to invoking `adapter.buildRequest` inside the override: a rebuild can repeat image normalization/translation, change dynamic OAuth headers and session IDs, and rerun request-budget accounting (`src/adapters/anthropic.ts:958-970,1086-1125`). The plan's `provider.headers`-last policy must still verify the *actual* final headers contain the Fast beta before treating a response as a Fast refusal (`src/adapters/anthropic.ts:1091-1110`). Do not mutate the original request in place until the fallback is admitted; keep the original refusal if a budget, abort or pacing gate refuses the new send. + +Schedule that standard request as a **second visible send** in the owning retry loop: reserve/check the shared budget, call the normal per-send pacing and deadline wrapper, call `noteRoutedAttemptSend`/`commitKeyAttemptSend` once for that physical send, update/invalidate the relevant request cache, and attach an explicit fallback outcome. The main adapter recovery loop (`src/server/responses/adapter-dispatch.ts:393-430,538-670`), continuation (`src/server/responses/adapter-continuation.ts:164-185,274-318`), and image/web-search iteration owners (`src/images/loop.ts:575-625`; `src/web-search/loop.ts:445-519`) need either this hook or a shared dispatch abstraction with callbacks they supply. Scope it to final adapter `anthropic`, an actual `speed:"fast"` body plus beta, and non-forward auth. Limit to one standard fallback per physical Fast request; mark/refuse further fallback after a standard send. This is a required expansion beyond a minimal `request-transport.ts` hook. A bounded two-response integration test should assert two real fetches, two attempt sends and budget charges, fresh pacing/deadline behavior, no Fast on any subsequent refetch, correct failure/header attribution, and 1x cost on the final standard echo. + +## Pricing namespace: PASS + +`formatAnthropicProviderForLog("anthropic", accountId)` returns `anthropic-p<hex6>` (`src/oauth/anthropic-routing.ts:859-870`; `src/codex/account-label.ts:7,49-50`). `baseProviderLabel` recognizes that suffix and returns `anthropic` (`src/providers/label.ts:26-38`; `tests/usage/usage-provider-label.test.ts:24-26`). `resolveMatchedPrice` collapses the label unless it is literally a configured provider name, then `estimateAttemptCost` passes the resulting `price.provider` into the multiplier lookup (`src/usage/cost.ts:183-202,543-552,486-497`). Thus ordinary pooled OAuth labels match `provider:"anthropic"` rules. Add a regression for `anthropic-pabcdef` with confirmed Fast, and for `anthropic-apikey` with the same model; keep a same-named configured provider and exact user-overlay precedence intact. The rule's `requiresResponseConfirmation` is essential because no OAuth account in the probe returned a confirmed Fast 200 (`020_probe-evidence.md:31-38`). diff --git a/devlog/_plan/260923_anthropic_fast_speed/040_audit.md b/devlog/_plan/260923_anthropic_fast_speed/040_audit.md new file mode 100644 index 0000000000..72c81b35c6 --- /dev/null +++ b/devlog/_plan/260923_anthropic_fast_speed/040_audit.md @@ -0,0 +1,56 @@ +# Anthropic Fast plan — independent source audit (2026-09-23) + +Scope: `010_plan.md` D2/D3/D5/D6 against the current source and the supplied probe, reflection, docs, and code map. Read-only source review; no tests or live requests run. **Verdict: FAIL as written**, with a localized repair path. The revised placement is sound, but the current generic Anthropic resend path does not meet D5's stated request/workflow budget guarantee, and the memo/tier/metrics semantics need explicit treatment before build. + +## Blocking findings + +1. **D5 physical-send budget is not automatically charged by `rebuildAndRefetch` for Anthropic.** `src/server/responses/adapter-dispatch.ts:470-513` constructs `refetchAllowance` and supplies `attempts`/`onSendsConsumed: noteTransientSends` only when `transientRetryPolicyFor` returns a policy. Ordinary Anthropic takes `fetchWithResetRetry` without those options, which gets its own default attempts (`src/lib/upstream-retry.ts:603-626`). The initial generic leg likewise omits the shared counter in the no-policy branch (`src/server/responses/adapter-dispatch.ts:313-343`). `noteRoutedAttemptSend` increments the durable attempt but does not charge `sendBudget.used` or the workflow counter (`src/server/responses/request-transport.ts:253-261`; `src/server/responses/request-send-budget.ts:55-64`). Thus inserting the new arm and calling `rebuildAndRefetch("anthropic-fast-downgrade")` yields a separately paced, timed, attempt-logged send, **but not the promised shared-budget admission/charge**; a reset inside that leg may add another send. The fast arm needs an explicit shared allowance/dispatch reservation and a physical-send accounting path, including the reset helper, with a test at a spent cap/workflow ceiling. Do not cancel the real refusal or consume its body until fallback admission is secured; preserve it when admission fails. Trigger: fast-specific 400/429 after earlier sends or a workflow near its ceiling. Impact: cap bypass and invisible extra upstream work. + +2. **D5 tier downgrade cannot be inferred from a null wire value.** For an eligible `set` decision, `createAdapterTierMetadata(..., null, null)` produces `fastOutcome: "downgraded"` but reason `wire-unavailable`, not `response-declined` (`src/providers/fastwire.ts:306-341,358-376`). If the memo-suppressed rebuild uses that factory unchanged, it misstates why the request went standard. `recordAdapterTier` *does* overwrite/delete the old attempt and active observer (`src/server/request-log.ts:744-767`) when `rebuildAndRefetch` takes its build branch (`src/server/responses/adapter-dispatch.ts:403-429`), but only after `invalidateSameTargetRequest()` bumps the token (`src/server/responses/request-transport.ts:112-117`); otherwise it reuses the refused Fast request and never records a new tier. Explicitly mark the rebuilt metadata `downgraded`/`response-declined` with null wire value, then attach it before `recordAdapterTier`, including a terminal standard echo and subsequent 401/429 refetch test. Trigger: first Fast refusal. Impact: inaccurate outcome/price provenance or Fast being reintroduced from cache. + +3. **Token-derived memo keys lose account identity on OAuth refresh.** The proposed `sha256(apiKey)+wireModelId` separates six pooled OAuth accounts correctly while their tokens remain stable, and separates static keys. However `oauth-401` refresh replaces `route.provider.apiKey` with `refreshed.accessToken` (`src/server/responses/adapter-dispatch.ts:553-587`); pool rotation also replaces it (`:755-762`). The same OAuth account with a refreshed access token therefore misses its earlier refusal memo, contradicting D5's claim that later builds remain standard. Key OAuth entries by stable account ID plus provider/endpoint/model, using the selected/sent snapshot (`src/server/responses/request-transport.ts:235-246,402-445`), and key static API keys by a sufficiently long hash plus endpoint/model. Keep the bounded TTL so newly granted entitlement can recover. Trigger: token refresh before TTL. Impact: repeated rejected Fast sends and inconsistent continuation behavior. A short unspecified SHA-256 *prefix* also needs a stated collision bound. + +4. **The promised fallback is limited to the main adapter recovery loop.** The selected arm after the OAuth/key 401 branches and before the same-target 429/key/OAuth rotation at `src/server/responses/adapter-dispatch.ts:599-687` is the correct ordering: a recognized Fast 429 should not wait, cool, or rotate. But continuation performs its own fetch/429 handling (`src/server/responses/adapter-continuation.ts:162-245,274-315`) and image/web-search iteration sends run through independent owners (`src/server/responses/sidecar-execution.ts:332-365,409-450`). A process memo suppresses their *later builds* after a main-loop refusal; it does not convert a Fast refusal on their first send. The plan names this as a residual. C4's unqualified "a recognized fast refusal is replaced" should be narrowed to main dispatch, or add a visible resend at each owner. Trigger: first Fast send on a continuation/sidecar iteration. Impact: the request fails despite an available standard-speed retry. + +## Required implementation/consumer inventory + +- **Recovery vocabulary:** add the literal to `src/usage/telemetry-contract.ts:25-41` (the single roster/type), `src/lib/request-failure-model.ts:214-232` (`satisfies Record<AttemptRecoveryKind, RequestFailureCause>`), and `gui/src/pages/Logs.tsx:329-343` (`satisfies Record<AttemptRecoveryKind,string>`). `src/usage/log.ts:539,673-674` derives its read-back whitelist automatically; do not add a second roster. The source-oracle/parity tests are `tests/usage/request-outcome-agreement.test.ts:179-210`, `tests/lib/failure-stage-model.test.ts:178-204`, `tests/lib/ambiguous-resend-gate.test.ts:154-162`, and `tests/lib/failure-attribution.test.ts:130-145`. Add behavioral cases for the new kind. A new regression file also needs entries in `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`; those files do not enumerate recovery kinds. +- **Status/failure attribution:** `src/lib/request-failure-attribution.ts:98-109` has a partial, status-confirmed recovery map; the request failure model only invokes it for final 400+ other than 429 (`:149-158`). Add a **400** mapping if a repaired Fast parameter is rejected again. Do not classify a final standard-speed 429 as `parameter-rejected`: that is an ordinary rate limit after the repair. Tests belong in `tests/lib/failure-attribution.test.ts:130-145` and `tests/lib/failure-stage-model.test.ts:178-204`. +- **Metrics export:** the cause map feeds `src/server/request-metrics.ts:127-146`. Today `parameter-rejected` maps to `effort_downgrade` (`:135`), so the new Fast recovery would be exported as a reasoning-effort downgrade. Use a semantically general closed class (or a recovery-kind-specific mapping) and update `REQUEST_METRICS_RECOVERY_CLASSES` at `:30-42`, `tests/server/management-metrics-export.test.ts:350-370,465`, `structure/gui-and-management-api.md:662-670`, and the closed-class list in `docs-site/src/content/docs/reference/management-api.md:290-305` if the class name changes. Do not export a model/account/refusal text label. +- **GUI localization:** add the new key in all ten catalogs: `gui/src/i18n/{en,ko,ja,zh,zh-TW,fr,de,ru,tr,vi}.ts` (current sibling `reasoningEffortDowngrade` at en:946, ko:928, ja:857, zh:909, zh-TW:2422, fr:922, de:897, ru:914, tr:933, vi:920). `tests/usage/request-outcome-agreement.test.ts:200-210` requires every `Logs.tsx` recovery key in all catalogs. The existing failure-cause labels remain valid if the cause remains `parameter-rejected`. +- **Docs:** `structure/transports/responses.md:272-289,1409` owns the recovery/FastWire transport contract; `structure/providers-and-adapters.md:20-23` owns the Anthropic adapter, `structure/gui-and-management-api.md:616-624,662-670` owns cost/metrics, and `docs-site/src/content/docs/reference/configuration/providers.md:482-510` explains FastWire capabilities. The management API metrics page above changes only if the closed metric class changes. `docs-site/src/content/docs/guides/claude-code.md` should describe the user-facing Anthropic Fast route and its entitlement/fallback limits, without claiming native raw passthrough is covered. No other docs-site page is an exhaustive recovery-kind roster in the searched source. + +## D2/D3/D6 checks + +- **D2 IDs: PASS.** `src/providers/registry/model-seeds.ts:12-13` spells exactly `claude-opus-5-5`, `claude-opus-5`, `claude-opus-4-8`, matching the plan. Both Anthropic entries currently lack Fast declarations (`src/providers/registry/entries-core.ts:389-425`); `src/providers/fastwire.ts:13-20,204-233` currently has an empty `anthropic-speed` adapter set. Enrichment is fill-only for `fastWire` (`src/providers/derive.ts:569-584`) and merges registry model defaults beneath saved explicit values (`:405-413`); `src/providers/service-tier.ts:82-107` gives configured model entries final precedence. The relevant flipping pins are `tests/routing/fastwire-policy.test.ts:201-218,649-660`. `tests/codex-integration/fast-row.test.ts:102-108` uses an implicit/no-declaration Anthropic fixture, so its negative expectation should remain, while its stale comment changes. `tests/service/service-tier-capability.test.ts:270-282,342-352` likewise has no FastWire declaration; keep those negatives. Add saved-config backfill and explicit `fastWire:null` / per-model `false` tests. A custom destination/renamed provider needs separate behavior: the registry match/enrichment is conditional (`src/providers/derive.ts:485-497`; `src/providers/registry.ts:75-81,103-116`), so do not claim unconditional backfill. +- **D3 header merge: needs a precise final-header rule.** `src/adapters/anthropic.ts:1091-1110` seeds OAuth `anthropic-beta`, then uses case-sensitive `Object.assign(headers, provider.headers)`. An override spelled `Anthropic-Beta` can coexist with `anthropic-beta`; `new Headers(...)` comma-joins rather than overriding, the exact defect described in `src/providers/registry.ts:55-59`. Merge case-insensitively into one header, dedupe beta tokens, preserve the OAuth set, and define how a user override interacts with the required Fast beta. If final user headers remove the Fast beta, do not send `speed:"fast"` or classify a subsequent error as a Fast refusal; the probe shows speed without beta is rejected (`020_probe-evidence.md:9-12`). The memo-suppressed build must remove only the Fast beta while retaining OAuth/user betas. Add different-case and explicit-override tests. +- **D6 pooled labels: PASS for rule lookup, with a pricing test gap.** `formatAnthropicProviderForLog` emits an `anthropic-p...` account label (`src/oauth/anthropic-routing.ts:863-869`); `baseProviderLabel` collapses recognized suffixes (`src/providers/label.ts:26-37`); `resolveMatchedPrice` uses the base unless a literal configured provider or exact user overlay owns the name (`src/usage/cost.ts:183-202`). `estimateAttemptCost` then looks up the multiplier against `price.provider` (`:543-552`). Thus rules for `anthropic` and `anthropic-apikey` reach ordinary pool and key rows without leaking to resellers. `requiresResponseConfirmation` prevents assumed/standard turns from receiving 2x (`src/usage/cost.ts:413-434,486-504`). Add tests for `anthropic-p...`, key auth, explicit user overlay precedence, and standard/absent echo. The multiplier requires a resolvable base `Cost4`; `src/usage/expected-prices.ts:222-228` has explicit Anthropic Opus 5.5 and OAuth Opus 5 rows but no `anthropic-apikey/claude-opus-5` or Anthropic Opus 4.8 row, so verify the bundled/vendor fallback for those two in a cost test and add exact base overlays if absent. Also consider marking an assumed Fast cost provisional: current `priorityLowerBound` special-cases OpenRouter only (`src/usage/cost.ts:507-526,554-566`). + +No tracked files changed by this audit. No local checks were run; all findings are source-level, with the supplied probe as external behavior evidence. + +VERDICT: FAIL + +## Round 2 + +- **#1 budget: unresolved.** D5 now states the limitation honestly (`010_plan.md:24,45`), but `!sendBudgetExhausted()` is ineffective on ordinary Anthropic: reset-only initial/refetch legs omit `onSendsConsumed`, so `sendBudget.used` does not advance (`src/server/responses/adapter-dispatch.ts:313-343,470-513`; `src/server/responses/request-send-budget.ts:55-64,90-95,140-141`). Calling the new resend once per request bounds this *arm*, not total physical sends or workflow spend; a reset inside its helper can send again. “Pre-existing in other arms” is a scope note, not a sound budget rebuttal. Either explicitly adopt that risk without describing the guard as effective budget admission, or account for the new leg and test the cap. Preserve the original cloned refusal until dispatch is admitted. +- **#2 tier and #3 memo: folded.** D5 switches to a request-local `drop` decision plus `upstreamDeclinedFast` (`010_plan.md:21-23`), invalidates the same-target cache, and plans the specific `response-declined` outcome. `tierObservationContext()` returns a plain mutable object; no freeze of it or `parsed.options` was found (`src/providers/fastwire.ts:265-279`; `src/server/responses/core-normalize.ts:215-222`). `recordAdapterTier` on the forced rebuild replaces the old observer (`src/server/responses/adapter-dispatch.ts:403-429`; `src/server/request-log.ts:753-767`). Replacing `tierDecision` does not mutate the prior decision or `_rawBody`; the “TierDecision immutability” test checks serialized behavior and raw-body identity (`tests/routing/fastwire-policy.test.ts:663-690`). Add the optional-property guard/type extension: `tierObservation?` is optional in `src/types/request.ts:295-298`, so a bare `.upstreamDeclinedFast = true` needs a narrowing or assignment and the new field in `src/types/provider.ts:225-245`. +- **#4 scope, metrics, header, pricing: folded.** C4 explicitly limits fallback to main dispatch, with first-send continuation/sidecar failures residual (`010_plan.md:26,37,46`). Continuations use the same `parsed` (`src/server/responses/adapter-continuation.ts:564-573`); image iterations reuse or spread `parsed.options` (`src/images/loop.ts:451-456,68-93`), and web-search iterations do likewise (`src/web-search/loop.ts:445-449`). These owners run before the main adapter exchange (`src/server/responses/core.ts:127-155`), so D5's claim about *later sidecar builds after a main-loop downgrade* is effectively vacuous, but does not introduce a defect. The additive `fast_downgrade` metric class, case-insensitive beta merge with required Fast beta last, and base-price tests for Opus 5/4.8 are concrete follow-ups (`010_plan.md:25,27-28`; `040_audit.md:33-39`). D3's older “provider overrides last” wording at `010_plan.md:18` should defer explicitly to D3a's Fast-beta-last rule. + +Residual: shared request/workflow budget is still uncharged on reset-only Anthropic refetches; first Fast refusals in continuation/sidecar owners do not downgrade. + +VERDICT: FAIL + +## Round 3 + +- **New leg:** The proposed `reserveCredentialHop("repair", ..., false)` admits and charges one request-budget send at reservation; `permit.use()` confirms it after rebuild/pacing, and `release()` refunds a pre-send failure (`src/lib/request-execution-budget.ts:305-380`; `src/server/responses/adapter-dispatch.ts:393-448,805-866`). Anthropic has no adapter-owned `fetchResponse` and cannot use `transientRetryPolicyFor`, so `countedExternally=false` is correct (`src/adapters/anthropic.ts:945-950,1125-1128`; `src/providers/key-failover.ts:609-623`). Its generic retry helper has no ambiguous-resend grant, so a reset does not automatically buy a second send (`src/lib/upstream-retry.ts:619-626,652-665`). The permit must be cleared in `finally`; a rejected reservation must leave the cloned original response untouched. D3 now explicitly defers to D3a (`010_plan.md:18,27`). +- **Residual budget gap:** This resolves the *new leg's own request-ledger charge*, but not the cumulative guarantee in blocker #1. Existing reset-only initial/refetch sends still omit `onSendsConsumed` (`src/server/responses/adapter-dispatch.ts:317-343,470-513`), so `reserveDispatch` sees a counter below the number of physical sends already made. Also `reserveCredentialHop` only calls `sendBudget.reserveDispatch` (`src/server/responses/request-send-budget.ts:242-249`); `chargeWorkflowSends` is invoked only by `noteTransientSends` (`:60-64`), which the Anthropic reset-only leg does not call. Thus the new send is booked in the request ledger but is not charged to the workflow send ceiling. A synthetic spent-budget test alone would miss both gaps. Either narrow D5's “real shared-budget charge” to the per-request reservation and record the workflow/cumulative limits as explicit residuals, or wire physical-send reporting once for all reset-only legs. + +VERDICT: NEAR-PASS + +## Fold (main, after round 1) + +- #1 budget: arm guarded by !sendBudgetExhausted(); reset-only budget charging is pre-existing for every rebuildAndRefetch arm, recorded as residual. +- #2 tier reason: TierObservationContext.upstreamDeclinedFast -> response-declined. +- #3 memo identity: memo removed; per-request decision on parsed.options. +- #4 scope: C4 narrowed to main dispatch; continuation/sidecar first-send residual. +- Metrics: additive fast_downgrade class via kind override. Header merge: D3a. Pricing: base tuple checks for anthropic-apikey opus-5/4-8 in cost tests. diff --git a/devlog/_plan/260923_anthropic_fast_speed/050_build.md b/devlog/_plan/260923_anthropic_fast_speed/050_build.md new file mode 100644 index 0000000000..064d2cfac4 --- /dev/null +++ b/devlog/_plan/260923_anthropic_fast_speed/050_build.md @@ -0,0 +1,32 @@ +# 050 Build and verification (wp1) + +## What landed + +- `src/providers/fastwire.ts`: `anthropic-speed` is available on the `anthropic` adapter; an observation with `upstreamDeclinedFast` reports `response-declined` instead of `wire-unavailable`. +- `src/providers/anthropic-fast.ts` (new): the beta constant, narrow refusal recognition, and a case-insensitive `anthropic-beta` merge. +- `src/adapters/anthropic.ts`: a `set` decision on the declared wire sends `speed: "fast"` with the beta; the adapter owns `tierLog`; `usage.speed` is observed in `message_start`, `message_delta` and buffered bodies. +- `src/providers/registry/entries-core.ts`: `anthropic` and `anthropic-apikey` declare the wire and classify `claude-opus-5-5`, `claude-opus-5`, `claude-opus-4-8`. +- `src/server/responses/adapter-dispatch.ts` + `core-opaque-recovery.ts`: one budget-reserved standard resend on a recognized fast refusal, before every 429 arm. The physical resend charges the root workflow once; the request permit is not charged twice. +- Recovery kind `anthropic-fast-downgrade` (cause `parameter-rejected`, metrics class `fast_downgrade`, log label in ten locales); 2x confirmation-gated pricing rules; docs and structure owners. + +## Live smoke (real OAuth token, repository adapter, 2026-09-23) + +| Leg | Sent | Result | Tier outcome | +|---|---|---|---| +| claude-opus-5-5, registry-eligible, set | `speed: fast` + beta | 429 "Usage credits are required for fast mode.", recognized as a fast refusal | applied / assumed at send | +| same request after downgrade | no speed, no fast beta | 200 "OK" | downgraded / response-declined | +| claude-opus-4-6 with an operator capability override, set | `speed: fast` + beta | 200 "OK", `usage.speed: "standard"` | downgraded / response-declined (live echo) | + +## Checks + +- Original implementation (prior head `5e4cb7ea77`): the earlier PR Verification section recorded `bun run typecheck`, `bun run structure:check`, `bun run privacy:scan`, `bun run lint:gui`, `bun run skill:surface:check`, `git diff --check`, focused tests, and directory runs. Those results do not certify the repair head. +- Repair checkout based on `ea0fab74a3ddb7b485452faa24a2d15b1036080b`: `bun install --frozen-lockfile` passed (104 packages); `bun test tests/responses/responses-anthropic-fast-downgrade.test.ts` passed (8 pass, 0 fail); `bun run typecheck` passed; `git diff --check` passed. Earlier focused-test runs during the repair failed on an assertion against the wrong public error string and then on a test-injected spend-home owner leak; both test defects were corrected before the final pass. +- The repair was deliberately limited to the focused test and typecheck. The full suite, docs build, privacy scan, structure check, and other original focused tests were not rerun on this repair head. Hosted exact-head CI remains required before landing; cancelled test shards on the prior head are missing evidence, not a pass. + +## Delegation + +gpt-6-sol leaves: Helmholtz (Aside docs research), Nash (code map, reflection), Kant (independent audit, three rounds), Avicenna (log label + locales), Heisenberg (pricing), Nietzsche (docs), Cicero (new tests). + +## Rendered request-log label + +An isolated in-process proxy (throwaway OPENCODEX_HOME, local fake Anthropic upstream answering the fast send with the credits 429) served one request end to end: the fast send was refused, the standard resend answered, and the request detail shows the new recovery label. Capture: `evidence/logs-fast-downgrade.png`. diff --git a/devlog/_plan/260923_anthropic_fast_speed/060_done.md b/devlog/_plan/260923_anthropic_fast_speed/060_done.md new file mode 100644 index 0000000000..c7607d2e55 --- /dev/null +++ b/devlog/_plan/260923_anthropic_fast_speed/060_done.md @@ -0,0 +1,21 @@ +# 060 Done (wp1) + +## Conclusion + +Claude fast mode is a native FastWire (`anthropic-speed`) on the Anthropic adapter for `claude-opus-5-5`, `claude-opus-5` and `claude-opus-4-8` on both the OAuth and API-key providers, with `usage.speed` confirmation, a budgeted one-shot standard-speed resend on a recognized fast refusal, a `fast_downgrade` metrics class and request-log label, and confirmation-gated 2x pricing. PR #5604 targets `dev`; the initial completion record described head `5e4cb7ea77`. The later repair and its validation are recorded in `050_build.md` and the PR Verification section. + +## Evidence + +- Live probe matrix (020) and live smoke through the new adapter code (050): refusal → standard resend → `downgraded/response-declined`; Opus 4.6 live standard echo downgrades. +- Isolated end-to-end through the real proxy pipeline with a fake upstream, rendered request-log label (050, `evidence/logs-fast-downgrade.png`). +- Original local gates and focused/directory tests were recorded for the initial implementation head. The repair's focused validation is in `050_build.md`; exact-head hosted CI is still required. + +## What did not complete + +- Hosted CI was not complete on the initial head: Cross-platform CI runs 35781678978 and 35783578147 attempt 2 were cancelled during the release window. The repair head needs its own successful exact-head checks; prior cancelled runs cannot certify it. +- No user account can currently serve a live `usage.speed: "fast"` 200 (four need usage credits, two orgs have fast disabled). +- The full local suite was not run for this repair. The scoped checkout ran only the focused regression file and typecheck; other gates remain for hosted CI. + +## Next + +Verify exact-head CI on the repair commit; merge is the maintainer's call. diff --git a/devlog/_plan/260923_anthropic_fast_speed/evidence/logs-fast-downgrade.png b/devlog/_plan/260923_anthropic_fast_speed/evidence/logs-fast-downgrade.png new file mode 100644 index 0000000000..4fb71bf3f5 Binary files /dev/null and b/devlog/_plan/260923_anthropic_fast_speed/evidence/logs-fast-downgrade.png differ diff --git a/devlog/_plan/260923_bundle_a_tests_hygiene/000_triage.md b/devlog/_plan/260923_bundle_a_tests_hygiene/000_triage.md new file mode 100644 index 0000000000..aba50e4151 --- /dev/null +++ b/devlog/_plan/260923_bundle_a_tests_hygiene/000_triage.md @@ -0,0 +1,20 @@ +# Lane A — tests hygiene triage + +Bundle lane A of the 260923 PR-consolidation round. One branch (codex/260923-bundle-a-tests-hygiene) from origin/dev 685321e297, one commit per carried PR, one PR to dev. Each candidate got a read-only gpt-6-sol soundness review against current dev. + +| PR | Author | Verdict | Carry notes | +|---|---|---|---| +| #5607 | FredAmartey | CARRY | Translator budgets disposed per test via onTestFinished; module-level afterEach only fires for the first importing file in a shared process. | +| #5605 | FredAmartey | CARRY-WITH-FIXES | Restore real modules after image-test mock.module overrides. Fix: capture each real module before its first override and restore only captured snapshots (a partial beforeAll must not install an empty module); z-handler-activation restores even if directory cleanup throws. | +| #5570 | FredAmartey | CARRY | Every test file that pins OPENCODEX_HOME restores the inherited value; commit 2 already folded the CodeRabbit ordering finding. | +| #5482 | FredAmartey | CARRY | Capture resolveAdapter before mock.module rewrites the live binding (three files). | +| #5630 | sh940701 | CARRY-WITH-FIXES | Guard the real desktop restart adapter when OCX_TEST_HOME_GUARD=1 and no execFile is injected. Fix: document the armed-test skip and CLI outcome in structure/runtime.md. | +| #5340 | codingbooo | CARRY-WITH-FIXES | README memory inventory counts derived from registries with a per-locale guard test. Fix: rebuild on dev after #5615 (keep its prose), retained stores are now 14 (native_control_replay is pinned, evictOldest returns 0, so the "all evicted" wording changes), recompute readme/i18n-manifest.json hash from the final README, register the new test in layout.json and test-layout-expected.json. | + +## Issue #5439 + +Part 1 (batched runner cannot run on macOS: GNU timeout, mapfile) is already fixed on dev by #5456 (portable process-group timeout fallback, no mapfile). Part 2 (failure counts depend on batch size) is caused by the cross-file leaks that #5570, #5605 and #5607 fix; tests/server/config.test.ts already restores its cwd. The single-owner spend ledger errors are the intended owned-spend-home contract. The PR references the issue with the evidence; closing is the coordinator's call. + +## Verification plan + +Focused files per carry, including non-isolated same-process pairs that reproduce the leak (the reviewers' named orderings), then bun run typecheck, bun run structure:check, bun run privacy:scan, tests/test-layout*.test.ts and the file-size ratchet test. No full local suite. diff --git a/devlog/_plan/260923_bundle_a_tests_hygiene/010_build.md b/devlog/_plan/260923_bundle_a_tests_hygiene/010_build.md new file mode 100644 index 0000000000..442c4ba771 --- /dev/null +++ b/devlog/_plan/260923_bundle_a_tests_hygiene/010_build.md @@ -0,0 +1,12 @@ +# Lane A — build order + +Commits on codex/260923-bundle-a-tests-hygiene, in order, each with the original author's Co-authored-by trailer: + +1. #5482 capture resolveAdapter before mocking — Fred Amartey <43480311+FredAmartey@users.noreply.github.com>. Applied as-is. +2. #5607 per-test translator budget disposal — Fred Amartey. Applied as-is. +3. #5570 restore inherited OPENCODEX_HOME in every test file — Fred Amartey. Both PR commits squashed into one carry. +4. #5605 restore real modules after image mocks — Fred Amartey. Fold: snapshot each module before its first override, restore only captured snapshots, and restore in z-handler-activation before the throwable directory cleanup. +5. #5630 guard the real desktop restart adapter in armed test processes — terin <100397903+sh940701@users.noreply.github.com>. Fold: structure/runtime.md documents the skipped outcome. +6. #5340 derived README memory inventory counts — codingbo <9621077+codingbooo@users.noreply.github.com>. Rebuilt on dev after #5615: 14 retained stores, pinned-store eviction wording, recomputed readme/i18n-manifest.json hash, new test registered in layout.json and test-layout-expected.json. + +Focused proof per commit: the PR's named files plus a non-isolated same-process ordering that reproduced the leak on dev (run before and after where cheap). diff --git a/devlog/_plan/260923_bundle_a_tests_hygiene/020_delivery.md b/devlog/_plan/260923_bundle_a_tests_hygiene/020_delivery.md new file mode 100644 index 0000000000..277e294f0e --- /dev/null +++ b/devlog/_plan/260923_bundle_a_tests_hygiene/020_delivery.md @@ -0,0 +1,11 @@ +# Lane A — delivery + +1. gpt-6-sol adversarial review of origin/dev..codex/260923-bundle-a-tests-hygiene; fold or rebut every finding. +2. Push the branch (no-verify), open one PR to dev from the repository template with Supersedes #5482 #5607 #5570 #5605 #5630 #5340, Refs #5439 with the findings, credit list, verification commands and the environment-only issue-914 note. +3. Watch exact-head CI; a run cancelled by the 2.64 release coordinator is re-dispatched after the release, never read as a failure. +4. Final report to the coordinator. +## Delivery record + +- PR #5672 to dev from codex/260923-bundle-a-tests-hygiene; supersedes #5482, #5607, #5570, #5605, #5630 and #5340; refs #5439. +- Adversarial review: P2 dashboard finding withdrawn (the gap predates this branch for every desktop skip reason, and test_environment only occurs under OCX_TEST_HOME_GUARD=1), P3 EOF nits fixed. +- Exact-head CI is read from the PR head only; a run cancelled by the 2.64 release coordinator is re-dispatched after the release. diff --git a/devlog/_plan/260923_bundle_f1_provider_registry/000_overview.md b/devlog/_plan/260923_bundle_f1_provider_registry/000_overview.md new file mode 100644 index 0000000000..71a994d510 --- /dev/null +++ b/devlog/_plan/260923_bundle_f1_provider_registry/000_overview.md @@ -0,0 +1,7 @@ +# 260923 bundle lane F1 — provider registry + +Lane of the 260923 PR lane bundle round (coordinator plan: devlog/_plan/260923_pr_lane_bundle/ in the coordinator worktree). One branch, codex/260923-bundle-f1-provider-registry, cut from origin/dev 685321e297, one PR to dev. Every carry was reviewed by a gpt-6-sol reviewer against current dev before it was rebuilt. + +Docs: + +- 010_carry_plan.md — per-PR verdicts, carry order, required fixes, exclusions. diff --git a/devlog/_plan/260923_bundle_f1_provider_registry/010_carry_plan.md b/devlog/_plan/260923_bundle_f1_provider_registry/010_carry_plan.md new file mode 100644 index 0000000000..1c91b4c028 --- /dev/null +++ b/devlog/_plan/260923_bundle_f1_provider_registry/010_carry_plan.md @@ -0,0 +1,14 @@ +# Carry plan + +| Order | Item | Verdict | Commit contents | +|---|---|---|---| +| 1 | #5362 cursor composer-2.5-fast external tool continuation | carry with fixes | Route composer-2.5-fast through cursorNeedsExternalToolContinuation; rewrite the contradicting assertions in cursor-blob (at its line cap, so replace, never add), cursor-live-transport and cursor-tool-continuation tests; keep a native resumeAction counterexample (composer-1); cover the clipped-invocation case for fast; update structure/providers/cursor.md. | +| 2 | #5314 Meta Muse web_search strip for every model id | carry with fixes | Dev already strips search_content_types and indexed_web_access for Contributor ids; extend it to every id on the direct Meta host, including a missing id and the muse-spark-1.3 default; document in structure/transports/responses-wire-shapes.md and correct the "unrelated models" wording; tests in muse-spark-web-search-compat.test.ts (openai-responses-passthrough.test.ts is at its cap). | +| 3 | #5349 connect deadline for provider artifact downloads | carry with fixes | 10 s connect deadline on the connectPublicHttps production path and a per-call option on pinnedHttpsGet; correct the rationale (a 60 s first-byte timer already exists, the new deadline bounds TCP/TLS setup specifically); document in the owning transport contract; register the new test file in both layout manifests. | +| 4 | #5188 Alibaba Token Plan Responses wire defaults (closes #5097) | carry with fixes | Default Responses pins for qwen3.8-flash, qwen3.7-plus and glm-5.3 on the Beijing preset for Responses inbound, keeping Chat/Anthropic routing and explicit modelAdapters overrides; fix the glm-5.3 test to assert the Responses default and assert the qwen3.7-plus reasoning.effort payload; rebuild the docs row and layout hunks; update responses-wire-shapes.md. | +| 5 | #5000 kiro short-name executable fallback | PARTIAL | Security review failed the Unix part: ~/.local/bin, /usr/local/bin and /opt/homebrew/bin are shared directories where an unrelated kiro binary (for example the Kiro IDE launcher) could be run for credential commands. Carry only the Windows fallback to kiro.exe inside the dedicated Kiro-Cli install folders that dev already trusts for kiro-cli.exe, skip it when the base path is not absolute, and keep canonical-name-first order. Not superseded; the original PR stays open. | +| — | #5147 CodeBuddy roster discovery (#5146) | excluded | codebuddy --help returns the home-logged-in account's roster regardless of the key passed, while the PR caches it under the configured key's hash, so it can advertise another account's models for a key. Needs a way to prove roster and request key belong to the same account first (owner design decision). #5146 stays open; its tool-bridge half is already on dev. | + +Residual for the owner: #5362 changes fast routing for every client; the maintainer review asked for one direct Cursor fast tool turn to confirm it still answers. This lane does not spend live Cursor calls. + +Verification per commit: focused test files for the touched area and their consumers, then bun run typecheck, bun run structure:check, bun run privacy:scan, layout guards (tests/test-layout.test.ts, tests/test-layout-tooling.test.ts) and the file-size ratchet test. No full local suite (reserved for the owner after all lanes land). diff --git a/devlog/_plan/260923_bundle_f1_provider_registry/020_delivery.md b/devlog/_plan/260923_bundle_f1_provider_registry/020_delivery.md new file mode 100644 index 0000000000..b404ae54a3 --- /dev/null +++ b/devlog/_plan/260923_bundle_f1_provider_registry/020_delivery.md @@ -0,0 +1,6 @@ +# Delivery + +1. Adversarial gpt-6-sol review of the whole branch and a security re-review of the rebuilt #5000 commit. Fold every accepted finding as a follow-up commit; record rebuttals in the PR. +2. Push codex/260923-bundle-f1-provider-registry and open one PR to dev with the repository template: Summary, Verification (exact focused commands and counts, full suite not run by owner instruction), Checklist, Closes #5097, Supersedes lines for the fully carried PRs (#5362, #5314, #5349, #5188), #5000 listed as partial (not superseded), #5147 and #5146 listed as excluded with the reason, and every Co-authored-by credit. +3. Watch exact-head CI. A run cancelled by the 2.64 release coordinator is re-dispatched after the release; real failures are fixed on the branch. +4. Final report to the coordinator. diff --git a/devlog/_plan/260923_bundle_lane_e/000_overview.md b/devlog/_plan/260923_bundle_lane_e/000_overview.md new file mode 100644 index 0000000000..b2582aaf0a --- /dev/null +++ b/devlog/_plan/260923_bundle_lane_e/000_overview.md @@ -0,0 +1,5 @@ +# 260923 bundle lane E — responses/combo + +Lane E of the coordinator round devlog/_plan/260923_pr_lane_bundle (coordinator task 01a0cda3-22de-7680-b771-f5338e65ec57). Task 01a0cdb0-b76c-7511-a70f-3bdee1788dc1, worktree .codex/worktrees/0ba7, branch codex/260923-bundle-e-responses-combo from origin/dev 685321e297. One branch, ordered commits, one PR to dev. + +Docs: 010_decisions.md (per-candidate verdict and carry order), 020_carry.md (commit plan with fixes folded from gpt-6-sol soundness reviews), 030_verify.md (focused tests and gates). diff --git a/devlog/_plan/260923_bundle_lane_e/010_decisions.md b/devlog/_plan/260923_bundle_lane_e/010_decisions.md new file mode 100644 index 0000000000..80f4f4c7ec --- /dev/null +++ b/devlog/_plan/260923_bundle_lane_e/010_decisions.md @@ -0,0 +1,20 @@ +# Decisions + +Soundness reviews by gpt-6-sol reviewers, notes in the lane worktree .tmp/lane-e/review-*.md (scratch, not committed). Every PR merges cleanly into dev, and the cumulative stack simulates cleanly with git merge-tree. Carry order is 5629 -> 5659 -> 5646 -> 5633 -> 5489: #5646 first so the WebSocket row #5633 adds is never exposed to the 2xx third-send gap. + +| Item | Verdict | Decision | +|---|---|---| +| #5629 Devin approximate retry delays | SOUND-WITH-FIXES | Carry; fix the stale "~ prevents re-parsing" comment in src/adapters/devin/cloud-direct/chat.ts and reject a repeated approximation marker (retry after ~1 minute ~30 seconds) with a negative parser case. | +| #5659 code-mode goal helpers | SOUND-WITH-FIXES | Carry; add the original guard-input assertion, an unrelated-name rejection and bare-goal precedence case; update stale authorization comments in src/types/tools.ts. Closes #5495. | +| #5646 stop failover once the replacement is spent | SOUND-WITH-FIXES (pair) | Carry first of the pair. Closes the shared resend-safety gap (CodeRabbit's #5633 2xx finding). | +| #5633 WebSocket retryOnReset replacement | SOUND-WITH-FIXES (pair); UNSOUND alone | Carry after #5646; drop the duplicate settleOperatorReplacement import in passthrough-dispatch.ts; rewrite structure/transports/responses-failover.md so the 2xx gap reads as settled. Partial for #4191 (not in this lane's list; not claimed). | +| #5489 undeclared zero-output tool failover | UNSOUND as submitted, salvageable | Carry with fix: non-streaming classification must not hop after a replayUnsafe heartbeat; add a non-streaming E2E proving no second dispatch; sync structure/runtime.md, structure/transports/responses-failover.md. Covers only the Responses path of #5407, so #5407 stays open. | +| #5221 sub-agent own-model identity | UNSOUND | Exclude. Routed raw-body repair never personalizes the neutral catalog line, native parent -> routed worker stays wrong, fenced identity sentences can be rewritten, new test is unregistered. | +| #5217 | not fixable in bounded effort | Exclude; needs a destination-aware identity design across catalog, parser and passthrough. | +| #5494 DeepSeek combo adapter_eof | NEEDS-REPRO | Exclude. Current dev already hops a zero-output adapter_eof; the final 502 and cooldown 503 follow existing rules; wire capture needed to separate upstream truncation from a relay/adapter terminal loss. | +| #5369 responses-state spill growth | not a defect | Exclude. Reporter's own re-measure stays under the 1 GiB / 1000-entry / 24 h bounds; the remaining unreferenced-file footprint is a design question for snapshot-omitted in-memory owners. | + + +## Closure claims (audit fold) + +The PR says Closes #5495 only. #5407 (Responses path covered, Claude Code/Anthropic path not), #5217 and #4191 are not claimed. Listed issues #5407 and #5217 are reported to the coordinator as unfixed with findings, as the lane packet allows ("fixed ... or excluded with findings"). Supersede claims: #5629, #5659, #5646, #5633 and #5489 are superseded only when their whole net contribution is on the branch; #5489 is carried whole (its issue coverage is what is partial). #5221 is excluded and not superseded. diff --git a/devlog/_plan/260923_bundle_lane_e/020_carry.md b/devlog/_plan/260923_bundle_lane_e/020_carry.md new file mode 100644 index 0000000000..7eed3c4601 --- /dev/null +++ b/devlog/_plan/260923_bundle_lane_e/020_carry.md @@ -0,0 +1,17 @@ +# Carry plan + +Commit order (one commit per item, Co-authored-by trailer for the original author): + +1. #5629 (luvs01) plus review fixes. +2. #5659 (Ingwannu) plus review fixes; Closes #5495. +3. #5646 (FredAmartey). +4. #5633 (FredAmartey) with the import and structure-doc cleanup. +5. #5489 (AaronZ345) net diff (its upstream/dev merge commit dropped) plus the replayUnsafe fix. + +Mechanism: cherry-pick each PR's own commits (squashed per PR) onto the lane branch, then apply the folded review fixes in the same commit. Registries (scripts/test-layout/layout.json, tests/fixtures/test-layout-expected.json) keep every dev entry. + + +## Documentation folded from the audit + +- #5489: one new row in the hop/terminal table of docs-site/src/content/docs/guides/combos.md (an undeclared first tool call before any output and without a replay-unsafe side effect hops; after a replay-unsafe side effect it stays terminal), mirrored in every translated combos.md that carries the table; structure/runtime.md and structure/transports/responses-failover.md updated. +- #5659: one sentence in the English code-mode section of docs-site/src/content/docs/guides/codex-integration.md, next to the existing shell/patch repairs. Locales do not describe these repairs, so they do not contradict it. diff --git a/devlog/_plan/260923_bundle_lane_e/030_verify.md b/devlog/_plan/260923_bundle_lane_e/030_verify.md new file mode 100644 index 0000000000..a1a5aac8d0 --- /dev/null +++ b/devlog/_plan/260923_bundle_lane_e/030_verify.md @@ -0,0 +1,12 @@ +# Verification + +Per item, after its commit: + +- #5629: bun test tests/server/retry-delay-hardening.test.ts tests/server/retry-after-429.test.ts tests/providers/devin-stated-reset-retry.test.ts tests/providers/devin-stated-reset-hardening.test.ts tests/providers/devin-hardening.test.ts tests/codex-integration/combo-authoritative-reset.test.ts (includes the new repeated-marker negative case). +- #5659: bun test tests/responses/responses-code-mode-goal-helpers.test.ts tests/responses/responses-undeclared-tool-guard.test.ts tests/responses/responses-custom-tool-repair.test.ts tests/responses/responses-bare-echo-helper-fence.test.ts tests/responses/responses-default-namespace-emit-normalize.test.ts tests/responses/legacy-shell-compat.test.ts tests/responses/responses-code-mode-shell-compile.test.ts tests/responses/responses-code-mode-patch-compile.test.ts (includes the raw guard-input, unrelated-name and bare-goal precedence cases). +- #5646 + #5633: bun test tests/responses/ws-ambiguous-resend.test.ts tests/responses/ws-failure-stage.test.ts tests/server/replay-refusal-parity.test.ts tests/lib/ambiguous-resend-composition.test.ts tests/routing/routing-policy-fallback.test.ts tests/lib/upstream-retry.test.ts tests/responses/responses-reset-replay.test.ts tests/responses/responses-opaque-blob-recovery.test.ts tests/server/server-combo-failover-e2e.test.ts. +- #5489: bun test tests/adapters/run-turn-queue.test.ts tests/server/server-combo-zero-output-failover.test.ts tests/server/server-combo-failover-e2e.test.ts tests/responses/responses-stream-tool-events.test.ts tests/adapters/bridge.test.ts tests/responses/responses-undeclared-tool-guard.test.ts, including the new non-streaming case: heartbeat(replayUnsafe) then an undeclared tool call returns the refusal with exactly one target dispatch. + +Branch gates: tests/test-layout.test.ts tests/test-layout-tooling.test.ts tests/ci-workflows/file-size-ratchet.test.ts tests/ci-workflows/structure-ssot.test.ts, bun run typecheck, bun run structure:check, bun run privacy:scan, git diff --check. No full local suite (owner runs it after every lane lands). + +Evidence required before the final report: exact-head hosted CI with every required job completed success at the PR head SHA (run ids and per-job conclusions recorded), and a gpt-6-sol adversarial review of the final diff with verdict PASS and findings folded. diff --git a/devlog/_plan/260923_claude_desktop_first_party_models/000_plan.md b/devlog/_plan/260923_claude_desktop_first_party_models/000_plan.md new file mode 100644 index 0000000000..cfa6c6dfe7 --- /dev/null +++ b/devlog/_plan/260923_claude_desktop_first_party_models/000_plan.md @@ -0,0 +1,62 @@ +# Claude Desktop first-party: Code tab model bindings + +First-party mode keeps Claude Desktop signed in to claude.ai and routes only the Code tab's +Claude Code through the local intercept. The routing works, but the Code tab picker is owned by +claude.ai, so none of opencodex's models can appear there and the operator has no way to reach +them from Desktop. This unit adds first-party model bindings: the operator binds a picker model +id (for example `claude-sonnet-4-6`) to an opencodex route, and only requests that arrive +through the intercept honour the binding. `ocx claude` sessions and the public Messages +endpoint are unaffected. Evidence for the constraint is in [001_probe_evidence.md](001_probe_evidence.md); +the decisions are in [010_roadmap.md](010_roadmap.md). + +## Loop spec + +- Loop archetype: satisfy-spec, three work-phases (docs-first, implementation, live proof + PR). +- Trigger: the user reported that first-party still does not work in the Claude app and asked + for a live probe of the injection mechanism with Computer Use, a working Claude app, and a PR + with screenshots. +- Goal: from the Desktop Code tab in first-party mode, the operator can pick a Desktop picker row + and be served by an opencodex route of their choice, with the binding visible in CLI, API, + dashboard and docs. +- Non-goals: adding rows to the Desktop picker (claude.ai owns it), changing the OS trust store or + system proxy, modifying Claude.app, changing gateway mode, merging or releasing the PR, + restarting the user's live service. +- Verifier: focused bun tests named in 020, `bun run typecheck`, `bun run structure:check`, + `bun run skill:surface:check`, `bun run lint:gui`, `bun run build:gui`, then the live + Desktop proof in 030 (usage.jsonl provider + app screenshot). +- Stop condition: PR opened against `dev` with screenshots and exact-head CI reported; no merge. +- Memory artifact: this unit directory; the goalplan at + `.codexclaw/goalplans/opencodex-claude-desktop-first-party-claude-code/`. +- Expected terminal outcomes: DONE when C1-C4 hold; BLOCKED if Desktop needs a login only the + user can perform; NEEDS_HUMAN if claude.ai changes the picker ids mid-run. +- Escalation condition: any need to touch the OS trust store, system proxy, Claude.app, or the + user's live service; any merge decision. +- Resource bounds: local worktree writes only; push limited to the PR branch and screenshot + assets; one live Desktop session probe per proof; no token or time budget was set by the user. + +## Work-phase map + +| Work-phase | Doc | Closes with | +| --- | --- | --- | +| wp1 docs-first | this unit, 001, 010 | roadmap locked, no code | +| wp2 bindings | [020_wp2_first_party_bindings.md](020_wp2_first_party_bindings.md) | focused tests, typecheck, structure/skill checks green | +| wp3 live proof + PR | [030_wp3_live_proof_and_pr.md](030_wp3_live_proof_and_pr.md) | Desktop Code tab served by a bound route, screenshots, PR open | + +## Architect consultation + +- Handle: `01a0cda6-d2ab-73d2-95ea-ead02c2cd992` (devin/swe-2, CXC-ROLE architect, read-only). +- Proposal decisions D1-D5 and main's dispositions are recorded in [010_roadmap.md](010_roadmap.md). +- Reflection on revision r1 (000/010/020/030): ALIGNED. One minor gap: rename migrations rewrite + only `modelMap` values. Disposition: 020 now rewrites `intercept.modelMap` in the provider, routing-profile + and combo rename paths, and records why the legacy OpenAI-id migration is excluded. + +## Audit record + +- Reviewer `01a0cdae-0b0c-7840-a1f4-30fdc5398854` (devin/swe-2, CXC-ROLE reviewer), round 1: + GO-WITH-FIXES (blockers=1). Blocker 1 (native/ targets only normalized in the 3P-alias branch; + PUT validation source unspecified) folded into 020: read-side normalization in + `claudeCodeForIngress` and validation against the unfiltered Desktop route vocabulary. Notes folded: + CSS in a new file (styles.css is at cap), picker-id keys are not migrated on renames, `ocx-route` + precedence documented, scratch-server safety argument written into 030. +- Round 2: reviewer PASS. Architect reflection on r2: ALIGNED with one residual (normalize only the + intercept entries, not merged global values), folded into 020. diff --git a/devlog/_plan/260923_claude_desktop_first_party_models/001_probe_evidence.md b/devlog/_plan/260923_claude_desktop_first_party_models/001_probe_evidence.md new file mode 100644 index 0000000000..44bef2708b --- /dev/null +++ b/devlog/_plan/260923_claude_desktop_first_party_models/001_probe_evidence.md @@ -0,0 +1,51 @@ +# 001 — Probe evidence (2026-09-23) + +All observations were made on the maintainer machine with Claude.app 1.18286.0, Desktop's +bundled Claude Code 2.1.197, standalone Claude Code 2.1.278 and the source-dogfooded proxy +(`/Users/jun/Developer/new/700_projects/opencodex` at 206fbc6b3f) on port 10100, intercept on 10200. +Screenshots and extracted bundles stay outside the repository under `/tmp/ocx-claude-probe/`. + +## Where the Desktop Code tab picker comes from + +- Desktop main process (`app.asar` `.vite/build/index.js`) exposes an IPC + `LocalSessions.setAvailableCodeModels(modelIds)` that the renderer calls; the renderer is the + claude.ai web app. +- The claude.ai bundle (`shared-16-*.js`) calls `setAvailableCodeModels(ae.map(e=>e.id))` with + `{selectableModels:ae}=oy("code")`, and `oy` builds the catalog from `modelSelectorConfig` + (`shared-0-*.js` `function Zj`). Unknown ids only ever become an "Unsupported model" entry for + the current selection; `Yj` adds `[1m]` rows only for models already in the catalog. +- The renderer reads Claude Code settings through `resolveLocalSettings` (`shared-4-*.js` `mN`): + `model`, `availableModels`, `fastMode`, effort and permission keys. `availableModels` only + disables rows; `model` does not add one. Claude Code 2.1.278 supports a `modelPicker` settings + key with labels and `behavesAs`, but Desktop does not read it. +- Live check: with `~/.claude/settings.json` `model` set to `claude-ocx-xai--grok-4.7` and a new + Code session, the picker still listed only Opus 5.5, Sonnet 5, Fable 5.1, Haiku 4.5 and More + models (Opus 5, Fable 5, Opus 4.8, Opus 4.7, Opus 4.6, Sonnet 4.6). The setting was restored. + +Conclusion: in first-party mode no local file can add an opencodex row to the Desktop picker. +The only lever is the request path: the Code tab sends the picker id and the intercept can route it. + +## The intercept path works + +- `ocx claude desktop apply --first-party` pivoted the Desktop library to the standard profile + and wrote only `HTTPS_PROXY`/`NODE_EXTRA_CA_CERTS` into `~/.claude/settings.json`. +- Standalone CLI: `claude -p ... --model claude-ocx-xai--grok-4.7` returned `PROBE-OK`; + usage.jsonl recorded `xai xai/grok-4.7 200 loopback messages`. +- Desktop Code tab, Haiku 4.5: reply `DESKTOP-1P-PROBE-HAIKU`; usage.jsonl recorded two + `anthropic-native claude-haiku-4-5-20251001 200 loopback` rows, so Desktop's Claude Code does go + through the intercept. +- Desktop Code tab, Sonnet 4.6 after a temporary global `claudeCode.modelMap` + `{"claude-sonnet-4-6":"xai/grok-4.7"}`: usage.jsonl recorded `xai xai/grok-4.7 grok-4.7 200 loopback`. + The global map was the only way to do this, and it also reroutes `ocx claude` sessions. + +## Picker ids observed on the wire + +`claude-opus-5-5`, `claude-opus-5`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001` (dated), plus the +catalog rows Opus 4.8/4.7/4.6 and Fable 5/5.1. Dated ids reach an undated key through the existing +date-suffix strip in `resolveInboundModel` (src/claude/inbound-model-options.ts). + +## Side observation, not in scope + +Before the probe the saved config said `desktopMode: first-party` while the Desktop library still +applied the opencodex gateway profile, and status reported `first_party_residue`. The running +proxy was 26 commits behind `dev`; this unit does not chase that state. diff --git a/devlog/_plan/260923_claude_desktop_first_party_models/010_roadmap.md b/devlog/_plan/260923_claude_desktop_first_party_models/010_roadmap.md new file mode 100644 index 0000000000..853bde6bab --- /dev/null +++ b/devlog/_plan/260923_claude_desktop_first_party_models/010_roadmap.md @@ -0,0 +1,22 @@ +# 010 — Roadmap and decisions + +Status: locked at the end of wp1 (reviewer PASS, architect ALIGNED on revision r2). + +Order follows the build dependency: the binding has to resolve on the request path before any +surface can edit it, and the live proof needs both. + +1. wp2 — storage, request-path resolution, API, CLI, dashboard, docs ([020](020_wp2_first_party_bindings.md)). +2. wp3 — live Desktop proof with screenshots and the PR ([030](030_wp3_live_proof_and_pr.md)). + +## Architect proposal (handle 01a0cda6) and dispositions + +| ID | Proposal | Disposition | +| --- | --- | --- | +| D1 storage | New `claudeCode.intercept.modelMap`; global `modelMap` stays untouched | Accepted. | +| D1 plumbing | Build a shallow `{...config, claudeCode: {...}}` copy in serve-options for intercept requests | Amended. A config copy can reach `saveConfig`/live-reconcile helpers keyed on the config object and would persist the merged map. Instead serve-options passes `claudeIntercept: true` to the two handlers, and the handlers derive a request-scoped `claudeCode` view (`claudeCodeForIngress`) that only the model-resolution calls read. | +| D2 matching | Reuse `resolveInboundModel` unchanged (exact, date-stripped, `[1m]`, `--fast`) | Accepted; the view overlays `modelMap` so every existing rule applies. Amended after audit round 1: intercept targets written as `native/<slug>` are normalized to the bare slug inside the view, because `resolveInboundModel` returns map values verbatim; global `modelMap` values are not normalized. | +| D3 surfaces | CLI `bind`/`unbind`, API, GUI first-party card, schema, docs | Accepted with a dedicated `PUT /api/claude-desktop/first-party-bindings` route instead of widening the gateway profile PUT, which carries conflict checks unrelated to bindings. The CLI drives that route so the running proxy adopts the change immediately. | +| D4 constraints | No capped file touched; register any new test file in both layout maps; lab boundary untouched | Accepted. | +| D5 honesty | Show "picker id → served route"; never claim the Desktop label changes | Accepted; docs and GUI copy say the picker keeps Anthropic's label. | +| Risk (c) | A picker id that is also an alias resolves alias-first | Accepted as is; picker ids are genuine Anthropic ids. | +| Risk (d) | Picker ids can change | Documented; the dashboard offers the observed ids as suggestions and accepts any `claude-` id. | diff --git a/devlog/_plan/260923_claude_desktop_first_party_models/020_wp2_first_party_bindings.md b/devlog/_plan/260923_claude_desktop_first_party_models/020_wp2_first_party_bindings.md new file mode 100644 index 0000000000..5f5c88ffe5 --- /dev/null +++ b/devlog/_plan/260923_claude_desktop_first_party_models/020_wp2_first_party_bindings.md @@ -0,0 +1,79 @@ +# 020 — wp2: first-party model bindings + +## Scope + +IN: storage, request-path resolution for intercepted Messages and count_tokens, status/PUT API, +CLI `bind`/`unbind`, dashboard first-party card, docs and structure. +OUT: Desktop picker rows or labels, OS trust store/system proxy, Claude.app, gateway-mode +behaviour, global `claudeCode.modelMap` semantics, the public `/v1/messages` path. + +## Contract + +- `claudeCode.intercept.modelMap?: Record<string, string>` — key: a Desktop picker model id + (`claude-` prefix, e.g. `claude-sonnet-4-6`); value: an opencodex route in the Desktop route + vocabulary (`provider/model`, or `native/<slug>` for the native OpenAI pool). +- A binding applies only to requests that arrive on the `claude-intercept` ingress. It is + overlaid on the global `modelMap` for that request (binding wins per key), so every existing + resolution rule applies unchanged: alias first, Desktop 3P alias, exact key, date-suffix-stripped + key, `[1m]` strip, `--fast` decode. A bound id is therefore never natively passed through. +- `native/<slug>` targets resolve to the bare slug, matching how Desktop 3P aliases resolve + (src/claude/inbound-model-options.ts:48-53). Normalization runs read-side inside + `claudeCodeForIngress`, so a hand-written config, the CLI, the API and the GUI all converge; + the stored value keeps the Desktop route vocabulary (`native/<slug>`) for round trips. Only the + intercept entries are normalized before the merge; global `modelMap` values keep today's verbatim + semantics on every path. +- An `ocx-route` body directive (src/server/claude-messages.ts:694-697) still wins over a binding, + because it rewrites the model before resolution. Bindings do not change that precedence. +- PUT validates a target against the whole Desktop route vocabulary from `buildClaudeDesktopState` + (`state.models`, available entries, native routes included). The apply path's filtered `routed` + list (agent-settings-routes.ts:1174) is not used, because it drops `native/` routes. + +## File change map + +| File | Change | +| --- | --- | +| `src/types/config.ts` (~146) | `intercept?: { enabled?; port?; modelMap?: Record<string, string> }` with doc comment. | +| `src/config/schema/config-schema.ts` (~285) | Validate `intercept.modelMap`: plain object; keys match the picker-id shape; values non-empty strings without whitespace. | +| `src/claude/intercept/model-bindings.ts` (new) | `INTERCEPT_BINDING_ID` shape, `normalizeBindingTarget(route)`, `claudeCodeForIngress(cc, claudeIntercept)` (request-scoped view, never persisted), `applyBindingPatch(current, {set, remove})` with validation errors. | +| `src/server/index/serve-options.ts` (~1478, ~1509) | Pass `{ claudeIntercept: ingress === "claude-intercept" }` to `handleClaudeCountTokens` and `handleClaudeMessages`. | +| `src/server/claude-messages.ts` (~96-210, ~634-790, ~1211-1270) | Derive `cc = claudeCodeForIngress(config.claudeCode, claudeIntercept)` once per request; use it in `decodeFablePickerAlias`, `decodeClaudeFastSelector`, capture, `wantsNativePassthrough` (new `cc` argument) and `anthropicToResponsesTranslation`. `config` itself is never copied. | +| `src/server/management/agent-settings-routes.ts` (~1237-1310) | Status adds `firstParty.modelBindings`; new `PUT /api/claude-desktop/first-party-bindings` taking `{ set?, remove? }`, validating routes against the available Desktop routes, committing through `mutatePersistedConfig` and adopting the committed `claudeCode` into the live config. | +| `src/server/management/route-registry.ts` | Declare the PUT route. | +| `src/cli/claude-desktop.ts` | `ocx claude desktop bind <picker-id> <route>` and `unbind <picker-id>` via `runtimeRequest`; help text. | +| `src/cli/capabilities.ts` + `skills/ocx` surface map | Register `claude desktop bind` and `claude desktop unbind`; regenerate with `bun run skill:surface`. | +| `gui/src/components/ClaudeFirstPartyBindings.tsx` (new) + `gui/src/pages/ClaudeDesktop.tsx` + `gui/src/i18n/*.ts` + `gui/src/styles/claude-first-party-bindings.css` (new; `gui/src/styles.css` is at its 2958-line cap and is not touched) | First-party card: rows "picker id → route", add/remove, suggestions of observed picker ids, route select from available Desktop routes, copy that the Desktop label stays Anthropic's. | +| `docs-site/src/content/docs/guides/claude-code.md` + locales | Section "Use opencodex models from the Desktop Code tab"; the CLI-compatibility bullet names bindings next to `modelMap`. | +| `structure/clients/claude-desktop.md`, `structure/runtime.md` | Contract above and the ingress-scoped overlay. | +| `src/providers/provider-id-rewrite.ts` (~97), `src/server/management/routing-profile-routes.ts` (~200), `src/server/management/combo-routes.ts` (~273) | Rewrite `intercept.modelMap` values alongside `modelMap` on provider, routing-profile and combo renames so a binding cannot go stale silently. Keys are not migrated: they are Anthropic picker ids, never opencodex public ids, so a routing-profile rename cannot rename them. `src/providers/openai-tiers.ts` legacy-id migration is left alone: it rewrites pre-existing legacy ids, and bindings are written after it with current ids. | +| Tests | `tests/claude-integration/claude-intercept-model-bindings.test.ts` (new, registered in both layout maps); an intercept-vs-public case in `tests/server/claude-intercept-integration.test.ts`; route/CLI cases next to the existing first-party tests. | + +Field chain for `intercept.modelMap`: creation — CLI `bind`, PUT route, GUI card, hand-written +config; serialization — `mutatePersistedConfig`; deserialization — schema validation on load +(invalid entries reported, never silently used); consumers — `claudeCodeForIngress` in both +handlers, status route, GUI card, CLI output, and the three rename migrations above. + +## Activation scenarios (C-ACTIVATION-GROUNDING-01) + +1. Bound id via intercept: a Messages request for `claude-sonnet-4-6` through the CONNECT proxy + with binding `xai/…` reaches the fake provider, not the fake Anthropic upstream. +2. Same request on the public listener: passes through to the fake Anthropic upstream. +3. Dated id: `claude-haiku-4-5-20251001` reaches a `claude-haiku-4-5` binding. +4. `native/<slug>` target resolves to `<slug>`. +5. count_tokens on a bound id via intercept is not natively passed through. +6. PUT rejects a non-`claude-` key, an unavailable route and a whitespace value with 400 and + leaves config unchanged; `remove` of an unknown id is a no-op. +7. Schema rejects a non-object `intercept.modelMap` and non-string values. + +## Verifiers (run before writing, PLAN-VERIFIER-REAL-01) + +- `bun test tests/server/claude-intercept-integration.test.ts tests/claude-integration/claude-desktop-first-party.test.ts` + — exit 0, 33 pass on the base; both files import the intercept pair and first-party module directly. +- New test file above, plus `bun run typecheck`, `bun run structure:check`, `bun run skill:surface:check`, + `bun run lint:gui`, `bun run build:gui`, `bun run test:changed` (run in B/C). + +## Delegation + +Main writes server, config, CLI, API, tests, structure and English/Korean docs. One devin/swe-2 +worker writes the GUI component, page wiring, CSS and all ten GUI locales against the API contract +above (disjoint write scope: `gui/` only). A second devin/swe-2 worker translates the new docs +section into fr, ja, ru, tr, zh-cn and zh-tw (write scope: those six files). Main reviews both diffs. diff --git a/devlog/_plan/260923_claude_desktop_first_party_models/030_wp3_live_proof_and_pr.md b/devlog/_plan/260923_claude_desktop_first_party_models/030_wp3_live_proof_and_pr.md new file mode 100644 index 0000000000..15c899159b --- /dev/null +++ b/devlog/_plan/260923_claude_desktop_first_party_models/030_wp3_live_proof_and_pr.md @@ -0,0 +1,52 @@ +# 030 — wp3: live Desktop proof and PR + +## Live proof + +The user's service runs from the main checkout; this worktree's code is proven without restarting +or reconfiguring it. A scratch server starts from this worktree through `startServer` directly +(the same entry the integration tests use, so no CLI ensure/sync step touches `~/.codex` or the +service manager), with `OPENCODEX_HOME` pointing at a scratch directory outside the repository. +Its config holds only one provider that authenticates with a static API key (`zai` or `aim`, +copied from the user's config; OAuth providers are excluded because a refresh in the copy could +rotate the user's token), `claudeCode.intercept.port` 10400 and public port 10300. The scratch +server mints its own intercept CA. The scratch config carries no `openai`/Codex provider, so the +Codex sync and quota paths stay inert (src/server/index.ts:276-287, src/codex/quota-auto-refresh.ts:279), +and the service-ownership check reads the default-home records and fails closed to "foreign" +(src/service/state.ts:156-157). The repointed `NODE_EXTRA_CA_CERTS` reads as foreign to the user's +own proxy, so its ensure step does not rewrite it mid-probe (src/claude/intercept/settings.ts:88-99). + +1. Back up `~/.claude/settings.json` (already at `/tmp/ocx-claude-probe/backup/`), then point + `HTTPS_PROXY` at 10400 and `NODE_EXTRA_CA_CERTS` at the scratch CA for the probe. +2. `ocx claude desktop bind claude-sonnet-4-6 <provider>/<model>` against the scratch server + (CLI targets it through the scratch home), and the dashboard card for the GUI screenshot. +3. Fully quit and reopen Claude Desktop (first-party), Code tab, pick Sonnet 4.6, send a probe; + pick Haiku 4.5 and send a second probe. +4. Evidence: scratch `usage.jsonl` shows the bound provider for the Sonnet 4.6 probe and + `anthropic-native` for Haiku 4.5; the user's own proxy log shows no new Messages rows for the + probes; screenshots of the Desktop conversation, the picker and the dashboard card, cropped to + exclude account names. +5. Restore: settings.json byte-for-byte from backup, stop the scratch server, delete the scratch + home, fully quit and reopen Desktop, confirm the user's proxy on 10100/10200 is untouched. + +## PR + +- Branch `codex/claude-desktop-first-party-models` → `dev`, repository template (Summary, + Verification, Checklist), screenshots uploaded to the `pr-assets` branch and linked by commit SHA. +- Report exact-head CI; do not merge. + +## Result (2026-09-23) + +- Scratch server from this branch on 10300/10400 (zai only, `claudeCode.desktopMode: first-party`), + binding set with the new CLI: `ocx claude desktop bind claude-sonnet-4-6 zai/glm-5.3-flash`. + The CLI refused `gpt-6` (not a picker id) and `nope/missing` (route not available). +- Claude Desktop 1.18286.0, first-party, Code tab, Sonnet 4.6 picked: the reply arrived, and the + scratch `usage.jsonl` recorded `zai zai/glm-5.3-flash glm-5.3-flash 200 loopback messages` for it + and `anthropic-native claude-haiku-4-5-20251001 200` for Desktop's own title call, so unbound ids + still pass through natively. The user's proxy recorded no Messages rows for the probes. +- The bound model still described itself as Sonnet, because Claude Code's system prompt tells it so. + The docs say this and recommend binding rows the operator does not otherwise use. +- Screenshots on `pr-assets` at 793b39d85d (`260923-claude-desktop-first-party-bindings/`). +- Restored afterwards: `~/.claude/settings.json` byte-identical to the backup, scratch server stopped + and its home deleted, the temporary global `modelMap` used during the probe removed from the + user's proxy, Desktop reopened. Desktop was left in first-party mode, which is the saved + `desktopMode`; before the probe it was running the gateway profile. diff --git a/devlog/_plan/260923_claude_reset_grants/010_plan.md b/devlog/_plan/260923_claude_reset_grants/010_plan.md new file mode 100644 index 0000000000..61eb469630 --- /dev/null +++ b/devlog/_plan/260923_claude_reset_grants/010_plan.md @@ -0,0 +1,237 @@ +# 260923 Claude reset grants — diff-level plan (wp1) + +Loop-spec: HOTL, single work-phase wp1. Write scope: this worktree only +(branch codex/anthropic-reset-grants). Tools: local fs, bun tests, GET-only +Anthropic probes with local OAuth tokens. Forbidden: any POST to +`/api/organizations/*/reset_rate_limits` during development or verification, +push/PR/merge, restarting the live ocx service. Budget: one PABCD cycle. + +## Problem + +Claude Pro/Max/Team subscriptions currently carry a one-time usage-limit reset +grant (Anthropic program `cedar_ember`, e.g. `opus55-launch-promax-20260921`), +valid until 2026-10-22T16:00Z. opencodex already shows reset tickets for Codex +(reset credits) and Grok (reset coupons) on the account rows, but Anthropic OAuth +rows show nothing, and the existing Anthropic quota probe sends +`claude-cli/2.1.63`, which upstream now answers with +`ineligible_reason: "cli_version"` for the grant block. + +## Upstream contract (evidence) + +- Read: `GET https://api.anthropic.com/api/oauth/usage?cedar_ember=1&skip_spend=1`, + bearer OAuth token, `anthropic-beta: oauth-2025-04-20`, + `User-Agent: claude-cli/2.1.280 (external, cli)`. Verified live (GET only) on six + local accounts on 2026-09-23: each returned one unused grant. +- Redeem (from the Claude Code 2.1.278 client, function `fJe`): + `POST https://api.anthropic.com/api/organizations/{orgUuid}/reset_rate_limits` + body `{program:"cedar_ember", grant_id, request_id}`; grant id + `/^[a-z0-9_-]{1,40}$/`, request id `/^[A-Za-z0-9_-]{1,64}$/`; response + `{result: reset|already_used|not_limited|cooldown|ineligible|unavailable, reason, + resets_left, cleared[], weekly_resets_at, cooldown_until}`; 429 → rate_limited, + 401/403 → auth_error. The client reuses the same request id when retrying an + unsettled claim for the same grant. orgUuid comes from `GET /api/oauth/profile` + (`organization.uuid`), which accepts the OAuth token. +- Not usable: `GET /api/organizations/{org}/usage` returns 403 + `oauth_token_not_accepted` for OAuth tokens. + +## Decisions (architect proposal D1–D6, main dispositions) + +- D1 accepted: `src/providers/anthropic-reset-grants.ts` (wire + parsing), + `src/providers/anthropic-reset-grant-ledger.ts` (journal), + `src/server/management/anthropic-reset-grant-routes.ts` (lazy-loaded). +- D2 accepted: fail-closed parse; malformed or missing `cedar_ember` is an error, + never zero grants; `event_props` and upstream bodies never leave the server. +- D3 amended: JSON journal with atomic write (not the Codex SQLite ledger), with + the D3 semantics. The single binding contract is the "Ledger contract" section + below; it supersedes the first-revision wording. +- D4 accepted: `GET /api/anthropic/reset-grants?accountId=`, + `POST /api/anthropic/reset-grants/consume` with required + `{accountId, grantId, operationId}`. New operations re-read eligibility first and + require the grant present, not paused, usable_now, resets_left > 0. Status codes + 400/401/409/429/502/503. Both routes registered as `deferred-verb` with this + document as ownerDoc (CLI verb out of scope). +- D5 accepted: `gui/src/hooks/useAnthropicResetGrants.ts`, + `gui/src/components/provider-workspace/AnthropicResetGrants.tsx`, wired into + healthy Anthropic OAuth rows in ProviderAuthPanel; i18n prefix `anthropicGrant.*`. +- D6 accepted: `src/providers/claude-cli-identity.ts` exports + `CLAUDE_CLI_USER_AGENT`; the quota probe imports it. + +## File change map + +| File | Change | +| --- | --- | +| src/providers/claude-cli-identity.ts | new: pinned UA constant | +| src/providers/quota/vendor-probes-oauth.ts | use the constant | +| src/providers/anthropic-reset-grants.ts | new: read, profile, redeem, parsers | +| src/providers/anthropic-reset-grant-ledger.ts | new: journal | +| src/server/management/anthropic-reset-grant-routes.ts | new: handlers | +| src/server/management-api.ts | lazy dispatch | +| src/server/management/route-registry.ts | two entries, deferred-verb | +| gui/src/hooks/useAnthropicResetGrants.ts | new | +| gui/src/components/provider-workspace/AnthropicResetGrants.tsx | new badge + dialog | +| gui/src/components/provider-workspace/ProviderAuthPanel.tsx | wire badge + modal | +| gui/src/i18n/*.ts (10 locales) | anthropicGrant.* keys | +| tests/adapters/anthropic/anthropic-reset-grants.test.ts | new (existing adapters/anthropic domain) | +| tests/server/management-anthropic-reset-grants.test.ts | new | +| gui/tests/anthropic-reset-grants.test.tsx | new: badge + abort → same-id retry | +| scripts/test-layout/layout.json, tests/fixtures/test-layout-expected.json | register | +| structure/gui-and-management-api.md | ownership row | +| docs-site reference/management-api.md (+7 locales) | route rows + dashboard note | + +## Acceptance (activation scenarios) + +1. Parser: valid block → grants; missing block, bad grant id, duplicate id, + negative counts → error (unit test with fixtures). +2. Redeem wire: exact URL/body/headers; each `result` and 429/401/403/500 + mapping (fake fetch records the request). +3. Ledger: new op executes; same op after terminal replays with no fetch; same op + while leased → 409 `in_flight`; same op open, lease lapsed, inside the 10 min + retry window → re-sends the same request id; outside it → 409 + `unknown_outcome_expired`, no fetch; other account/grant/org → 409; corrupt + journal → 503, no fetch; settlement write failure after an upstream answer → + 500 `journal_write_failed` and the record stays open. + A different operation id for the same account + grant + org while an open + record younger than 10 min exists → 409 `unresolved_prior_operation`, no fetch. +4. Routes: missing grantId/operationId → 400; ineligible or unusable grant → 409 + with no redeem call; token failure → 401; happy path returns code `reset`; + unknown outcome (fetch throws) → 502 `unknown_outcome` and record stays open. +5. GUI: badge renders count, error, loading; dialog two-step confirm; screenshot. +6. Gates: `bun run typecheck`, focused tests, route registry, i18n parity, test + layout, file-size ratchet, structure:check, core-lab boundary, GUI lint/build. +7. Live GET through the new module for local accounts; no POST. + +## Out of scope + +CLI verb (owed; this doc is the deferred-verb owner), auto-redeem, push/merge. + +## Build deviations + +- The consume route requires the `gui-session` principal and is registered as + `session-only` instead of `deferred-verb`. AGENTS.md ("User-consent actions") + asks that any new action spending the user's identity or credits be gated rather + than left to a prompt an agent can answer; a one-time subscription reset is such + a spend. The read route stays `deferred-verb` with this doc as owner. +- New CSS lives in `gui/src/styles/anthropic-reset-grants.css` (imported from + `gui/src/main.tsx`) because `gui/src/styles.css` sits exactly at its + file-size cap. +- Test files: `tests/adapters/anthropic/anthropic-reset-grants.test.ts`, + `tests/server/management-anthropic-reset-grants.test.ts`, + `gui/tests/anthropic-reset-grants.test.tsx`. + +## Check-phase code review (Mill, gpt-6-sol) + +Round 1 found five issues; four were fixed in `13c15083d3`: the journal now +publishes through `atomicWriteFileStreamed` (temp fsync plus parent-directory +sync) so the open record is on disk before the claim; settlement returns the +stored answer and a missing record fails closed; a replayed refusal renders as +that refusal; a same-id retry refused for a transient reason keeps the attempt +held. Trailing blank lines were removed. + +Round 2 residual, accepted: `syncParentDirectory` in +`src/config/atomic-write.ts` is best-effort by platform (no directory +descriptor on Windows; some filesystems refuse the open), and this unit does not +change the shared primitive. Losing a just-renamed open record needs a power cut +in that window on such a platform, and a second spend then still needs the grant +to report `resets_left > 0` to the pre-spend gate after the first claim; every +grant observed today has `resets_total: 1`. + +## Architect reflection (MISALIGNED → folded) + +The same architect flagged three gaps against the first revision; all accepted: + +1. D3 identity and concurrency. The journal record binds account + grant + a + SHA-256 digest of the organization UUID (the raw UUID is never stored). Every + journal read-modify-write runs inside a cross-process lock: a sibling + `<journal>.lock.sqlite` opened with `busy_timeout=0; BEGIN IMMEDIATE` (same + OS-backed pattern as `src/config/mutation-lock.ts`). Details, lease length + and recovery rules: see "Ledger contract (binding)"; the older wording in this + item is superseded. +2. D4 spend gate. A new operation requires `eligible === true`, the grant + present, not paused, `usable_now`, `resets_left > 0`, and `at_limit` when + `use_requires_limit` is true. A retry of an open record re-checks that the + profile organization digest matches the journal (mismatch → 409). +3. GUI same-id retry. After an unknown outcome the dialog keeps the operationId + and its retry sends the same id; a happy-dom test drives abort → retry and + asserts both POST bodies carry the identical operationId. + +## Audit round 1 (Arendt, FAIL → folded) + +Four blockers, all accepted. The first two are resolved by the ledger contract +below; test placement is fixed in the change map; privacy is item P below. + +## Ledger contract (binding) + +Journal: `<configDir>/anthropic-reset-grant-ledger.json`, version 1, atomic write. +Lock: every read-modify-write runs synchronously inside a sibling +`<journal>.lock.sqlite` `BEGIN IMMEDIATE` transaction with `busy_timeout=0`. +Nothing asynchronous runs inside the lock; the upstream POST always runs after the +open record is durably written and the lock is released. Busy → 503 +`ledger_busy`; unreadable or corrupt journal → 503 `ledger_unavailable`; no +spend in either case. + +Record: `{accountId, grantId, orgDigest (sha256 of org uuid), status: +open|settled, code?, resetsLeftAtOpen, leaseUntil, attempts, createdAt, +updatedAt}`. The operationId (UUIDv4) is the upstream `request_id`. + +Begin (under lock): +- no record → write `open` with `leaseUntil = now + 90 s` (25 s client timeout + plus margin) → execute. +- record for another account/grant/org → 409 `operation_identity_mismatch`. +- settled → replay the stored code, no upstream call. +- open with an unexpired lease → 409 `in_flight`, no upstream call. +- open with an expired lease, less than 10 minutes after `createdAt` → renew the + lease and POST again with the same request id. This is only reached by an + explicit user retry of the same operation; nothing retries automatically. +- open with an expired lease, 10 minutes or more after `createdAt` → 409 + `unknown_outcome_expired`; no upstream call for this operation ever again. +- a NEW operation for the same account + grant + org while another record for + that triple is still `open` and younger than 10 minutes → 409 + `unresolved_prior_operation` (response carries only that code); the dashboard + steers the user to the same-id retry instead. + +Why same-id retry is allowed (audit round 2): the vendor client does exactly +this. Claude Code 2.1.278 keeps `unsettledClaimRequestId` per grant and reuses it +for a retry while `now - unsettledClaimAtMs < 600000` (`b5o=600000`, functions +`Fqt`/`TJe`/`Dqt`), and after a second unconfirmed attempt tells the user +"nothing more was used" (`stillUnconfirmedLine`). The server-side dedup on +`request_id` is therefore the vendor's designed recovery contract, and the +10-minute window mirrors it. No settlement is ever inferred from a re-read: an +unknown outcome stays open until an upstream answer settles it or the window +closes. After the window, a new operation still has to pass the pre-spend gate +(eligible, usable_now, resets_left > 0), so a first attempt that did consume the +grant blocks the new one. + +Audit round 3 (new-id bypass): folded for the 10-minute window as above. After +the window, a new operation id is allowed through the pre-spend gate. Rebuttal +for blocking it forever: the vendor client does the same — once +`now - unsettledClaimAtMs >= 600000`, `Fqt` stops reusing the old id and returns +a fresh `d5o()` UUID for the next claim, and a permanent block would strand a +grant that was never spent with no way to release it. Accepted residual (not a +guarantee): the 25 s timeout is client-side only, so an upstream that is still +processing the first POST, or has not reflected it in `/api/oauth/usage`, after +ten minutes could let a second reset be spent when `resets_left > 1`. Every grant +observed on 2026-09-23 had `resets_total: 1`; that is an observation, not a +parser-enforced limit. + +Settle (under lock): first terminal settlement wins; a later or stale +settlement for an already-settled record is ignored. Terminal codes: every +upstream `result` value, plus `rate_limited` and `auth_error` (the spend was +refused before it ran). A thrown fetch, timeout, or unreadable response leaves the +record `open` and the route answers 502 `unknown_outcome`. If the upstream +answered but the settlement write fails, the route fails closed with 500 +`journal_write_failed` and no upstream code; the record stays open, the GUI +treats it as an unknown outcome, and a same-id retry inside the window gets the +upstream's deduplicated answer. + +Residual risk: two processes that both find an expired lease serialize on the +lock, so only one renews it. A first attempt still in flight after 90 s could +overlap an explicit same-id retry; both carry the same request id, which is the +case the vendor dedup exists for. + +## Privacy (P) + +Routes return fixed codes and fixed messages only; no exception text, upstream +body, token, organization UUID, or email reaches a response or log line. Route +tests assert that response bodies do not contain the fake token, org uuid, or +email used by the fakes, and capture console output during the route tests to +assert none of those values is logged. diff --git a/devlog/_plan/260923_configured_native_gpt_models/010_plan.md b/devlog/_plan/260923_configured_native_gpt_models/010_plan.md new file mode 100644 index 0000000000..3ddf8e794c --- /dev/null +++ b/devlog/_plan/260923_configured_native_gpt_models/010_plan.md @@ -0,0 +1,79 @@ +# 010 — Configured native GPT models (plan) + +## Problem + +Claude models on the Anthropic provider load from config: listing `claude-opus-5-5` under +`providers.anthropic.models` (plus optional `modelContextWindows`) is enough, and live discovery +fills in the rest. Native GPT models on the ChatGPT/Codex login do not. Every native slug is +hard-coded in `src/codex/catalog/native-models.ts` (`NATIVE_OPENAI_MODELS`), and the four GPT-6 +rows each repeat the literal 272,000 / 872,000 context pair in `NATIVE_OPENAI_CONTEXT_OVERRIDES`. +Adding GPT-6 Sol and Luna took #5580 (763 lines). `providers.openai.models` is ignored for the +forward provider (`provider-models.ts` returns `[]` for `authMode: "forward"`). + +Routing already works: `router.ts` sends any bare `gpt-*` id down the `native-family` route. What +is missing is the catalog: the id is dropped by `isUnsupportedOpenAiNativeSlug` and gets no +capabilities or context. + +## Contract + +A bare id listed under `providers.openai.models` becomes a *configured native* when: + +- `providers.openai` exists, is not disabled, and is the canonical Codex forward provider + (adapter `openai-responses`, authMode `forward` or omitted, canonical Codex base URL); +- the id matches `^gpt-[a-z0-9][a-z0-9.-]*$` (no `/`); +- it is not built in, not retired (`RETIRED_NATIVE_OPENAI_MODELS`), and not `gpt-reserve`. + +A configured native: + +- joins `NATIVE_OPENAI_MODELS` / `SUPPORTED_NATIVE_OPENAI_SLUGS`, so the canonical catalog backfill, + `/v1/models`, dashboard native rows and desktop projections list it. It is never account-gated. +- inherits capability metadata from its own pinned upstream row when one exists, otherwise from + the pinned `gpt-6-sol` row (reasoning ladder, modalities, instructions, speed tiers). The display + name comes from the slug (`gpt-6-nova` -> `GPT-6-Nova`); instructions are retargeted with + `identifyRoutedModel`. +- uses the GPT-6 family context default: 272,000 window, 872,000 opt-in ceiling, 872,000 max input + (clamped to the resolved window). `providers.openai.modelContextWindows[id]`, `contextWindow` + and `providerContextCaps.openai` apply unchanged through `narrowToLimits`. +- Removing the id from config unregisters it; the persisted catalog row then counts as unsupported + and is dropped on the next canonical write, like any other unsupported native. + +## Diff-level changes + +1. `src/codex/catalog/native-models.ts`: mutable `NATIVE_OPENAI_MODELS` seeded from a frozen built-in + list; `NATIVE_GPT6_CONTEXT`; `CONFIGURED_NATIVE_OPENAI_TEMPLATE_MODEL = "gpt-6-sol"`; + `configuredNativeOpenAiModelIds(config)` (pure filter), `setConfiguredNativeOpenAiModels(ids)` + (diff-applies, notifies), `configuredNativeOpenAiModels()`, `isConfiguredNativeOpenAiModel()`, + `subscribeConfiguredNativeOpenAiModels()` (fires immediately), `refreshConfiguredNativeOpenAiModels(config)`. + `nativeOpenAiCapabilitySourceSlug` maps a configured slug to itself when pinned, else to the + template; alias check and presentation cover template-borrowing configured slugs. +2. `src/codex/catalog/metadata.ts`: GPT-6 overrides spread `NATIVE_GPT6_CONTEXT`; + `upstreamNativeEntryForSlug` admits configured slugs; a subscription after + `UPSTREAM_NATIVE_ENTRIES` keeps `PINNED_NATIVE_CAPABILITY_ENTRIES`, `UPSTREAM_NATIVE_ENTRIES` and + `NATIVE_OPENAI_CONTEXT_OVERRIDES` in step; `nativeOpenAiSlugs` / `listCatalogNativeSlugs` include + configured ids. +3. `src/codex/catalog/build-entries.ts`: `CANONICAL_NATIVE_CATALOG_CONTENT_POLICY.nativeBackfillSlugs` + becomes a getter over the current list. +4. `src/vision/reasoning.ts`: consult `SUPPORTED_NATIVE_OPENAI_SLUGS` at call time. +5. New `src/config/derived-registries.ts`: `refreshConfigDerivedRegistries(config)` runs + `refreshUserCostOverlays` then `refreshConfiguredNativeOpenAiModels`, replacing the + `refreshUserCostOverlays` calls in `config.ts` (line count unchanged), `config/load-degrade.ts`, + `config/persist-unlocked.ts` and `usage/user-cost-overlay-reconciler.ts`, so load, save and + external-edit reconcile all register. +6. Tests: new `tests/codex-integration/configured-native-models.test.ts` (registered in layout.json + and test-layout-expected.json). +7. Docs: `structure/catalog.md` and docs-site `reference/configuration/providers.md`. + +## Out of scope + +Routing, account gating, bare listing of unknown roster observations, prices, GUI, API-key rows. + +## Verification + +`bun run typecheck`; the new test file plus `gpt6-native-rows.test.ts`, `native-model-toggle.test.ts`, +layout and file-size guards; `bun run test:changed`; `bun run structure:check`; `git diff --check`; +then exact-head hosted CI on the PR. + +## HOTL bounds + +Write scope: files above plus this devlog unit. Tools: local git/bun, gh for one PR to dev. +Merge only after exact-head required CI succeeds. diff --git a/devlog/_plan/260923_configured_native_gpt_models/020_audit.md b/devlog/_plan/260923_configured_native_gpt_models/020_audit.md new file mode 100644 index 0000000000..5989e99748 --- /dev/null +++ b/devlog/_plan/260923_configured_native_gpt_models/020_audit.md @@ -0,0 +1,18 @@ +# 020 — Audit (plan 010) + +Reviewer: read-only subagent (the `devin/swe-2` attempt failed with Devin `resource_exhausted`, retry ~1740s; +rerun on the inherited model). Verdict: NEAR-PASS. All points folded; nothing rebutted. + +| Finding | Resolution | +| --- | --- | +| Own-pinned-row branch (e.g. `gpt-daybreak-red-latest`) is not self-described, so `isGpt56NativeSlug`/`nativeLadderIncludesUltra` would synthesize `ultra` and strip Lite/WebSocket flags | Dropped. Every configured slug borrows `gpt-6-sol`; its source is self-described, so both predicates follow Sol. Context is always the GPT-6 default. | +| Combo `nativeAlias` validation (combos/types.ts:228) runs during schema parse, before registration | Documented limitation: a `nativeAlias` onto a configured slug is refused. Out of scope. | +| Process-global registry leaks across test files | `resetConfiguredNativeOpenAiModelsForTests()`; the new test calls it in `afterEach`. | +| Mutate the shared array, Set and three maps in place; never delete built-ins | Registration diff-applies in place; built-in ids are ineligible, so they are never added or removed. | +| Import cycle / GUI bundle | `native-models.ts` stays import-free. The config filter (`configuredNativeOpenAiModelIds`) lives in server-only `src/config/derived-registries.ts`, importing `openai-tiers-destination`. | +| Missed refresh sites: config/live-reconcile.ts, server/management/provider-routes.ts (2) | Added to the replacement list. | +| `ENTITLEMENT_PREFERRED_NATIVE_OPENAI_MODELS` / `NATIVE_MAIN_DRAIN_SENTINEL_MODELS` exclude configured slugs | Deliberate: both sets are reasoned per model. Configured slugs behave like `gpt-5.5` there. Documented in structure/catalog.md. | +| Every pool selector gets the slug (metadata.ts:831) | Consistent with "never gated"; asserted in the test. | +| Alias branch needs a presentation | Presentation generated from the slug. | +| Module-state vs argument rule (metadata.ts:238) | Registration happens inside `loadConfig`, so every process that loads config (including `ocx ensure`) sees it. Documented. | +| `refreshConfiguredNativeOpenAiModels` redundant | Folded into `refreshConfigDerivedRegistries`. | diff --git a/devlog/_plan/260923_docs_polish_404/000_plan.md b/devlog/_plan/260923_docs_polish_404/000_plan.md new file mode 100644 index 0000000000..35315b92ae --- /dev/null +++ b/devlog/_plan/260923_docs_polish_404/000_plan.md @@ -0,0 +1,113 @@ +# 260923 docs polish and 404 repair — roadmap + +A reader on X reported that https://opencodex.me/guides/macos-menu-bar/ returned 404. The page +existed on `main`; the Deploy Docs run for #5510 (run 35714540735) was cancelled by hand on +2026-09-22, so the live site stayed on the 2.60.0 build. The maintainer approved a redeploy and +run 35778046041 (workflow_dispatch on main 2f8216792f) published it; the reported routes and their +ko/ja/fr variants now answer 200. This unit fixes what the redeploy could not: source links that +404 even when deployed, a macOS guide describing a retired companion app, README claims that +drifted from the code, and 18 English pages missing from some locales. It lands as one PR to +`dev`; the live site changes again only at the next `dev → main` promotion and Deploy Docs run. + +## Loop spec + +| Field | Value | +| --- | --- | +| Loop archetype | satisfy-spec, multi-cycle HOTL (5 work-phases, one PR) | +| Trigger | User request: fix the reported 404, polish README and docs, open a docs PR, use gpt-6-sol subagents freely | +| Goal | No docs link in the repository points at a route the site does not build; macOS/desktop docs and README match current code; every locale carries the guides, reference and troubleshooting pages English has (`contributing/` excluded: open PR #5593 owns it, so `contributing/pr-quality.md` stays missing in ja/ko/ru/zh-cn) | +| Non-goals | Runtime code, dependencies, CI workflow edits (security-review boundary), file-size cap raises, `devlog/_fin` and ADR history, merge/release/promotion, the memory-inventory README section owned by open PR #5340, contributing pages and `docs-site/public/pr-screenshots` owned by open PR #5593 | +| Verifier | `bun test tests/ci-workflows/docs-link-targets.test.ts` (new; reads every docs content file and the URL surfaces), `bun test tests/ci-workflows/docs-readme-translation-parity.test.ts` (reads README.md, readme/*), test-layout tests, `cd docs-site && bun run build` (reads all content + astro.config.mjs), a scratch rendered-anchor audit over `docs-site/dist`, `bun run typecheck`, `bun run structure:check`, `bun run privacy:scan` | +| Stop condition | PR open against dev with exact-head CI inspected per job and all goalplan criteria met | +| Memory artifact | this unit + the `.codexclaw/goalplans/` goalplan and ledger | +| Expected terminal outcomes | DONE with PR URL and CI evidence; BLOCKED if push/PR refused; UNSAFE if a fix needs runtime or workflow changes | +| Escalation condition | Anything needing `.github/workflows`, a runtime change, or edits overlapping #5340/#5593; a subagent packet that fails with two distinct agents is reclaimed by main | + +HOTL resource bounds: repo-local shell, gh (read, push this branch, create the PR), gpt-6-sol +subagents with disjoint write scopes; write scope is the file map in 010-040; no user-set token or +wall-clock budget. + +## Work-phase map (dependency order) + +| wp | Doc | Outcome | +| --- | --- | --- | +| wp1 | this file + 010-040 | roadmap locked (docs only) | +| wp2 | [010](010_wp2_link_integrity.md) | broken links fixed, guard test landed and driven red | +| wp3 | [020](020_wp3_english_polish_readme_resync.md) | English README + desktop/macOS guides match code; 7 README locales resynced | +| wp4 | [030](030_wp4_locale_coverage.md) | missing locale pages translated, sidebar labels complete | +| wp5 | [040](040_wp5_delivery.md) | gates green, PR open, exact-head CI inspected | + +wp3 depends on wp2 (the guard checks the rewritten links). wp4 depends on wp3 (translations are made +from the final English pages). wp5 depends on everything. + +## Evidence already gathered + +- Live: before the redeploy `guides/macos-menu-bar/`, `guides/desktop-app/` and `ko/guides/macos-menu-bar/` + returned 404 and the sitemap lacked both pages. After run 35778046041 they return 200. +- Scratch audit `.tmp/link-audit.ts` (gitignored): one root-relative break + (`guides/desktop-app.md:104` `/opencodex/guides/macos-menu-bar/`) and three relative breaks + (`reference/configuration/server.md:717-719` `../../guides/codex-integration.md#…`, which resolves to + `/reference/guides/codex-integration.md`, live 404). +- `lidge-jun.github.io/opencodex/…` 301-redirects to `opencodex.me/…`, so the README links work today. + +## Architect consultation (formal P) + +- Architect handle `01a0caba-ab24-76c1-a2a6-710fbb3f6d9a` (gpt-6-sol, high effort, `CXC-ROLE: architect`, + V1 transport). Proposal decisions D1-D6. +- Main dispositions: + - D1 guard at `tests/ci-workflows/docs-link-targets.test.ts`: ACCEPT, amended to also resolve + relative links against the page URL, which is where the three real breaks were. + - D2 English fallback is a valid locale route: ACCEPT. + - D3 expand `.github/workflows/ci.yml` for rendered-anchor proof: REJECT the workflow edit (a + security-review boundary, AGENTS.md "Security boundary"), ACCEPT the need. The rendered check moves + into the Astro build as a local integration, which the existing CI `docs` job and Deploy Docs + already run; the Bun test covers README/source URLs, which the `ci` filter already matches. + - D4 README links to canonical opencodex.me, localized per README: ACCEPT. + - D5 ordered commits: ACCEPT, amended so the English README edit ships in the same commit as the + seven locale resyncs and every commit keeps the parity test green. + - D6 ratchet/union risks: ACCEPT; also fix the stale locale sentence in + `structure/ops/docs-and-release.md`. +- Open question from the architect (translate all pages or a subset): all 18, main decision, because + the user granted unlimited subagents and fallback pages read as untranslated in the nav. +- Reflection round 1: MISALIGNED, six gaps (image/src links, trailing-slash wording, hand slugger and + MDX fragments unproven, URL exception in the translation contract, wrong structure binding + mechanism, docs-only PRs never run the Bun suite). Dispositions: all folded; 010 rewritten around a + build-time check that reads rendered ids and every href/src, 030 gains the URL exception, 040 the + README fragment audit, 010 drops the manifest binding. +- Reflection round 2: MISALIGNED, two gaps. Confirmed sound: `astro:build:done` receives `dir`; CI + `docs` job runs `bun run build` in docs-site (`ci.yml:1008-1014`); plain .mjs import into Bun is + fine. Gaps folded: absolute same-host links count as internal in Layer A; existing fragment failures + are fixed, never globally disabled; 030 acceptance attributes fragment proof to Layer A. + +## Explorer evidence + +- Meitner `01a0caba-ac00-7700-8bcb-6c51d5fdfda9` (guides audit): 7 findings; main spot-checked + `desktop/src-tauri/src/resolve.rs:3-7`, `auth.rs:19-24`, `tray.rs:236-240`, `desktop/package.json:4-11` and + `desktop/README.md:28-45`; all confirmed. +- Turing `01a0caba-acf8-77d2-a378-b7af9f77a883` (README audit): 5 findings; #4 (memory counts) is left to + open PR #5340; #2 confirmed (`gh api repos/lidge-jun/opencodex --jq .default_branch` is main while the + README says a plain clone runs dev). + +## Audit round 1 (reviewer Volta `01a0cac3-2bd5-73b1-a667-1d7a02948ec8`, gpt-6-sol high) + +VERDICT: GO-WITH-FIXES (blockers=7). Synthesis and dispositions: + +1. Layer A missed relative and fragment-only links: FOLDED (010 resolves every rendered href against the page URL; `#frag` checks the same page). +2. Legacy host `lidge-jun.github.io/opencodex/…` (used by `src/server/management/cursor-integration-routes.ts:24`) is valid via GitHub's redirect: FOLDED (010 strips the project prefix for that host only; a canonical-host `/opencodex/` stays broken; fixtures for both). +3. `contributing/pr-quality.md` missing in four locales was not listed: FOLDED by narrowing the goal (contributing is #5593's area) and excluding `contributing/` from the scan explicitly. +4. DMG vs app signing: FOLDED (020 wording distinguishes the notarized app from the unnotarized DMG container; evidence in 020). +5. Bypass record lacked tier and final layer: FOLDED (E8 for both layers; final layer named). +6. 020/030 not diff-level: FOLDED. 020's replacement text lives in `021_wp3_macos_menu_bar_draft.md` (written by a gpt-6-sol worker, verified by main before A>B); 030 records a delegation output contract with mechanical acceptance, because translated prose cannot be pre-written without doing the translation. +7. A combined Bun command hides an absent test: FOLDED (010 and 040 run the new test alone and count its pass lines). + +Non-blocking anchor corrections are folded into 020. + +## Audit round 2 (same reviewer) + +VERDICT: GO-WITH-FIXES (blockers=3). Five round-1 folds confirmed closed; 021 sampled claims confirmed. + +1. Signing wording pinned to v2.61.0 and README overstated local builds: FOLDED (021 and 020 say "Release builds of OpenCodex.app…", local builds ad-hoc). +2. Stop proxy described as conditional: FOLDED (021: always listed, enabled only when the app started the proxy; tray.rs:61,279). +3. 030 lacks pre-written translated content: REBUTTED. The translated prose is the wp4 deliverable itself; writing it into the plan would perform wp4 inside the docs-only cycle, which LOOP-DOCS-FIRST-01 forbids ("no production patches" in the roadmap cycle). DIFFLEVEL-ROADMAP-01's purpose, an executable PRD per phase, is met by the fixed NEW-file list, the pinned source (English at the wp4 P revision), the byte-identity rules for code, links and frontmatter keys, and the mechanical parity script. Sidebar labels are defined as each new page's translated frontmatter title, so they follow from the worker output with no further judgment. wp4's own P re-verifies this doc against the tree, as the rule requires. + +Main verdict for A>B: near-pass. diff --git a/devlog/_plan/260923_docs_polish_404/001_existing_link_failures.md b/devlog/_plan/260923_docs_polish_404/001_existing_link_failures.md new file mode 100644 index 0000000000..d0e4c8cfaf --- /dev/null +++ b/devlog/_plan/260923_docs_polish_404/001_existing_link_failures.md @@ -0,0 +1,112 @@ +# 001 existing internal-link failures (rendered build, dev a4bdc03054) + +Produced by a scratch scan of docs-site/dist after bun run build (497 pages, 65438 internal href/src). Links whose source page is 404.html are excluded (Starlight language picker on the 404 page points at /<locale>/404/, which Starlight does not build). Columns: kind, source file under docs-site/src/content/docs (FALLBACK = English source rendered at a locale URL), href as written. + +| kind | source | href | +| --- | --- | --- | +| FRAG | fr/guides/codex-integration.md | /fr/reference/cli/lifecycle/#ocx-service-installrepairstartstopstatusuninstallremove | +| FRAG | guides/claude-code.md | #mcp-tool-schemas-fill-the-context-on-turn-one | +| FRAG | guides/claude-code.md | /reference/configuration/#anthropicaccountpool-experimental | +| FRAG | guides/claude-code.md | /reference/configuration/#sidecars | +| FRAG | guides/codex-integration.md | /reference/cli/#ocx-service | +| FRAG | guides/opencode.md | /reference/configuration/#remote-access | +| FRAG | guides/pi.md | /reference/configuration/#remote-access | +| FRAG | guides/providers.md | /reference/cli/#ocx-account-subcommand | +| FRAG | guides/providers.md | /reference/configuration/#cursor-provider-adapter-cursor | +| FRAG | guides/sidecars.md | /reference/configuration/#sidecars | +| FRAG | ja/guides/claude-code.md | /ja/reference/configuration/#sidecars | +| FRAG | ja/guides/codex-integration.md | /reference/cli/#ocx-service | +| FRAG | ja/guides/grok-build.md | #manual-recipe-without-auto-registration | +| FRAG | ja/guides/image-bridge.md | #configuration | +| FRAG | ja/guides/opencode.md | /reference/configuration/#remote-access | +| FRAG | ja/guides/pi.md | /reference/configuration/#remote-access | +| FRAG | ja/guides/providers.md | /ja/reference/cli/#ocx-account-subcommand | +| FRAG | ja/guides/providers.md | /ja/reference/configuration/#cursor-provider-adapter-cursor | +| FRAG | ja/guides/sidecars.md | /ja/reference/configuration/#sidecars | +| FRAG | ja/reference/adapters.md | /ja/reference/configuration/providers/#cursor-provider-adapter-cursor | +| FRAG | ja/reference/architecture.md | /ja/guides/codex-integration/#the-subagent-picker | +| FRAG | ja/reference/cli/agents.md | /reference/configuration/#remote-access | +| FRAG | ja/reference/configuration/providers.md | /reference/configuration/server/#claude-code | +| FRAG | ko/guides/claude-code.md | /ko/reference/configuration/#sidecars | +| FRAG | ko/guides/codex-integration.md | /reference/cli/#ocx-service | +| FRAG | ko/guides/grok-build.md | #manual-recipe-without-auto-registration | +| FRAG | ko/guides/image-bridge.md | #configuration | +| FRAG | ko/guides/opencode.md | /reference/configuration/#remote-access | +| FRAG | ko/guides/pi.md | /reference/configuration/#remote-access | +| FRAG | ko/guides/providers.md | /ko/reference/cli/#ocx-account-subcommand | +| FRAG | ko/guides/providers.md | /ko/reference/configuration/#cursor-provider-adapter-cursor | +| FRAG | ko/guides/sidecars.md | /ko/reference/configuration/#sidecars | +| FRAG | ko/reference/adapters.md | /ko/reference/configuration/providers/#cursor-provider-adapter-cursor | +| FRAG | ko/reference/architecture.md | /ko/guides/codex-integration/#the-subagent-picker | +| FRAG | ko/reference/cli/agents.md | /reference/configuration/#remote-access | +| FRAG | ko/reference/configuration/providers.md | /reference/configuration/server/#claude-code | +| FRAG | reference/cli/agents.md | /reference/configuration/#remote-access | +| FRAG | reference/configuration/providers.md | /reference/configuration/server/#claude-code | +| FRAG | ru/guides/claude-code.md | /ru/reference/configuration/#sidecars | +| FRAG | ru/guides/codex-integration.md | /reference/cli/#ocx-service | +| FRAG | ru/guides/grok-build.md | #manual-recipe-without-auto-registration | +| FRAG | ru/guides/image-bridge.md | #configuration | +| FRAG | ru/guides/opencode.md | /reference/configuration/#remote-access | +| FRAG | ru/guides/pi.md | /reference/configuration/#remote-access | +| FRAG | ru/guides/providers.md | /ru/reference/cli/#ocx-account-subcommand | +| FRAG | ru/guides/providers.md | /ru/reference/configuration/#cursor-provider-adapter-cursor | +| FRAG | ru/guides/sidecars.md | /ru/reference/configuration/#sidecars | +| FRAG | ru/reference/adapters.md | /ru/reference/configuration/providers/#cursor-provider-adapter-cursor | +| FRAG | ru/reference/architecture.md | /ru/guides/codex-integration/#the-subagent-picker | +| FRAG | ru/reference/cli/agents.md | /reference/configuration/#remote-access | +| FRAG | ru/reference/configuration/providers.md | /reference/configuration/server/#claude-code | +| FRAG | tr/guides/claude-code.md | /tr/reference/configuration/#anthropicaccountpool-experimental | +| FRAG | tr/guides/claude-code.md | /tr/reference/configuration/#sidecars | +| FRAG | tr/guides/codex-app-models.md | /tr/guides/combos/#codex-desktop-native-allowlist-compatibility | +| FRAG | tr/guides/codex-app-models.md | /tr/reference/configuration/routing/#exact-codex-account-selectors | +| FRAG | tr/guides/codex-integration.md | /tr/guides/codex-app-models/#subagent-selection | +| FRAG | tr/guides/codex-integration.md | /tr/reference/cli/#ocx-service | +| FRAG | tr/guides/grok-build.md | #otomatik-kayit-olmadan-manuel-tarif | +| FRAG | tr/guides/image-bridge.md | /tr/guides/codex-integration/#dahili-g%C3%B6rsel-%C3%BCretimi-image_gen | +| FRAG | tr/guides/opencode.md | /tr/reference/configuration/#remote-access | +| FRAG | tr/guides/pi.md | /reference/configuration/#remote-access | +| FRAG | tr/guides/providers.md | /tr/guides/web-dashboard/#codex-auth-ve-hesap-havuzlari | +| FRAG | tr/guides/providers.md | /tr/reference/cli/#ocx-account-subcommand | +| FRAG | tr/guides/providers.md | /tr/reference/configuration/#cursor-saglayicisi-adapter-cursor | +| FRAG | tr/guides/sidecars.md | /tr/reference/configuration/#sidecars | +| FRAG | tr/guides/sub-agent-surface.md | /tr/reference/configuration/agents/#encrypted-v2-task-recovery | +| FRAG | tr/guides/web-dashboard.md | /tr/guides/providers/#saglayicilar-genel-bakis-havuz-kapasitesi | +| FRAG | tr/reference/adapters.md | /tr/reference/architecture/#kopru-bridge | +| FRAG | tr/reference/architecture.md | /tr/guides/codex-integration/#the-subagent-picker | +| FRAG | tr/reference/cli/agents.md | /tr/reference/configuration/#remote-access | +| FRAG | tr/reference/configuration/agents.md | #sifrelenmis-v2-gorev-kurtarma | +| FRAG | tr/reference/configuration/providers.md | /tr/reference/configuration/server/#claude-code | +| FRAG | tr/reference/configuration/server.md | /tr/guides/claude-code/#auth-mode | +| FRAG | zh-cn/guides/claude-code.md | /zh-cn/reference/configuration/#sidecars | +| FRAG | zh-cn/guides/codex-integration.md | /reference/cli/#ocx-service | +| FRAG | zh-cn/guides/grok-build.md | #manual-recipe-without-auto-registration | +| FRAG | zh-cn/guides/image-bridge.md | #configuration | +| FRAG | zh-cn/guides/opencode.md | /reference/configuration/#remote-access | +| FRAG | zh-cn/guides/pi.md | /reference/configuration/#remote-access | +| FRAG | zh-cn/guides/providers.md | /zh-cn/reference/cli/#ocx-account-subcommand | +| FRAG | zh-cn/guides/providers.md | /zh-cn/reference/configuration/#cursor-provider-adapter-cursor | +| FRAG | zh-cn/guides/sidecars.md | /zh-cn/reference/configuration/#sidecars | +| FRAG | zh-cn/reference/adapters.md | /zh-cn/reference/configuration/providers/#cursor-provider-adapter-cursor | +| FRAG | zh-cn/reference/architecture.md | /zh-cn/guides/codex-integration/#subagent-%E9%80%89%E6%8B%A9%E5%99%A8 | +| FRAG | zh-cn/reference/cli/agents.md | /reference/configuration/#remote-access | +| FRAG | zh-cn/reference/configuration/providers.md | /reference/configuration/server/#claude-code | +| FRAG | zh-tw/guides/claude-code.md | /zh-tw/reference/configuration/#anthropicaccountpool-experimental | +| FRAG | zh-tw/guides/claude-code.md | /zh-tw/reference/configuration/#sidecars | +| FRAG | zh-tw/guides/codex-integration.md | /zh-tw/guides/codex-app-models/#subagent-selection | +| FRAG | zh-tw/guides/codex-integration.md | /zh-tw/reference/cli/#ocx-service | +| FRAG | zh-tw/guides/grok-build.md | #manual-recipe-without-auto-registration | +| FRAG | zh-tw/guides/image-bridge.md | /zh-tw/guides/codex-integration/#built-in-image-generation-image_gen | +| FRAG | zh-tw/guides/pi.md | /zh-tw/reference/configuration/#remote-access | +| FRAG | zh-tw/guides/providers.md | /zh-tw/guides/web-dashboard/#codex-auth-and-account-pools | +| FRAG | zh-tw/guides/providers.md | /zh-tw/reference/cli/#ocx-account-subcommand | +| FRAG | zh-tw/guides/providers.md | /zh-tw/reference/configuration/#cursor-provider-adapter-cursor | +| FRAG | zh-tw/guides/sidecars.md | /zh-tw/reference/configuration/#sidecars | +| FRAG | zh-tw/reference/cli/agents.md | /zh-tw/reference/configuration/#remote-access | +| FRAG | zh-tw/reference/configuration/providers.md | /zh-tw/reference/configuration/server/#claude-code | +| FRAG | zh-tw/reference/configuration/server.md | /zh-tw/guides/claude-code/#auth-mode | +| ROUTE | FALLBACK:guides/desktop-app.md | /opencodex/guides/macos-menu-bar/ | +| ROUTE | guides/desktop-app.md | /opencodex/guides/macos-menu-bar/ | +| ROUTE | reference/configuration/server.md | ../../guides/codex-integration.md#rich-tool-results-and-explicit-approvals-after-response-completion | +| ROUTE | reference/configuration/server.md | ../../guides/codex-integration.md#steering-confirmation-deadlines-and-retained-context | +| ROUTE | reference/configuration/server.md | ../../guides/codex-integration.md#steering-continuation-settings | +| ROUTE | reference/configuration/server.md | providers.md#provider-entries-ocxproviderconfig | diff --git a/devlog/_plan/260923_docs_polish_404/010_wp2_link_integrity.md b/devlog/_plan/260923_docs_polish_404/010_wp2_link_integrity.md new file mode 100644 index 0000000000..c0e1700f7e --- /dev/null +++ b/devlog/_plan/260923_docs_polish_404/010_wp2_link_integrity.md @@ -0,0 +1,144 @@ +# 010 wp2 — link integrity + +Two layers, because CI reaches the two kinds of source through different jobs. The `ci` change filter +(`.github/workflows/ci.yml:225-243`) matches `README.md`, `src/**`, `gui/**`, `tests/**` but not +`docs-site/**`; a docs-only pull request runs only the `docs` job, which builds the Astro site +(`ci.yml:261-274,993-1014`). A Bun test alone would never see a docs-only PR, so the docs-site check lives +inside the Astro build, and the Bun test covers the URLs other surfaces hard-code. + +## Link repairs + +MODIFY `docs-site/src/content/docs/guides/desktop-app.md:104` + +```diff +-[macOS Menu Bar App guide](/opencodex/guides/macos-menu-bar/) for widget setup and the ++[macOS Menu Bar App guide](/guides/macos-menu-bar/) for widget setup and the +``` + +MODIFY `docs-site/src/content/docs/reference/configuration/server.md:717-719`: the three +`../../guides/codex-integration.md#<frag>` links become `/guides/codex-integration/#<frag>`, same fragments +(headings at `guides/codex-integration.md:968,1092,1129`). Same rewrite in any locale copy carrying the +`.md` form (grep `codex-integration.md#` under `docs-site/src/content/docs`), with that locale's prefix. + +Any further break the new build check reports on the current corpus is fixed in the same commit and +listed in the commit body. + +## Layer A: build-time check (docs-site) + +NEW `docs-site/src/integrations/internal-links.mjs` (plain ESM, no dependency): an Astro integration with an +`astro:build:done` hook that receives `dir` (the output URL). + +- Walk every `*.html` under the output dir. For each page collect `id="…"` values and every + `href="…"` / `src="…"`. Resolve each with `new URL(value, pageUrl)`, where `pageUrl` is + `https://opencodex.me/<page dir>/` (the URL Starlight's own nav and sitemap use). The result is internal + when its host is `opencodex.me`, or `lidge-jun.github.io` with a leading `/opencodex` segment, which is + stripped once for that host only (GitHub's redirect does the same). Skip other hosts, `mailto:`, + `data:`, `/_astro/`, `/pagefind/`. Relative links and fragment-only `#frag` (same page) are checked. +- On the generated `404.html` only, skip unresolvable links whose path is `/<locale>/404/`: Starlight's + language picker there points at locale 404 pages it never builds (measured: 14 such links on dev + a4bdc03054). Every other link on the 404 page, and picker links on ordinary pages, stay checked. +- Resolve a path: strip query; decode; if the path names an existing file in dist, it resolves; + else try `<path>/index.html` (with or without the trailing slash, matching `trailingSlash: "ignore"` at + `astro.config.mjs:18`), else `<path>.html`. No match is a broken link. +- Fragment: when the link carries `#frag` and resolves to an HTML page, `frag` (decoded) must be one of + that page's ids. These are the rendered Starlight heading ids, so no slugger is reimplemented. +- Report as `page → href` lines. Broken routes and broken fragments both throw and fail the build. + There is no global fragment switch and no allowlist. + +## Pre-existing failures (measured, see [001](001_existing_link_failures.md)) + +A scratch scan of the current build found 106 unique failures (the 404-page picker excluded): 5 route +breaks and 11 fragment breaks in English sources, about 90 fragment breaks in locale sources. Most come +from the configuration reference split into subpages (`/reference/configuration/#remote-access` now lives +on `/reference/configuration/server/`) and translated headings whose slugs changed. All are fixed in this +commit so Layer A can throw on the first build: + +- English sources (`guides/{claude-code,codex-integration,desktop-app,opencode,pi,providers,sidecars}.md`, + `reference/cli/agents.md`, `reference/configuration/{providers,server}.md`): main fixes each by finding + the heading's current page and slug in the rendered build. +- Each locale's rows: one gpt-6-sol worker per locale, write scope = the listed source files in + `docs-site/src/content/docs/<locale>/`, rule = point the link at the rendered id that now carries that + heading (in the same locale when the target page exists there, else the fallback page's English id); + never delete a link to silence it. When the fragment names prose with no heading (for example + `guides/claude-code.md` `#mcp-tool-schemas-fill-the-context-on-turn-one` points at bold text), link the + nearest enclosing heading's id. Workers run no builds; main rebuilds once after all return. +- Link ledger (audit round 3): before and after the repair, main's scratch script records per edited source + file the ordered list of Markdown link texts and the count of links; the two lists must be equal, only + hrefs may differ, and every row of 001 must map to a changed href whose new target resolves in the rebuilt + dist. A shrunken list fails the phase. +- `.github/ISSUE_TEMPLATE/documentation.yml:31` placeholder `https://opencodex.me/providers/` is broken. Open PR + #5593 changes that line to + `"https://opencodex.me/guides/providers/ or docs-site/src/content/docs/guides/providers.md"`; this commit + makes the byte-identical change so either merge order is clean, and Layer B scans + `.github/ISSUE_TEMPLATE/`. +- Export the pure resolver (`resolveInternalLink(distFiles, idsByPage, fromPage, href)`) so the Bun test + can exercise it with fixtures without building. + +MODIFY `docs-site/astro.config.mjs`: import the integration and append it to `integrations` after +`starlight(...)`. + +Bypass record (PLAN-BYPASS-NAMED-01): tier E8 (out-of-band build gate). Executing surface: the Astro +build, run by the CI `docs` job on every pull request touching `docs-site/**` and by `deploy-docs.yml` +before publishing. Known bypasses: links assembled by client-side JavaScript, links to other hosts, a +maintainer merging over a red `docs` check, and editing the integration out of `astro.config.mjs`. +Residual risk: broken external links and client-rendered links. Wording: "checked at build time" for +hrefs and srcs present in the generated HTML, not "every link". Final layer: the Deploy Docs build, +which refuses to publish a site with a broken internal link. + +## Layer B: Bun test (hard-coded URLs elsewhere) + +NEW `tests/ci-workflows/docs-link-targets.test.ts` (target under 300 lines; new-file threshold 2000): + +- Route table from `docs-site/src/content/docs/**/*.md[x]`: path minus extension, `index` collapses to its + directory, lowercased. Locale keys parsed from the `locales: {…}` block of `docs-site/astro.config.mjs` + (root excluded). `<locale>/<rest>` with `<rest>` in the table resolves as a Starlight fallback page. + For `lidge-jun.github.io` the leading `/opencodex` project segment is stripped once (redirect + semantics); on `opencodex.me` a leading `opencodex/` is never stripped. Files under `docs-site/public` + resolve as assets. +- URL surfaces: `README.md`, `readme/*.md`, root `*.md`, `package.json`, `src/**`, `gui/src/**`, + `skills/**`, `docs-site/src/components/**`, for `https://opencodex.me/<path>` and + `https://lidge-jun.github.io/opencodex/<path>`. Template-literal paths and sitemap/robots/image URLs are + skipped. +- Fragments on these URLs are not checked here: only the rendered pages carry the real ids. C runs a + one-off scratch audit of every such `#frag` against `docs-site/dist` and fixes what it finds; after + that, a fragment-only regression in a README is not guarded (stated limit). +- Tests: (1) every URL resolves, failure lists `file:line url`; (2) fixtures for the Layer A resolver + imported from `docs-site/src/integrations/internal-links.mjs`: `/opencodex/guides/macos-menu-bar/` + broken, `/guides/macos-menu-bar/` and `/guides/macos-menu-bar` resolve, `/guides/x/#missing` broken + fragment, same-page `#missing` broken, relative `../../guides/codex-integration.md` from + `/reference/configuration/server/` broken, `https://opencodex.me/guides/macos-menu-bar/` internal and + resolving, `/favicon.png` asset; (3) route-table fixtures: `/ko/guides/desktop-app/` resolves (file or + fallback), `https://opencodex.me/opencodex/guides/x/` broken, + `https://lidge-jun.github.io/opencodex/guides/cursor-private-inference/` resolves; (4) sanity: the scan + finds more than 20 URLs so an empty extractor fails. + +Layer B bypass record: tier E8 (CI suite). Executing surface: the Bun test shards, which the `ci` filter +starts for `README.md`, `src/**`, `gui/**`, `tests/**`, `package.json`. Known bypasses: `readme/**` and +`skills/**` are not in that filter, so a PR touching only those skips it; fragments are not checked. +Residual risk: a broken URL in a locale README or skill lands and is caught on the next run that does +start the suite. Wording: "guarded in the CI suite". Final layer: none beyond the suite. + +MODIFY `scripts/test-layout/layout.json` explicit map and `tests/fixtures/test-layout-expected.json`: add +`"docs-link-targets.test.ts": "ci-workflows"` beside `docs-readme-translation-parity.test.ts`. + +## Structure doc + +MODIFY `structure/ops/docs-and-release.md` "Public docs": the locale sentence lists all seven locales (adds +French `/fr` and Turkish `/tr`, matching `astro.config.mjs:61-72`); one paragraph names both layers and +their limits. No manifest or `INV-*` binding is added (`structure/AGENTS.md:98-102` owns that mechanism +and this unit makes no invariant claim). + +## Acceptance + +- `cd docs-site && bun run build` exit 0 with the integration active, and it reports the number of pages + and links checked. +- Driven red, Layer A: revert desktop-app.md:104 locally; the build fails naming + `/guides/desktop-app/ → /opencodex/guides/macos-menu-bar/`; restore. +- Driven red, Layer B: add `https://opencodex.me/opencodex/guides/x/` to a scratch copy path the scanner + reads (temporary edit to README.md, reverted); test (1) fails; restore. +- `bun test tests/ci-workflows/docs-link-targets.test.ts` run alone exits 0 and its output lists at + least four `(pass)` lines from this file; then + `bun test tests/test-layout.test.ts tests/test-layout-tooling.test.ts` and `bun run structure:check` exit 0. +- Driven red, fragment: a scratch edit adding `[x](#does-not-exist)` to one English page fails the build; + reverted. +- Commit: `docs: repair stale docs links and check every internal link at build time`. diff --git a/devlog/_plan/260923_docs_polish_404/020_wp3_english_polish_readme_resync.md b/devlog/_plan/260923_docs_polish_404/020_wp3_english_polish_readme_resync.md new file mode 100644 index 0000000000..6d3cb86ab8 --- /dev/null +++ b/devlog/_plan/260923_docs_polish_404/020_wp3_english_polish_readme_resync.md @@ -0,0 +1,88 @@ +# 020 wp3 — English polish and README locale resync + +## English guides (verified against code) + +MODIFY `docs-site/src/content/docs/guides/macos-menu-bar.md`: replaced by the full text in +[021](021_wp3_macos_menu_bar_draft.md) (main verifies its evidence list before A>B). Topics it covers: + +- Intro and "What it shows": the macOS tray belongs to the Tauri desktop app + (`desktop/src-tauri/src/tray.rs:45`); the native usage panel shows Today, 30 days, a chart, models and + account limits (`app/Sources/NativeTray/UsageView.swift:25`). Drop the "separate read-mostly app" and + the four-section/provider-switch claims. +- Proxy discovery: the app asks the bundled CLI (`ocx resolve --json`, `resolve.rs:3-7,209`) instead of + reading `runtime-port.json` itself. +- Auth: token from the environment or `$OPENCODEX_HOME/admin-api-token` (`auth.rs:19-24`), retried on + 401 (`proxy.rs:183-186`); no Keychain claim. +- Widget refresh: the tray polls every 60 s and refreshes the widget snapshot every fifth tick, about + every five minutes (`tray.rs:236`). +- Build from source: `cd desktop`, `bun install`, `bun run prepare-sidecar`, `bun run prepare-widget`, + `bun run build:local` for a workstation build; `bunx tauri build` needs `TAURI_SIGNING_PRIVATE_KEY` + (`desktop/package.json:4-11`, `desktop/README.md:28-45`). Link `/guides/desktop-app/` for install. +- Gatekeeper and signing, verified on the v2.61.0 release asset (main, 2026-09-23): + `spctl -a -t open --context context:primary-signature OpenCodex-2.61.0-macos.dmg` gives rejected, + `source=Unnotarized Developer ID`; `xcrun stapler validate` on the DMG finds no ticket. On the mounted + `OpenCodex.app`: `spctl -a -t exec` gives accepted, `source=Notarized Developer ID`; `stapler validate` + works; `codesign -dv` shows `flags=0x10000(runtime)`, Developer ID Application, TeamIdentifier + U9ATA49N28. Wording: "Release builds of OpenCodex.app are signed with a Developer ID and notarized by + Apple, so macOS normally shows only the standard confirmation for an app downloaded from the internet." + Fallback: "If macOS still refuses to open it, open System Settings → Privacy & Security and choose + Open Anyway." No claim that the DMG is notarized. Local builds are ad-hoc signed unless + `MACOS_SIGN_IDENTITY` is set. +- Keep verified claims: asset names (`desktop/scripts/collect-release-assets.ts:22`), macOS 13 app / + macOS 14 widget (`desktop/src-tauri/tauri.conf.json:32`, `app/Widget-Info.plist:22`), Show Usage + (`menu.rs:95-98`), six-hour update check (`updater.rs:81`). + +MODIFY `docs-site/src/content/docs/guides/desktop-app.md`: exact before/after snippets in 021 for the +dashboard URL sentence (line 8), the Gatekeeper paragraph (lines 19-21, same wording as above) and the +first-launch discovery paragraph (lines 49-51, `ocx resolve`). + +## README + +MODIFY `README.md`: + +- Desktop section (README.md:97-114), after: + + > A native shell around the same dashboard, plus a WidgetKit extension that shows proxy status, + > today's usage and provider quotas without opening a browser. The proxy is unchanged: the app + > finds a running one or starts the bundled `ocx` sidecar, and the dashboard stays on the proxy's + > port (**http://localhost:10100** unless you configured another). + > + > It is beta. Release builds of the macOS app are signed with a Developer ID and notarized (local + > builds are ad-hoc signed); the Windows installer is + > not code-signed yet, so SmartScreen warns on first run. The widget needs macOS 14 or newer; the + > snapshot model it renders lives in [`app/`](./app) (`MenuBarCore`). + > + > Download it from the [latest release](https://github.com/lidge-jun/opencodex/releases), or build + > it locally: run `bun install && bun run build:gui` at the repository root, then + > `bun install && bun run prepare-sidecar && bun run prepare-widget && bun run build:local` in `desktop/` + > (`prepare-sidecar` bundles `gui/dist`, `desktop/scripts/prepare-sidecar.ts:58`). + > + > Install locations, service files and everything else written to disk are listed in + > [`AGENTS_INSTALL.md`](./AGENTS_INSTALL.md#where-things-are-installed). The + > [Desktop App guide](https://opencodex.me/guides/desktop-app/) and the + > [macOS Menu Bar App guide](https://opencodex.me/guides/macos-menu-bar/) cover per-platform + > installation and first launch. + + Windows evidence (wp3 P, 2026-09-23): the v2.61.0 MSI has no `DigitalSignature` or + `MsiDigitalSignatureEx` stream, and neither `release.yml` nor the Tauri config configures Windows code + signing; it ships only a Tauri updater `.sig`. The "not code-signed yet" sentence is accurate. +- Source install (README.md:207-226): both clone commands become + `git clone -b dev https://github.com/lidge-jun/opencodex.git`; after `bun install` add + `bun run build:gui` (macOS/Linux: `~/.bun/bin/bun run build:gui`) so `GET /` serves the dashboard + (`docs-site/src/content/docs/getting-started/installation.md:89`). Closing sentence unchanged. +- Memory inventory section: untouched (open PR #5340). + +MODIFY `readme/README.{fr,ja,ko,ru,tr,zh-CN,zh-TW}.md`: the same edits translated; docs links in the +localized form `https://opencodex.me/<docsPath>/guides/…` the parity test requires; commands +byte-identical. +MODIFY `readme/i18n-manifest.json`: every locale `sourceSha256` is the LF-normalized SHA-256 of the new +README.md. + +Dispatch: seven gpt-6-sol workers, one README locale each, write scope = that one file; main computes the +manifest hash and runs the parity test. + +## Acceptance + +- `bun test tests/ci-workflows/docs-readme-translation-parity.test.ts tests/ci-workflows/docs-link-targets.test.ts` exit 0. +- Commits: `docs(desktop): describe the shipped macOS tray and build path` (guides) and + `docs(readme): correct desktop and source-install claims in every locale` (README + 7 + manifest). diff --git a/devlog/_plan/260923_docs_polish_404/021_wp3_macos_menu_bar_draft.md b/devlog/_plan/260923_docs_polish_404/021_wp3_macos_menu_bar_draft.md new file mode 100644 index 0000000000..273d5c07aa --- /dev/null +++ b/devlog/_plan/260923_docs_polish_404/021_wp3_macos_menu_bar_draft.md @@ -0,0 +1,173 @@ +# 021 wp3 — replacement text for guides/macos-menu-bar.md and desktop-app.md edits + +Replace `docs-site/src/content/docs/guides/macos-menu-bar.md` with the following complete Markdown: + +````markdown +--- +title: macOS Menu Bar App +description: Use the OpenCodex desktop app's macOS tray, native usage panel, and widget. +--- + +The macOS menu bar item is part of the OpenCodex desktop app. It shows usage from the local proxy and opens a native usage panel. The same app also contains the dashboard and a WidgetKit extension. See the [desktop app guide](/guides/desktop-app/) for installation on other platforms. + +## Install + +Download `OpenCodex-<version>-macos.dmg` from the [latest release](https://github.com/lidge-jun/opencodex/releases). Open the DMG and drag `OpenCodex.app` to Applications. The desktop app requires macOS 13 or later; its widget requires macOS 14 or later. + +## First launch + +Release builds of `OpenCodex.app` are signed with a Developer ID, use the hardened runtime, and are notarized by Apple with the ticket stapled to the app. On first launch, macOS normally asks only for the standard confirmation for an app downloaded from the internet. If macOS still blocks it, open **System Settings → Privacy & Security** and choose **Open Anyway** for OpenCodex. Apps you build yourself are ad-hoc signed; see [Build from source](#build-from-source). + +The app shows its startup progress in a window when you open it. It enables **Start at Login** once on first launch; you can turn that off from the tray menu. Later launches from the login item start with the window hidden while the tray remains available. + +## Menu bar and usage panel + +The menu bar headline shows today's total tokens by default. In the dashboard's **Menu bar & widget** settings, you can choose requests, tokens, estimated cost, quota, or icon only. + +Use **Show Usage** in the tray menu to open the native panel. The panel shows today's and 30-day totals, a usage chart, a model list, and provider and account limits according to your display settings. Totals include tokens and requests, with estimated cost when enabled. Quota rows show their window, percentage, and reset time. Missing measurements appear as `—`, and partial usage is marked as incomplete. + +The panel has **Refresh**, **Dashboard**, and **Settings** controls. **Dashboard** opens the usage view in the desktop window; **Settings** opens the companion settings there. The tray menu also offers **Open Dashboard**, **Open in Browser**, **Start at Login**, **Stop proxy**, **Check for Updates…**, an **Install update** item when one is available, and **Quit**. **Stop proxy** is always listed but is enabled only when the app started the proxy itself; a proxy you started separately keeps running. Closing the window or using Command-Q hides the app when its tray is available; use the tray's **Quit** to exit it. + +The tray headline refreshes every 60 seconds. While the native panel is open, its data refreshes every 60 seconds; **Refresh** requests an immediate update. + +## Widget + +On macOS 14 or later, open **Edit Widgets** from the desktop and add **OpenCodex**. Widget sizes show different combinations of proxy status, today's tokens and requests, estimated cost, quotas, and a usage chart. The extension reads a local snapshot written by the desktop app; that snapshot contains display data, not API keys or raw account data. The app refreshes the widget snapshot on every fifth 60-second tray tick, about every five minutes while the proxy is connected. WidgetKit also requests a new timeline after five minutes. + +## Connecting to the proxy + +The desktop app asks its bundled CLI to run `ocx resolve --json`. It attaches to an existing reachable local proxy, or starts its bundled runtime only when the CLI proves no runtime is listening. If discovery is uncertain, startup reports the problem instead of starting a second proxy. The app talks to the resolved port on `127.0.0.1`. + +For management requests, the app first tries without a token. If the proxy returns HTTP 401, it retries using `OPENCODEX_ADMIN_AUTH_TOKEN` from the app's environment or the resolved configuration home's `admin-api-token` file. It does not use the macOS Keychain for this token. A proxy bound only to an address the app cannot reach on loopback cannot be attached to by the desktop shell. + +## Build from source + +On macOS 13 or later, with Bun, Rust, and the macOS Swift/Xcode tools available, build the dashboard from the repository root, then run the desktop commands from `desktop/`: + +```bash +bun install +bun run build:gui +cd desktop +bun install +bun run prepare-sidecar +bun run prepare-widget +bun run build:local +``` + +`build:local` produces the local app and DMG without requiring a Tauri updater signing key. A direct `bunx tauri build` requires `TAURI_SIGNING_PRIVATE_KEY` because it also produces an updater artifact. The widget build uses an ad-hoc signature unless `MACOS_SIGN_IDENTITY` is set; local desktop bundles are also ad-hoc signed. + +## Uninstall + +Turn off **Start at Login** in the tray menu if you enabled it, then move `OpenCodex.app` from Applications to the Trash. This removes the bundled CLI and widget extension, but does not remove the proxy's `$OPENCODEX_HOME` state or a separately installed `ocx` service. The desktop app also writes an installation ID and login-item markers in its app configuration directory, plus a widget snapshot under `~/Library/Containers/com.opencodex.desktop.widget/Data/Library/Application Support/OpenCodex/snapshot.json`; moving the app to the Trash does not delete those files. +```` + +Apply these exact replacements in `docs-site/src/content/docs/guides/desktop-app.md`: + +### Opening description (current lines 6–11) + +Before: + +```markdown +The OpenCodex desktop app combines a native tray with the web dashboard. It discovers an +existing local proxy, or starts the bundled `ocx` sidecar when no proxy is running. + +The dashboard remains available at [http://127.0.0.1:10100](http://127.0.0.1:10100). +The desktop app does not replace the proxy; it is a local shell around the dashboard and +its bundled runtime. +``` + +After: + +```markdown +The OpenCodex desktop app combines a native tray with the web dashboard. Its bundled CLI +resolves an existing local proxy; the app starts its bundled runtime only when absence is proven. + +The dashboard is served from the resolved local proxy endpoint (port `10100` by default). +The desktop app is a local shell around that dashboard and its bundled runtime. +``` + +### macOS first launch (current lines 17–23) + +Before: + +```markdown +Download `OpenCodex-<version>-macos.dmg` from the +[latest release](https://github.com/lidge-jun/opencodex/releases). Open the DMG and drag +`OpenCodex.app` to Applications. + +On first launch, macOS Gatekeeper may warn that the developer cannot be verified. Right-click +the app, choose **Open**, and confirm **Open**. This build is signed for integrity but is not +yet notarized. +``` + +After: + +```markdown +Download `OpenCodex-<version>-macos.dmg` from the +[latest release](https://github.com/lidge-jun/opencodex/releases). Open the DMG and drag +`OpenCodex.app` to Applications. The app requires macOS 13 or later. + +Release builds of `OpenCodex.app` are signed with a Developer ID and notarized by Apple, so on +first launch macOS normally asks only for the standard confirmation for a downloaded app. If macOS +still blocks it, use **System Settings → Privacy & Security → Open Anyway**. +``` + +### Proxy startup (current lines 51–55) + +Before: + +```markdown +## First launch + +The app first looks for an existing `ocx` proxy on loopback, using the runtime port +metadata when available and falling back to port `10100`. If no proxy answers, it starts +the bundled sidecar. The dashboard is then opened inside the app's webview. +``` + +After: + +```markdown +## First launch + +The app asks its bundled CLI to run `ocx resolve --json` and attaches to a reachable local +proxy if one is already running. It starts the bundled runtime only when the CLI proves +absence; an uncertain result is shown as a startup failure. The dashboard then opens in +the app's webview at the resolved loopback endpoint. +``` + +### Widget guide link (current lines 101–105) + +Before: + +```markdown +## Widget + +The macOS app includes the OpenCodex WidgetKit extension. See the +[macOS Menu Bar App guide](/guides/macos-menu-bar/) for widget setup and the +privacy-safe snapshot details. +``` + +After: + +```markdown +## Widget + +The macOS app includes the OpenCodex WidgetKit extension. See the +[macOS Menu Bar App guide](/guides/macos-menu-bar/) for widget setup and the +local snapshot details. +``` + +## Evidence + +- Desktop shell and bundled CLI/widget: `desktop/README.md:3-19`; `desktop/src-tauri/tauri.conf.json:17-36`; `desktop/src-tauri/src/native_tray.rs:1-4`. +- macOS and widget minimum versions: `desktop/src-tauri/tauri.conf.json:32-36`; `app/Widget-Info.plist:17-26`. +- Release-signing procedure: `.github/workflows/release.yml:275-299`, `:323-354`, `:369-386`. The specific v2.61.0 app/DMG signature, notarization, and stapling status comes from the artifact verification supplied in this task; repository code alone cannot prove the published artifact's state. +- Startup window and login-item behavior: `desktop/src-tauri/src/lib.rs:221-244`; `desktop/src-tauri/src/first_run.rs:32-69`; `desktop/src-tauri/src/startup.rs:88-96`; `desktop/src-tauri/src/tray.rs:48-55`, `:181-188`. +- Tray labels, actions, ownership gate, and quitting: `desktop/src-tauri/src/tray.rs:45-95`, `:147-228`, `:279-284`; `desktop/src-tauri/src/exit.rs:296-325`. +- Headline metrics and refresh: `desktop/src-tauri/src/tray.rs:233-265`, `:330-374`; `gui/src/pages/usage-companion-panel.tsx:24-35`, `:49-55`, `:481`. +- Native panel contents, actions, and incomplete data: `app/Sources/NativeTray/UsageView.swift:23-89`, `:93-116`; `app/Sources/NativeTray/UsageSections.swift:4-47`, `:50-109`; `desktop/src-tauri/src/native_tray.rs:83-106`, `:125-175`; `desktop/src-tauri/src/native_tray_data.rs:53-160`. +- Widget contents, local snapshot, and refresh cadence: `app/Sources/OpenCodexWidget/Views.swift:32-129`; `app/Sources/OpenCodexWidget/Provider.swift:21-36`; `app/Sources/OpenCodexWidget/SnapshotReader.swift:15-25`; `desktop/src-tauri/src/widget.rs:17-69`, `:233-278`, `:280-319`; `desktop/src-tauri/src/tray.rs:233-265`. +- Discovery, loopback, and unknown-state handling: `desktop/src-tauri/src/resolve.rs:65-76`, `:103-170`, `:173-225`; `desktop/src-tauri/src/startup.rs:515-585`. +- Token source and HTTP 401 retry: `desktop/src-tauri/src/auth.rs:9-25`; `desktop/src-tauri/src/proxy.rs:181-208`. The active desktop auth implementation reads only these sources; it has no Keychain lookup. +- Source-build commands and signing distinction: `desktop/package.json:4-12`; `desktop/scripts/prepare-sidecar.ts:37-59`; `desktop/README.md:26-54`, `:64-73`; `desktop/scripts/build-local.ts:43-63`, `:97-110`; `desktop/scripts/build-widget.sh:22-39`. +- Uninstall and remaining state: `AGENTS_INSTALL.md:61-85`, `:99-112`; `desktop/src-tauri/src/identity.rs:21-53`; `desktop/src-tauri/src/first_run.rs:5-9`, `:49-67`, `:83-110`; `desktop/src-tauri/src/widget.rs:233-237`, `:253-277`; `desktop/src-tauri/src/tray.rs:181-188`. diff --git a/devlog/_plan/260923_docs_polish_404/030_wp4_locale_coverage.md b/devlog/_plan/260923_docs_polish_404/030_wp4_locale_coverage.md new file mode 100644 index 0000000000..f259c2acaa --- /dev/null +++ b/devlog/_plan/260923_docs_polish_404/030_wp4_locale_coverage.md @@ -0,0 +1,122 @@ +# 030 wp4 — locale coverage + +## Missing pages (English source to NEW locale file) + +| Page | Missing in | +| --- | --- | +| guides/codex-log-guard-reclaim.md | fr ja ko ru tr zh-cn zh-tw | +| guides/codex-log-guard.md | fr ja ko ru tr zh-cn zh-tw | +| guides/codex-native-context.md | fr ja ru tr zh-cn zh-tw | +| guides/cursor-private-inference.md | fr ja ko ru tr zh-cn zh-tw | +| guides/desktop-app.md | fr ja ko ru tr zh-cn zh-tw | +| guides/factory-droid.md | ja ru tr zh-cn zh-tw | +| guides/integrations.md | ja ko ru zh-cn | +| guides/macos-menu-bar.md | fr tr zh-tw | +| guides/minimax.md | ja ko ru tr zh-cn zh-tw | +| guides/native-main-profiles.md | fr ja ko ru tr zh-cn zh-tw | +| guides/remote-workspace.md | fr ja ko ru tr zh-cn zh-tw | +| guides/response-inspection.md | fr ja ko ru tr zh-cn zh-tw | +| guides/routing-profile-editor.md | ja ko ru zh-cn | +| guides/subagent-v1-default.md | fr ja ko ru tr zh-cn zh-tw | +| reference/inbound-body-admission.md | fr ja ko ru tr zh-cn zh-tw | +| reference/platform-support.md | fr ja ko ru tr zh-cn zh-tw | +| troubleshooting/codex-cannot-sign-in.md | fr ja ko ru tr zh-cn zh-tw | +| troubleshooting/disk-usage-temp-files.md | fr ja ko ru tr zh-cn zh-tw | + +Also MODIFY the existing `{ko,ja,zh-cn,ru}/guides/macos-menu-bar.md` to the wp3 English rewrite. + +Excluded: `contributing/**` (open PR #5593). #5593 appends a "GitHub Copilot App" section to English +`guides/integrations.md`; the new ja/ko/ru/zh-cn copies will lag it if #5593 lands later. The PR notes this. + +## Delegation output contract (DIFFLEVEL-ROADMAP-01 for translated prose) + +Translated prose is the build output itself, so this doc fixes inputs and mechanical acceptance instead +of pre-writing ~110 pages. Per NEW file `docs-site/src/content/docs/<locale>/<page>`: + +- Source: the English file at the wp4 P revision (after wp3 lands). +- Mechanical parity, checked by main with a scratch script and by the verifier lane: same count and + levels of headings; identical fenced code blocks byte for byte; same number of Markdown links and + images; every site link either locale-prefixed or an identical external URL; identical frontmatter + keys; no paragraph over 80 characters that is byte-identical to an English paragraph. +- Build: Layer A passes with the file present. + +## Translation contract (per file) + +- Frontmatter `title` and `description` translated; every other frontmatter key identical. +- Headings, prose, table text and alt text translated; code fences, inline code, commands, config keys, + URLs, file paths, env vars, numbers and product names byte-identical, with one exception: site links + in prose are rewritten as the next rule says. +- Site links gain the locale prefix (`/guides/x/` to `/<locale>/guides/x/`). A fragment pointing into a + page that exists in that locale uses that page's translated heading slug; otherwise keep the English + fragment on the fallback route. The Layer A build check (010) verifies every resulting fragment against + the rendered ids. +- Relative image paths gain one `../` because the file sits one directory deeper. +- Match the register of existing pages in that locale (read two sibling pages first). + +## Sidebar + +MODIFY `docs-site/astro.config.mjs`: every slug in the table gets all seven `translations` labels +(missing today on Response Inspection, Factory Droid, Cursor Private Inference, Native Context +Compatibility, and any other slug lacking a full set). Main edits this file alone after the workers +return, using the titles they chose. + +## Dispatch + +Seven gpt-6-sol workers, one locale each; write scope = the listed files under +`docs-site/src/content/docs/<locale>/` only; read scope = the English sources plus sibling pages in that +locale. Then a separate read-only gpt-6-sol verifier per locale checks structure parity: same heading +count and levels, identical fenced blocks, identical link count, no untranslated English paragraphs. + +## Acceptance + +- The missing-page scan (every English page outside `contributing/`) prints nothing. +- `cd docs-site && bun run build` exit 0; its Layer A check (010) proves localized links and rendered + fragments. `docs-link-targets` still passes. +- Commits: one per locale, `docs(<locale>): translate the pages English had and <locale> lacked`, then + `docs(site): label every sidebar entry in all locales`. + +## wp4 P revision (2026-09-23, source pinned at `7b60e10439`) + +Re-verified: the missing-page scan over every English page outside `contributing/` prints exactly the +18 rows above (112 copies). English sources are final after wp3; `guides/macos-menu-bar.md` and +`guides/desktop-app.md` changed in wp3, and the wp3 C review requires the four existing +`{ko,ja,zh-cn,ru}/guides/macos-menu-bar.md` to be retranslated before push. + +Dispatch: 14 gpt-6-sol workers, two per locale, disjoint write sets: + +- Group A (per locale): `guides/{codex-log-guard-reclaim,codex-log-guard,codex-native-context,cursor-private-inference,desktop-app,factory-droid,macos-menu-bar}.md` + — only the ones missing in that locale, plus a full retranslation of `macos-menu-bar.md` where it exists (ko ja zh-cn ru). +- Group B (per locale): `guides/{integrations,minimax,native-main-profiles,remote-workspace,response-inspection,routing-profile-editor,subagent-v1-default}.md`, + `reference/{inbound-body-admission,platform-support}.md`, `troubleshooting/{codex-cannot-sign-in,disk-usage-temp-files}.md` — only missing ones. + +Workers do not build (one shared `docs-site/dist`); main builds once after all return, runs +`.tmp/trans-parity.ts <locale> <pages…>` (mechanical contract above) and the Layer A build check, and sends +failures back to the same worker. Main then edits `docs-site/astro.config.mjs` alone: each of the 18 slugs +gets all seven `translations` labels equal to that locale file's frontmatter `title`. Existing labels that +already match are left alone. + +Reflection (Plato) folds: + +- Sidebar `translations` keys are `fr ko "zh-CN" "zh-TW" ru ja tr` (`astro.config.mjs:68-69,77`), not the + `zh-cn`/`zh-tw` directory names. +- Inbound links: once a translated page replaces English fallback, existing locale pages that link into it + with an English fragment break. Main repairs those inbound hrefs itself after the first build (Layer A names + them), under the wp2 ledger rule (hrefs only, link text and count unchanged); workers only fix failures in + their own files. +- `guides/subagent-v1-default.md`'s relative SVG gains one `../`; root-relative public images keep their paths. + +Audit (Volta) folds: .tmp/trans-parity.ts now rejects an empty page list, checks the full 116-copy inventory when run bare, and adds title/description, admonition, table-row and inline-code checks; the verifier lane inspects table cells and admonitions explicitly. + +## wp4 outcome (C) + +- 14 worker packets; five hit a provider 429 at spawn time (fr A/B, zh-tw A/B, zh-cn B) and were re-dispatched + unchanged with lower concurrency. Bare `.tmp/trans-parity.ts`: checked 116, failing 0. +- Inbound fix: ja/ko/zh-cn/ru `guides/integrations.md` link the English `/reference/management-api/#aside-profile-controls`; + their translated management-API pages never gained that section (pre-existing drift, out of scope). +- Seven read-only gpt-6-sol verifiers: ja PASS; fr, ko, ru, tr, zh-cn GO-WITH-FIXES (0 blockers); zh-tw 1 blocker + ("authenticated" rendered as "verified" in `codex-native-context.md`) fixed. Applied: ru and tr "ad-hoc signing" + terminology, ru inference-endpoint and catalog sentence, tr grace-period sentence. Rejected by contract: English + labels inside fenced diagrams and code comments (fr, ko, zh-cn, zh-tw), which stay byte-identical to English. +- Sidebar: 24 labels added from the translated titles. Five pages (codex-log-guard, codex-log-guard-reclaim, + native-main-profiles, routing-profile-editor, inbound-body-admission) have no sidebar entry; Platform Support uses a + `link:` entry that already had all seven labels. diff --git a/devlog/_plan/260923_docs_polish_404/040_wp5_delivery.md b/devlog/_plan/260923_docs_polish_404/040_wp5_delivery.md new file mode 100644 index 0000000000..d20f41e3f2 --- /dev/null +++ b/devlog/_plan/260923_docs_polish_404/040_wp5_delivery.md @@ -0,0 +1,26 @@ +# 040 wp5 — verification and delivery + +1. Rebase check: `git fetch origin dev`; if dev moved, rebase `codex/docs-polish-404` and rerun the parity + hash (README.md may have moved) and the link guard. +2. Gates, fresh, exit codes recorded: `bun test tests/ci-workflows/docs-link-targets.test.ts` alone, with + its pass lines counted; `bun run typecheck`; + `bun test tests/ci-workflows/docs-*.test.ts tests/test-layout.test.ts tests/test-layout-tooling.test.ts tests/ci-workflows/file-size-ratchet.test.ts`; + `bun run structure:check`; `bun run privacy:scan`; `git diff --check origin/dev...HEAD`; + `cd docs-site && bun install --frozen-lockfile && bun run build` (the Layer A check verifies every + internal href, src and rendered fragment); a scratch audit of README/source `https://opencodex.me/…#frag` + URLs against `docs-site/dist` ids; dist contains + `guides/macos-menu-bar/index.html` and `guides/desktop-app/index.html` for all eight locales. + The full `bun run test` runs locally at the rebased head (AGENTS.md default before review readiness); + its pass/fail/skip counts go into the PR Verification section, with any environment-only failures named. +3. Push `codex/docs-polish-404` to origin and open a PR to `dev` with the repository template (Summary, + Verification, Checklist). No GUI change, so no screenshot requirement; title and body avoid the word + "gui". +4. Inspect exact-head CI per job (`gh pr view --json headRefOid,statusCheckRollup`, check-runs API); + queued, skipped or cancelled is missing evidence. +5. The final report names the live step: after merge, `dev → main` promotion triggers Deploy Docs. + +## wp5 P revision + +origin/dev is at 6d5d501a6d (two commits past the base, #5595 and #5601), neither touching a file this branch changes. B rebases onto it, reruns the 040 gates at the rebased head, pushes `codex/docs-polish-404` to origin, opens one PR to dev, attaches it, and inspects exact-head CI. The PR body notes: #5593 overlap (identical ISSUE_TEMPLATE line; it appends a Copilot section to English integrations.md that the new ja/ko/ru/zh-cn copies will then lag), #5340 overlap (README and all seven locale READMEs plus the manifest: whichever lands second must regenerate the manifest hash), and that the live 404 was fixed by the approved redeploy (run 35778046041) and future link breaks now fail the docs build. + +Reflection fold (Plato): if #5340 lands first, resync its README prose into all seven locale READMEs before recomputing the manifest hash; if #5593 lands first, translate its added English integrations section into the four new locale copies. A test-layout JSON rebase conflict keeps both entries and reruns the layout tests. No screenshot is required: the gate keys on changed gui/ paths (.github/scripts/pr-quality.cjs:540-549), not on the word. diff --git a/devlog/_plan/260923_docs_retirement/000_plan.md b/devlog/_plan/260923_docs_retirement/000_plan.md new file mode 100644 index 0000000000..4f0e5f05f0 --- /dev/null +++ b/devlog/_plan/260923_docs_retirement/000_plan.md @@ -0,0 +1,59 @@ +# 000 — Retire `docs/` and move PR evidence images off `dev` + +Status: open (wp1, single PABCD cycle). Branch `codex/remove-docs-pr-assets`, base `origin/dev` `746c7386e6`. + +## Problem + +The root `docs/` folder describes itself as historical notes, yet it grew to 4.9 MB, and 4.5 MB of +that is PR screenshot evidence. The same kind of image also piles up in `.github/pr-assets/`, +`assets/pr-screenshots/` and `docs-site/public/pr-screenshots/`. The GUI screenshot rule in +`enforce-target` sends authors to commit images on their PR branch; squash merges then carry every +image into `dev`. Moving the folder does not fix this, as `docs-site/public/pr-screenshots/` +already shows. Worse, anything under `docs-site/public/` is published to GitHub Pages. + +## Inventory at `746c7386e6` + +| Path | Live references outside `devlog/` | Disposition | +| --- | --- | --- | +| `docs/README.md` | `CONTRIBUTING.md:11`, `structure/ops/docs-and-release.md:232`, `scripts/structure-ssot.ts:223` → `structure/INDEX.md:5` | delete; rewrite the three references | +| `docs/design-system/*` (Korean GUI token/component contract) | none | move to `gui/design-system/` (current contract, lives next to `gui/src/styles.css`) | +| `docs/adr/0004`, `docs/adr/0005` (GUI toggle contrast, design tokens) | linked by design-system README | move to `gui/design-system/decisions/` | +| `docs/adr/0001-0003, 0006, 0007` | none | delete; git history keeps them at `746c7386e6` | +| `docs/superpowers/**` (16 dated plans/specs) | none | delete | +| `docs/codex-app-model-catalog.md`, `docs/codex-path-investigation.md` | devlog links only | delete; devlog history links go stale by design | +| `docs/qoder-cli-provider.md` | devlog only | delete; covered by `docs-site/.../guides/providers.md` "Official Qoder CLI"; #3010 credit already sits in the carry trailer (`devlog/_fin/260908_provider_runtime_stack/050_delivery_record.md:11`) | +| `docs/shadow-call-intercept.md` | devlog only | delete; covered by `docs-site/.../reference/configuration/server.md` "Shadow calls" | +| `docs/github-copilot-app.md` | devlog only | port to `docs-site/.../guides/integrations.md` (not covered anywhere in docs-site) | +| `docs/pr-assets/**`, `docs/screenshots/**` | none | delete | +| `.github/pr-assets/**` (31), `assets/pr-screenshots/**` (6), `docs-site/public/pr-screenshots/**` (20), `assets/pr2950-capacity-expiry.png`, `assets/pr715-selection-order.png`, `assets/request-pacing-dashboard.jpg`, `assets/zh-tw-providers.png`, `assets/pr-gate-screenshot-required.png` | none (`rg -F -f` over all 62 basenames, excluding `devlog/` and `docs/`) | delete | + +Other `docs/` strings in the tree are unrelated: synthetic paths in `.github/scripts/pr-hygiene.test.cjs`, +`.github/scripts/issue-quality.test.cjs:148` and `tests/ci-workflows/privacy-scan-meta-key.test.ts:50`, +upstream vendor paths in `src/adapters/*` comments, and `gui/public/provider-icons/README.md`. + +## Replacement workflow + +An orphan branch `pr-assets` on `lidge-jun/opencodex` holds PR evidence images. It shares no history +with `dev`, so nothing on it can reach a squash merge. Authors link images by commit SHA +(`https://raw.githubusercontent.com/lidge-jun/opencodex/<sha>/<path>`), which keeps the link stable. +A branch ruleset blocks deletion and force-push so pinned SHAs stay reachable. Contributors without +push access use GitHub's drag-and-drop attachment, which the `enforce-target` message already suggests +(`.github/scripts/pr-quality-messages.cjs:234`). + +CI impact: every workflow `push:` trigger is pinned to `main`, `preview` or `dev` (`ci.yml`, +`issue-quality-tests.yml`, `deploy-docs.yml`, `react-doctor.yml`, `cleanup-orphaned-workflows.yml`, +`service-lifecycle.yml`); `pr-hygiene.yml` is `pull_request_target` only. A push to `pr-assets` +triggers nothing, so no workflow edit is needed. + +## Constraints + +- `.github/PULL_REQUEST_TEMPLATE.md` stays byte-identical: `PR_TEMPLATE_BOILERPLATE_LINES` in + `.github/scripts/pr-quality.cjs` matches its lines literally. +- `.gitignore` entries are root-anchored. A bare `docs/` would ignore `docs-site/src/content/docs/`. +- No file-size cap changes (`tests/fixtures/file-size-baseline.json`); none of the edited files is capped. +- Security scratch rule untouched; no security content moves. + +## Out of scope + +Rewriting devlog links, CI gate semantics, and translated docs-site pages other than the one +"Structure SOT" contributing bullet (D9 in `010`). diff --git a/devlog/_plan/260923_docs_retirement/010_phase1_docs_retirement.md b/devlog/_plan/260923_docs_retirement/010_phase1_docs_retirement.md new file mode 100644 index 0000000000..20e0dc11a7 --- /dev/null +++ b/devlog/_plan/260923_docs_retirement/010_phase1_docs_retirement.md @@ -0,0 +1,107 @@ +# 010 — Phase 1: retire `docs/`, seed `pr-assets`, add guards + +Diff-level change map. Paths are relative to the repository root. + +## DELETE + +- `docs/` entirely, except the files moved below (`git rm -r docs`). +- `.github/pr-assets/`, `assets/pr-screenshots/`, `docs-site/public/pr-screenshots/`. +- `assets/pr2950-capacity-expiry.png`, `assets/pr715-selection-order.png`, `assets/request-pacing-dashboard.jpg`, + `assets/zh-tw-providers.png`, `assets/pr-gate-screenshot-required.png`. + +## MOVE (git mv, then edit links) + +- `docs/design-system/{README,foundations,components,contributing}.md` → `gui/design-system/`. +- `docs/adr/0004-gui-toggle-contrast-and-nav-spacing.md`, `docs/adr/0005-gui-design-token-system.md` + → `gui/design-system/decisions/`. +- `gui/design-system/README.md`: source tree block `docs/design-system/` → `gui/design-system/`; + ADR links `../adr/000N-…` → `./decisions/000N-…`. +- Any `../../gui/…` style relative link inside the moved files is re-resolved against the new location. + +## MODIFY + +- `gui/AGENTS.md`: one bullet pointing to `gui/design-system/` as the token/component contract. +- `CONTRIBUTING.md:11`: replace the `docs/` bullet with the PR screenshot hosting rule (drag-and-drop + attachment first; maintainers with push access may use the `pr-assets` branch; never commit + evidence images on the PR branch). AGENTS.md and docs-site use the same order. +- `AGENTS.md` "Issues and pull requests (agents)": after the screenshot sentence, add where the + image goes (`pr-assets` branch, SHA-pinned raw URL) and that PR branches must not add evidence images. +- `docs-site/src/content/docs/contributing.md:157`: same hosting rule, one sentence. +- `docs-site/AGENTS.md:8` (A-phase blocker 1, Planck): "historical `docs/` or `devlog/` material" → + "`devlog/` notes or older revisions in git history". +- D9 (architect addition): the "Structure SOT" bullet in `docs-site/src/content/docs/contributing.md:180` + and its seven translations (`tr:193`, `fr:169`, `ko:131`, `ja:132`, `ru:133`, `zh-tw:138`, `zh-cn:120`) + sends historical notes to `docs/`. Each becomes "`devlog/`" (the tracked home for investigation and + planning notes), translated in place. +- `docs-site/src/content/docs/guides/integrations.md`: new `## GitHub Copilot App` section at the end, + ported from `docs/github-copilot-app.md` (manual setup; not an Integrations-tab switch). +- `structure/ops/docs-and-release.md` "Historical docs": state that `docs/` is retired, where each kind + of material now lives, and the `pr-assets` branch. +- `scripts/structure-ssot.ts:223`: INDEX header drops the `docs/` clause; then `bun run structure:index` + regenerates `structure/INDEX.md`. +- `structure/manifest.json` `absentPaths`: add `{ "path": "docs/", "reason": … }` next to `go/`. +- `scripts/privacy-scan.ts:151`: drop the `docs/` username allowance (no file there any more). +- `.github/ISSUE_TEMPLATE/documentation.yml:31`: placeholder `docs/providers.md` → + `docs-site/src/content/docs/guides/providers.md`. +- `tests/ci-workflows/repo-hygiene.test.ts`: `RETIRED_TRACKED_DIRS` gains `docs`, `.github/pr-assets`, + `assets/pr-screenshots`, `docs-site/public/pr-screenshots`, with a comment naming the cause. +- `.gitignore`: root-anchored `/docs/`, `/.github/pr-assets/`, `/assets/pr-screenshots/`, + `/docs-site/public/pr-screenshots/` (the gitignore assertion in the same test requires them). + +## Architect dispositions (Dalton, gpt-6-sol high) + +- D1, D3, D5 ACCEPT. D5 audit done: 268 PRs mention these image paths; none links them through a + `dev`, `main` or `preview` ref, so deleting them from `dev` breaks no PR description. +- D2 AMEND accepted: ADR 0005 keeps its historical text (it names `docs/design-system` as of its date); + the moved README is the current pointer. +- D4 AMEND accepted: the ported section says it is a client setup, separate from the upstream + `github-copilot` provider, and its auth/field claims are rechecked against `src/server/chat-completions.ts` + and `src/server/auth-cors.ts` before publishing. +- D6 AMEND accepted (INDEX is regenerated, never hand-edited). +- D7 AMEND accepted after reflection: the four directories join `RETIRED_TRACKED_DIRS`; the five loose + images get a separate `RETIRED_TRACKED_FILES` assertion in the same test (no gitignore line, since + the directory mechanism's `<dir>/` gitignore check does not fit single files). +- D8 AMEND accepted: ruleset is created and verified before any doc tells authors to pin SHAs; the + drag-and-drop attachment is presented first; the PR template stays byte-identical. +- D9 ADD accepted (above). + +## C-phase verifier findings + +- Content verifier: ADR 0007 (CLI parity), ADR 0002 (doctor proxy-env disclosure) and the CL-10 + closure contracts had no `structure/` home; carried into `structure/ops/docs-and-release.md`, + `structure/config.md` and `structure/adapters/compatibility-lab.md`. ADR 0006 was already covered + (`structure/config.md` provider output defaults, `structure/transports/streaming-health.md` replay). +- Follow-up outside this unit: `docs-site/.../reference/configuration/server.md` "Remote access" table + (English and seven translations) still says `/v1/responses` and `/v1/chat/completions` reject Bearer + admission. `src/server/auth-cors.ts` `AUTH_MATRIX` and `reference/proxy-formats.md` accept it since + #1686. The new Copilot guide links the correct matrix; the stale table predates this unit. + +## Remote (outside the PR diff) + +1. Orphan branch `pr-assets` with one `README.md` explaining layout (`<pr-number-or-slug>/<file>`), + SHA-pinned linking, and that the branch is append-only. Pushed from a temporary clone so this + worktree's HEAD never moves. +2. Branch ruleset "Protect pr-assets" on `refs/heads/pr-assets`: `deletion`, `non_fast_forward`, + enforcement active. Verified with `gh api repos/lidge-jun/opencodex/rulesets`. + +## Enforcement and bypass (PLAN-BYPASS-NAMED-01) + +- Tier: CI test (`repo-hygiene`) plus `structure:check` `absentPaths`. Executing surface: hosted CI. +- Known bypass: images committed under any other path (for example `gui/public/`), or a maintainer + merging with red CI. Residual risk: re-accumulation elsewhere; review catches it. +- Wording: called a guard for these paths, not a general image ban. +- Ruleset: repository admins can still edit or disable the ruleset. + +## Acceptance and verifiers (PLAN-VERIFIER-REAL-01) + +| Criterion | Command | Reads the target | +| --- | --- | --- | +| `docs/` untracked | `git ls-files docs \| wc -l` → 0 | yes, index | +| structure gate | `bun run structure:check` | yes: manifest `absentPaths`, INDEX, ops doc | +| privacy | `bun run privacy:scan` and `bun test tests/ci-workflows/privacy-scan*.test.ts` | yes: `scripts/privacy-scan.ts` | +| guards | `bun test tests/ci-workflows/repo-hygiene.test.ts tests/ci-workflows/structure-ssot.test.ts` | yes | +| issue template | `node --test .github/scripts/issue-quality*.test.cjs` | yes, template read by tests | +| typecheck | `bun run typecheck` | yes, `scripts/*.ts` | +| docs-site | `cd docs-site && bun run build` | yes, `integrations.md`, `contributing.md` | +| import graph | `bun run test:changed` | partial; source-read tests listed above run explicitly | +| guard activation | stage a throwaway `docs/x.md` with `git add -f`, run repo-hygiene → red, unstage | yes | diff --git a/devlog/_plan/260923_gpt6_catalog_lanes/000_plan.md b/devlog/_plan/260923_gpt6_catalog_lanes/000_plan.md new file mode 100644 index 0000000000..ca1a8fcba5 --- /dev/null +++ b/devlog/_plan/260923_gpt6_catalog_lanes/000_plan.md @@ -0,0 +1,36 @@ +# 260923 GPT-6 catalog lanes — plan + +## Objective + +Two independent pull requests against `dev`: + +- **PR 1** (`codex/upstream-catalog-resync`): re-pin `src/codex/data/upstream-models.json` to openai/codex main `6cfe29984` and add an opt-in flag that admits unknown bare native models from the authenticated ChatGPT Codex `/models` roster. +- **PR 2** (`codex/gpt6-sol-luna-astra-minor`): register `gpt-6-sol`, `gpt-6-luna` and `gpt-6-astra-minor`. + +## Evidence (2026-09-23) + +- openai/codex `deb0a08f2` (#47085, 2026-09-21) describes GPT-5.6-Sol as "Reliable agentic workhorse for everyday tasks." and moves its priority 6 → 4. `cf6754e68` (#47130) removes `ultrafast` from Sol. `eb7bd64ef` (#44250) removed the retired `gpt-5.4-mini` / `gpt-5.2` rows. `gpt-daybreak-blue-latest` / `gpt-daybreak-red-latest` rows now ship in the bundled catalog. +- openai/codex contains no `gpt-6-sol`, `gpt-6-luna` or `gpt-6-astra-minor` at `6cfe29984`. +- Live roster probe (`https://chatgpt.com/backend-api/codex/models`, main + 5 pool accounts): `gpt-6-sol` (priority 2) and `gpt-6-luna` (priority 3) are served as full native rows only when `client_version >= 0.155.0`; `0.154.0` returns `gpt-6-astra` only. Rows claim `minimal_client_version: "0.153.0"`. Sol is absent on 2 of 6 accounts; Luna is on all; `gpt-6-astra-minor` is on none. Installed Codex is 0.154.0; npm latest is 0.155.1. +- OpenAI announced GPT-6 Sol and Luna on 2026-09-22 (https://openai.com/index/introducing-gpt-6-sol-and-luna/, API changelog Sep 22). `gpt-6-astra-minor` appeared only in an Azure AI Playground config snapshot (pl4nty/data `2b6a2351`, 2026-09-22T04:34Z), was withdrawn by ~14:56Z and has no OpenAI doc page (404). + +## Constraints + +- File-size ratchet: `tests/codex-integration/codex-catalog.test.ts` is at its 7985-line cap; new cases go into sibling files registered in `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`. +- PR 1 rewrites `upstream-models.json` wholesale, so PR 2 keeps its rows in a sibling data file to stay conflict-free. +- Flagship natives are listed unconditionally (owner decision 2026-09-04, `structure/providers/openai-tiers.md`); only unconfirmed ids are account-gated. +- Per user steering on 2026-09-23: no local test runs; push with `--no-verify`; hosted CI is the verification of record. `bun run typecheck` only. + +## Phase map (dependency order) + +| Work-phase | Doc | Branch | Depends on | +|---|---|---|---| +| wp0 roadmap | this unit | — | — | +| wp1 snapshot re-pin | `010_phase1_upstream_repin.md` | PR 1 | wp0 | +| wp2 opt-in roster admission | `020_phase2_roster_admission.md` | PR 1 | wp1 | +| wp3 GPT-6 rows | `030_phase3_gpt6_rows.md` | PR 2 | wp0 | + +## SoT sync + +`structure/catalog.md` (snapshot and admission), `structure/config.md` (new flag), `structure/providers/openai-tiers.md` (flagship roster), `docs-site/src/content/docs/guides/codex-app-models.md`, `docs-site/src/content/docs/reference/configuration/providers.md`. + diff --git a/devlog/_plan/260923_gpt6_catalog_lanes/010_phase1_upstream_repin.md b/devlog/_plan/260923_gpt6_catalog_lanes/010_phase1_upstream_repin.md new file mode 100644 index 0000000000..1c416a1c19 --- /dev/null +++ b/devlog/_plan/260923_gpt6_catalog_lanes/010_phase1_upstream_repin.md @@ -0,0 +1,50 @@ +# 010 Phase 1 — re-pin the upstream snapshot + +## Change map + +| Path | Action | +|---|---| +| `src/codex/data/upstream-models.json` | MODIFY: byte-for-byte copy of openai/codex `6cfe29984` `codex-rs/models-manager/models.json` | +| `src/codex/catalog/metadata.ts` | MODIFY: `PINNED_UPSTREAM_MODELS` maps every row through `withDerivedBaseInstructions` | +| `tests/codex-integration/codex-catalog.test.ts` | MODIFY in place (net 0 lines, file is at cap) | +| `tests/codex-integration/reserve-catalog-lifecycle.test.ts` | MODIFY: fixture derives `base_instructions` | +| `tests/codex-integration/codex-model-entitlements.test.ts` | MODIFY: Daybreak Blue now has a pinned row | +| `structure/catalog.md`, `docs-site/src/content/docs/guides/codex-app-models.md` | MODIFY: snapshot facts | + +## Why metadata.ts changes + +Upstream #43604 stopped shipping top-level `base_instructions`; rows keep `model_messages.instructions_template`. `hasNativeCatalogRowShape` (`metadata.ts:677`) requires `base_instructions`, and the alias branch of `upstreamNativeEntryForSlug` only rewrites `base_instructions` when present. `withDerivedBaseInstructions` already fills it from the template for `gpt-6-astra`; apply it to the whole map so the pinned JSON stays byte-identical to upstream. + +```diff + const PINNED_UPSTREAM_MODELS: Map<string, RawEntry> = new Map( + ((upstreamModelsSnapshot as unknown as { models?: RawEntry[] }).models ?? []) +- .flatMap(model => typeof model.slug === "string" ? [[model.slug, model] as const] : []), ++ .flatMap(model => typeof model.slug === "string" ? [[model.slug, withDerivedBaseInstructions(model)] as const] : []), + ); +``` + +## Test repairs (measured in a scratch worktree before this plan) + +Replacing only the JSON turns 9 tests red beyond the 30 environment-only `codex-cooldown-recovery` failures that also fail on the base (worktree under `~/.codex` trips the test home guard): + +| Test | Line | Old → new | +|---|---|---| +| gpt-5.6 natives come from the pinned upstream snapshot | `codex-catalog.test.ts:3563`, `:4272` | Sol description → "Reliable agentic workhorse for everyday tasks." | +| Daybreak Blue inherits Sol capabilities | `codex-catalog.test.ts:3757` | 372_000/372_000 → 272_000/872_000 | +| configured ChatGPT-forward Daybreak | `codex-catalog.test.ts:3950` | fixed by the metadata.ts derivation | +| catalog sync upgrades fallback-quality gpt-5.6 entries | `codex-catalog.test.ts:4252` | Luna priority 3 → 8 | +| reserve lifecycle x3 | `reserve-catalog-lifecycle.test.ts:39` | fixture clones Luna and adds `base_instructions` from `model_messages.instructions_template` | +| ungating the 5.6 family empties the derivation | `codex-model-entitlements.test.ts:1674` | derivation now returns `"0.142.2"` from the shipped Daybreak Blue row; composed floor stays `"0.144.0"` | + +## Out of scope + +## Audit round 1 dispositions (reviewer Peirce, grok-4.7, VERDICT FAIL) + +- B1 blast radius — rebutted with measurement: the scratch run replaced only the JSON and ran the eight touching files (`.tmp/resync-probe-full.log`: 470 pass / 39 fail; base `.tmp/resync-base-full.log`: 479 pass / 30 fail, the same 30 cooldown-recovery environment failures). `codex-catalog.test.ts:3497`, `:3524` and `:3865` passed in that run. Folded the part that is right: the comments that describe the old pin are updated — `native-models.ts:74-79` (the pin no longer holds `gpt-5.2` / `gpt-5.4-mini`), `model-entitlements.ts:93-111` and `:123-129` (the snapshot now records 0.144.0 for the 5.6 family and 0.142.2 for Daybreak Blue). Raw-row assertion at `:3757` is kept as a raw-row assertion (272_000/872_000) with its comment corrected; the effective 922_000 override is asserted separately at `:3939` and is unchanged. +- Capped file: every `codex-catalog.test.ts` edit is an in-place value swap, net 0 lines. + +Daybreak Blue stays a capability alias of Sol (its pinned row is data only); making it self-described would drop the Fast tier and is a separate decision. + +## Verification + +`bun run typecheck` (exit 0). Tests: NOT RUN locally per user steering; hosted CI on the PR head. diff --git a/devlog/_plan/260923_gpt6_catalog_lanes/020_phase2_roster_admission.md b/devlog/_plan/260923_gpt6_catalog_lanes/020_phase2_roster_admission.md new file mode 100644 index 0000000000..2b5a21b4d4 --- /dev/null +++ b/devlog/_plan/260923_gpt6_catalog_lanes/020_phase2_roster_admission.md @@ -0,0 +1,35 @@ +# 020 Phase 2 — opt-in admission of unlisted native models + +> Decision 2026-09-23 (user, async answer): PR 1 ships the re-pin only; opt-in roster admission moves to a separate follow-up PR. This document is the design input for that PR, including the round-1 audit findings below. + +> Status after audit round 1: BLOCKED ON USER DECISION. Reviewer B2 showed the admission has to reach both catalog writers (`retained-sync.ts` and `convergence.ts:268`) and six allowlists (`build-entries.ts:385`, `:674-681`, `:700-703`, `retained-sync.ts:326`, `metadata.ts:510` `applyNativeVisibility`, `subagent-roster.ts:102`), and `parseAccountModels` (`model-entitlements.ts:632`) must keep bodies. B3: the new module must take a structural `Record<string, unknown>` and import neither `parsing.ts` nor `metadata.ts` (cycle through `parsing.ts:34`). With local tests forbidden this is not shippable in PR 1 without tests being run somewhere; the user was asked whether to move it to a follow-up PR. + +## Behaviour + +New config field `codexAdmitUnlistedNativeModels?: boolean` (absent or malformed = off). When on, the authenticated main-account roster fetch keeps full rows for bare ids that are `gpt-*`/`o1-`/`o3-`/`o4-`, not in `SUPPORTED_NATIVE_OPENAI_SLUGS`, not in `RETIRED_NATIVE_OPENAI_MODELS`, `supported_in_api === true`, `visibility === "list"` and pass `hasNativeCatalogRowShape`. Catalog sync then emits those rows as bare picker rows. The roster is asked at the same client version as today (installed runtime raised to the gated floor), so a row appears only once the installed Codex can be served it; nothing raises the floor. + +## Change map + +| Path | Action | +|---|---| +| `src/codex/catalog/unlisted-natives.ts` | NEW: `hasNativeCatalogRowShape` (moved from metadata.ts, re-exported), `admissibleUnlistedNativeRow(row)`, main-account row store with `recordUnlistedNativeRows(accountId, clientVersion, rows)` / `unlistedNativeRowsForCatalog()` / reset for tests | +| `src/codex/model-entitlements.ts` | MODIFY: `parseAccountModels` also returns candidate rows; `fetchAccountModels` records them for `MAIN_CODEX_ACCOUNT_ID` on a confirmed roster | +| `src/codex/catalog/retained-sync.ts` | MODIFY: when the flag is on, bare rows from the store join the catalog with `visibility: "list"` and their slugs join `observedNativeSlugs` | +| `src/codex/catalog/metadata.ts` | MODIFY: import `hasNativeCatalogRowShape` from the new module | +| `src/types/config.ts`, `src/config/schema/config-schema.ts`, `src/config/feature-flags.ts`, `src/config/diagnostics.ts` | MODIFY: field doc, `z.boolean().optional().catch(false)`, `admitUnlistedNativeModelsEnabled()`, own-boolean validation | +| `tests/codex-integration/unlisted-native-admission.test.ts` | NEW, registered in layout.json + test-layout-expected.json | +| `structure/config.md`, `structure/catalog.md`, `docs-site/.../reference/configuration/providers.md` | MODIFY | + +## Field chain (PLAN-FIELD-CHAIN-01) + +creation: config.json / PUT /api/settings passthrough → schema `.catch(false)` → consumer `admitUnlistedNativeModelsEnabled(config)` in retained-sync; serialization N/A (not written by code); no GUI control (out of scope). + +## Activation scenarios + +- Flag off: store may hold rows, catalog emits none (test). +- Flag on + roster row `gpt-future-x` with full shape: bare row listed (test). +- Retired `gpt-5.4`, hidden row, row without `base_instructions`, `supported_in_api:false`: refused (test). + +## Limits + +`/v1/models` and dashboard rows are unchanged in this phase; the Codex picker is the target surface. diff --git a/devlog/_plan/260923_gpt6_catalog_lanes/030_phase3_gpt6_rows.md b/devlog/_plan/260923_gpt6_catalog_lanes/030_phase3_gpt6_rows.md new file mode 100644 index 0000000000..96e4a4029d --- /dev/null +++ b/devlog/_plan/260923_gpt6_catalog_lanes/030_phase3_gpt6_rows.md @@ -0,0 +1,29 @@ +# 030 Phase 3 — gpt-6-sol, gpt-6-luna, gpt-6-astra-minor (PR 2) + +Branch `codex/gpt6-sol-luna-astra-minor` from `origin/dev`; independent of PR 1. + +## Decisions + +- D1 `gpt-6-sol` and `gpt-6-luna` are SELF-DESCRIBED natives. Their rows are the live-roster rows captured on 2026-09-23 (client_version 0.155.0), stored verbatim in a NEW sibling file `src/codex/data/roster-pinned-models.json` so PR 1's wholesale re-pin of `upstream-models.json` does not conflict. +- D2 Both are listed unconditionally like `gpt-6-astra` (flagship owner decision). They are not account-gated. +- D3 `gpt-6-astra-minor` is an ACCOUNT-GATED capability alias of `gpt-6-astra` with hand-written presentation (display "GPT-6-Astra-Minor"). It stays hidden and request-refused until an authenticated roster lists it. +- D4 No API-key registry rows or pricing in this PR. +- D5 (audit B4) `isGpt56NativeSlug`/`ensureGpt56ReasoningLevels` (`effort.ts:281-287`) grant the full ladder only when the slug's own or source pinned row ships `ultra`; Luna keeps low..max. `finishUpstreamNativeEntry` (`derive-entry.ts:35`) and the sync branch (`build-entries.ts:688`) inherit this through the predicate. +- D6 (audit B4, rebutted in part) The first-five spawn roster is `config.subagentModels` (`src/config/subagent-models.ts:8`, migrated once); adding natives does not reorder it, so `DEFAULT_SUBAGENT_MODELS` stays unchanged in this PR. Users who want GPT-6 Sol/Luna as subagents pick them in the dashboard. + +## Change map + +| Path | Action | +|---|---| +| `src/codex/data/roster-pinned-models.json` | NEW: `{ "source": ..., "models": [gpt-6-sol row, gpt-6-luna row] }` copied from `.tmp/gpt6-roster-rows.json` | +| `src/codex/catalog/pinned-models.ts` | NEW: `pinnedNativeModelRows()` = upstream snapshot rows followed by roster rows whose slug the snapshot lacks | +| `src/codex/catalog/native-models.ts` | MODIFY: constants `NATIVE_GPT6_SOL_MODEL`, `NATIVE_GPT6_LUNA_MODEL`, `NATIVE_GPT6_ASTRA_MINOR_MODEL`; add all three to `NATIVE_OPENAI_MODELS`; minor to `ACCOUNT_GATED_NATIVE_OPENAI_MODELS`; sol+luna to `SELF_DESCRIBED_NATIVE_OPENAI_MODELS`; minor → astra in `NATIVE_OPENAI_CAPABILITY_SOURCES` + presentation; sol+luna to `NATIVE_MAIN_DRAIN_SENTINEL_MODELS` | +| `src/codex/catalog/metadata.ts` | MODIFY: `PINNED_UPSTREAM_MODELS` built from `pinnedNativeModelRows()`; sol+luna in `DOCUMENTED_NATIVE_OPENAI_ADDITIONS`; context overrides 272_000/872_000/872_000 for sol, luna, minor | +| `src/codex/catalog/effort.ts` | MODIFY: `isGpt56NativeSlug` full ladder only when the source row ships `ultra` — Luna ships low..max | +| `src/codex/model-entitlements.ts` | MODIFY: floor derivation reads `pinnedNativeModelRows()` | +| `tests/codex-integration/gpt6-native-rows.test.ts` | NEW (layout registered) | +| `structure/catalog.md`, `structure/providers/openai-tiers.md`, `docs-site/.../reference/configuration/providers.md` | MODIFY | + +## Verification + +`bun run typecheck` only; tests NOT RUN locally per user steering; hosted CI on the PR head. diff --git a/devlog/_plan/260923_grok47_parity/000_plan.md b/devlog/_plan/260923_grok47_parity/000_plan.md new file mode 100644 index 0000000000..36e98e1cdb --- /dev/null +++ b/devlog/_plan/260923_grok47_parity/000_plan.md @@ -0,0 +1,122 @@ +# 260923 grok-4.7 parity — plan + +grok-4.7 shipped on 2026-09-21 and already answers through xAI, Devin, Command Code and Cursor, but OpenCodex has no +registry entry for it: the picker shows it without a context window, reasoning ladder, image input, Fast row or +Responses wire, and cost estimates are unavailable. This unit gives grok-4.7 the same declarations grok-4.6 carries, +using values measured with real grok-4.7 calls (010_probe-evidence.md) and xAI's published model page, and applies +them to the other providers that serve it where their evidence supports each declaration. + +## Loop spec + +- Archetype: satisfy-spec, single work-phase (wp1), one PABCD cycle, one PR to dev. +- Trigger: user request 2026-09-23 "grok-4.7 모델 피커 컨텍스트 fast 와이어를 실제 토큰응답으로 ... grok-4.6과 같이 패치하고 다른 프로바이더들에도 적용하는 pr". +- Goal: grok-4.7 has grok-4.6-equivalent picker/context/effort/image/Fast/wire/price metadata on xAI, plus Devin, + Command Code, Cursor, OpenCode Go and bundled gateway metadata where evidenced. +- Non-goals: changing default or sidecar models (web-search defaults stay grok-4.6); grok-4.7-build-fast; GitHub + Copilot wire pin and OpenCode Go hosted web_search strip for 4.7 (no probe possible, not configured locally); + merge, release, service restart. The user forbade local tests: no bun test, typecheck or build runs locally. +- Verifier: static only locally — `git diff --check`, JSON parse of edited JSON, `rg` roster consistency (every + grok-4.6 xAI-block key has a grok-4.7 sibling), byte-equality of regenerated metadata via the generator script + (codegen, not a test); then exact-head hosted CI on the PR (typecheck + 4 test shards + file-size + layout). +- Stop: PR open, independent review has no unresolved blocker, exact-head CI reported. +- Memory artifact: this unit (000/010), goalplan add-grok-4-7-to-opencodex-with-the-same-first-cl. +- Expected terminal outcomes: DONE (PR open, CI green or failures fixed); BLOCKED if push refused. +- Escalation: a CI failure that needs a local run to diagnose, or a design dispute that requires changing a default. +- HOTL bounds: tools = repo edits, gh, live proxy probes already done; write scope = files listed below; no token or + time budget was set by the user. + +## Architect consultation + +Handle 01a0caad-1195-74b3-8338-e7a354313b5d (Banach, gpt-6-sol). Proposal D1–D8. Dispositions: + +- D1 accept (xAI declarations, grok-4.7 ahead of 4.6 in XAI_MODELS). +- D2 amended in revision 3 (see Audit round 1 synthesis): toggle set unchanged; 4.7 gets the OAuth Responses default + through modelWireDefaults only. +- D3 accept (Devin roster, 500k, measured low..max ladder, default medium). +- D4 accept: add xai/grok-4.7 and xai/grok-4.6 to COMMAND_CODE_IMAGE_MODELS; the 4.6 negative is contradicted by the + same two-path grid probe the header demands. +- D5 amend: OpenCode Go wire/efforts/default ARE mirrored — opencode.ai/docs/go lists "Grok 4.7 grok-4.7 + https://opencode.ai/zen/go/v1/responses @ai-sdk/openai", the same documented evidence the 4.6 pin (#3394) used. + The Go web_search strip and the Copilot Responses pin stay 4.6-only (unprobed; recorded as follow-ups). +- D6 amend: Cursor's live GetUsableModels roster (explorer 01a0caad-9fb2-70a1-9cf5-120ad392775f) lists + grok-4.7-{low,medium,high,xhigh} and the same with -fast, no cursor- prefix, no max. Mirror with no wirePrefix and + keep the prefix condition 4.5/4.6-only. Context: Cursor's API reports no window; 4.6's 500k is likewise the model's + published window, so 4.7 gets 500k from xAI's page and the measured xAI limit. +- D7 accept for xAI prices; OpenRouter's distinct prices arrive through the regenerated bundled metadata rather than + a new overlay (4.6 has no OpenRouter overlay either). Devin-cli gets a derived row only if DEVIN_GROK equals xAI's + list price, labeled derived like the GPT-6 rows. +- D8 accept. + +Reflection: see "Reflection" below. + +## File change map (dependency order) + +1. src/providers/registry/model-seeds.ts — XAI_MODELS: insert "grok-4.7" before "grok-4.6". + COMMAND_CODE_IMAGE_MODELS: add "xai/grok-4.6" and "xai/grok-4.7" with the probe note; drop xai/grok-4.6 from the + verified-negative header list (both mentions). +2. src/providers/registry/entries-core.ts, xai block: modelSupportsServiceTier, modelWireDefaults (oauth, responses + inbound), modelInputModalities, preserveReasoningContentModels, modelReasoningEfforts [low..xhigh], + modelDefaultReasoningEfforts high, modelContextWindows 500_000 — each with a grok-4.7 sibling of 4.6; comments cite + devlog/_plan/260923_grok47_parity/010_probe-evidence.md. Devin block: add "grok-4-7" after "grok-4-6" in models. + OpenCode Go block: modelWireDefaults, modelReasoningEfforts, modelDefaultReasoningEfforts for grok-4.7. +3. src/adapters/devin/live-models.ts — DEVIN_MODEL_CONTEXT_WINDOWS "grok-4-7": 500_000; DEVIN_MODEL_EFFORTS (if it + has per-model entries) "grok-4-7": low..max, default medium if a default map exists. +4. (removed in revision 3 — see Audit round 1 synthesis; xai-responses-opt-in.ts is unchanged) +5. src/usage/expected-prices.ts — xai grok-4.7 base {2,6,0.5,0}, priority 2x rule list gains grok-4.7, >=200k + UNIFORM_DOUBLE row with confirmedPriorityRelation lower-bound; devin-cli grok-4-7 conditional (D7). +6. src/adapters/cursor/{catalog.ts,effort-map.ts,discovery.ts} — "grok-4.7" capability (displayName "Cursor Grok + 4.7", CONTEXT_500K, no wirePrefix, regular+fast low..xhigh), tiers for "grok-4.7" and "grok-4.7-fast", heuristic + window 500_000 for grok-4.7 ids. Verify the Fast path emits flattened grok-4.7-<effort>-fast (accepted live) and + not the bare grok-4.7-fast (not_found live). +7. scripts/model-metadata.source.json + src/generated/model-metadata.ts — add grok-4.7 rows beside each existing + grok-4.6 row for providers whose current models.dev entry lists 4.7 (xai, opencode-go, opencode, openrouter, kilo, + vercel, zenmux if present), copying that provider's live models.dev record; regenerate with + scripts/generate-model-metadata.ts. +8. Tests (hosted CI runs them): tests/providers/provider-registry-parity.test.ts:1259 default-effort map; + tests/service/service-tier-capability.test.ts:125; tests/usage/usage-cost.test.ts:939; an xAI + wire-default case for 4.7 (OAuth Responses inbound resolves openai-responses; explicit modelAdapters Chat wins); command-code vision assertion + (tests/providers/command-code-provider.test.ts:219 flips 4.6 to image-capable); cursor effort/Fast wire-id cases for + 4.7; opencode-go Responses wire case for 4.7. codex-catalog.test.ts is at its cap: no edits there. +9. Docs: docs-site guides/codex-app-models.md model table (+ locales), reference/configuration/providers.md xAI Responses + default note if it lists models (the opt-in toggle list stays 4.5/4.6); structure/providers/xai-grok.md Fast set, structure/transports/responses.md:418, + structure/providers/cursor.md grok row. + +## Acceptance + +- A1 every xai-block map that names grok-4.6 also names grok-4.7 with the measured value (rg check). +- A2 Cursor 4.7 wire ids equal the live roster: regular grok-4.7-<e>, Fast grok-4.7-<e>-fast, no cursor- prefix. +- A3 (removed in revision 3): toggle set unchanged; xai-transport and management toggle tests stay as they are. +- A4 generated metadata byte-matches the generator output (codegen run + model-metadata-sync test in CI). +- A5 hosted CI green at the PR head, or failures diagnosed and fixed. + + +## Reflection + +Architect 01a0caad-1195-74b3-8338-e7a354313b5d on revision 1: MISALIGNED (narrow), D1–D8 all mapped. Gaps and +dispositions: + +- OpenRouter >=200k band rule: rebutted. `src/usage/expected-prices.ts` carries no OpenRouter context tier for any + model, including grok-4.6 whose OpenRouter entry publishes the same kind of override band. Adding one only for 4.7 + would create a new, inconsistent pattern; it belongs in a separate change that covers OpenRouter tiers as a whole. + Recorded as a follow-up. +- Broken evidence pointer (010 -> 010_plan.md): fixed to 000_plan.md, and the Cursor Fast success / bare-id + rejection recorded in 010_probe-evidence.md. +- Missing Reflection section: this section (revision 2). + + +## Audit round 1 synthesis (revision 3) + +Reviewer 01a0cab4-38a2-7ec0-8de2-02c1982a4345: FAIL, 2 High, both caused by D2 (4.7 joining the Responses toggle). + +Root cause: `XAI_RESPONSES_OPT_IN_MODELS` is the scope of a legacy compatibility switch (dashboard copy +"Grok 4.5 and 4.6", management write path provider-routes.ts:449, v1 migration). grok-4.6's actual Responses +default comes from `modelWireDefaults` (entries-core.ts:264), and that is the declaration parity requires. + +D2 amended (before -> after): before, 4.7 joins the toggle set and the migration gets a separate legacy list; after, +the toggle set, its migration, the GUI copy and the management tests stay unchanged, and 4.7 receives the same OAuth +Responses default through `modelWireDefaults` only. A user who wants 4.7 on Chat sets +`modelAdapters["grok-4.7"]="openai-chat"`, which always wins (the same escape hatch the Go/Copilot pins document). +Consequences: blocker 1 (xai-transport.test.ts:85, management-provider-validation.test.ts:4102/4123) and blocker 2 +(gui/src/i18n copy, ProviderAuthPanel mixed state) no longer arise; plan step 4 is removed, and so is acceptance A3. +Non-blocking note folded: DEVIN_STATIC_MODELS (src/adapters/devin/live-models.ts:18) gains "grok-4-7". +The proposed "grok-4-6" fallback entry was removed because its Devin-specific ladder was not measured. diff --git a/devlog/_plan/260923_grok47_parity/010_probe-evidence.md b/devlog/_plan/260923_grok47_parity/010_probe-evidence.md new file mode 100644 index 0000000000..86340dad25 --- /dev/null +++ b/devlog/_plan/260923_grok47_parity/010_probe-evidence.md @@ -0,0 +1,51 @@ +# Live probe evidence — grok-4.7 (2026-09-23 KST) + +Mechanics: POST /v1/responses on the running proxy (127.0.0.1:10100, ocx 2.62.0), then `ocx logs --json` for the +matching attempt (adapter, credentialSource, reasoningWireField/Value, tierOutcome, usage). xAI traffic used the Grok +OAuth lane (credentialSource "grok-oauth"); no API key was involved. The Responses-wire and `--fast` rows used a +temporary `providers.xai.modelAdapters["grok-4.7"]="openai-responses"` plus +`modelSupportsServiceTier["grok-4.7"]=true` override, applied through the attested provider reload +(`notifyRunningProxy("xai")`, the path `ocx login` uses) and removed the same way afterwards; the restored maps were +compared against a pre-probe backup. Scratch scripts lived in the gitignored `.tmp/`. + +## xAI grok-4.7 (Grok OAuth) + +| Probe | Chat wire (provider default) | Responses wire (temporary override) | +|---|---|---| +| effort low / medium / high / xhigh | 200, sent as `reasoning_effort` | 200 on all four | +| effort max | 400 `Invalid reasoning effort.` | 400 `Invalid reasoning effort.` | +| effort none | 200, but 640 reasoning tokens: the proxy omits the field and the model still reasons | not probed | +| image, user message (3x3 random color grid, 180x180 PNG) | 9/9 | — | +| image, tool result (same grid inside function_call_output) | 9/9 | — | +| caller `service_tier: "priority"` | 200, wire service-tier priority, response tier priority | 200, applied/confirmed, response tier priority | +| `xai/grok-4.7--fast` | — | 200, fastOutcome applied, confirmation confirmed, response tier priority | +| 530,000-word prompt | 400 `context_length_exceeded`: "531243 tokens > 500000 tokens" | — | + +Upstream model name on the Responses wire is `grok-4.7-build` (grok-4.6 reports `grok-4.6-build` the same way). + +Billing parity: the Responses `cost_in_usd_ticks` fits these per-token rates exactly across every probe, for both models: +default input 6800, cached input 1700, output 20400 ticks; priority input 40000, cached 10000, output 120000 ticks. +grok-4.6 probed in the same window produced identical rates (e.g. 83 uncached + 128 cached input, 58 output = +1,965,200 ticks). The OAuth subscription is not per-token billed, so these ticks are recorded as parity evidence only; +the key-auth prices below come from xAI's published page. + +Published (docs.x.ai/developers/models/grok-4.7, read 2026-09-23): 500,000 context; reasoning effort +low/medium/high (default)/xhigh, reasoning cannot be disabled; text+image input; $2.00 input, $0.50 cached, +$6.00 output per 1M; prompts over 200k tokens $4.00 / $1.00 / $12.00; Responses and Chat Completions. +models.dev `xai/grok-4.7`: output limit 500,000 (same as grok-4.6), released 2026-09-21. + +## Other providers + +| Provider | Evidence | Result | +|---|---|---| +| devin (`grok-4-7`) | live probe 200 at low and xhigh; tool-result grid 9/9; proxy /v1/models from Devin's live catalog: context 500000, input text+image, efforts low/medium/high/xhigh/max, default medium | exposes | +| command-code (`xai/grok-4.7`) | live probe 200; grid 9/9 on user-message and tool-result paths; COMMAND_CODE_TEXT_ONLY_MODELS is empty and the logs show no vision-sidecar request, so the route read the image natively | exposes, native image | +| cursor (`grok-4.7`) | live probe 200 through the cursor adapter; live GetUsableModels lists grok-4.7-{low,medium,high,xhigh} and the same ids with -fast (no cursor- prefix, no max); probes: grok-4.7-low and grok-4.7-xhigh-fast accepted, bare grok-4.7-fast rejected not_found (see 000_plan.md D6) | exposes | +| opencode-go / opencode-zen | public `/zen/go/v1/models` and `/zen/v1/models` list `grok-4.7` | listed (not configured locally, not probed) | +| openrouter | public API `x-ai/grok-4.7`: 500000 ctx, max completion 450000, $1.6/$4.8/$0.4, >=200k $3.2/$9.6/$0.8, text+image+file | listed (not probed) | +| github-copilot | models.dev `grok-4.7`: ctx 500000, input 372000, output 128000 | listed (not configured locally) | +| kilo, vercel | models.dev lists kilo `x-ai/grok-4.7` and vercel `spacexai/grok-4.7` | listed | + +Command Code `xai/grok-4.6` is included in `COMMAND_CODE_IMAGE_MODELS`: it read the grids 9/9 +(user message) and 8/9 (tool result) without a vision sidecar. The registry accepts native image +input; 8/9 remains the measured limitation on the tool-result path. diff --git a/devlog/_plan/260923_luvs_l5_responses_usage/000_plan.md b/devlog/_plan/260923_luvs_l5_responses_usage/000_plan.md new file mode 100644 index 0000000000..5c14f03220 --- /dev/null +++ b/devlog/_plan/260923_luvs_l5_responses_usage/000_plan.md @@ -0,0 +1,97 @@ +# L5 luvs01 bundle: Responses continuation, retry and usage boundaries + +Lane L5 of the luvs01 contributor-PR bundling. Eight open originals become one branch, +`codex/260923-luvs-l5-responses-usage`, cut from `origin/dev` at `a4bdc03054`, with ordered +attributable commits and one pull request to `dev`. Landing is decided by the maintainer; this +unit never merges. + +## Constraints + +- Local verification is not run in this lane (no test, typecheck, build, install, CLI or service + commands). Hosted exact-head CI is the verifier; every report says "local checks: NOT RUN". +- Push only `HEAD:codex/260923-luvs-l5-responses-usage`. Never write to contributor branches, to + `stack/*` branches, or to `dev`. +- `tests/fixtures/file-size-baseline.json` caps never move up. Overflow moves byte for byte to a + sibling file registered in `scripts/test-layout/layout.json` `explicit` and + `tests/fixtures/test-layout-expected.json`. +- Security-sensitive review notes stay in scratch space, never in this directory. + +## Dispositions (pinned heads, re-checked 2026-09-23) + +| Original | Head | Disposition | Evidence | +|---|---|---|---| +| #5474 cursor replay bound | `f4eab495c3` | ALREADY ON DEV (index) + DROP (cutoff) | The constant-time replacement index landed in `74490eee36` (#5507), which says it partially carries #5474. The remaining 4,096-message cutoff can begin inside a user turn, and its new test expects the initiating user root to vanish; #5507 deferred it for that reason. | +| #5305 usage.jsonl size cap | `fa7f53fee3` | DROP | Unconditional 64 MiB rotation and legacy-ledger deletion contradict the documented opt-in `usageLedgerMaxBytes` retention (`src/usage/ledger-retention.ts`, configuration reference). Readers only read `usage.jsonl`, so rotated rows disappear from totals. | +| #5434 OAuth rotation attribution | `f6778bfb70` | CHERRY-PICK | Both commits apply cleanly; `hasEligibleGenericOAuthFailoverTarget` is absent from dev. | +| #5560 continuation boundaries | `2ec0cd12f5` | REIMPLEMENT (net) + CHERRY-PICK | Final tree merges cleanly. The xAI empty-catalog selector part is already on dev in `b20acc79d2` (#5376); the first two commits are combined into their net change. The other nine commits carry in order. | +| #5542 tool normalization | `b57d7c5da0` | REIMPLEMENT (selective) | The four native-Responses commits are on dev in `53654291cd` (#5508). The five tool-normalization commits carry, with ADR-0097 renamed to ADR-0099 and dev's newer #5508 docs/tests kept on the three conflicts. | +| #5553 retry/compaction/account | `67c4f579e4` | REIMPLEMENT (selective) | `35fb727ddf` and `940b318292` are on dev in `b7351ddef3` (#5575), which widened the replacement fence. Fifteen commits carry; the retry conflicts keep dev's side. | +| #5562 search replay boundaries | `6ea3a95c21` | REIMPLEMENT (selective) | `76aa665e64` and `7e826dc089` are on dev in `b7351ddef3` (#5575). Combo isolation and terminal repair carry. Dev's caller-principal and single send-budget contracts are kept. `421ba780ae` and the lifecycle helper from `6b122cd2f0` are carried by open #5549 (another lane); the key-failover fixture adoption that depends on that helper is dropped from this lane and handed back to the maintainer. | +| #5556 usage observation | `d3589638a8` | CHERRY-PICK + REIMPLEMENT (one hunk) | Ten commits carry. The attribution-timestamp check is tightened to the producer's canonical ISO form. The merge-only commit and the screenshot-only commit are omitted. | + +## Commit ledger for the dropped originals + +| Commit | Disposition | Reason | +|---|---|---| +| #5474 `49a9c15988` | ALREADY ON DEV (index) + DROP (cutoff) | `entryIndex` replacement is in `74490eee36`; the raw 4,096-message cutoff is dropped. | +| #5474 `68f74eb844` | DROP | The test asserts that the initiating user root disappears. | +| #5474 `f4eab495c3` | DROP | Merge from dev; no own change. | +| #5305 `fa7f53fee3` | DROP | Conflicts with the opt-in ledger retention contract. | + +## Transitive provenance + +| Carrier | Source PRs and authors | +|---|---| +| #5474 | contributor fork PR #348 (luvs01) | +| #5560 | #5350 (Yeonwoo Choi / twoimo), #5420 (maosisheng, Cursor co-author), `82a5f6da81` (Epinephrine), `aac783fe8d` (Devin AI, Epinephrine co-author) | +| #5542 | #5508 (already on dev; itself carried #5479, #5470, #5492 by luvs01), #5230 (kosta), #5352 (Flowershangfromthebranches), `7cbbf44f6c` (Epinephrine), `19a2005e41` (Devin AI) | +| #5553 | #5446, #5423, #5415 (luvs01), compaction identity and scoped quota series (Epinephrine, Devin AI) | +| #5562 | #5480, #5365 (luvs01), `973a4ac702` (Devin AI, Epinephrine co-author) | +| #5556 | #5358, #5283, #5275, #5255 (luvs01) | + +Cherry-picked commits keep their authors and gain `-x` source trailers. Reimplemented commits +carry `Co-authored-by` trailers for every source author. + +## Work-phase map + +| Phase | Doc | Content | +|---|---|---| +| wp1 | this file | roadmap (docs only) | +| wp2 | `010_phase1_small_units.md` | #5434 | +| wp3 | `020_phase2_responses_sequence.md` | #5560, #5542, #5553 on the shared dispatch file | +| wp4 | `030_phase3_search_usage.md` | #5562, #5556 | +| wp5 | `040_phase4_pr_ci_review.md` | push, PR, review waves, exact-head CI, security verdict | +| wp6 | `050_phase5_close_originals.md` | close superseded originals with credit | + +## Shared files + +- `src/server/responses/passthrough-dispatch.ts`: #5560 (error mapping near the custom-tool + admission), #5542 (native-control authorization), #5553 (OpenCode Go reset exception), #5434 + (OAuth budget-denial attribution). Disjoint hunks, applied in wp2 then wp3 order. +- `structure/transports/responses.md`: every carrier except #5562 edits a separate paragraph; + union the paragraphs and keep dev's #5575 status table. +- `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`: additive + entries only. +- Capped files touched: `src/server/responses/core.ts` (210/210, one-line re-export kept), + `tests/responses/responses-compaction-routing.test.ts` (2776 cap, carry reaches 2772), + `tests/server/server-auth.test.ts` (shrinks), `tests/providers/cursor/cursor-blob.test.ts` + (net zero), `tests/responses/openai-responses-passthrough.test.ts` (net zero after extraction), + `gui/src/pages/Models.tsx` (2792 cap, carry reaches 2783). + +## Cross-lane seams + +`src/server/responses/request-prepare.ts`, `passthrough-delivery.ts`, `src/codex/auth-context.ts`, +`src/server/responses/compact.ts`, `core-codex-account.ts`, `src/usage/log.ts`, +`src/bridge/sse.ts`, `structure/ops/docs-and-release.md`, and both test-layout registries. +`src/responses/parser.ts` and `src/responses/plaintext-v2-agent-messages.ts` are not touched (the +#5542 hunk on the latter is already on dev). `421ba780ae` and the whole of `6b122cd2f0`/ +`6ea3a95c21` depend on the sandbox-cleanup helper that open #5549 carries; they stay out of this +lane so no change is applied twice. + +## Re-pin: #5553 moved (2026-09-23) + +#5553's head moved from `67c4f579e4` to `cc466ed9c0` by fast-forward. The four new commits are not +carried by this lane: `f732aa4689` and `6e6bd22f3b` are the whole of #5307, which lane L2 carries in +#5600 (`c448a49794`); `7aaf9594ec` is the Kiro part of #5310, which belongs to lane L7; `cc466ed9c0` +adds tests and structure notes for those two carries. The merge of `a077087b74` is already on dev. +Every earlier #5553 commit is carried as planned. diff --git a/devlog/_plan/260923_luvs_l5_responses_usage/010_phase1_small_units.md b/devlog/_plan/260923_luvs_l5_responses_usage/010_phase1_small_units.md new file mode 100644 index 0000000000..3bcfe86075 --- /dev/null +++ b/devlog/_plan/260923_luvs_l5_responses_usage/010_phase1_small_units.md @@ -0,0 +1,23 @@ +# wp2: #5434 OAuth rotation attribution + +Source: #5434 head `f6778bfb70`. #5474 (`49a9c15988`, `68f74eb844`, merge `f4eab495c3`) and +#5305 (`fa7f53fee3`) close without a carry; the commit ledger is in `000_plan.md`. + +## Recipe + +```sh +git cherry-pick -x 1ac1ba0c8f f6778bfb70 +``` + +Files (MODIFY): `src/oauth/generic-account-failover.ts` (new non-mutating +`hasEligibleGenericOAuthFailoverTarget` using the same eligibility predicate as rotation), +`src/server/responses/adapter-continuation.ts`, `src/server/responses/passthrough-dispatch.ts`, +`src/server/responses/run-turn-execution.ts` (gate `noteAttemptRecoveryWithheld` on the probe), +`structure/transports/responses.md` (cooldown-aware attribution sentence), +`tests/oauth/generic-oauth-failover.test.ts` (negative cooldown case, positive eligible case, +source-oracle assertion over the three call sites). + +## Check + +Static: `git diff --check origin/dev...HEAD`; merge preview clean. A reviewer confirms the probe +matches `rotateGenericOAuthAccountOn429`'s predicate and that the three sites are gated. diff --git a/devlog/_plan/260923_luvs_l5_responses_usage/020_phase2_responses_sequence.md b/devlog/_plan/260923_luvs_l5_responses_usage/020_phase2_responses_sequence.md new file mode 100644 index 0000000000..908468232a --- /dev/null +++ b/devlog/_plan/260923_luvs_l5_responses_usage/020_phase2_responses_sequence.md @@ -0,0 +1,53 @@ +# wp3: #5560, #5542, #5553 on the shared Responses dispatch path + +Applied after wp2 in this order. Each hunk on `passthrough-dispatch.ts` is disjoint. + +## #5560 (head `2ec0cd12f5`) + +1. Combine `71a9fe575b` and `ffd50f485c` (both Yeonwoo Choi) into one commit authored by + Yeonwoo Choi: `git cherry-pick -n 71a9fe575b ffd50f485c`, then restore + `src/adapters/xai-web-search.ts` to `origin/dev` (dev `b20acc79d2` already owns the selector + rule), keeping the net Cursor continuation, blob-estimate, xAI custom item-ID repair, extracted + tests and docs. The net diff leaves `tests/providers/cursor/cursor-blob.test.ts` at 3,657 lines + and `tests/responses/openai-responses-passthrough.test.ts` at 4,809. +2. `git cherry-pick -x 31f21f0370 a326b67338 7e8fb09b39 57407be416 c7781bf81c fbecefa18b 82a5f6da81 aac783fe8d 2ec0cd12f5` + Layout-map conflicts resolve by union. + +## #5542 (head `b57d7c5da0`) + +Skip `43f1c19fbe`, `10bf60cea3`, `d61ec2e603` (on dev in `53654291cd`) and `7f3f18aed8` (merge). + +1. `git cherry-pick -x e555e7305b`. Its decision record is added as ADR-0097 and renamed to + ADR-0099 by `b57d7c5da0` below (dev's ADR-0097 is unrelated); the head has no duplicate. +2. `git cherry-pick -x 7cbbf44f6c 19a2005e41 9662528195`. +3. `git cherry-pick -x b57d7c5da0`. As a single-commit pick it carries only its own delta (the + ADR rename and the combined JSON/SSE regression), so dev's #5508 versions of + `docs-site/.../guides/codex-integration.md`, `structure/transports/streaming-health.md` and + `tests/responses/ws-native-injection.test.ts` stay intact. A dry run on `a4bdc03054` applied + every wp3 commit without conflict. + +## #5553 (head `67c4f579e4`) + +Skip `35fb727ddf`, `940b318292` (on dev in `b7351ddef3`) and `67c4f579e4` (merge). + +1. `git cherry-pick -x b8f9a45761 808dd85a9f db854bf306 b037810fe2 e6f9339f83 f86a53437c 76b40f9fd0 385f338d82 feb0c160aa 466c75c89c 9050722914 b2eda92b1b 1069b541f7 37a006e223` +2. Conflicts in `src/lib/upstream-retry.ts`, `src/lib/errors.ts`, `tests/lib/upstream-retry.test.ts`, + `tests/usage/request-log.test.ts` keep dev's #5575 side (`invitesResendAfterReplacement`, the + whole-sentence refusal matcher and its status table). +3. `git cherry-pick -x be1fee99aa`, then a follow-up commit (luvs01 co-author trailer) rewrites + the transport-doc paragraph so it references dev's broader replacement fence instead of a + 5xx-only rule. The dry run applied it without conflict; the wording is the only repair. + +Cap checks after the phase: `src/server/responses/core.ts` 210, +`tests/responses/responses-compaction-routing.test.ts` at most 2,776. + +## Outcome (wp3) + +All planned commits applied on `a077087b74` after resolving the conflicts above. Review follow-ups: +`test(cursor): pin exact host-wrapper classification in continuation scope` (exact summary and +ambient wrappers are classified by shape, matching the Codex client; documented in +`structure/providers/cursor.md`) and `test(server): prove a suppressed same-workspace alternate is +never sent` (exact one-send assertion; the transport contract now limits suppression to the +in-request move). A later request can still select a same-workspace sibling that was not itself +refused; that selection behavior predates this carry and is reported to the maintainer. +Local checks: NOT RUN. Static gate passed; hosted CI verifies in wp5. diff --git a/devlog/_plan/260923_luvs_l5_responses_usage/030_phase3_search_usage.md b/devlog/_plan/260923_luvs_l5_responses_usage/030_phase3_search_usage.md new file mode 100644 index 0000000000..fd845d8313 --- /dev/null +++ b/devlog/_plan/260923_luvs_l5_responses_usage/030_phase3_search_usage.md @@ -0,0 +1,64 @@ +# wp4: #5562 and #5556 + +## #5562 (head `6ea3a95c21`) + +Skip `76aa665e64`, `7e826dc089` (on dev in `b7351ddef3`), `65c3477dd2` (merge), and +`421ba780ae`, `6b122cd2f0`, `6ea3a95c21`: open #5549 carries the sandbox-cleanup helper, +`createTestCaseLifecycle` and their tests (its `ef5c002220` and `8dc4050fad`). The key-failover +fixture adoption in `6b122cd2f0` imports that helper, so it cannot land here without applying the +helper twice; it is dropped from this lane and reported to the maintainer for a follow-up after +#5549. + +1. `git cherry-pick -x 7f45883fb5 c4fa8c8d8f`. +2. `git cherry-pick -x 8d46989165 3f3fdf17f4`. Once the two already-landed commits are skipped, + both apply cleanly (dry run on `a4bdc03054`): `request-prepare.ts` keeps dev's caller-principal + block from #5575 and gains only the early combo intersection and shadow marker. +3. `git cherry-pick -x 973a4ac702 bb49c9f582`, then a follow-up commit (luvs01 co-author) adapts + `tests/web-search/web-search-passthrough-bridge.test.ts` (the `clientPrincipalId: "loopback"` + expectation) to dev's documented rule that keyless callers get no bridged replay: configure an + inbound API key, assert the derived principal, keep a keyless miss control. +4. `ae52669293`: cherry-pick. +5. Keep dev's `src/web-search/executor.ts`, `tests/web-search/web-search-sidecar-429.test.ts`, the + negative controls in `tests/web-search/web-search-bridge-replay.test.ts`, and the single + physical-send budget wording in `structure/runtime.md` and `structure/providers-and-adapters.md`. + +## #5556 (head `d3589638a8`) + +1. `git cherry-pick -x 0f0ef96ea3 83514c382f`. +2. `138069331f` reimplemented: in `src/cli/access.ts` treat `attributionSince` as valid only when + it round-trips through `new Date(value).toISOString()`; add a malformed-but-parseable case + (for example `"0"`) next to the invalid-string case in `tests/cli/cli-dto-fidelity.test.ts`. +3. `git cherry-pick -x 5563577fc2 c8a9d1a75e 823a7d2d9f 22ee516602 1a8d5f7ded 2241d03f44 ddfef1320b`. +4. Omit `96602cd13d` (merge of `41ec40f7e3`, already an ancestor of dev) and `d3589638a8` + (screenshot asset only; the PR description links the existing capture). + +Cap check: `gui/src/pages/Models.tsx` at most 2,792. + +## Amendments after review (wp4 P) + +- #5562 `3f3fdf17f4`: drop its early combo intersection hunk in + `src/server/responses/request-prepare.ts`. It sampled a combo target with `routeModel` before + dispatch, so the decision could follow a different pick than the one sent and could advance + round-robin or random state. Dev's #4129 rule stays: a shadow call rewritten to a combo enters + the combo and carries `shadowCallIntercepted`. The test + `a combo whose first target intersects the source still routes as a combo` keeps dev's + assertions. The combo-child isolation marker and its tests remain. +- #5562 `bb49c9f582` follow-up: the bridge replay test configures an inbound API key, derives the + principal with `resolveContextPrincipal`, passes the full loopback admission, and adds a keyless + miss control. +- #5556 `138069331f` follow-up: accept `attributionSince` only in canonical + `toISOString()` form; positive fixtures use `.000Z`; malformed cases include `"0"`. +- #5556 selector encoding: `encodePersistedRequestedModel` must stay idempotent because rows are + normalized again on read, so a literal selector equal to another selector's encoded form + aliases it. Document the limitation in the code comment and pin it with a test; a digest column + would remove it and is reported to the maintainer. + +## Outcome (wp4) + +#5562: seven commits carried; follow-ups `fix(responses): keep combo shadow interception on the +dispatch pick` and `test(web-search): bind repaired-leg replay to a keyed caller principal`. +`421ba780ae`, `6b122cd2f0` and `6ea3a95c21` stay with #5549. +#5556: ten commits carried; follow-ups `fix(cli): accept only an ISO-8601 UTC attributionSince` +and `docs(usage): state the aliasing limit of the idempotent selector encoding`. The screenshot +commit is not carried; the PR links the existing capture. +Local checks: NOT RUN. Static gate passed; hosted CI verifies in wp5. diff --git a/devlog/_plan/260923_luvs_l5_responses_usage/040_phase4_pr_ci_review.md b/devlog/_plan/260923_luvs_l5_responses_usage/040_phase4_pr_ci_review.md new file mode 100644 index 0000000000..57e770b4dc --- /dev/null +++ b/devlog/_plan/260923_luvs_l5_responses_usage/040_phase4_pr_ci_review.md @@ -0,0 +1,18 @@ +# wp5: pull request, review, CI, security verdict + +1. `git push --no-verify -u origin HEAD:codex/260923-luvs-l5-responses-usage`. +2. Open one ordinary pull request to `dev` (not draft) with every section of + `.github/PULL_REQUEST_TEMPLATE.md`, the disposition table, a "Cross-lane seams" section, + "local checks: NOT RUN", and a screenshot link for the Models tab lifecycle change. +3. Independent reviewers read each carried unit; each confirmed defect gets a fix and a focused + regression test in a new commit. +4. CI is judged on the latest run per job at the current head. Missing, queued, skipped or + cancelled jobs are not success. If no cross-platform run appears after a push, close and reopen + once. +5. An independent security reviewer reads the final diff and posts a short verdict comment. + +## Outcome (wp5, in progress) + +PR #5608 opened to `dev` from `codex/260923-luvs-l5-responses-usage` (not draft). An +integration review of the combined branch passed before the push. Hosted CI at the PR head is the +verifier; local checks: NOT RUN. diff --git a/devlog/_plan/260923_luvs_l5_responses_usage/050_phase5_close_originals.md b/devlog/_plan/260923_luvs_l5_responses_usage/050_phase5_close_originals.md new file mode 100644 index 0000000000..aa1a0f7c3e --- /dev/null +++ b/devlog/_plan/260923_luvs_l5_responses_usage/050_phase5_close_originals.md @@ -0,0 +1,18 @@ +# wp6: close superseded originals + +For each original: re-read its head; if it moved past the pinned SHA, re-carry first. Then close +with a short credit comment naming the bundle PR (ALREADY ON DEV names the dev commit; DROP gives +the reason). Transitive source PRs owned by other contributors (#5350, #5420, #5230, #5352) are not +closed by this lane; the bundle description credits them. + +## Outcome (wp6) + +All eight originals were closed on 2026-09-23 after a final head re-pin, each with a credit comment +naming #5608: #5474 (index already on dev in 74490eee36; cutoff dropped), #5305 (dropped), #5434, +#5560, #5542, #5553 (four later commits left to #5307/#5600 and #5310), #5562 and #5556. None was +merged by this lane. + +#5549 was closed after the roadmap was written; its sandbox cleanup and lifecycle helper now travel +in #5597. The key-failover fixture lifecycle from `6b122cd2f0` and the documentation in +`6ea3a95c21` build on that helper and are not carried by any open PR; they can be re-offered once +#5597 lands. diff --git a/devlog/_plan/260923_models_catalog_delivery_disclosure/000_plan.md b/devlog/_plan/260923_models_catalog_delivery_disclosure/000_plan.md new file mode 100644 index 0000000000..eeefbf3e6c --- /dev/null +++ b/devlog/_plan/260923_models_catalog_delivery_disclosure/000_plan.md @@ -0,0 +1,61 @@ +# Models page: collapsible "How changes reach Codex" disclosure + +The Models tab shows a static three-row card ("Saved on hub / Fetched by this client / +Active in a running client") between the subtitle and the workspace. Two rows are fixed +sentences, the middle row reads "not reported" on every standalone install because +`catalogSyncedAt` only exists in `ocx connect` client mode, and the subtitle repeats the +same caveat. The card pushes the workspace down and does not explain the process. This unit +replaces it with a one-line `<details>` disclosure that names the real delivery steps for +the current mode and expands into a detailed explanation, trims the duplicated subtitle +sentences, and fixes the save toast that says "hub" on standalone installs. + +## Loop spec + +- Loop archetype: satisfy-spec, single work-phase (wp1), C2 GUI change. +- Trigger: user request 2026-09-23 ("반영 과정 이렇게 해놓고 한줄로 접기, 설명은 더 자세하게", cxc-loop, mimo subagents). +- Goal: collapsed one-line summary of the delivery steps; expanded detail per step with what it means, how OpenCodex knows, and what to do. +- Non-goals: no `src/` runtime change, no change to CodexStaleBanner/appServerState logic, no docs-site edit, no push/PR/merge/service restart. +- Verifier: `bun x tsc --noEmit -p tsconfig.app.json` in gui/ (reads every locale file and the component via tsconfig.app include of src); `bun test tests/models-status-toast.test.tsx tests/codex-stale-banner.test.ts tests/i18n-locales.test.ts tests/i18n-language-switch.test.tsx` in gui/ (key-set and placeholder parity read all ten catalogs; the toast test mounts Models); render grounding on Vite dev server proxied to :10100 (collapsed + expanded screenshots, ko). Client-mode branch has no live hub here: it is exercised by a focused render assertion in gui/tests (see activation below). +- Stop condition: all six goalplan criteria met with fresh evidence and a local commit. +- Memory artifact: this directory; goalplan `.codexclaw/goalplans/opencodex-gui-models-page-worktree-users-jun-cod/`. +- Expected terminal outcomes: DONE; BLOCKED if the GUI cannot render or typecheck cannot run; NEEDS_HUMAN if copy would contradict the #5031 honesty contract. +- Escalation: two distinct translator agents failing the same locale -> main translates it directly. +- Resource bounds: no user token/time budget; mimo subagents (aim/mimo-v2.6-flash-free), write scope one locale file each. + +## Facts the design rests on + +- `catalogSyncedAt` origin: gui/src/App.tsx:506 -> gui/src/api-targets.ts:172 (only when connected) -> src/client/connect.ts:601/684 (hub catalog download written to DEFAULT_CATALOG_PATH). +- Standalone: `targets.connected === false`; Models edits this proxy's catalog directly. +- Codex reads the catalog when its app-server starts; the page-head button (`dash.codexRestart` "Codex 모델 목록 새로고침") stops app-servers and the user reopens Codex (ko.ts:342-346). +- CodexStaleBanner appears above the tabs when the running app-server is older than the catalog. + +## File change map + +| File | Change | +|---|---| +| gui/src/pages/models-catalog-state.tsx | Replace `ModelCatalogStateSummary` with `ModelCatalogDelivery({ connected, catalogSyncedAt })`: `<details className="models-delivery">` closed by default, `<summary>` = title + step chips joined by arrows, body = `<ol>` of 2 (standalone) or 3 (client) steps + hint. | +| gui/src/pages/Models.tsx | Props gain `connected?: boolean` (same line); call site renders `<p className="page-sub">` then `{tab === "catalog" && <ModelCatalogDelivery .../>}`. Net line delta <= +2 (cap 2792, now 2784). | +| gui/src/App.tsx | Pass `connected={targets.connected}` on the existing Models line (0 lines). | +| gui/src/styles-models-workspace.css | `.models-delivery*` rules (styles.css is at cap and unchanged). | +| gui/src/i18n/{en,ko,de,fr,ja,ru,tr,vi,zh,zh-TW}.ts | Remove `models.catalogState.*` (8 keys); add `models.delivery.*` (listed below); rewrite `models.subtitle` (drop last two sentences) and `models.applied` (mode-neutral). | +| gui/tests/models-catalog-delivery.test.tsx (new) | Render assertions for standalone (2 steps, no sync step) and client (3 steps, time and unknown variants), closed by default. gui/tests has no layout registry; tsconfig.app.json does not compile tests, so running the test is its verifier. | + +## i18n keys (en source) + +- models.delivery.title: "How changes reach Codex" +- models.delivery.chip.saved: "Saved"; chip.savedHub: "Saved on hub"; chip.synced: "Synced {time}"; chip.syncedUnknown: "Sync not recorded"; chip.loaded: "Loaded when Codex restarts" +- models.delivery.saved.title / .body (standalone save) +- models.delivery.savedHub.title / .body (client-mode save) +- models.delivery.synced.title / .bodyAt ({time}) / .bodyUnknown +- models.delivery.loaded.title / .body +- models.delivery.hint + +## Conditional paths and activation + +- `connected` true vs false: activation = new test renders both; standalone also observed live. +- `catalogSyncedAt` valid / missing / unparsable: test renders valid and missing; unparsable goes through the same `formatFetchTime` null path. +- Tab gate: disclosure only on catalog tab; observed live by switching to Combos. + +## Architect consultation + +See 010_architect.md. diff --git a/devlog/_plan/260923_models_catalog_delivery_disclosure/010_architect.md b/devlog/_plan/260923_models_catalog_delivery_disclosure/010_architect.md new file mode 100644 index 0000000000..0b0c0f8239 --- /dev/null +++ b/devlog/_plan/260923_models_catalog_delivery_disclosure/010_architect.md @@ -0,0 +1,19 @@ +# Architect consultation (wp1) + +- Handle: grok-4.7 subagent `01a0ca58-e95c-73b1-9d91-6f07e61251af` (Tesla). Earlier attempts: mimo `01a0ca43-d5eb-7953-a025-2bfc6be823fb` produced no proposal after ~25 min and was closed; two gpt-5.6-sol agents failed with HTTP 429 before starting. +- Combined proposal + reflection against 000_plan.md (one packet, because the plan already existed when a responsive architect became available). + +| ID | Proposal | Main disposition | +|---|---|---| +| D1 | `ModelCatalogDelivery({ connected, catalogSyncedAt })`, reuse `formatFetchTime` (missing and unparsable both null), caller owns subtitle | Accepted | +| D2 | Closed `<details>`, summary = title + arrow-joined chips, body `<ol>` + hint; rely on global `:focus-visible` (styles.css:242); copy marker pattern styles.css:2041 | Accepted; no custom focus style | +| D3 | Delete 8 `models.catalogState.*`, add `models.delivery.*` in 10 catalogs, trim `models.subtitle`, mode-neutral `models.applied` (en.ts:746) | Accepted | +| D4 | CSS only in styles-models-workspace.css; drop inline styles | Accepted | +| D5 | Render beside subtitle only when `tab === "catalog"` (panels stay mounted hidden, Models.tsx:2654); pass `connected` from App.tsx:506 | Accepted | +| D6 | New gui/tests/models-catalog-delivery.test.tsx; no layout.json entry for gui/tests | Accepted | + +Reflection: **ALIGNED**. Gaps and dispositions: + +- "registered wherever gui test layout requires" is a no-op for gui/tests -> plan wording corrected. +- `tsconfig.app.json` includes only `src`, so typecheck does not compile the new test -> the test run itself is its verifier. + diff --git a/devlog/_plan/260923_models_catalog_delivery_disclosure/020_done.md b/devlog/_plan/260923_models_catalog_delivery_disclosure/020_done.md new file mode 100644 index 0000000000..6fe54d8907 --- /dev/null +++ b/devlog/_plan/260923_models_catalog_delivery_disclosure/020_done.md @@ -0,0 +1,18 @@ +# Done: wp1 + +The Models tab no longer opens with a three-row card of fixed sentences. Under the subtitle there is now one line, "How changes reach Codex", folded closed, that names the real steps for the current mode: two on a standalone install (saved here -> Codex loads it on restart) and three for an `ocx connect` client (saved on hub -> synced <time> -> Codex loads it on restart). Expanded, each step explains what it means, how OpenCodex knows it or why it cannot, and the hint names the page's own Reload Codex models button. The subtitle lost the two sentences this now covers, and the save toast no longer says "hub" on installs that have none. + +## Evidence + +- gui `bun x tsc --noEmit -p tsconfig.app.json`: exit 0. +- gui tests: models-catalog-delivery (new, 3), models-status-toast, codex-stale-banner, i18n-locales, i18n-language-switch: 54 pass. +- root `bun test tests/ci-workflows/file-size-ratchet.test.ts tests/gui`: 439 pass. Models.tsx 2785/2792; styles.css untouched. +- `bun run lint:gui` exit 0; `bun run privacy:scan` passed. +- Render: Vite dev server proxied to :10100. Collapsed height 27px; expanded shows 2 steps + hint in en and ko; Combos tab renders no disclosure. Screenshots: evidence/ko-collapsed.png, evidence/ko-expanded.png. +- Audit: grok-4.7 reviewer PASS (no blockers); architect ALIGNED (010_architect.md). + +## What did not improve + +- The client-mode (3-step) branch was not observed live; this machine is standalone. It is covered by the render test only. +- Codex activation is still unverifiable from OpenCodex; the step says so instead of claiming it. Using CodexStaleBanner's app-server age to colour that step is a possible follow-up, not done here. +- Eight locales were translated by subagents and checked for key/placeholder parity and type errors, not by native readers. diff --git a/devlog/_plan/260923_models_catalog_delivery_disclosure/evidence/ko-collapsed.png b/devlog/_plan/260923_models_catalog_delivery_disclosure/evidence/ko-collapsed.png new file mode 100644 index 0000000000..adfc113545 Binary files /dev/null and b/devlog/_plan/260923_models_catalog_delivery_disclosure/evidence/ko-collapsed.png differ diff --git a/devlog/_plan/260923_models_catalog_delivery_disclosure/evidence/ko-expanded.png b/devlog/_plan/260923_models_catalog_delivery_disclosure/evidence/ko-expanded.png new file mode 100644 index 0000000000..f18aa47fa6 Binary files /dev/null and b/devlog/_plan/260923_models_catalog_delivery_disclosure/evidence/ko-expanded.png differ diff --git a/devlog/_plan/260923_opus_5_5_catalog/010_plan.md b/devlog/_plan/260923_opus_5_5_catalog/010_plan.md new file mode 100644 index 0000000000..f5b14230c1 --- /dev/null +++ b/devlog/_plan/260923_opus_5_5_catalog/010_plan.md @@ -0,0 +1,80 @@ +# 260923 Claude Opus 5.5 catalog, pricing and 1M context + +## Problem + +Anthropic released Claude Opus 5.5 (`claude-opus-5-5`) on 2026-09-22. Live Anthropic discovery +already lists it, so the picker shows the row, but nothing in the static catalog knows it: + +- `/v1/models` on the running proxy returns `anthropic/claude-opus-5-5` with no + `max_output_tokens`, no input modalities, `supports_reasoning: false` and no effort ladder, + while `anthropic/claude-opus-5` carries all of them. +- The price resolver has no overlay, no bundled metadata row and no vendor fallback for the id, + so the ~$ column and every price surface render blank. 560 logged anthropic requests and 10 + `anthropic-native` requests (`claude-opus-5.5` / `claude-opus-5-5`) are currently unpriced. + +## Evidence (collected 2026-09-23) + +| Source | Surface | Facts | +|---|---|---| +| platform.claude.com/docs/en/about-claude/pricing | Aside repl | Opus 5.5: $4 in, $5 5m write, $8 1h write, $0.20 cache hit (0.05x footnote), $20 out; batch $2/$10; fast $8/$40. Opus 5 now listed at $5/$6.25/$10/$0.50/$25 | +| platform.claude.com/docs/en/models/opus-5-5/overview | Aside repl | id `claude-opus-5-5`, 1M context, 128K output (300K batch beta), adaptive thinking always on, default effort medium, Bedrock `anthropic.claude-opus-5-5` | +| platform.claude.com/docs/en/build-with-claude/effort | Aside repl | Opus 5.5 supports all five levels (low..max) | +| platform.claude.com/docs/en/models/opus-5-5/migration-guide | Aside repl | thinking disabled/enabled 400; tool_choice any/tool 400 | +| cursor.com/docs/models/claude-opus-5-5 | Aside repl | Cursor id `claude-opus-5-5`, 300K default / 1M max context, no long-context multiplier, $4/$5/$0.2/$20; fast `claude-opus-5-5-fast` $8/$10/$0.4/$40; thinking variant | +| docs.devin.ai/desktop/models | Aside repl | Opus 5.5 not in the modelCostData table yet | +| running proxy `/v1/models` | curl | `devin/claude-opus-5-5` is live, context 1_000_000, efforts low..max | +| kiro.dev/docs/models | Aside repl | no Opus 5.5 | +| models.dev/api.json | curl | anthropic, amazon-bedrock (6 ids, regional 1.1x), kilo `anthropic/claude-opus-5.5`, venice `claude-opus-5-5` (4.8/24/0.24/6), openrouter `anthropic/claude-opus-5.5`, all 1M/128K | +| openrouter.ai/api/v1/models | curl | `anthropic/claude-opus-5.5` 1M/128K, 4/20/0.2/5 | +| ai-gateway.vercel.sh/v1/models | curl | `anthropic/claude-opus-5.5` 4/20/0.2/5 and `-fast` 8/40/0.4/10 | + +GitHub Copilot and opencode-zen have no Opus 5.5 upstream yet; Kiro has none; they stay unchanged. + +## Diff-level plan (wp1, one PABCD cycle) + +1. `src/usage/expected-prices.ts` + - Add `CLAUDE_OPUS_55: Cost4 = { input: 4, output: 20, cacheRead: 0.2, cacheWrite: 5 }` with a + comment naming the 0.05x cache-hit footnote. + - Overlays: `anthropic` and `anthropic-apikey` (verified, ANTHROPIC_PRICING), + `cursor` (verified-derived, cursor.com/docs/models/claude-opus-5-5 vendor page), + `devin` and `devin-cli` (verified-derived, absent from Devin table, Anthropic list price). + - Promote the three Opus 5 rows from the user-confirmed derived source to the now-published + Anthropic list price (`CLAUDE_OPUS_5` constant, anthropic row `verified`, cursor/kiro keep + `verified-derived`). Tuple unchanged. +2. `scripts/model-metadata.source.json` (snapshot rows copied from each provider's claude-opus-5 + row shape, values from models.dev / vendor APIs) then regenerate `src/generated/model-metadata.ts`: + anthropic `claude-opus-5-5`; amazon-bedrock `anthropic.`, `global.`, `us.`, `eu.`, `jp.`, + `au.anthropic.claude-opus-5-5`; openrouter `anthropic/claude-opus-5.5`; vercel-ai-gateway + `anthropic/claude-opus-5.5` and `anthropic/claude-opus-5.5-fast`; kilo + `anthropic/claude-opus-5.5`; venice `claude-opus-5-5`. +3. `src/providers/registry/model-seeds.ts`: `claude-opus-5-5` first among Opus in + `ANTHROPIC_MODELS`, `1_000_000` in `ANTHROPIC_MODEL_CONTEXT_WINDOWS`, provenance comment. +4. Devin: `claude-opus-5-5` in the devin registry seed list (entries-core) and + `DEVIN_MODEL_CONTEXT_WINDOWS` = 1_000_000 (value read from the live catalog). +5. Cursor: `CURSOR_CAPABILITIES["claude-opus-5-5"]` (Claude Opus 5.5, 1M window, thinking default, + regular/thinking/fast/thinkingFast FULL ladder, preemptive until the live roster is measured); + effort-map tiers for base, -fast, -thinking, -thinking-fast and thinking families; + `models-capabilities.ts` local picker family regex `^claude-opus-5(?:-5)?$`. +6. Docs: `reference/configuration/providers.md` Cursor Fast base list gains `claude-opus-5-5` + in English and every locale carrying the same token list. +7. Tests: add focused assertions next to existing ones (usage-cost overlay/resolver, anthropic seed, + cursor capability) without growing any file past its ratchet cap; update pinned rosters. + +Adapter check: `claudeFamilyVersion("claude-opus-5-5")` = opus 5.5, so adaptive wire is used and +explicit `thinking: disabled` (sonnet >= 5 only) is never sent. Forced `tool_choice` mapping is a +generic path shared by all models; out of scope, reported as a residual. + +## Verification + +- `bun run generate:model-metadata` then `bun test tests/codex-integration/model-metadata-sync.test.ts` +- `bun test tests/usage tests/providers/cursor tests/providers/model-presets.test.ts tests/providers/devin-adapter.test.ts` plus + registry/anthropic tests selected by `bun run test:changed` +- `bun run typecheck`, file-size ratchet + test-layout tests, `bun run structure:check`, + `bun run privacy:scan` +- Post-change: resolver probe prints the Opus 5.5 tuple for anthropic, anthropic-native dot id, cursor, devin. + +## Bounds + +Write scope: files above plus this devlog unit. No push, merge, release or service restart inside +the loop. Wall clock: single session. + diff --git a/devlog/_plan/260923_opus_5_5_catalog/020_audit.md b/devlog/_plan/260923_opus_5_5_catalog/020_audit.md new file mode 100644 index 0000000000..81e1ab5925 --- /dev/null +++ b/devlog/_plan/260923_opus_5_5_catalog/020_audit.md @@ -0,0 +1,24 @@ +# 020 Audit (wp1) + +Reviewer: independent read-only subagent (Hooke), static reads only. Verdict NEAR-PASS. + +## Folded + +- B1: promoting the Opus 5 overlay source breaks `tests/usage/usage-cost.test.ts` (asserts `user-confirmed`). + Kept the promotion because the pricing page now publishes Opus 5 at the same tuple; the test is + rewritten to assert the published Anthropic source, which strengthens provenance rather than weakening it. +- Scope: `src/adapters/devin/live-models.ts` added explicitly for `DEVIN_MODEL_CONTEXT_WINDOWS`. +- Anthropic resolves through the bundled row first, so new tests expect `source: "jawcode"`, `verified`; + the anthropic/anthropic-apikey overlays cover account-label namespaces. +- Cursor: `CURSOR_THINKING_FAMILIES` entries for `claude-opus-5-5-thinking` (source `claude-opus-5-5`) and + `-thinking-fast` (source `claude-opus-5-5-fast`), thinking-then-effort, required by the catalog oracle. +- Cursor fast ladder mirrors the measured `claude-opus-5-fast` (low/medium/high) instead of FULL. +- `models-capabilities.ts` regex change dropped: that table mirrors Cursor's shipped 3.18.25 bundle. +- Docs: only English providers.md carries the Cursor Fast base list. + +## Residuals (reported, not fixed here) + +- Cursor Fast rows price at the base $4/$20 rather than $8/$40; same pre-existing gap as `claude-opus-5-fast`. +- Anthropic adapter maps `tool_choice` required/named to `any`/`tool`, which Opus 5.5 rejects with 400. +- The running proxy memoizes prices; the new rows appear only after a service restart. + diff --git a/devlog/_plan/260923_opus_5_5_catalog/030_done.md b/devlog/_plan/260923_opus_5_5_catalog/030_done.md new file mode 100644 index 0000000000..30791aa426 --- /dev/null +++ b/devlog/_plan/260923_opus_5_5_catalog/030_done.md @@ -0,0 +1,41 @@ +# 030 Done (wp1) + +## Conclusion + +Claude Opus 5.5 (`claude-opus-5-5`) now carries its official price, 1M context, 128K output, +text+image input and the low..max effort ladder everywhere Claude Opus 5 is represented and an +upstream source shows Opus 5.5 exists. Commit `9cff9b8c52` on `codex/opus-5-5-catalog`. + +## What changed + +- Anthropic seed (`model-seeds.ts`): the id and its 1M window; the seed drives the effort ladder + and modalities the live picker was missing. +- Metadata snapshot and regenerated table: anthropic, six Bedrock ids, openrouter, vercel (+fast), + kilo, venice. +- Price overlays: anthropic, anthropic-apikey, cursor (verified), devin and devin-cli (derived). + Opus 5 overlays now cite the published Anthropic price. +- Cursor capability, effort tiers and thinking families; Devin seed and context window; docs. + +## Evidence + +- Receipt `.codexclaw/evidence/01a0ca42-e654-76f1-a399-b6003105d628/test-receipt.json`: + 1542 pass / 0 fail across 73 files at `9cff9b8c52`. `bun run typecheck` exit 0. +- Verifier subagent (gpt-5.6-sol): cursor 1309 pass, provider/usage 215 pass, layout 18 pass, + ratchet 9 pass, `structure:check` and `privacy:scan` pass. +- Fresh-process probe: anthropic, anthropic-native `claude-opus-5.5`, cursor + `claude-opus-5-5-thinking-high`, devin, openrouter resolve to 4 / 20 / 0.2 / 5; Bedrock US to + 4.4 / 22 / 0.22 / 5.5. + +## What did not improve + +- `bun run test:changed` in this worktree: 896 failures, all from the test-home guard refusing + cleanup under `/Users/jun/.codex` (this checkout lives in `~/.codex/worktrees`). No failure + touches a changed file. Hosted CI is the real full-suite signal. +- Residuals from 020: Cursor Fast rows price at base; forced `tool_choice` on Opus 5.5 returns 400 + through the Anthropic adapter; the running proxy needs a restart to load the new rows. +- Kiro, GitHub Copilot and opencode-zen stay without Opus 5.5 until their catalogs list it. + +## Next + +No further work-phase under this goal. Push/PR and service restart need separate approval. + diff --git a/devlog/_plan/260923_opus_5_5_catalog/040_plan_providers.md b/devlog/_plan/260923_opus_5_5_catalog/040_plan_providers.md new file mode 100644 index 0000000000..df86681fdf --- /dev/null +++ b/devlog/_plan/260923_opus_5_5_catalog/040_plan_providers.md @@ -0,0 +1,31 @@ +# 040 Plan (wp2): Opus 5.5 on non-Anthropic providers + +Continues 030: the Anthropic-family rollout is committed; the user asked for every other provider +that carries claude-opus-5. Five parallel xai/grok-4.7 research leaves checked each one on 2026-09-23. + +## Classification + +| Provider | Evidence | Decision | +|---|---|---| +| Opper | api.opper.ai/v3/models?limit=2000: pool `claude-opus-5-5` (anthropic, aws eu, vertex, vertex-eu), all 1M / 128K, vision, efforts low..max on the Anthropic member | ADD to `OPPER_MODELS`, context 1_000_000, max output 128_000 | +| GitHub Copilot | github.blog changelog 2026-09-22 (GA); docs models-and-pricing $4 / $0.20 cached / $5 write / $20; 1M context in VS Code and CLI; API id unpublished | ADD metadata row `claude-opus-5-5` (catalog's hyphen convention, same as `claude-opus-5`), shape of the Opus 5 row (64K output, effort minimal..high, openai-completions), GitHub-published price. Live discovery owns the roster, so a wrong id stays inert | +| Command Code | api.commandcode.ai/provider/v1/models lists `claude-opus-5-5`, 1M | NO static site (live-only registry entries). Price resolves through the Anthropic vendor fallback; add a regression assertion | +| Kiro | kiro.dev models, available-models, effort, changelog: no Opus 5.5 | NOT ADDED: `KIRO_MODELS` is a static user-visible list; an unpublished id would be a dead selection | +| opencode-zen / opencode-go | live /zen/v1/models and /zen/go/v1/models: no Opus 5.5 | NOT ADDED | +| OpenRouter / Venice / Kilo fast | live lists: no 5.5 fast | NOT ADDED (Vercel fast already present) | +| google-antigravity | only claude-opus-4-6-thinking live | NOT ADDED | +| Docs/README | `anthropic/claude-opus-5` used as routing examples | LEAVE (prose) | + +## Diff + +1. `src/providers/registry/model-seeds.ts`: `claude-opus-5-5` above `claude-opus-5` in the three Opper maps. +2. `scripts/model-metadata.source.json`: github-copilot `claude-opus-5-5` cloned from its `claude-opus-5` + row with cost 4 / 20 / 0.2 / 5; regenerate `src/generated/model-metadata.ts`. +3. `tests/usage/usage-cost.test.ts`: Opus 5.5 test also asserts command-code, opper and github-copilot. + +## Verification + +model-metadata sync, usage-cost, opper/commandcode/registry tests, typecheck, ratchet, layout, +structure:check, privacy:scan, fresh-process probe. Then push, PR to dev, exact-head CI, merge +(user authorized merge on 2026-09-23). + diff --git a/devlog/_plan/260923_opus_5_5_catalog/050_done_providers.md b/devlog/_plan/260923_opus_5_5_catalog/050_done_providers.md new file mode 100644 index 0000000000..6a6b1e01a5 --- /dev/null +++ b/devlog/_plan/260923_opus_5_5_catalog/050_done_providers.md @@ -0,0 +1,16 @@ +# 050 Done (wp2) + +Opus 5.5 now reaches every non-Anthropic provider that publishes it. Opper gains the +`claude-opus-5-5` pool (1M / 128K); GitHub Copilot's snapshot gains a `claude-opus-5-5` row at +GitHub's published $4 / $20 / $0.20 / $5. Command Code lists the id live and prices through the +Anthropic vendor row, now pinned in `tests/usage/usage-cost.test.ts` together with Opper and Copilot. + +Evidence: receipt 1557 pass / 0 fail across 75 files at `50be43a386` after rebasing onto dev +`b7351ddef3`; typecheck, structure:check and privacy:scan pass; fresh-process probe prices all three. +Reviews: grok-4.7 plan audit PASS, meta-muse final branch review PASS. + +Not added, with reason: Kiro (no Opus 5.5 on kiro.dev; static user-visible list), opencode-zen/go +(absent from live lists), OpenRouter/Venice/Kilo fast tiers (absent upstream). +What would show this wrong: a provider shipping a different id spelling. Copilot's id is the +catalog's hyphen convention, unconfirmed by GitHub; the row is inert until live discovery lists it. + diff --git a/devlog/_plan/260923_opus_5_5_catalog/060_plan_preemptive.md b/devlog/_plan/260923_opus_5_5_catalog/060_plan_preemptive.md new file mode 100644 index 0000000000..cec7c2ad97 --- /dev/null +++ b/devlog/_plan/260923_opus_5_5_catalog/060_plan_preemptive.md @@ -0,0 +1,24 @@ +# 060 Plan (wp3): preemptive Opus 5.5 rows + +050 closed with Kiro, opencode-zen/go and three fast tiers "not added" because the providers had +not published Opus 5.5. On 2026-09-23 the user asked to add them ahead of the providers +("우리가 선으로"). Direction changes from evidence-gated to preemptive for these rows only, each +labelled as such. + +| Site | Entry | Numbers | Note | +|---|---|---|---| +| `src/providers/kiro-models.ts` | `claude-opus-5.5` in `KIRO_MODELS` and context map | 1_000_000 | Kiro spelling (dot), mirrors claude-opus-5; visible in the static Kiro list, so a call can fail until Kiro ships it | +| `src/adapters/kiro/reasoning.ts` | `KIRO_NATIVE_EFFORT_FIELDS["claude-opus-5.5"] = "output_config"` | low..max | same native field as Opus 5 | +| snapshot `opencode-zen` | `claude-opus-5-5` from its claude-opus-5 row | 4 / 20 / 0.2 / 5, 1M / 128K | Zen roster is live, row inert until listed | +| snapshot `openrouter` | `anthropic/claude-opus-5.5-fast` from its opus-5-fast row | 8 / 40 / 0.4 / 10 | Anthropic fast price | +| snapshot `kilo` | `anthropic/claude-opus-5.5-fast` | 8 / 40 / 0.4 / 10, 1M / 128K | Kilo's 5.5 row carries list price | +| snapshot `venice` | `claude-opus-5-5-fast` | 9.6 / 48 / 0.48 / 12, 1M / 128K | Venice 1.2x markup derived from its 5.5 row | + +opencode-go is excluded: it carries no Claude model at all, so there is no Opus row to follow. +Kiro price: no overlay; `claude-opus-5.5` resolves through the vendor fallback's dot-to-dash step. + +Tests: kiro-adapter.test.ts is at its 2050-line cap, so `claude-opus-5.5` joins the existing +native-effort and 1M-context id lists in place. usage-cost asserts kiro, opencode-zen and the +OpenRouter fast row. Verify: kiro dir, usage-cost, metadata sync, registry parity, ratchet, +layout, typecheck, structure:check, privacy:scan. Then PR to dev and merge. + diff --git a/devlog/_plan/260923_p1_provider_state_continuity/000_overview.md b/devlog/_plan/260923_p1_provider_state_continuity/000_overview.md new file mode 100644 index 0000000000..55ee918775 --- /dev/null +++ b/devlog/_plan/260923_p1_provider_state_continuity/000_overview.md @@ -0,0 +1,63 @@ +# P1: provider state continuity + +Issues: #5563 (explicit compatibility settings lost on provider save) and #5618 (shadow-call +intercept keeps targeting a disabled or deleted provider). Carries #5614. + +## Problem + +Saving a provider rebuilds the stored row from the submitted body. `POST /api/providers` carries a +list of fields the add/edit form cannot send (`apiKeyPool`, `modelCosts`, context windows, pacing, +and others), but none of the five operator compatibility settings: + +- `preserveReasoningContentModels` +- `requiresReasoningPlaceholderModels` +- `foldDeveloperRoleToSystem` +- `reasoningWireFormat` +- `omitReasoningEffortWithToolsModels` + +After an unrelated overwrite a custom provider loses them, and a registry provider gets the +registry seed back through `enrichProviderFromCatalog`. `src/config/live-reconcile.ts` cannot +restore them: the disk row equals its baseline, so the three-way merge keeps the live value the +save just wrote. The symptom reads as an old bug returning, for example a gateway that rejects the +`developer` role failing Native Chat after an edit to something unrelated. + +Separately, `shadowCallIntercept.model` can name a provider that is later disabled or deleted. +Disabled: `routeModel` throws `Provider is disabled` and the request fails with a generic 404, +logged as `http_404`, on every helper call. Deleted: a `provider/model` target whose provider no +longer exists falls through to the terminal default-provider fallback and is sent, unannounced, +to a different destination with different credentials and cost. Neither management route says +that the intercept depended on the provider. + +## Decisions + +1. **Survive/reset contract (Part 1).** The five settings are operator compatibility choices + about one upstream. A POST overwrite that keeps the same destination carries each stored value + the request omits, including an explicit `[]` or `false`. A POST overwrite that changes the + destination carries none of them, and it also stops carrying the stored `apiKeyPool`, since + those keys were issued for the previous destination. Destination means the adapter, the + normalized base URL and the auth mode. The whole old row is never merged into the candidate. +2. **PATCH is a field mask.** It already keeps every field it does not name. It gains write + branches for the four settings it rejects today (the two reasoning lists from #5614, plus + `foldDeveloperRoleToSystem` and `reasoningWireFormat`), with `null` clearing a field. PATCH does + not apply the destination reset; it only changes the fields a request names. +3. **Shadow target lifecycle (Part 2).** Disabling or deleting a provider that the intercept + target resolves to succeeds, and the response carries `dependentShadowIntercept`, which + the dashboard shows as a warning. At request time an unavailable target (disabled provider, + unknown combo, or a slash-qualified target that would only resolve through the terminal + default-provider fallback) returns one `409 intercept_target_unavailable` without an upstream + attempt. It never passes through to native and never falls back to the default provider. + Combo and routing-profile targets keep their declared failover, which is the fallback the + operator approved. + +## Phases + +- [010_part1_compat_contract.md](./010_part1_compat_contract.md): Part 1 implementation and tests. +- [020_part2_shadow_target_lifecycle.md](./020_part2_shadow_target_lifecycle.md): Part 2 implementation and tests. +- [030_delivery.md](./030_delivery.md): PR, docs parity and exact-head CI. + +## Constraints + +No local suite, typecheck, build or proxy run in this lane (local checks: NOT RUN); evidence is +static reading plus hosted CI on the exact head. `src/server/management/provider-routes.ts` is +1892 lines and the ratchet flags any file above 2000 lines, so new logic goes into sibling modules +and the route keeps only call sites. diff --git a/devlog/_plan/260923_p1_provider_state_continuity/010_part1_compat_contract.md b/devlog/_plan/260923_p1_provider_state_continuity/010_part1_compat_contract.md new file mode 100644 index 0000000000..6457349cca --- /dev/null +++ b/devlog/_plan/260923_p1_provider_state_continuity/010_part1_compat_contract.md @@ -0,0 +1,70 @@ +# 010: Part 1, explicit compatibility settings survive provider saves + +## Field contract + +| Field | Unrelated POST overwrite (same destination) | POST overwrite to a new destination | PATCH | +|---|---|---|---| +| `preserveReasoningContentModels` | stored value carried when omitted, `[]` included | not carried (registry seed may fill) | set, `[]` kept, `null` clears | +| `requiresReasoningPlaceholderModels` | same | not carried | set, `[]` kept, `null` clears | +| `foldDeveloperRoleToSystem` | stored `true`/`false` carried when omitted | not carried | boolean, `null` clears | +| `reasoningWireFormat` | stored value carried when omitted | not carried (registry seed may fill) | `"gateway-object"`, `null` clears | +| `omitReasoningEffortWithToolsModels` | stored value carried when omitted | not carried | unchanged (already handled) | +| `apiKeyPool` (credential) | carried, as today | not carried | not writable (key endpoints) | + +A value the request sends always wins. All other fields the POST handler already carries keep +their current behavior; this unit does not widen or narrow them. + +Destination is the tuple (adapter, normalized base URL, auth mode). The base URL is normalized by +lowercasing the scheme and host and dropping trailing slashes. Auth mode counts only when the +request names one: the dashboard form sends `authMode` only for `key` and `forward`, and +registry enrichment never sets it, so an omitted value is not evidence of a new destination. + +## Diff plan + +- New `src/server/management/provider-overwrite-carry.ts`: + - `PROVIDER_COMPAT_CARRY_FIELDS`: the five field names, as a readonly tuple the tests import. + - `sampleSubmittedCompatFields(prov)`: `Object.hasOwn` snapshot, taken before + `enrichProviderFromCatalog` like the other `submitted*` samples. + - `sameProviderDestination(a, b)`: the destination comparison above. + - `compatFieldConfigError(raw)`: POST type checks when a client does send a field + (non-blank string arrays for the three lists, boolean for fold, `"gateway-object"` for wire + format). + - `carryProviderCompatFields(prov, live, submitted)`: when the destination is unchanged, copy + each omitted field from the live row (read after the DNS await, so a PATCH landing during the + wait is kept). When it changed, copy nothing: the candidate keeps only what the request sent + and what registry enrichment filled for the new destination. + - `applyProviderCompatPatchFields(rawBody, next)`: the PATCH branches for the reasoning lists + (from #5614), `foldDeveloperRoleToSystem` and `reasoningWireFormat`. +- `provider-routes.ts`: call the sampler before enrichment, run `compatFieldConfigError` beside + the other body checks, gate the existing `apiKeyPool` carry on `sameProviderDestination`, call + `carryProviderCompatFields` next to `restorePersistedAliasOverlays`, and call the PATCH helper from + `applyProviderPatchFields`. +- Carry #5614: its test file `tests/server/management-provider-reasoning-lists.test.ts` (PATCH + and dashboard-save cases, registry seed case, concurrent PATCH during DNS) is kept, adapted to + the helper. The branch commit carrying it has a `Co-authored-by` trailer for the #5614 author. +- New `tests/server/management-provider-compat-carry.test.ts`: for each of the five fields, seed a + custom provider, send an unrelated POST overwrite with the same name that omits the field, + reload with `loadConfig()`, route with `routeModel`, build the outgoing request with the chat + adapter, and assert the field's effect on the body. A second group changes the base URL, adapter + or auth mode and asserts none of the five fields and no `apiKeyPool` carry. + Both test files are registered in `scripts/test-layout/layout.json` and + `tests/fixtures/test-layout-expected.json`. + +## Effects asserted on the next outgoing request + +- fold: a native-chat passthrough body with a `developer` message goes out as `system` when the + stored value is `true`; or on the translated path `false` keeps `developer`. +- wire format: `reasoning: { enabled, effort }` instead of `reasoning_effort`. +- omit list: a tool-bearing request for a listed model carries neither `reasoning_effort` nor + `reasoning`. +- preserve list: an assistant tool-call continuation carries `reasoning_content` when the list + names the model. +- placeholder list: an explicit `[]` stops the placeholder that the preserve list would otherwise + imply. + +## Docs + +`docs-site/src/content/docs/reference/configuration/providers.md` gains a short "What a provider +save keeps" section with the table above, and the five rows say how PATCH clears them. The seven +translated pages get the same section. `structure/gui-and-management-api.md` records the contract +next to the provider routes. diff --git a/devlog/_plan/260923_p1_provider_state_continuity/020_part2_shadow_target_lifecycle.md b/devlog/_plan/260923_p1_provider_state_continuity/020_part2_shadow_target_lifecycle.md new file mode 100644 index 0000000000..d1b99801ea --- /dev/null +++ b/devlog/_plan/260923_p1_provider_state_continuity/020_part2_shadow_target_lifecycle.md @@ -0,0 +1,53 @@ +# 020: Part 2, shadow-call target lifecycle + +## Diff plan + +- `src/server/management/shadow-call-validation.ts`: add + `shadowInterceptProviderDependency(config, providerName)`, returning + `{ model, enabled }` when `config.shadowCallIntercept.model` resolves to that provider, else + `null`. Combo and routing-profile selectors return `null`: their pickers already skip a disabled + or missing member, and deleting a provider that a combo uses is refused today. A + `provider/model` or `alias/model` prefix naming the provider counts. Otherwise the target is + resolved with `routeModel` and the resolved provider name is compared. The dependency is computed + before the mutation, while the provider still resolves. +- `provider-routes.ts`: PATCH with `disabled: true` and DELETE add + `dependentShadowIntercept` to their success response when a dependency exists. Neither refuses: + the operator's choice stands, and the report tells them what it affects. +- New `src/server/responses/shadow-target-availability.ts`: + `shadowTargetUnavailableReason(config, model, route | error)` classifies a disabled provider, + an unknown combo, and a slash-qualified target whose route reason is the terminal + `default-provider` fallback (its prefix names no configured provider or alias). A bare target that + resolves through the default provider stays valid; one existing test depends on that. + `interceptTargetUnavailableResponse(model, reason)` returns `409` with + `{ error: { type: "invalid_request_error", code: "intercept_target_unavailable", message } }`. + The message names the target and says to pick another target or re-enable the provider. The + request log records the code. A server-log warning is printed once per target and reason. +- `request-prepare.ts` late intercept site: resolve the target inside its own try. Admission, + combo-exhaustion and policy errors keep their current handling; any other target failure, and + the default-fallback case, return the new response before any upstream attempt. The early combo + site already requires the combo to exist; a deleted canonical `combo/<id>` reaches the late site + and is classified there. +- `shadowCallTargetError` (PUT `/api/shadow-call-settings`) rejects the same default-fallback case, + so the dashboard cannot save a dangling target. +- GUI: `gui/src/pages/use-providers-crud.ts` reads `dependentShadowIntercept` from the disable + and delete responses and shows a warning notice. It adds one `prov.*` key, translated in every + locale catalog. + +## Tests + +New `tests/responses/shadow-intercept-target-lifecycle.test.ts`: + +- disable: the PATCH response reports the dependency; the next intercepted helper call returns + 409 `intercept_target_unavailable` with no fetch. +- delete: the DELETE response reports the dependency; the next helper call returns the same + error and is not sent to the default provider. +- re-enable: PATCH `disabled: false` has no report, and the next helper call is intercepted to + the target again. +- a disabled provider behind a combo target still fails over inside the combo. +- a bare target resolving through the default provider is still intercepted. + +## Docs + +`docs-site/src/content/docs/reference/configuration/server.md` (shadow-call section) and its +seven translations describe the report and the error. `structure/gui-and-management-api.md` +records the response field. diff --git a/devlog/_plan/260923_p1_provider_state_continuity/030_delivery.md b/devlog/_plan/260923_p1_provider_state_continuity/030_delivery.md new file mode 100644 index 0000000000..7e2706fa44 --- /dev/null +++ b/devlog/_plan/260923_p1_provider_state_continuity/030_delivery.md @@ -0,0 +1,12 @@ +# 030: Delivery + +- One branch, `codex/260923-p1-provider-state-continuity`, with ordered commits: roadmap, + Part 1 (the #5614 carry with its co-author trailer first), Part 2, docs. +- One PR to `dev` using `.github/PULL_REQUEST_TEMPLATE.md`, with `Closes #5563` and + `Closes #5618`. Verification records local checks as NOT RUN and cites hosted CI on the exact + head. +- GUI: the Part 2 change in `gui/src/pages/use-providers-crud.ts` is logic plus locale strings. + The lane cannot run the dashboard to take a screenshot, so the PR asks the coordinator for the + screenshot or the waiver. +- Completion: every required check on the head SHA completes successfully. On a failure, read the + failing job log, fix the cause, push again and re-read. diff --git a/devlog/_plan/260923_p2_update_continuity/000_roadmap.md b/devlog/_plan/260923_p2_update_continuity/000_roadmap.md new file mode 100644 index 0000000000..f7536654c4 --- /dev/null +++ b/devlog/_plan/260923_p2_update_continuity/000_roadmap.md @@ -0,0 +1,42 @@ +# 000 — Update and package-replacement continuity (roadmap) + +Issues: #5496 (a replaced package tree fences `/healthz`, so `ocx restart` cannot find the proxy it +tells the user to restart) and #5624 (a Windows `ocx update` fails during the npm install, keeps +the old version, and leaves a temp tree behind with a locked `bunx.exe`). + +Objective: an installed proxy that is updated or has its package tree replaced ends up on exactly +one healthy new runtime, or keeps the old install and service intact with a clear next step. + +## Work phases + +| Phase | Doc | Outcome | +|---|---|---| +| wp0 | this unit | Roadmap locked; no production change | +| wp1 | [010](./010_fenced_proxy_identity.md) | A fenced proxy is discoverable through attested identity by `ocx restart`, `ocx stop` and service stop; a real-order test and the structure invariant | +| wp2 | [020](./020_updater_leftovers_and_guidance.md) | Updater-owned staging leftovers are marked, swept and never block; unowned trees are never deleted; every failure names its next step; docs-site troubleshooting page | +| wp3 | — | One PR to `dev`, required CI green on the exact head | + +## Constraints for this lane + +- Local suite, focused tests, typecheck, build, install and the proxy itself are not run. + Evidence is static reading plus hosted CI on the exact head. +- File-size: `src/cli/index.ts` sits at 1992 lines against the 2000-line implicit cap and + `src/update/job.ts` at 1987. Changes there stay inline and near zero net lines; new logic + lives in sibling modules. +- New tests go in new sibling files registered in `scripts/test-layout/layout.json` and + `tests/fixtures/test-layout-expected.json`. + +## Out of scope + +- Redesigning the automatic drain-and-restart guard (`src/server/index/package-tree-guard.ts`). +- Changing the 2.59.0 updater that the #5624 report ran; only the current updater is hardened. +- Source checkouts and standalone binaries remain outside the fence. + +## Status + +| Phase | State | +|---|---| +| wp0 | Done: roadmap audited (near-pass, no blockers) and locked | +| wp1 | Done: ca7f4e3024 (fenced identity, tests, INV-FENCE-01) | +| wp2 | Done: 91a2f3c47f (owned leftovers, failure guidance) and 9d80cc61ac (troubleshooting page) | +| wp3 | PR #5643; hosted CI on the exact head is the proof (local checks: NOT RUN) | diff --git a/devlog/_plan/260923_p2_update_continuity/010_fenced_proxy_identity.md b/devlog/_plan/260923_p2_update_continuity/010_fenced_proxy_identity.md new file mode 100644 index 0000000000..456c53284c --- /dev/null +++ b/devlog/_plan/260923_p2_update_continuity/010_fenced_proxy_identity.md @@ -0,0 +1,86 @@ +# 010 — Fenced proxy identity for manual restart and stop (#5496) + +## What current `dev` already does + +`src/server/index/package-tree-guard.ts` wires the integrity guard to `acceptSystemRestart`: a +replacement that stays stable for the debounce enters drain-and-restart, a service child re-checks +service-home ownership before draining and before handoff, a failed admission retries, and +`server.stop()` vetoes a pending callback. `tests/ci-workflows/package-tree-restart-ownership.test.ts` +covers those edges with fakes. That part is kept as is. + +## Remaining gap + +The fenced `/healthz` answers 503 `restart_required`. Three things then refuse it: + +1. `proxyIdentityAt` (`src/server/proxy-liveness.ts`) returns null on any non-OK status, so + `findLiveProxy` reports no proxy and `ocx restart` falls through to a start that cannot bind. +2. The fenced body carries no attestation proof, and `requestBoundSystemRestart` + (`src/cli/system-restart-client.ts`) requires `proofResponse.ok` and `restartCapability`. +3. The version-skew guard compares the CLI version with the proxy's boot version. After a + replacement those differ by construction, although the in-place respawn runs the files now on + disk at the same path. + +A PID in a 503 body is attacker-controlled for anyone holding the port, so it cannot be trusted by +itself. + +## Design + +Identity is bound to owned state the CLI can check, separately from readiness. + +Server (`src/server/index/serve-options.ts`, fenced branch, `/healthz` only): +- add the attestation proof header when the request carries a challenge, computed exactly like + the healthy branch (`createLocalAttestationProof(secret, challenge, process.pid, port)`), over + the same port value the fenced body reports, which is the port the runtime record holds; +- add `restartCapability` and `installedVersion` (the version in the package manifest now on + disk, read through the guard) to the body; +- the message names the command that works: `ocx restart` (or `ocx service restart`). + +Guard (`src/lib/package-tree-integrity.ts`): optional `installedVersion()` on the guard interface; +the runtime guard reads `package.json` version (bounded semver or undefined); the option +`readInstalledVersion` is a test seam. `package-tree-guard.ts` forwards it. +Review follow-up: a readable manifest is not an install-completion signal, so `installedVersion()` +stays undefined until the guard's stability debounce has seen the same replacement identity for the +full interval, and again whenever the tree has moved since. + +CLI liveness (`src/server/proxy-liveness.ts`): +- `LivenessIo.acceptPackageTreeFenced` (opt-in). When set and the 503 body is an opencodex + `restart_required` body with `error.code === "package_tree_changed"` and an integer pid, + `proxyIdentityAt` reads the owned runtime record for that pid (`readRuntimeFn(pid)`), requires + `record.pid === pid`, `record.port === port` and an attestation secret, sends a fresh + challenge to `/healthz`, and verifies the proof with `verifyLocalAttestationProof`. + Any failure returns null (unverifiable identity is refused). +- the result and `LiveProxy` carry `packageTreeFenced: true`. +- default callers keep today's behaviour: a fenced proxy is not "live" for ensure, update health + waits or replacement waits. + +Restart client (`src/cli/system-restart-client.ts`): +- accept a 503 proof response only for a fenced body; everything else in the proof check is + unchanged; +- for a fenced body compare the CLI version with `installedVersion`; a missing or unbounded value + rejects with `restart_package_tree_unsettled` (retry after the install finishes); +- the pre-POST recheck passes `acceptPackageTreeFenced`. + +Callers that opt in: `ocx restart` discovery (`src/cli/index.ts` `handleProxyRestart`), the +`ocx stop` orphan fallback, service stop's orphan fallback and post-stop liveness in +`src/service/orchestration.ts`. `reportRestartFailure` gets a line for the new rejection code +through a helper in `system-restart-client.ts` so `src/cli/index.ts` stays under its cap. + +## Tests (new sibling files) + +- `tests/server/proxy-liveness-package-tree-fence.test.ts`: attested fenced identity is found only + with the opt-in; wrong secret, missing record, record pid/port mismatch, expected-pid mismatch and + a non-fence 503 are all refused. +- `tests/cli/system-restart-client-package-tree.test.ts`: fenced restart is accepted when the + installed version matches, rejected on skew and on an unsettled tree, and a non-fence 503 is + still rejected. +- `tests/ci-workflows/package-tree-fenced-restart.test.ts`: real `startServer` in order — boot + (200), replace the observed tree, fenced 503, automatic admission, manual discovery through the + real `findLiveProxy` and real `requestBoundSystemRestart` against the live listener (accepted as + already draining), one scheduled handoff, exactly one exit, and a service child that lost the + service home never hands off. + +## Structure + +`structure/ops/docs-and-release.md` "Package-tree integrity fence" gains the identity rule: the +fenced `/healthz` stays attestable, liveness accepts it only by opt-in and only with a proof +from the owned runtime record, and restart compares against the installed version. diff --git a/devlog/_plan/260923_p2_update_continuity/020_updater_leftovers_and_guidance.md b/devlog/_plan/260923_p2_update_continuity/020_updater_leftovers_and_guidance.md new file mode 100644 index 0000000000..eb2015e22a --- /dev/null +++ b/devlog/_plan/260923_p2_update_continuity/020_updater_leftovers_and_guidance.md @@ -0,0 +1,64 @@ +# 020 — Updater leftovers and failure guidance (#5624) + +## What is and is not established + +- The current npm updater (`bin/ocx.mjs` -> `transactionalNpmUpdate` in + `src/update/transactional-install.mjs`) stages into a sibling `.ocx-staging-<timestamp>` + prefix, verifies, and swaps with rollback. Our code never creates `@bitkyc08/.opencodex-*`; + that name is npm's own rename-aside during a direct global install. +- Stage cleanup is a `rmSync(..., { force: true })` whose errors are swallowed. A locked file (the + reported `bunx.exe`, EPERM) leaves the staging tree in place silently and nothing sweeps it. +- `mkdirSync(stageRoot, { recursive: true })` would reuse an existing directory of the same + name instead of failing. +- The ENOTDIR on `mkdir` in the report is not explained by this code. The job log withholds the + path, so the component that was a file is unknown. It is not attributed to the leftover tree. +- The GUI job records only `update command failed (N)`; the one recovery command is printed on + the child's stderr. + +## Design + +`src/update/transactional-install.mjs`: +- Staging directories are created exclusively (`mkdirSync` without `recursive`, a random suffix + after the timestamp) and immediately receive an ownership marker `.ocx-update-owner.json` + (`{ schema: 1, kind: "staging", pkgName, pid, createdAt }`). +- `removeOwnedStage(dir)` deletes every entry except the marker, retrying EPERM/EBUSY/EACCES + briefly; the marker is removed last and only when everything else is gone, so a partly locked + tree stays provably owned for the next sweep. +- `sweepUpdateLeftovers({ scopeDir, pkgName })` runs before staging. It removes only + `.ocx-staging-*` real directories whose marker parses with `kind: "staging"` and the same + `pkgName` and whose `createdAt` is older than a floor well above the npm install timeout, so a + concurrent update from another home sharing the prefix never loses its in-flight stage. It never + follows symlinks or junctions, never touches `.ocx-backup-*` (the boot + probe owns those), and reports unmarked `.ocx-staging-*` and npm's `.<name>-*` rename-asides as + not owned with their paths. A locked owned tree is reported and retried next time. The sweep never + fails the update. +- A post-swap verification failure moves the rejected tree into the owned stage before restoring the + backup, instead of deleting it in place, so a locked file there cannot turn a rollback into a + double fault. +- `launcherUsableAfterNpmUpdate(tx)` replaces the inline expression in `bin/ocx.mjs`. + +`src/update/update-failure-guidance.mjs` (new): `npmUpdateFailureGuidance({ phase, rolledBack, +pkgName, version })` returns the next step. Previous version kept: run `ocx update` again; if it +fails the same way, `ocx stop`, `npm install -g --allow-scripts=bun <pkg>@<version>`, then +`ocx service restart` (or `ocx start`). Double fault: restore with the command in +`.ocx-recovery.json` or reinstall. + +`bin/ocx.mjs` prints the guidance at the end of a failed npm update in place of the old +"Try manually" line. `src/update/job.ts` appends a +short fixed next step to `update command failed (N)` (net +1 line, file stays under 2000). + +## Tests (new sibling file `tests/update/update-transactional-leftovers.test.ts`) + +- Injected failure at each step (stage mkdir, npm stage install, staged verify, live->backup rename, + stage->live rename, post-swap verify, double fault): the live tree keeps the old version or is + rolled back, `launcherUsableAfterNpmUpdate` and `planStoppedRuntimeRecovery` restore the + previous service, and the guidance names the next command. +- A stage left behind by a locked file keeps its marker; the next update removes it and succeeds. +- Unmarked `.ocx-staging-*`, npm's `.opencodex-*`, a marker for another package, and a symlink or + junction named like a stage are all left intact. + +## Docs + +- `docs-site/src/content/docs/troubleshooting/update-failed.md` (English) with the sidebar entry, + plus locale pages that say the same thing. +- `reference/cli/lifecycle.md` links to it from the update section. diff --git a/devlog/_plan/260923_p3_azure_opaque_recovery/000_roadmap.md b/devlog/_plan/260923_p3_azure_opaque_recovery/000_roadmap.md new file mode 100644 index 0000000000..a3345170a9 --- /dev/null +++ b/devlog/_plan/260923_p3_azure_opaque_recovery/000_roadmap.md @@ -0,0 +1,71 @@ +# 260923 P3 — Azure OpenAI opaque reasoning recovery (#5583) + +## Problem + +Moving an existing Responses conversation onto an `azure-openai` provider replays reasoning state +the previous provider minted. Azure answers `400 invalid_encrypted_content`, and the proxy never +recovers, although the same rejection on `openai-responses` gets one sanitized resend. + +Two defects, both confirmed from source on `origin/dev` e9643875f0: + +1. **Gate by name.** `src/server/responses/core-opaque-recovery.ts` admits recovery only when + `adapterName === "openai-responses"`, in `shouldAttemptOpaqueBlobRecovery` and in + `opaqueBlobRejectionBodyForRecovery`. `createAzureAdapter` spreads the Responses passthrough + and renames it `azure-openai`, and both dispatchers pass `adapter.name` into the gate + (`passthrough-dispatch.ts` HTTP and streamed recovery, `adapter-dispatch.ts`). The adapter + registry already declares the relationship: `azure` and `azure-openai` carry + `contractParent: "openai-responses"`, and `resolvedAdapterWire()` resolves it. +2. **The item id survives the blob.** Recovery sets `_stripReasoningEncryptedContent`; the + passthrough's `sanitizeReasoningInputContent` then deletes `encrypted_content`, but + `stripInvalidItemIds` keeps a well-formed `rs_*` id and `stripItemIdsWhenUnstored` removes ids + only when `store === false`. With `store` omitted or true, the resend still names an item the + previous identity stored, and a stateful Responses destination resolves it against its own store: + `Item with id 'rs_…' not found.` The reporter observed exactly this; the code path confirms it. + +## Decisions + +- Gate on the wire the registry declares (`resolvedAdapterWire(adapterName) === "openai-responses"`), + keeping the `adapterName` argument. A future wrapper that declares the same `contractParent` is + covered by construction; `openai-chat` and every other wire stay excluded. +- Drop the reasoning item `id` together with the blob only after the destination itself rejected + foreign opaque state: the recovery rebuild (`prepareOpaqueBlobRecovery`) and the five-minute + rejection memo that strips pre-flight on later turns. A new request flag + `_dropForeignReasoningItemIds` carries that signal. A plain proven route switch keeps its + current behaviour (blob removed, item and id kept), which an existing passthrough test pins. +- Keep the item and its summary, as openai-responses recovery already does; only the two opaque + fields minted by the rejected identity go. +- Recovery stays narrow: status, rejection identities, the single-shot guard and the send budget + are unchanged. + +Out of scope, recorded for the coordinator: `transientRetryPolicyFor` and the Lab +`upstreamProtocolForAdapter` table also key on adapter names and treat `azure-openai` +differently from its declared Responses contract. Neither affects this recovery path. + +## Work-phases + +| Id | Doc | Outcome | +| --- | --- | --- | +| wp-1 | this file | Roadmap locked (docs only) | +| wp-2 | [010_source.md](010_source.md) | Contract gate and rejected-id drop | +| wp-3 | [020_tests.md](020_tests.md) | Sibling regression file, layout registration | +| wp-4 | [030_docs.md](030_docs.md) | structure and docs-site sync | +| wp-5 | [040_delivery.md](040_delivery.md) | One PR to dev, exact-head CI | + +Review follow-up inside wp-5: [050_review_followup.md](050_review_followup.md). + +## Verification policy + +Local checks: NOT RUN (lane rule: no local suite, focused tests, typecheck, build, install or +proxy). Evidence is static source reading plus hosted CI on the exact PR head SHA. + +## Audit (wp-1 A) + +An independent read-only audit confirmed every claim above from source: the two name gates +(`core-opaque-recovery.ts` `shouldAttemptOpaqueBlobRecovery` and +`opaqueBlobRejectionBodyForRecovery`), the `contractParent` declarations and +`resolvedAdapterWire`, the surviving `rs_*` id when `store` is not `false`, and the import edge +already present through `request-prepare.ts`. No existing recovery or memo test carries a reasoning +id, so the rejected-id drop changes no current expectation. Two adjacent name checks sit outside the +recovery path and stay unchanged: `mandatoryResponsesReasoningReplayUnavailable` in +`core-replay.ts` (combo plaintext eligibility) and the OAuth-pool 429 scope rebind in +`passthrough-dispatch.ts`, which a key-auth Azure provider never reaches. Verdict: PASS. diff --git a/devlog/_plan/260923_p3_azure_opaque_recovery/010_source.md b/devlog/_plan/260923_p3_azure_opaque_recovery/010_source.md new file mode 100644 index 0000000000..899c495e7d --- /dev/null +++ b/devlog/_plan/260923_p3_azure_opaque_recovery/010_source.md @@ -0,0 +1,41 @@ +# 010 — Source: contract gate and rejected reasoning id drop (wp-2) + +## `src/server/responses/core-opaque-recovery.ts` + +- Import `resolvedAdapterWire` from `../../responses/continuation-ownership` (already in the + `core.ts` graph through `request-prepare.ts`; not a `./` sibling, so the owner graph is unchanged). +- Add `adapterSpeaksResponsesWire(adapterName: string): boolean` returning + `resolvedAdapterWire(adapterName) === "openai-responses"`, with a comment naming #5583 and the + `contractParent` inheritance. +- Replace `args.adapterName === "openai-responses"` in `shouldAttemptOpaqueBlobRecovery` and + `adapterName !== "openai-responses"` in `opaqueBlobRejectionBodyForRecovery` with the helper. +- In `prepareOpaqueBlobRecovery`, set `parsed._dropForeignReasoningItemIds = true` beside + `_stripReasoningEncryptedContent`. + +## `src/types/request.ts` + +- Add `_dropForeignReasoningItemIds?: boolean` after `_stripReasoningEncryptedContent`, documenting + the two setters and the `Item with id … not found` failure it prevents. + +## `src/server/responses/core-replay.ts` + +- In `bindRouteReasoningReplayScope`, the `reasoningReplayOpaqueBlobRejectionMemoized` branch also + sets `_dropForeignReasoningItemIds`. The serving-identity-change branch does not. + +## `src/adapters/openai-responses/reasoning.ts` + +- `sanitizeReasoningInputContent` gains `dropForeignItemId?: boolean` (named `dropStrippedItemId` in + the first revision; see 050). When true and the item's + `encrypted_content` is being removed, also delete `id`. Items that keep their blob, or never had + one, keep their id. + +## `src/adapters/openai-responses/passthrough.ts` + +- Pass `dropForeignItemId: parsed._dropForeignReasoningItemIds === true` into the existing + `sanitizeReasoningInputContent` call. Azure inherits this through `inner.buildRequest`. + +## Invariants kept + +- 5xx other than the exact function-output 502, non-self-identified 4xx, blobless bodies and a + second rejection never enter recovery. +- The call sites and `rebuildAndRefetch` budget plumbing are untouched. diff --git a/devlog/_plan/260923_p3_azure_opaque_recovery/020_tests.md b/devlog/_plan/260923_p3_azure_opaque_recovery/020_tests.md new file mode 100644 index 0000000000..bc5c004252 --- /dev/null +++ b/devlog/_plan/260923_p3_azure_opaque_recovery/020_tests.md @@ -0,0 +1,31 @@ +# 020 — Tests (wp-3) + +New sibling file `tests/responses/responses-azure-opaque-blob-recovery.test.ts` (the existing +recovery file is 1,852 lines against the 2,000-line new-file threshold). Register it in +`scripts/test-layout/layout.json` `explicit` and `tests/fixtures/test-layout-expected.json` under +`responses`. + +Fixture: two key-auth providers, `azure` (`adapter: "azure-openai"`, +`https://azure.example.test/openai/v1`) and `responses` (`adapter: "openai-responses"`). The +conversation carries a reasoning item with `id: "rs_foreign_backend"`, a summary and a foreign +`encrypted_content`, with `store` omitted so ids are not stripped by the unstored rule. + +Cases, parameterized over both adapters where the criterion says "the same test": + +1. One recovered send: first body carries blob and `rs_*` id, rejection is the reported Azure + `invalid_encrypted_content` body, second body carries neither, reasoning summary and user + message survive, `sendCount` 2, `recoveryKinds` `["opaque-blob-rejection"]`. +2. Later turns: a fake upstream rejects blobs with the opaque identity and rejects a surviving + `rs_*` id with `Item with id … not found`; three turns in one session give sends + `[blob+id, clean, clean, clean]` and every turn returns 200 (memo path drops the id too). +3. Trigger unit: `shouldAttemptOpaqueBlobRecovery` accepts `azure` and `azure-openai`, rejects + `openai-chat`, an unknown adapter, an ordinary 400, a 429 and 500/503 for `azure-openai`. +4. Path unit: `attemptOpaqueBlobRecovery` with an `azure-openai` 400 unrelated body, a 429 and a + 503 never calls `rebuild`; with the opaque rejection it calls it once and a second call on the + same guard is skipped. +5. Budget: a request budget with one total send yields exactly one upstream send and no second + send on Azure (the rebuild is refused by the shared ledger, not granted a fresh allowance). +6. Second rejection: repeated Azure rejection surfaces 400 after exactly two sends. + +Constants are derived: rejection bodies are local fixtures; the wire name comes from +`resolvedAdapterWire`, not restated. diff --git a/devlog/_plan/260923_p3_azure_opaque_recovery/030_docs.md b/devlog/_plan/260923_p3_azure_opaque_recovery/030_docs.md new file mode 100644 index 0000000000..ab889d0c91 --- /dev/null +++ b/devlog/_plan/260923_p3_azure_opaque_recovery/030_docs.md @@ -0,0 +1,14 @@ +# 030 — Docs (wp-4) + +- `structure/providers/chat-compat.md`: in the opaque-blob recovery paragraph, state that recovery + covers every adapter whose registry contract resolves to the Responses wire (`openai-responses` + and its `contractParent` wrappers, today `azure`/`azure-openai`), and that after the destination + itself rejected foreign opaque state (recovery rebuild and rejection memo) the reasoning item's + `id` is removed with its blob; a proven route switch alone keeps the id. +- `structure/adapters/registry.md`: note that behaviour keyed to a wire resolves the adapter through + `effectiveAdapterContract()` / `resolvedAdapterWire()`, with opaque-blob recovery as a consumer. +- `docs-site/src/content/docs/reference/proxy-formats.md` (Encrypted-content hygiene): add a + troubleshooting paragraph for switching providers mid-conversation, including Azure. +- `docs-site/src/content/docs/reference/adapters.md` `azure-openai` section: one bullet that it + shares the Responses opaque-state recovery. Translated locales carry no statement about this + recovery, so none contradicts the English source; they are left for the translation workflow. diff --git a/devlog/_plan/260923_p3_azure_opaque_recovery/040_delivery.md b/devlog/_plan/260923_p3_azure_opaque_recovery/040_delivery.md new file mode 100644 index 0000000000..eeec5e7b33 --- /dev/null +++ b/devlog/_plan/260923_p3_azure_opaque_recovery/040_delivery.md @@ -0,0 +1,8 @@ +# 040 — Delivery (wp-5) + +- Commits in order: roadmap docs, source, tests + layout, docs. Push with `--no-verify` to + `codex/260923-p3-azure-opaque-recovery`. +- One PR to `dev` using `.github/PULL_REQUEST_TEMPLATE.md` (Summary, Verification, Checklist), + `Closes #5583`, local checks recorded as NOT RUN. +- Read required CI on the exact head SHA; read failing job logs, fix causes, push, re-read. +- Do not merge, close issues or delete branches. diff --git a/devlog/_plan/260923_p3_azure_opaque_recovery/050_review_followup.md b/devlog/_plan/260923_p3_azure_opaque_recovery/050_review_followup.md new file mode 100644 index 0000000000..fdfc4ca7f7 --- /dev/null +++ b/devlog/_plan/260923_p3_azure_opaque_recovery/050_review_followup.md @@ -0,0 +1,43 @@ +# 050 — Review follow-up on PR #5639 (wp-5) + +Hosted CI on 873c0e04a1 passed every required job. Automated review raised four findings, all +verified against source before changing anything. + +| Finding | Verdict | Change | +| --- | --- | --- | +| Docs said no 5xx is retried; `shouldAttemptOpaqueBlobRecovery` admits the exact encrypted function-output 502 | Valid | English subsection names the one 502 exception | +| Locale `proxy-formats.md` pages lacked the new subsection while their adapter notes described it | Valid | Translated subsection appended to all seven locales | +| (duplicate of the 502 finding) | Valid | Same change | +| A proven route switch strips the blob but keeps the `rs_*` id; with `store` omitted Azure answers `Item with id … not found`, and recovery cannot run because the body no longer carries a blob | Valid | See below | + +## Proven switch across stores + +The reviewer proposed a second single-shot recovery keyed on the "Item not found" answer. That adds +a rejection identity and a round trip, and this lane keeps recovery to the existing opaque-state +identities. The id is foreign exactly when the item store changed, and the serving record already +knows that: its tuple is provider, durable destination, adapter, model, durable credential. + +- `reasoningReplayItemStoreChanged(scope)` in `src/responses/reasoning-replay-cache.ts` compares the + durable destination and credential of the last serving route with the current one. +- `bindRouteReasoningReplayScope` sets the id-drop flag, renamed `_dropForeignReasoningItemIds`, when + the serving identity changed and the store changed. A model or adapter change on the same + destination and credential still strips the blob and keeps the id, which that store can resolve. +- The existing passthrough test that sets `_stripReasoningEncryptedContent` directly is untouched. + +New cases in `tests/responses/responses-azure-opaque-blob-recovery.test.ts`, on both adapters: a +session first served by the other provider moves over with one clean send; a model change on the +same destination keeps the id and drops the blob. + +## Second round (head 2987b26b9e) + +CI on 2987b26b9e passed every required job. Two further findings: + +- **Id-only reasoning items kept their id** after a store change, because the drop was tied to + stripping a blob. Valid: a stateful destination resolves a blobless `rs_*` id the same way. The + sanitizer option is now `dropForeignItemId` and removes the id of every reasoning item while the + flag is set; the item and its summary stay. New case: an id-only item after a proven switch. +- **`responsesPath` is not part of the durable destination identity.** Declined here. + `durableReplayDestinationIdentity` also keys the persisted thought-signature store, so widening it + re-keys durable state repository-wide. The gap needs two providers with the same base URL and + credential whose different `responsesPath` values back separate item stores. It is a separate + change to the replay identity contract, not part of this recovery fix. diff --git a/devlog/_plan/260923_p4_ci_pr_finish/000_plan.md b/devlog/_plan/260923_p4_ci_pr_finish/000_plan.md new file mode 100644 index 0000000000..0be631ad15 --- /dev/null +++ b/devlog/_plan/260923_p4_ci_pr_finish/000_plan.md @@ -0,0 +1,50 @@ +# 000 — Finish three contributor CI pull requests in place + +## Objective + +Land three open contributor pull requests on their own branches instead of replacing them: +#5469 (privacy scan on devlog-only changes, issue #5468), #5456 (Bun test batches without GNU +`timeout`) and #5471 (latest-dev readiness wording, issue #4443). Each branch receives +maintainer-edit commits, is brought onto current `dev` by merge, and is judged by the full +required CI at its exact head. No new pull request is opened; merge, labels that attest a +maintainer review, and issue closure stay with a maintainer. + +## Constraints + +- Local checks are not run (no suite, focused test, typecheck, build or install). Evidence is + the diff read statically plus hosted CI at the exact head SHA. +- A hosted failure is fixed at its cause, never by weakening the assertion. +- `.github/workflows/ci.yml` is a security boundary: no new permissions, no secrets in logs, + no mutable action refs, no new `pull_request_target` surface. +- Test files at a size cap get a sibling file registered in `scripts/test-layout/layout.json` + and `tests/fixtures/test-layout-expected.json`. Counts and constants are derived, not restated. + +## Starting state (dev at e9643875f0) + +| PR | Head | Conflicts with dev | Hosted state | +|---|---|---|---| +| #5469 | f32b7cafca | `ci.yml`, `ci-structure-gate.test.ts` | `hygiene` and `enforce-target`: `unsponsored_surface`; Cross-platform CI awaiting approval | +| #5456 | ec95a6029b | `run-bun-test-batches.sh`, `ci-crash-disposition.test.ts` | review open: the fallback drops the batch deadline | +| #5471 | a5810630eb | `pr-quality.cjs`, `pr-quality.test.cjs` | review open: the #4443 overlap wording is a maintainer call | + +Both #5469 failures are one rule: a `.github/workflows/` change without the +`maintainer-sponsored` label. That label records a maintainer security review, so it is +reported for a maintainer decision rather than applied by this lane. + +## Work-phase map + +The three pull requests share no file, so after this roadmap they are independent. + +| Doc | Work-phase | Verifiable close | +|---|---|---| +| [010](010_pr5469_privacy_gate.md) | #5469 privacy gate onto dev | Cross-platform CI at the new head, including `ci-privacy-gate`, `ci-review-lanes`, `ci-scope-reduction` and the test-layout guards | +| [020](020_pr5456_batch_deadline.md) | #5456 portable batch deadline onto dev | Cross-platform CI at the new head, including the executed `ci-crash-disposition` runner cases | +| [030](030_pr5471_latest_dev_wording.md) | #5471 derived latest-dev wording onto dev | Cross-platform CI at the new head, including the `.github/scripts` node suites and the enforce-pr-target harness tests | + +## Delivery per pull request + +1. Local branch from the contributor head, merge `origin/dev`, resolve, commit the fix. +2. `git push --no-verify` to the contributor branch (maintainer edits are allowed on all three). +3. Approve the awaiting Cross-platform CI run for that exact run id, then read every job. +4. Leave a short English comment on the pull request stating what was pushed and why, and + which workflow security checks were made where applicable. diff --git a/devlog/_plan/260923_p4_ci_pr_finish/010_pr5469_privacy_gate.md b/devlog/_plan/260923_p4_ci_pr_finish/010_pr5469_privacy_gate.md new file mode 100644 index 0000000000..be604423fa --- /dev/null +++ b/devlog/_plan/260923_p4_ci_pr_finish/010_pr5469_privacy_gate.md @@ -0,0 +1,110 @@ +# 010 — #5469: privacy scan on devlog-only changes + +## Acceptance + +- (a) A devlog-only pull request runs `privacy:scan`, through `privacy-gate` alone. +- (b) No event runs the scan twice: `gates` and `privacy-gate` are never both selected. +- (c) The aggregate `ci` cannot conclude success on a devlog change whose privacy gate did + not run, including when the `privacy` filter output is missing or malformed. + +## Findings against the branch head + +1. `privacy-gate` stands down only on `ci == 'true'`, but `gates` also runs on every + non-pull-request event. Today every push that starts this workflow and every dispatch + reads `ci=true`, so no double run is reachable yet; making the job the exact complement of + `gates` keeps that true if a trigger or the push `paths` list changes. +2. `privacy` is consumed straight from the filter. A missing value reads as "not changed", + skips the job and leaves the aggregate green. `dev` already re-emits `ci` through a + validating step for this reason; `privacy` needs the same. +3. `ci-review-lanes.test.ts` executes the real aggregate step under `set -u` with every + `needs` job successful and no `CHANGES_PRIVACY`; it would abort on the unbound variable. +4. `ci-privacy-gate.test.ts` is not registered in the two test-layout files. +5. Extending the pinned `GATED_JOBS` line forces an edit to `ci-structure-gate.test.ts`; a + separate line for `privacy-gate` keeps that sibling test untouched. + +## Changes + +MODIFY `.github/workflows/ci.yml` + +```diff + outputs: +- privacy: ${{ steps.filter.outputs.privacy }} ++ privacy: ${{ steps.scope.outputs.privacy }} + ... + - name: Assert the scope output is usable + env: + CI_SCOPE: ${{ steps.filter.outputs.ci }} ++ PRIVACY_SCOPE: ${{ steps.filter.outputs.privacy }} + run: | + ...existing ci case unchanged... ++ case "$PRIVACY_SCOPE" in ++ true|false) printf 'privacy=%s\n' "$PRIVACY_SCOPE" >> "$GITHUB_OUTPUT" ;; ++ *) printf '::error::changes.outputs.privacy was %q, expected true or false\n' "$PRIVACY_SCOPE"; exit 1 ;; ++ esac + ... + privacy-gate: +- if: needs.changes.outputs.privacy == 'true' && needs.changes.outputs.ci != 'true' ++ if: github.event_name == 'pull_request' && needs.changes.outputs.ci != 'true' && needs.changes.outputs.privacy == 'true' + ... +-if [ "$CHANGES_PRIVACY" = "true" ] && [ "$CHANGES_CI" != "true" ]; then ++if [ "$scoped" = not-requested ] && [ "$CHANGES_PRIVACY" = "true" ]; then + privacy=requested + fi + ... +-GATED_JOBS="$GATED_JOBS structure-gate privacy-gate widget" ++GATED_JOBS="$GATED_JOBS structure-gate widget" ++GATED_JOBS="$GATED_JOBS privacy-gate" +``` + +The conflict with `dev` is resolved by keeping every `dev` change (native outputs and +matrices, `CHANGES_NATIVE`, the native `expected_for` arm) and re-applying the lines above. + +MODIFY `tests/ci-workflows/ci-structure-gate.test.ts`: take `dev`'s version unchanged. + +MODIFY `tests/ci-workflows/ci-review-lanes.test.ts`: in the executed release-gates case, add +`results["privacy-gate"] = { result: "skipped" }` and `CHANGES_PRIVACY: "false"`. The job is +a producer the dispatch did not request, exactly like `structure-gate`. + +MODIFY `tests/ci-workflows/ci-privacy-gate.test.ts` + +- Evaluate the checked-in `if:` of `gates` and `privacy-gate` over every + event, `ci` and `privacy` combination: at most one selected, exactly one when + `privacy` is true, and `privacy-gate` alone for a devlog-only pull request. +- Execute the scope step: a `PRIVACY_SCOPE` of `""` or `maybe` exits 1; `true` writes + `privacy=true`. +- Execute the aggregate step for a devlog-only pull request: `privacy-gate` success passes, + `privacy-gate` skipped fails naming the job. For a `ci.yml` pull request, `privacy-gate` + reporting success fails the aggregate as a run it did not request. +- Replace the verbatim `GATED_JOBS` and aggregate-condition string pins with those executions. + +MODIFY `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`: add +`"ci-privacy-gate.test.ts": "ci-workflows"` in sorted position. + +## Out of scope + +The `maintainer-sponsored` label: it attests a maintainer security review and is left for a +maintainer. The `push` trigger's `paths` stay equal to the `ci` filter. + +`docs`, `structure`, `gui` and `packaging` are also consumed unvalidated from the filter. +Closing that class belongs in its own change; it would alter pins in `ci-structure-gate.test.ts` +and widen this pull request beyond the privacy gate. + +## Outcome + +Delivered on the contributor branch as the merge commit `3f4d7ad4fe` (dev `0f9254b564` merged). +Cross-platform CI run 35819557849 at that head concluded success on every job, with +`privacy gate` skipped as expected: this pull request edits `ci.yml`, so `ci` is true and +`gates` ran the scan. The new executed cases passed on Linux and macOS: + +- `runs at most once for every event and scope, and exactly once for a devlog change` (b, a) +- `is green on a devlog-only pull request only when the privacy gate ran` (a, c) +- `a missing or malformed privacy output fails the changes job` (c) +- `rejects a second scan on a pull request that gates already scans` (b) + +One more defect surfaced while merging: the branch's test pinned `ci.yml` as a literal entry of +the `ci` filter, which on `dev` lists `.github/workflows/**` instead. The condition matrix +replaced it. + +Left for a maintainer: the `maintainer-sponsored` label (`hygiene`, `enforce-target`), and the +author's re-attestation of the readiness checklist, which `enforce-target` now requires because +the body still carries the previous first item. diff --git a/devlog/_plan/260923_p4_ci_pr_finish/020_pr5456_batch_deadline.md b/devlog/_plan/260923_p4_ci_pr_finish/020_pr5456_batch_deadline.md new file mode 100644 index 0000000000..b54b7a3f1a --- /dev/null +++ b/devlog/_plan/260923_p4_ci_pr_finish/020_pr5456_batch_deadline.md @@ -0,0 +1,52 @@ +# 020 — #5456: keep the batch deadline without GNU timeout + +## Acceptance + +Without GNU `timeout`, each batch keeps the same `BUN_TEST_BATCH_TIMEOUT_SECONDS` ceiling and +the same exit disposition as `timeout --signal=TERM --kill-after=GRACE SECS`: 124 when the +command ends after TERM, 137 when KILL was needed. A hung batch is killed together with its +child processes. With neither GNU `timeout` nor a way to start a process group, the runner +still exits 69 as before. + +## Review items on the pull request + +- The fallback runs Bun with no batch deadline. Valid; fixed below. +- `sort -z` is GNU-only. Not reproduced: the macOS system `sort` (2.3-Apple) accepts `-z` + and sorts NUL-delimited input. +- The test covers only a green run. Valid; a hung-batch case is added. +- `mapfile` is Bash 4 only. Already replaced on `dev`; the merge takes `dev`'s loop. + +## Changes + +MODIFY `scripts/ci/run-bun-test-batches.sh` + +- Replace the hard `command -v timeout` requirement with a probe: + GNU (`timeout --signal=TERM --kill-after=1s 1s true` succeeds) selects `gnu`; + otherwise `perl` present selects `portable` with a `::notice::` naming the kept deadline; + otherwise exit 69 with the old message extended to name the portable option. +- Add `run_with_batch_deadline SECS GRACE cmd...`: + - start `cmd` through `perl -e 'setpgrp(0, 0) ...; exec { $ARGV[0] } @ARGV'` in the + background, so the batch leads its own process group as it does under GNU `timeout`; + - a watchdog subshell with output sent to `/dev/null` (it must not hold the `tee` pipe) + sleeps SECS, marks the timeout, sends TERM then CONT to the group, waits up to GRACE + seconds for the group to empty, then sends KILL to whatever is left; + - the wrapper forwards INT, TERM and HUP to the group, waits for the command, and on a + timeout waits for the watchdog too; it returns 137 when the command itself needed KILL, + 124 for any other timed-out end, and the command's own status otherwise. +- `run_test_once` calls GNU `timeout` or the wrapper with the same arguments, keeping + `PARALLEL_ARG` from `dev`. +- The wrapper stays a pipeline element (`wrapper ... 2>&1 | tee`) rather than writing through a + process substitution: bash waits for every pipeline member, so the log the crash classifier + reads is complete when `PIPESTATUS[0]` is taken. + +MODIFY `tests/ci-workflows/ci-crash-disposition.test.ts` + +- A `timeoutTool` option selects the fake `timeout`; the non-GNU fake rejects GNU options + the way a BSD `timeout` does. +- Two hang modes for the fake `bun`, multi-file batches only so the attribution singletons + stay clean: `hang` starts a child that ignores TERM, records its pid and blocks while the + batch process itself still dies on TERM; `hang-ignore-term` makes the batch process itself + ignore TERM. +- New cases without GNU `timeout`, under a 1 s deadline and 1 s grace: a clean run is green; + `hang` exits 124 and the TERM-ignoring child is gone afterwards; `hang-ignore-term` exits + 137, the GNU status when KILL was needed. diff --git a/devlog/_plan/260923_p4_ci_pr_finish/030_pr5471_latest_dev_wording.md b/devlog/_plan/260923_p4_ci_pr_finish/030_pr5471_latest_dev_wording.md new file mode 100644 index 0000000000..952a6063cb --- /dev/null +++ b/devlog/_plan/260923_p4_ci_pr_finish/030_pr5471_latest_dev_wording.md @@ -0,0 +1,29 @@ +# 030 — #5471: the latest-dev box states the condition the gate enforces + +## Acceptance + +The latest-dev readiness item is built from `READINESS_LATEST_DEV_BEHIND_MAX` and never +restates the number. Rewording it leaves open pull requests' checklists and ticks untouched. + +## Verified on dev + +- The untick message already derives the number: `pr-quality-messages.cjs` interpolates + `READINESS_LATEST_DEV_BEHIND_MAX`. +- The re-attestation migration compares only the first item + (`firstReviewReadinessItem(body) === REVIEW_READINESS_ITEMS[0]`), so rewording item 1 does + not trigger it. +- `extractReviewReadiness` reads box count and checked state, never item text. + +## Changes + +MODIFY `.github/scripts/pr-quality.cjs`: merge `dev`, keeping `dev`'s item 0 +(`Required local validation passed; ...`) and the branch's `latestDevReadinessItem()` for item 1. + +MODIFY `.github/scripts/pr-quality.test.cjs`: merge `dev`; in the legacy-checklist case use +`REVIEW_READINESS_ITEMS[0]` for the first line, so the case isolates the old item 1 wording +from the separate item 0 migration. + +## Left for a maintainer + +Whether #4443 also needs a line asking authors to check intervening `dev` changes for +overlap, or whether this pull request closes it as written. diff --git a/devlog/_plan/260923_p5_ci_release_gaps/000_plan.md b/devlog/_plan/260923_p5_ci_release_gaps/000_plan.md new file mode 100644 index 0000000000..21b63b8fae --- /dev/null +++ b/devlog/_plan/260923_p5_ci_release_gaps/000_plan.md @@ -0,0 +1,135 @@ +# 260923 P5 — CI and release scope gaps + +## Objective + +Remove the CI and release waste and blind spots found after the 2.61.0 and 2.63.0 releases +without widening routine pull-request CI. One branch, `codex/260923-p5-ci-release-gaps`, cut from +`origin/dev`, ordered commits, one pull request to `dev`. + +## Evidence that shaped the plan + +### Release version conflict (run 35783865160) + +| Event | Time (UTC, 2026-09-22) | +| --- | --- | +| Preview release run 35781975066 created | 20:42:00 | +| Stable release run 35783865160 (2.62.0, main) created | 21:00:02 | +| `v2.63.0-preview.20260923` GitHub release published by the preview run | 21:06:11 | +| Preview run finished | 21:06:14 | +| Stable run's first job (`validate-dispatch`) started | 21:06:35 | +| Stable run's `publish` failed at "Refuse a release the current tag set already outranks" | 21:25:36 | + +The stable run did not start work until the preview run released the workflow-level +`concurrency: { group: release, cancel-in-progress: false }` slot. The runs were already +serialised. The conflicting tag existed before the stable run's first job, and the ordering gate +that rejects it ran only in `publish`, after about nineteen minutes of packaging and +verification. The waste is check placement, not a missing lock. The fix is a cheap +`preflight` job that runs first, while the `publish` copy stays as the final authority because +tags, releases and npm state can still move while a run packages (a manual tag push, a first +local publish). No new coordination is added; the existing shared group is pinned by a test so +it cannot silently become per-ref, which is what would let a stable and a preview run overlap. + +### Registry confirmation + +`Post-publish registry smoke` retries `npm view <pkg>@<version> version` six times, warns +"Registry lookup not confirmed", records `verification=pending`, and the run continues to the +GitHub release. `npm dist-tag ls` is printed to the log only. A green run therefore reads the +same whether or not the registry and the dist-tag were ever read back. + +### Shard imbalance (run 35816902207, four Linux shards) + +Per-file durations were measured from the timestamps of Bun's per-file `##[group]` and +`##[endgroup]` lines in the hosted job logs (1,558 files, 1,261 s of test time). Sorted +round-robin by count gives 364 / 394 / 248 / 254 s of test time per shard. Greedy +longest-first assignment to the least-loaded shard gives 315 / 315 / 315 / 315 s. With equal +weights the same greedy pass over path-sorted files reproduces round-robin exactly, which +makes "unknown duration" a deterministic, backwards-compatible fallback. + +Duration-balanced shards change which files share a twelve-file process. The simulated largest +batch rises from 74 s to 89 s against the 120 s Linux process timeout. Closing a batch once its +predicted duration would pass half the process timeout brings the largest batch back to 70 s +(one file that is already about 70 s on its own) at a cost of one or two extra processes per +shard. + +### Scope gaps + +The `changes` job's `ci` filter omits `.github/actions/**` and `native/**`. A pull request +that touches only `.github/actions/setup-project-bun/action.yml` or only +`native/remote-workspace-helper/**` runs no job that exercises what it changed, and the +aggregate `ci` check reports success over skips. + +## Constraints + +- Security boundary (AGENTS.md, MAINTAINERS.md): no new workflow or job permission beyond + `contents: read` for new jobs, no secret in a log, only SHA-pinned actions that the repository + already uses, no `pull_request_target` surface. +- Ordinary PR CI time must not grow; test shard count stays 4 / 2 / 1 / 9; fresh-process + isolation and every time ceiling stay. +- `scripts/ci/run-bun-test-batches.sh` changes stay inside the selection and batching loops so + they stay separable from open PR #5456 (timeout probe and Bun invocation lines). +- The `devlog/**` privacy gap belongs to open PR #5469 and is not touched. +- Lane rule: no local suite, focused test, typecheck, build, install, proxy or service run. + Evidence is static reading plus hosted CI on the exact pull-request head. + +## Work-phase map (dependency order) + +| ID | Doc | Outcome | +| --- | --- | --- | +| wp0 | this unit | Roadmap locked before implementation | +| wp1 | [010](010_release_preflight.md) | `preflight` job before packaging; final check kept in `publish` | +| wp2 | [020](020_release_outcome_report.md) | GitHub release and npm version / dist-tag reported as separate outcomes | +| wp3 | [030](030_shard_balance.md) | Shards assigned by recorded duration; batches bounded by predicted time | +| wp4 | [040](040_scope_gap_checks.md) | Narrow jobs for the setup action and the remote-workspace helper | +| wp5 | [050](050_delivery.md) | Push, one PR, exact-head hosted CI | + +wp1 precedes wp2 because both edit `release.yml` and the report reads the publish job's outputs. +wp3 and wp4 are independent of the release work and of each other; wp4 edits `ci.yml`, which wp3 +touches only in a comment. + +## Acceptance criteria (goal level) + +1. Each item has a `tests/ci-workflows` test that fails on the old workflow shape and passes on + the new one. +2. Ordinary PR CI time does not grow. +3. The release changes pass the security-boundary checklist above. +4. `structure/ops/cross-platform-ci.md` and `structure/ops/docs-and-release.md` describe the new + order. +5. One pull request to `dev` with every template section filled; exact-head hosted CI complete. + +## Verification policy + +Local checks: NOT RUN (lane rule). The one exception considered is generating the committed +duration table with the repository's own refresh tool from downloaded hosted logs; it is a data +generator, not a check, and it is recorded where it happens (030). + +## Architect consultation record + +Proposal (read-only architect, decisions D1-D4) and main dispositions: + +| ID | Proposal | Disposition | +| --- | --- | --- | +| D1 | Ordering gate as an extra step in `validate-dispatch`, run from the default-branch checkout; publish gate stays | Amended: a separate `preflight` job on the dispatched commit that also checks channel, version sources, tag, GitHub release, npm and dev readiness. `validate-dispatch` stays minimal (its permissions and shape are pinned) and the scripts that decide are the ones `publish` will run anyway. Accepted: the late gate never moves. | +| D2 | Make the registry smoke fail the run after unconfirmed reads | Rejected: the lane requires publishing behaviour to stay as it is. Adopted the underlying concern instead: the unconfirmed state becomes an explicit, separately reported outcome (020). | +| D3 | Weighted LPT in the selection loop, pure shell, equal weights reproduce round-robin, re-sort each shard by path | Accepted. Amended with a coverage guard (whole assignment must tile the general files) and a predicted-duration batch budget, both from the simulation in 030. | +| D4 | Two narrow filters and jobs mirroring `structure-gate`; PR scope only; aggregate wiring | Accepted, with the filter narrowed to `native/remote-workspace-helper/**` and, from the risk list, a validation step that fails loud on a malformed filter output. | + +Reflection check and independent audit: not obtained. The lane allows one subagent model, and +every call to it after the proposal (one reflection, seven audit attempts over about an hour) +returned a provider `resource_exhausted` error. The audit was performed by the main agent against +the same checklist and is recorded below; it is not independent. + +## Audit record (main agent) + +Verdict: near-pass. Blocker folded: the preflight test's fixture e-mail used `example.invalid`, +which `scripts/privacy-scan.ts` rejects (allowed fixture domains are `example.com`, +`example.test` and `*.test`); changed to `example.test`. Checked without finding a defect: +generic `ci.yml` job rules (numeric `timeout-minutes`, aggregate `needs` equals every job, +checkout never persists credentials, no `@vN`/`@main` action refs); `GATED_JOBS` equals every job +but `ci`; the pinned `structure-gate widget` line and push-paths-equal-`ci` rule; publish-job +step lookups by first name and by `assert-releasable` stay inside `publish`; `attach-release` +step-order assertions; the registry smoke executed test's npm call shapes; file-size ratchet +(threshold 2,000 lines, the table is below it); root typecheck covers `src` only; bash 3.2 and Git +Bash constructs in the runner block (no associative arrays, `$'\t'`, arithmetic assignment forms, +`getline` table load that works when the table is `/dev/null`). Residual: the new jobs' real +runner behaviour (live confinement on hosted macOS/Windows, the composite action on three OSes) is +provable only by this pull request's own hosted run. diff --git a/devlog/_plan/260923_p5_ci_release_gaps/010_release_preflight.md b/devlog/_plan/260923_p5_ci_release_gaps/010_release_preflight.md new file mode 100644 index 0000000000..45ac00de1d --- /dev/null +++ b/devlog/_plan/260923_p5_ci_release_gaps/010_release_preflight.md @@ -0,0 +1,251 @@ +# 010 — Release preflight before packaging (wp1) + +## Change map + +| Path | Action | +| --- | --- | +| `scripts/ci/release-preflight.sh` | NEW — every precondition decidable at dispatch | +| `.github/workflows/release.yml` | MODIFY — new `preflight` job; both packaging jobs need it | +| `tests/ci-workflows/release-preflight.test.ts` | NEW — structure + execution contract | +| `tests/ci-workflows/release-pipeline-contract.test.ts` | unchanged (`verify-release`/`publish`/`attach-release` needs are unchanged) | +| `scripts/test-layout/layout.json`, `tests/fixtures/test-layout-expected.json` | MODIFY — register the new test | +| `structure/ops/cross-platform-ci.md`, `structure/ops/docs-and-release.md` | MODIFY — describe the order (see 050 for the text owner) | + +## Job order after the change + +``` +validate-dispatch -> preflight -> package-standalone ┐ + -> package-desktop ┴-> verify-release -> publish -> attach-release +``` + +`publish` keeps every existing step, in order, including "Preflight release metadata" and +"Refuse a release the current tag set already outranks". Those are the final check; the new job +never replaces them. New step names are distinct from the publish-job names because +`tests/ci-workflows/ci-workflows.test.ts` locates those by first occurrence. + +## release.yml diff + +```diff ++ # Every publication precondition the dispatch can already decide, checked before any runner ++ # starts packaging. <incident + coordination comment, see script header> ++ preflight: ++ name: release preflight ++ needs: validate-dispatch ++ runs-on: ubuntu-latest ++ timeout-minutes: 5 ++ permissions: ++ contents: read ++ steps: ++ - name: Checkout ++ uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 ++ with: ++ persist-credentials: false ++ fetch-tags: true ++ - name: Setup project Bun ++ uses: ./.github/actions/setup-project-bun ++ - name: Fetch the dev line ++ run: git fetch --no-tags --depth=1 origin +refs/heads/dev:refs/remotes/origin/dev ++ - name: Refuse a release that cannot publish ++ env: ++ GH_TOKEN: ${{ github.token }} ++ RELEASE_VERSION: ${{ inputs.version }} ++ NPM_DIST_TAG: ${{ inputs.tag }} ++ DRY_RUN: ${{ inputs.dry-run }} ++ RESUME: ${{ inputs.resume-after-npm-publish }} ++ run: bash scripts/ci/release-preflight.sh + package-standalone: +- needs: validate-dispatch ++ needs: [validate-dispatch, preflight] + package-desktop: +- needs: validate-dispatch ++ needs: [validate-dispatch, preflight] +``` + +Inputs reach shell only through `env` (repository rule enforced by `ci-workflows.test.ts`). +`GH_TOKEN` is the job token with `contents: read`; nothing prints it. `npm view` needs no +credential. No new action is introduced. + +## Coordination decision + +Kept: the workflow-level `concurrency: { group: release, cancel-in-progress: false }`. It is a +constant string, so every dispatch on every ref shares one slot, and run 35783865160 shows it +worked. A new version reservation would add a write surface for no gain. The residual known +limitation is GitHub's single pending slot per group: a third dispatch replaces a pending one. +That is unchanged and is stated in the structure doc. The test pins the group as a constant with +`cancel-in-progress: false`, because a per-ref group is the one edit that would let a stable and a +preview run overlap. + +## Enforcement record (PLAN-BYPASS-NAMED-01) + +| Field | Value | +| --- | --- | +| Tier | CI job gate (early warning) | +| Executing surface | `preflight` job in `release.yml` | +| Known bypass | State change after the preflight (manual tag, first local publish); npm/GitHub read failure is treated as absent/unknown | +| Residual risk | Covered by the unchanged publish-job checks, which remain the final layer | +| Wording | "preflight", not "enforcement"; the final layer is the publish job | + +## scripts/ci/release-preflight.sh (full text) + +```bash +#!/usr/bin/env bash +# Release preflight: every publication precondition the dispatch can already decide, checked +# before any runner starts packaging. +# +# The publish job repeats these checks immediately before `npm publish`, and that copy stays the +# final authority: tags, releases and registry state can still move while a run packages. This +# copy exists so a release that can never publish fails in its first minute. Run 35783865160 +# packaged 2.62.0 for nineteen minutes and then failed the ordering gate on a preview tag that +# already existed when its first job started. +# +# Environment: RELEASE_VERSION, NPM_DIST_TAG, GITHUB_REF, GITHUB_SHA (required); DRY_RUN, RESUME. +# Reads the checkout's tags and refs/remotes/origin/dev, `gh release view` and `npm view`. +# Every problem is reported before the script exits, so one run names all of them. +set -euo pipefail + +: "${RELEASE_VERSION:?RELEASE_VERSION is required}" +: "${NPM_DIST_TAG:?NPM_DIST_TAG is required}" +: "${GITHUB_REF:?GITHUB_REF is required}" +: "${GITHUB_SHA:?GITHUB_SHA is required}" +dry_run="${DRY_RUN:-false}" +resume="${RESUME:-false}" +repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +release_tag="v${RELEASE_VERSION}" + +problems=0 +problem_list="" +fail() { + problems=$((problems + 1)) + problem_list="${problem_list}- $1 +" + echo "::error::$1" +} + +# Channel and dist-tag, exactly as the publish job derives them from the dispatched ref. +expected_tag="" +case "$GITHUB_REF" in + refs/heads/main) + expected_tag="latest" + [[ "$RELEASE_VERSION" != *-* ]] \ + || fail "main releases must use a stable semver version; got ${RELEASE_VERSION}" + ;; + refs/heads/preview) + expected_tag="preview" + [[ "$RELEASE_VERSION" == *-preview.* ]] \ + || fail "preview releases must use a preview prerelease version; got ${RELEASE_VERSION}" + ;; + *) + fail "Release must run from main or preview; got ${GITHUB_REF}" + ;; +esac +if [[ -n "$expected_tag" && "$NPM_DIST_TAG" != "$expected_tag" ]]; then + fail "${GITHUB_REF#refs/heads/} releases must publish with npm dist-tag '${expected_tag}', got '${NPM_DIST_TAG}'" +fi + +if [[ "$resume" == "true" && "$dry_run" == "true" ]]; then + fail "resume-after-npm-publish is a real-publication recovery path and cannot combine with dry-run" +fi + +# Every version source (package.json and the desktop manifests) must already name the release. +if ! bun "$repo_root/scripts/release-version-sources.ts" check "$RELEASE_VERSION"; then + fail "a version source does not match ${RELEASE_VERSION}; run scripts/release.ts on the release branch first" +fi + +# Git tag. A tag at another commit is always fatal; one at this commit is expected only when a +# dry run is repeated or a partial publication is resumed. +existing_tag_sha="$(git rev-parse -q --verify "refs/tags/${release_tag}^{commit}" || true)" +if [[ -n "$existing_tag_sha" && "$existing_tag_sha" != "$GITHUB_SHA" ]]; then + fail "${release_tag} already points at ${existing_tag_sha}, not ${GITHUB_SHA}" +elif [[ -n "$existing_tag_sha" && "$resume" != "true" && "$dry_run" != "true" ]]; then + fail "${release_tag} already exists. Refusing to publish a version with pre-existing Git metadata." +fi + +# GitHub release. An unreadable answer counts as absent here; the publish job reads it again. +if gh release view "$release_tag" >/dev/null 2>&1 && [[ "$resume" != "true" && "$dry_run" != "true" ]]; then + fail "GitHub Release ${release_tag} already exists. Choose the next unused version." +fi + +# npm. Only an exact version answer counts as present and only E404 counts as absent; anything +# else is a registry read failure, which warns rather than blocking a release it cannot judge. +pkg_name="$(node -p "require(process.argv[1]).name" "$repo_root/package.json")" +npm_error="$(mktemp)" +trap 'rm -f -- "$npm_error"' EXIT +npm_state="unknown" +if npm_answer="$(npm view "${pkg_name}@${RELEASE_VERSION}" version --fetch-retries=0 --fetch-timeout=8000 2>"$npm_error")"; then + if [[ "$npm_answer" == "$RELEASE_VERSION" ]]; then npm_state="present"; else npm_state="absent"; fi +elif grep -q "E404" "$npm_error"; then + npm_state="absent" +fi +case "$npm_state" in + present) + if [[ "$resume" == "true" ]]; then + echo "${pkg_name}@${RELEASE_VERSION} is on npm; the publish job verifies its source before resuming." + elif [[ "$dry_run" == "true" ]]; then + echo "::notice::${pkg_name}@${RELEASE_VERSION} already exists on npm; dry-run only" + else + fail "${pkg_name}@${RELEASE_VERSION} already exists on npm. Re-dispatch with resume-after-npm-publish: true if a previous run acknowledged it; otherwise choose the next unused version." + fi + ;; + absent) + [[ "$resume" != "true" ]] \ + || fail "resume-after-npm-publish is set, but ${pkg_name}@${RELEASE_VERSION} is not on npm" + ;; + *) + echo "::warning::Could not read npm for ${pkg_name}@${RELEASE_VERSION}; the publish job checks again before publishing." + ;; +esac + +# Cross-channel ordering against the whole tag set: the gate run 35783865160 reached too late. +allow="" +if [[ ( "$dry_run" == "true" || "$resume" == "true" ) && -n "$existing_tag_sha" && "$existing_tag_sha" == "$GITHUB_SHA" ]]; then + allow="--allow-existing-tag-at-head" +fi +if ! git tag --list 'v*' | bun "$repo_root/scripts/version-line.ts" assert-releasable "$RELEASE_VERSION" ${allow:+"$allow"}; then + fail "${RELEASE_VERSION} does not outrank the current tag set" +fi + +# dev must already carry a higher version (the pre-move). +if dev_package="$(git show refs/remotes/origin/dev:package.json 2>/dev/null)"; then + dev_version="$(printf '%s' "$dev_package" | node -p "JSON.parse(require('fs').readFileSync(0, 'utf8')).version")" + bun "$repo_root/scripts/version-line.ts" assert-ahead "$dev_version" "$RELEASE_VERSION" \ + || fail "dev carries ${dev_version}, which does not outrank ${RELEASE_VERSION}; merge the dev pre-move first" +else + fail "cannot read package.json from refs/remotes/origin/dev" +fi + +if (( problems > 0 )); then + if [[ -n "${GITHUB_STEP_SUMMARY:-}" ]]; then + printf '### Release preflight refused %s\n\n%s' "$RELEASE_VERSION" "$problem_list" >> "$GITHUB_STEP_SUMMARY" + fi + echo "Release preflight found ${problems} blocking problem(s); nothing was packaged." + exit 1 +fi +echo "Release preflight passed for ${RELEASE_VERSION} at ${GITHUB_SHA}; the publish job repeats these checks before publishing." +``` + +## Tests (`tests/ci-workflows/release-preflight.test.ts`) + +Structure (fail on the old shape): + +1. `jobs.preflight` exists, `needs: validate-dispatch`, `permissions == { contents: read }`, runs + `bash scripts/ci/release-preflight.sh` with inputs passed through `env`. +2. `package-standalone` and `package-desktop` both list `preflight` in `needs`. +3. `publish` still contains "Refuse a release the current tag set already outranks" running + `assert-releasable` (the final check stays). +4. `concurrency.group` is a constant containing no `${{` and `cancel-in-progress` is false. + +Execution (Linux/macOS; `bash`, real `bun scripts/version-line.ts`, fake `gh`/`npm` on PATH, +temporary git repository with tags and `refs/remotes/origin/dev`; `RELEASE_VERSION` is the +checkout's own `package.json` version so the real version-source check passes): + +| Scenario | Expect | +| --- | --- | +| Incident replay: stable `X.Y.0` on main, tag `vX.(Y+1).0-preview.20260923` exists | exit 1, ordering error, nothing else blocking | +| Clean stable: lower tags only, dev ahead, npm E404, no GitHub release | exit 0 | +| npm already has the version, no resume, not dry-run | exit 1 naming npm | +| Same, dry-run | exit 0 with notice | +| dev not ahead | exit 1 naming the dev pre-move | +| Two problems at once (wrong dist-tag and npm present) | exit 1, both reported | + +Activation: the incident replay is the conditional path; its observable effect is a non-zero exit +naming the blocking tag before any packaging job could start (the workflow `needs` edge). diff --git a/devlog/_plan/260923_p5_ci_release_gaps/020_release_outcome_report.md b/devlog/_plan/260923_p5_ci_release_gaps/020_release_outcome_report.md new file mode 100644 index 0000000000..dbf0b6342c --- /dev/null +++ b/devlog/_plan/260923_p5_ci_release_gaps/020_release_outcome_report.md @@ -0,0 +1,171 @@ +# 020 — Separate release outcomes (wp2) + +## Amendment after audit + +The audit found that a report step inside `attach-release` can never run when `publish` fails, +because `attach-release` needs `publish`. The report is therefore its own job, +`release-outcomes`: `needs: [publish, attach-release]`, `if: always() && inputs.dry-run != true`, +`permissions: contents: read`, and it also prints both job results. Under a read token a draft +release is invisible, so the GitHub row reads `published` or `not public (draft, missing or +unreadable)`; no write permission is added to observe drafts. The sections below describe the +original step placement; the job shape above supersedes it. + +## Change map + +| Path | Action | +| --- | --- | +| `.github/workflows/release.yml` | MODIFY — registry smoke records `npm_version` and `npm_dist_tag`; `publish` exposes them as job outputs; `attach-release` ends with an always-run report | +| `scripts/ci/release-outcome-report.sh` | NEW — writes one summary row per outcome | +| `tests/ci-workflows/release-outcome-report.test.ts` | NEW | +| layout files | MODIFY — register the test | + +## Registry smoke diff (publish job, step `registry-smoke`) + +Publishing behaviour is unchanged: the same six bounded version reads, the same single +`npm dist-tag ls`, the same continuation to the GitHub release when reads stay pending, the same +hard failure on an unexpected version. The existing executed test in +`tests/ci-workflows/ci-workflows.test.ts` pins every npm call shape; nothing here adds a call. + +```diff + - name: Post-publish registry smoke + id: registry-smoke + env: + RELEASE_VERSION: ${{ inputs.version }} ++ NPM_DIST_TAG: ${{ inputs.tag }} + PUBLISHED: ${{ steps.publication.outputs.published }} + ... + echo "verification=verified" >> "$GITHUB_OUTPUT" ++ echo "npm_version=confirmed" >> "$GITHUB_OUTPUT" + echo "Registry verified ${pkg_name}@${RELEASE_VERSION}." >> "$GITHUB_STEP_SUMMARY" +- timeout ... npm dist-tag ls "$pkg_name" ... || echo "::warning::Could not read npm dist-tags; exact version was verified" ++ dist_tag_state="unconfirmed" ++ if dist_tags="$(timeout ... npm dist-tag ls "$pkg_name" ...)"; then ++ printf '%s\n' "$dist_tags" ++ tagged="$(printf '%s\n' "$dist_tags" | awk -F': ' -v tag="$NPM_DIST_TAG" '$1 == tag { print $2; exit }')" ++ if [ "$tagged" = "$RELEASE_VERSION" ]; then dist_tag_state="confirmed" ++ elif [ -n "$tagged" ]; then dist_tag_state="mismatch"; echo "::warning::npm dist-tag ... points at ${tagged}" ++ else echo "::warning::npm dist-tag ${NPM_DIST_TAG} is not listed"; fi ++ else ++ echo "::warning::Could not read npm dist-tags; exact version was verified" ++ fi ++ echo "npm_dist_tag=${dist_tag_state}" >> "$GITHUB_OUTPUT" ++ echo "npm dist-tag ${NPM_DIST_TAG}: ${dist_tag_state}." >> "$GITHUB_STEP_SUMMARY" + exit 0 + ... + echo "verification=pending" >> "$GITHUB_OUTPUT" ++ echo "npm_version=unconfirmed" >> "$GITHUB_OUTPUT" ++ echo "npm_dist_tag=unconfirmed" >> "$GITHUB_OUTPUT" +``` + +```diff + publish: + needs: [validate-dispatch, verify-release] ++ outputs: ++ npm_version: ${{ steps.registry-smoke.outputs.npm_version }} ++ npm_dist_tag: ${{ steps.registry-smoke.outputs.npm_dist_tag }} +``` + +```diff + attach-release: (last step) ++ - name: Report release outcomes ++ if: always() ++ env: ++ GH_TOKEN: ${{ github.token }} ++ RELEASE_VERSION: ${{ inputs.version }} ++ NPM_DIST_TAG: ${{ inputs.tag }} ++ NPM_VERSION_STATE: ${{ needs.publish.outputs.npm_version }} ++ NPM_DIST_TAG_STATE: ${{ needs.publish.outputs.npm_dist_tag }} ++ run: bash scripts/ci/release-outcome-report.sh +``` + +`attach-release` already holds `contents: write` for the upload; the report only reads +(`gh release view`). Job permissions are unchanged. The step runs after "Attach to the release", +so the GitHub row reflects the draft flip, and `always()` keeps the row visible when the attach +step failed. The resume path (`published=true` without a new publish) reaches the same smoke and +report. + +## scripts/ci/release-outcome-report.sh (full text) + +```bash +#!/usr/bin/env bash +# Report what a release run established, one fact per row. +# +# npm publication, the registry read-back, the dist-tag and the public GitHub release are four +# separate facts, and a green run used to read the same whichever of them were true: the registry +# smoke warns "Registry lookup not confirmed" and the run still continues to the GitHub release. +# This writes each outcome as its own row of the job summary and adds a warning annotation for +# every row that is not confirmed. It never changes the run's result; publishing behaviour is +# owned by the publish and attach steps. +# +# Environment: RELEASE_VERSION, NPM_DIST_TAG, GITHUB_STEP_SUMMARY (required); +# NPM_VERSION_STATE and NPM_DIST_TAG_STATE from the publish job (confirmed | mismatch | +# unconfirmed; empty when the smoke never ran). +set -uo pipefail + +: "${RELEASE_VERSION:?RELEASE_VERSION is required}" +: "${NPM_DIST_TAG:?NPM_DIST_TAG is required}" +: "${GITHUB_STEP_SUMMARY:?GITHUB_STEP_SUMMARY is required}" +release_tag="v${RELEASE_VERSION}" + +github_state="not found" +if draft="$(gh release view "$release_tag" --json isDraft --jq .isDraft 2>/dev/null)"; then + case "$draft" in + false) github_state="published" ;; + true) github_state="draft (not public)" ;; + *) github_state="unreadable" ;; + esac +fi + +describe() { + case "$1" in + confirmed) echo "confirmed" ;; + mismatch) echo "points at another version" ;; + *) echo "not confirmed" ;; + esac +} +npm_version_state="$(describe "${NPM_VERSION_STATE:-}")" +npm_tag_state="$(describe "${NPM_DIST_TAG_STATE:-}")" + +{ + echo "### Release outcomes for ${RELEASE_VERSION}" + echo "" + echo "| Outcome | State |" + echo "| --- | --- |" + echo "| GitHub release \`${release_tag}\` | ${github_state} |" + echo "| npm version \`${RELEASE_VERSION}\` read back from the registry | ${npm_version_state} |" + echo "| npm dist-tag \`${NPM_DIST_TAG}\` points at \`${RELEASE_VERSION}\` | ${npm_tag_state} |" + echo "" + echo "Each row is read separately. A row that is not confirmed is not a failure of this run; inspect it before announcing availability, and never republish the version." +} >> "$GITHUB_STEP_SUMMARY" + +[[ "$github_state" == "published" ]] \ + || echo "::warning::GitHub release ${release_tag} is ${github_state}" +[[ "$npm_version_state" == "confirmed" ]] \ + || echo "::warning::npm version ${RELEASE_VERSION} was not read back from the registry" +[[ "$npm_tag_state" == "confirmed" ]] \ + || echo "::warning::npm dist-tag ${NPM_DIST_TAG} ${npm_tag_state} for ${RELEASE_VERSION}" +exit 0 +``` + +## Tests (`tests/ci-workflows/release-outcome-report.test.ts`) + +Structure (fail on the old shape): the smoke step writes `npm_version=` and `npm_dist_tag=`; +`publish.outputs` maps both from `steps.registry-smoke.outputs`; the last `attach-release` step +has `if: always()`, reads both job outputs through `env`, and runs the report script. + +Execution, smoke (fake `npm`/`timeout` shell functions as in the existing executed test): + +| npm dist-tag ls output | `npm_dist_tag` | +| --- | --- | +| `latest: 9.8.7` | confirmed | +| `latest: 9.8.6` | mismatch + warning | +| read fails | unconfirmed + warning | +| version reads stay pending | `npm_version=unconfirmed`, `npm_dist_tag=unconfirmed` | + +Execution, report (fake `gh` on PATH): + +| gh answer | NPM states | Summary rows | +| --- | --- | --- | +| `false` | confirmed / confirmed | published / confirmed / confirmed; no warning | +| `true` | confirmed / mismatch | draft / confirmed / points at another version; two warnings | +| exit 1 | empty / empty | not found / not confirmed / not confirmed; exit 0 | diff --git a/devlog/_plan/260923_p5_ci_release_gaps/030_shard_balance.md b/devlog/_plan/260923_p5_ci_release_gaps/030_shard_balance.md new file mode 100644 index 0000000000..6af5a0ddcb --- /dev/null +++ b/devlog/_plan/260923_p5_ci_release_gaps/030_shard_balance.md @@ -0,0 +1,343 @@ +# 030 — Duration-weighted shard assignment (wp3) + +## Change map + +| Path | Action | +| --- | --- | +| `scripts/ci/run-bun-test-batches.sh` | MODIFY — replace the round-robin selection loop and the batching loop (lines ~214-257 on `dev` at e9643875f0) with the block below; nothing else in the file moves | +| `scripts/ci/test-durations.ts` | NEW — log parser, table merge and renderer, `refresh` CLI | +| `scripts/ci/test-durations.tsv` | NEW — committed table generated by the tool from run 35816902207's four Linux shard logs | +| `.github/workflows/ci.yml` | MODIFY — the `test` job comment only ("sorted round-robin" becomes duration assignment) | +| `tests/ci-workflows/ci-shard-balance.test.ts` | NEW | +| layout files | MODIFY — register the test | + +Separation from PR #5456: that PR edits the GNU-timeout probe (~lines 56-59), the Bun invocation +inside `run_test_once` (~107-109) and the file enumeration. This change touches none of those +lines, and it adds no Bun invocation: the fake-Bun harness in `ci-crash-disposition.test.ts` +counts every Bun call except `-e`, so the balancer is POSIX `awk` and `sort` (bash 3.2 safe, no +associative arrays). + +## Design decisions + +- Table format `<ms>\t<path>`, comment lines start with `#`, sorted by path, generated only by the + tool. CRLF is tolerated (Windows checkouts may convert it). +- Weight of an unknown file = median of the table, derived at run time; empty or missing table = + every file weighs the same = exactly the old sorted round-robin (greedy, path order, lowest + index on ties). +- Coverage guard: each shard computes the whole assignment and exits 1 unless it covers every + general file exactly once (a pipeline failure can otherwise produce a partial list that stays + green). +- Batch budget = half of `BUN_TEST_BATCH_TIMEOUT_SECONDS`, derived, no new variable. It only ever + adds process boundaries, so isolation increases and every ceiling (`timeout-minutes`, per-process + timeout, batch size) is unchanged. +- Data source: hosted job logs, which every run already keeps; the Bun invocation is unchanged, + so the test legs carry no new reporter or artifact step and PR CI time cannot grow from it. +- Override `BUN_TEST_DURATIONS_FILE` exists for the executed tests. + +Simulation on run 35816902207 (1,558 files, 1,261 s): + +| | shard test time (s) | largest batch (s) | processes | +| --- | --- | --- | --- | +| round-robin (today) | 364 / 394 / 248 / 254 | 74 | 33 each (plus serial singletons) | +| duration, no budget | 315 / 315 / 315 / 315 | 89 | 33 each | +| duration + half-timeout budget | 315 / 315 / 315 / 315 | 70 (one file alone) | 34 / 35 / 33 / 33 | + +## Replacement block for run-bun-test-batches.sh + +```bash +# Shard ownership by recorded duration. +# +# Sorted round-robin split the suite evenly by COUNT, while file durations differ by three orders +# of magnitude: one Linux shard carried 394 s of tests and another 248 s (run 35816902207). Each +# file now weighs the milliseconds recorded for it in scripts/ci/test-durations.tsv, and the +# heaviest file goes first to the least-loaded shard, lowest index on a tie. A file the table does +# not know weighs the table's median, so with no usable table every file weighs the same and the +# result is exactly the old sorted round-robin. Every shard computes the whole assignment and +# refuses to run unless it covers every general file exactly once, because a shard that silently +# drops files is the one failure here that stays green. Each shard still runs its files in sorted +# order. Refresh the table with scripts/ci/test-durations.ts from hosted job logs. +batch_script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +readonly DURATIONS_FILE="${BUN_TEST_DURATIONS_FILE:-$batch_script_dir/test-durations.tsv}" +durations_source="/dev/null" +if [[ -f "$DURATIONS_FILE" ]]; then + durations_source="$DURATIONS_FILE" +fi + +GENERAL_FILES=() +for path in "${ALL_TEST_FILES[@]}"; do + if is_general_test_file "$path"; then + GENERAL_FILES+=("$path") + fi +done +if (( ${#GENERAL_FILES[@]} == 0 )); then + echo "No tests selected for shard ${SHARD_SPEC}." >&2 + exit 1 +fi + +# The median of the recorded milliseconds; any constant would do for an empty table. +fallback_ms="$( + awk -F '\t' '{ sub(/\r$/, "") } !/^#/ && NF == 2 && $1 ~ /^[0-9]+$/ { print $1 }' "$durations_source" \ + | LC_ALL=C sort -n \ + | awk '{ values[NR] = $1 } END { value = (NR > 0 ? values[int((NR + 1) / 2)] : 1000); print (value > 0 ? value : 1) }' +)" +readonly FALLBACK_MS="$fallback_ms" + +assignment_file="$(mktemp -t ocx-bun-test-shards.XXXXXX)" +printf '%s\n' "${GENERAL_FILES[@]}" \ + | awk -F '\t' -v table="$durations_source" -v fallback="$FALLBACK_MS" ' + BEGIN { + while ((getline line < table) > 0) { + sub(/\r$/, "", line) + if (line ~ /^#/ || split(line, field, "\t") != 2 || field[1] !~ /^[0-9]+$/) continue + weight[field[2]] = (field[1] > 0 ? field[1] : 1) + } + close(table) + } + { print (($0 in weight) ? weight[$0] : fallback) "\t" $0 } + ' \ + | LC_ALL=C sort -t $'\t' -k1,1nr -k2,2 \ + | awk -F '\t' -v shards="$SHARD_COUNT" ' + BEGIN { for (shard = 1; shard <= shards; shard += 1) load[shard] = 0 } + { + best = 1 + for (shard = 2; shard <= shards; shard += 1) if (load[shard] < load[best]) best = shard + load[best] += $1 + print best "\t" $1 "\t" $2 + } + ' > "$assignment_file" +assigned_count="$(awk 'END { print NR }' "$assignment_file")" +if (( assigned_count != ${#GENERAL_FILES[@]} )); then + rm -f -- "$assignment_file" + echo "Shard assignment covered ${assigned_count} of ${#GENERAL_FILES[@]} test files; refusing to run a partial suite." >&2 + exit 1 +fi + +SELECTED_FILES=() +SELECTED_WEIGHTS=() +predicted_ms=0 +while IFS=$'\t' read -r owner weight path; do + if [[ "$owner" == "$SHARD_INDEX" ]]; then + SELECTED_FILES+=("$path") + SELECTED_WEIGHTS+=("$weight") + predicted_ms=$((predicted_ms + weight)) + fi +done < <(LC_ALL=C sort -t $'\t' -k3,3 "$assignment_file") +rm -f -- "$assignment_file" + +if (( ${#SELECTED_FILES[@]} == 0 )); then + echo "No tests selected for shard ${SHARD_SPEC}." >&2 + exit 1 +fi + +# Keep shard ownership and sorted execution order; split only the process boundary. A batch also +# closes before its predicted duration would pass half the process timeout: balancing by duration +# changes which files share a process, and without this bound one twelve-file batch was predicted +# at 89 of its 120 seconds. A file heavier than the budget still runs, alone. +readonly BATCH_BUDGET_MS=$(( BATCH_TIMEOUT_SECONDS * 1000 / 2 )) +BATCH_STARTS=() +BATCH_LENGTHS=() +pending_start=0 +pending_count=0 +pending_ms=0 +for ((index = 0; index < ${#SELECTED_FILES[@]}; index += 1)); do + if is_serial_test_file "${SELECTED_FILES[$index]}"; then + if (( pending_count > 0 )); then + BATCH_STARTS+=("$pending_start"); BATCH_LENGTHS+=("$pending_count") + pending_count=0 + fi + BATCH_STARTS+=("$index"); BATCH_LENGTHS+=(1) + else + weight="${SELECTED_WEIGHTS[$index]}" + if (( pending_count > 0 && pending_ms + weight > BATCH_BUDGET_MS )); then + BATCH_STARTS+=("$pending_start"); BATCH_LENGTHS+=("$pending_count") + pending_count=0 + fi + if (( pending_count == 0 )); then pending_start=$index; pending_ms=0; fi + pending_count=$((pending_count + 1)) + pending_ms=$((pending_ms + weight)) + if (( pending_count == BATCH_SIZE )); then + BATCH_STARTS+=("$pending_start"); BATCH_LENGTHS+=("$pending_count") + pending_count=0 + fi + fi +done +if (( pending_count > 0 )); then + BATCH_STARTS+=("$pending_start"); BATCH_LENGTHS+=("$pending_count") +fi +readonly TOTAL_BATCHES=${#BATCH_STARTS[@]} +echo "Shard ${SHARD_SPEC}: ${#SELECTED_FILES[@]} files in ${TOTAL_BATCHES} primary Bun processes (scope ${TEST_FILE_SCOPE}, batch size <= ${BATCH_SIZE}, timeout ${BATCH_TIMEOUT_SECONDS}s)." +echo "Predicted shard time from recorded durations: $((predicted_ms / 1000))s; a file without a record weighs ${FALLBACK_MS}ms and a batch closes before ${BATCH_BUDGET_MS}ms." +``` + +## scripts/ci/test-durations.ts (full text) + +```ts +/** + * Per-file Bun test durations for shard assignment. + * + * `scripts/ci/run-bun-test-batches.sh` weighs every test file by the milliseconds recorded for it in + * `scripts/ci/test-durations.tsv` and assigns the heaviest file first to the least-loaded shard. This + * tool keeps that table honest by reading it back from hosted CI job logs, where every line carries + * a runner timestamp and Bun wraps each file's output in its own `##[group]<file>:` ... + * `##[endgroup]` pair. A file's duration is the time from the previous file's end (or its batch + * header, for the first file of a process) to its own end, so process start and module loading are + * charged to the file that caused them. + * + * Refresh from a green run on dev (all four Linux shards): + * + * gh run view <run-id> --log > .tmp/ci-run.log + * bun scripts/ci/test-durations.ts refresh --source "run <run-id>" .tmp/ci-run.log + * + * Files measured in the logs replace their rows; rows for files that still exist are kept; rows for + * files that no longer exist are dropped. Attribution sweeps (a failed shard re-running files one at a + * time) are ignored because they do not measure the batch shape the table is used for. + */ +import { existsSync, readFileSync, writeFileSync } from "node:fs"; +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; + +const REPO_ROOT = dirname(dirname(dirname(fileURLToPath(import.meta.url)))); +export const DURATIONS_TABLE = join(REPO_ROOT, "scripts", "ci", "test-durations.tsv"); + +const LOG_LINE = /^(?:(.*?)\t[^\t]*\t)?\uFEFF?(\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d)(\.\d+)?Z (.*)$/; +const BATCH_HEADER = /^##\[group\]shard \S+ batch \d+\/\d+(.*)$/; +const FILE_HEADER = /^##\[group\](tests\/.+):$/; +const ANSI = /\u001b\[[0-9;]*m/g; + +function timestampMs(seconds: string, fraction: string | undefined): number { + const milliseconds = (fraction ?? ".0").slice(1, 4).padEnd(3, "0"); + return Date.parse(`${seconds}.${milliseconds}Z`); +} + +/** Every measured duration per file, in milliseconds, from one or more concatenated job logs. */ +export function parseJobLog(text: string): Map<string, number[]> { + type JobState = { previousEnd: number | null; file: string | null; attribution: boolean }; + const jobs = new Map<string, JobState>(); + const samples = new Map<string, number[]>(); + for (const rawLine of text.split(/\r?\n/)) { + const match = LOG_LINE.exec(rawLine.replace(ANSI, "")); + if (!match) continue; + const job = match[1] ?? ""; + const at = timestampMs(match[2]!, match[3]); + const body = match[4]!; + let state = jobs.get(job); + if (!state) { + state = { previousEnd: null, file: null, attribution: false }; + jobs.set(job, state); + } + const batch = BATCH_HEADER.exec(body); + if (batch) { + state.previousEnd = at; + state.file = null; + state.attribution = batch[1]!.includes("attribution"); + continue; + } + const file = FILE_HEADER.exec(body); + if (file) { + state.file = state.attribution || state.previousEnd === null ? null : file[1]!; + continue; + } + if (body.startsWith("##[endgroup]") && state.file !== null && state.previousEnd !== null) { + const list = samples.get(state.file) ?? []; + list.push(Math.max(0, at - state.previousEnd)); + samples.set(state.file, list); + state.previousEnd = at; + state.file = null; + } + } + return samples; +} + +/** The recorded table, path -> milliseconds. Comment lines and malformed rows are ignored. */ +export function parseTable(text: string): Map<string, number> { + const table = new Map<string, number>(); + for (const line of text.split(/\r?\n/)) { + if (line.startsWith("#")) continue; + const fields = line.split("\t"); + if (fields.length !== 2 || !/^\d+$/.test(fields[0]!)) continue; + table.set(fields[1]!, Number(fields[0])); + } + return table; +} + +/** New measurements win; kept rows must still name a file; every duration is at least 1 ms. */ +export function mergeDurations( + previous: ReadonlyMap<string, number>, + measured: ReadonlyMap<string, readonly number[]>, + exists: (path: string) => boolean, +): Map<string, number> { + const merged = new Map<string, number>(); + for (const [path, value] of previous) if (exists(path)) merged.set(path, value); + for (const [path, values] of measured) { + if (!exists(path) || values.length === 0) continue; + const mean = values.reduce((sum, value) => sum + value, 0) / values.length; + merged.set(path, Math.max(1, Math.round(mean))); + } + return merged; +} + +export function renderTable(durations: ReadonlyMap<string, number>, source: string): string { + const rows = [...durations.entries()] + .sort(([left], [right]) => (left < right ? -1 : left > right ? 1 : 0)) + .map(([path, value]) => `${value}\t${path}`); + return [ + "# Per-file Bun test durations in milliseconds, read from hosted CI job logs.", + "# Consumed by scripts/ci/run-bun-test-batches.sh to balance shards; a file without a row weighs the median.", + "# Regenerate with scripts/ci/test-durations.ts (usage in its header); do not edit by hand.", + `# Source: ${source}`, + ...rows, + "", + ].join("\n"); +} + +if (import.meta.main) { + const [command, ...rest] = process.argv.slice(2); + let source = "unspecified"; + const logs: string[] = []; + for (let index = 0; index < rest.length; index += 1) { + if (rest[index] === "--source") source = rest[++index] ?? source; + else logs.push(rest[index]!); + } + if (command !== "refresh" || logs.length === 0) { + console.error("usage: bun scripts/ci/test-durations.ts refresh [--source <text>] <job-log>..."); + process.exit(64); + } + const measured = new Map<string, number[]>(); + for (const log of logs) { + for (const [path, values] of parseJobLog(readFileSync(log, "utf8"))) { + measured.set(path, [...(measured.get(path) ?? []), ...values]); + } + } + if (measured.size === 0) { + console.error("No per-file durations found; pass hosted logs of the Linux test shards."); + process.exit(1); + } + const previous = existsSync(DURATIONS_TABLE) ? parseTable(readFileSync(DURATIONS_TABLE, "utf8")) : new Map<string, number>(); + const merged = mergeDurations(previous, measured, path => existsSync(join(REPO_ROOT, path))); + writeFileSync(DURATIONS_TABLE, renderTable(merged, source)); + console.log(`Recorded ${merged.size} files (${measured.size} measured) in ${DURATIONS_TABLE}.`); +} +``` + +## Tests (`tests/ci-workflows/ci-shard-balance.test.ts`) + +Executed against the real runner with a fake `bun` and `timeout` (same shape as the existing +disposition harness): + +1. Skewed table (alpha 10 s, five files 1 s), shard 1/2 runs only alpha, 2/2 runs the other five. + Old shape (round-robin) runs alpha, charlie, echo on 1/2: fails. +2. No table: 1/2 and 2/2 equal the old `i % 2` membership (fallback reproduces round-robin). +3. Shards 1..3 with a skewed table together run every general file exactly once. +4. Budget: timeout 2 s (budget 1,000 ms), every file 600 ms, batch size 3, shard 1/1: six + one-file processes. Old shape: two three-file processes: fails. +5. Tool: `parseJobLog` on synthetic logs in both formats (job-log API, and `gh run view --log` + with job/step prefixes), attribution groups ignored; `mergeDurations` drops missing files and + floors at 1 ms; `renderTable` round-trips through `parseTable`. +6. The committed table is well formed: every row `<digits>\t tests/...`, sorted, unique, non-empty. + Existence of listed files is deliberately not asserted: deleting a test must not fail CI. + +## Generation of the committed table + +`bun scripts/ci/test-durations.ts refresh --source "run 35816902207 (test 1/4-4/4)" <four job logs>`, +run once in the lane checkout. This is the only local execution in the lane; it writes a data file +and is not a test, typecheck, build or install. Recorded in the PR. diff --git a/devlog/_plan/260923_p5_ci_release_gaps/040_scope_gap_checks.md b/devlog/_plan/260923_p5_ci_release_gaps/040_scope_gap_checks.md new file mode 100644 index 0000000000..a251c934ae --- /dev/null +++ b/devlog/_plan/260923_p5_ci_release_gaps/040_scope_gap_checks.md @@ -0,0 +1,116 @@ +# 040 — Narrow checks for the setup action and the remote-workspace helper (wp4) + +## Change map + +| Path | Action | +| --- | --- | +| `.github/workflows/ci.yml` | MODIFY — two filters, a validated output step, two jobs, aggregate wiring | +| `tests/ci-workflows/ci-scope-gaps.test.ts` | NEW | +| layout files | MODIFY — register the test | + +## Filters (changes job) + +```diff ++ # The composite action every Bun job runs. The ci filter above omits .github/actions/** ++ # on purpose, so an edit that changes only the action gets this narrow check instead of ++ # the full matrix. ++ setup_action: ++ - '.github/actions/**' ++ - '.github/workflows/ci.yml' ++ # The Rust helper crate. Nothing else builds it, and its sandbox is real only on macOS ++ # and Windows. ++ remote_helper: ++ - 'native/remote-workspace-helper/**' ++ - '.github/workflows/ci.yml' +``` + +Both stay pull-request scope, like `docs` and `structure`: the push trigger's `paths:` is pinned +to equal the `ci` filter, and neither path is added to `ci` (that would start the full suite). + +New step after "Assert the native and matrix outputs are usable": + +```yaml + - name: Assert the narrow scope outputs are usable + id: narrow + shell: bash + env: + SETUP_ACTION: ${{ steps.filter.outputs.setup_action }} + REMOTE_HELPER: ${{ steps.filter.outputs.remote_helper }} + run: | + set -euo pipefail + for pair in "setup_action=$SETUP_ACTION" "remote_helper=$REMOTE_HELPER"; do + case "${pair#*=}" in + true|false) printf '%s\n' "$pair" >> "$GITHUB_OUTPUT" ;; + *) printf '::error::changes.outputs.%s was %q, expected true or false\n' "${pair%%=*}" "${pair#*=}"; exit 1 ;; + esac + done +``` + +Outputs: `setup_action: ${{ steps.narrow.outputs.setup_action }}`, +`remote_helper: ${{ steps.narrow.outputs.remote_helper }}`. + +## Jobs + +```yaml + setup-action: + name: setup action ${{ matrix.os }} + needs: changes + if: needs.changes.outputs.setup_action == 'true' + runs-on: ${{ matrix.os }} + timeout-minutes: 5 + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest, macos-latest] + steps: + - Checkout (persist-credentials: false) + - name: Setup project Bun + id: bun + uses: ./.github/actions/setup-project-bun + - name: Require the runtime package.json declares + shell: bash + env: + RESOLVED: ${{ steps.bun.outputs.version }} + run: declared (node -p package.json dependencies.bun) == RESOLVED == bun --version, else ::error and exit 1 + + remote-helper: + name: remote helper ${{ matrix.os }} + needs: changes + if: needs.changes.outputs.remote_helper == 'true' + runs-on: ${{ matrix.os }} + timeout-minutes: 15 + strategy: { fail-fast: false, matrix: { os: [ubuntu-latest, macos-latest, windows-latest] } } + steps: + - Checkout (persist-credentials: false) + - dtolnay/rust-toolchain@02cb101ec7c40f2c49e1d9714d64511d8e1b74de # master (stable, rustfmt, clippy) + - cargo fmt --check (Linux leg) + - cargo clippy --locked --all-targets -- -D warnings + - cargo test --locked (live confinement tests compile only on macOS/Windows) +``` + +Each leg is one small job, not a suite: the action check is about a minute per runner and the +helper check compiles a four-dependency crate. Neither job runs for an ordinary pull request, because +neither filter matches ordinary source paths. Both run on this lane's own pull request because it +edits `ci.yml`, which is how the jobs get proven. + +## Aggregate gate + +`needs` gains `setup-action, remote-helper`; env gains `CHANGES_SETUP_ACTION` and +`CHANGES_REMOTE_HELPER`; two derived states (`requested` iff the output is `true`); `expected_for` +gains `setup-action) echo "$setup_action" ;;` and `remote-helper) echo "$remote_helper" ;;`; +`GATED_JOBS` gains a new line `GATED_JOBS="$GATED_JOBS setup-action remote-helper"` (the existing +lines are pinned verbatim by `ci-structure-gate.test.ts`). + +## Tests (`tests/ci-workflows/ci-scope-gaps.test.ts`) + +1. `.github/actions/setup-project-bun/action.yml` matches a filter whose job's `if` reads it and + whose steps use `./.github/actions/setup-project-bun`. Old shape: no filter matches: fails. +2. `native/remote-workspace-helper/src/main.rs` matches a filter whose job runs `cargo test` and + `cargo clippy` against `native/remote-workspace-helper/Cargo.toml`. Old shape: fails. +3. Neither path matches the `ci` or `native` filters (no full suite, no macOS suite). +4. The executed aggregate block expects each job `requested` when its output is `true` and + `not-requested` when `false`, and `GATED_JOBS` names both. +5. Both outputs come from the validation step, which can exit 1. + +Path matching uses `Bun.Glob`, which follows the same `**` semantics as the filter for these +patterns. diff --git a/devlog/_plan/260923_p5_ci_release_gaps/050_delivery.md b/devlog/_plan/260923_p5_ci_release_gaps/050_delivery.md new file mode 100644 index 0000000000..f97e75847c --- /dev/null +++ b/devlog/_plan/260923_p5_ci_release_gaps/050_delivery.md @@ -0,0 +1,15 @@ +# 050 — Delivery (wp5) + +1. Ordered commits on `codex/260923-p5-ci-release-gaps`: roadmap (wp0), release preflight (wp1), + release outcomes (wp2), shard balance (wp3), scope-gap checks (wp4), structure docs (with the + phase that changes the described behaviour). +2. Structure text owners: `structure/ops/cross-platform-ci.md` (release order paragraph, batch + runner paragraph, narrow jobs) and `structure/ops/docs-and-release.md` (workflow map rows for + `ci.yml` and `release.yml`, the round-robin sentence). Stay under the manifest line budgets. +3. `git push --no-verify -u origin codex/260923-p5-ci-release-gaps`; open one PR to `dev` with + every section of `.github/PULL_REQUEST_TEMPLATE.md`, including the security-boundary checklist + and "local checks: NOT RUN". +4. Read exact-head hosted CI. On failure read the failing job log and fix the cause. Expected new + legs on this PR: `setup action` x3 and `remote helper` x3 (because `ci.yml` changed). +5. Optional follow-up only if the first CI run shows drift: refresh the table from this PR's own + Linux shard logs and push once more. diff --git a/devlog/_plan/260923_subagent_roster_defaults_autosave/000_README.md b/devlog/_plan/260923_subagent_roster_defaults_autosave/000_README.md new file mode 100644 index 0000000000..6f966dbca3 --- /dev/null +++ b/devlog/_plan/260923_subagent_roster_defaults_autosave/000_README.md @@ -0,0 +1,12 @@ +# Subagent roster: GPT-6 defaults and dashboard autosave + +Two stacked pull requests to `dev`, squash-merged bottom-up. + +| Doc | Work-phase | Branch | Base | +|---|---|---|---| +| [010](010_gpt6_default_roster.md) | wp1 | `codex/subagent-roster-gpt6-defaults` | `dev` | +| [020](020_roster_autosave.md) | wp2 | `codex/subagent-roster-autosave` | wp1 branch | +| [030](030_delivery.md) | wp3 | — | — | + +Out of scope: fallback-list autosave, delegation settings, provider catalogs, +pricing, other dashboard pages. diff --git a/devlog/_plan/260923_subagent_roster_defaults_autosave/010_gpt6_default_roster.md b/devlog/_plan/260923_subagent_roster_defaults_autosave/010_gpt6_default_roster.md new file mode 100644 index 0000000000..7c00c0e02a --- /dev/null +++ b/devlog/_plan/260923_subagent_roster_defaults_autosave/010_gpt6_default_roster.md @@ -0,0 +1,59 @@ +# 010 — GPT-6 three-model default roster + +## Problem + +`DEFAULT_SUBAGENT_MODELS` still ships the Astra-plus-GPT-5.6 roster +(`gpt-6-astra, gpt-5.6-sol, gpt-5.6-terra, gpt-5.6-luna, gpt-5.5`). The default +should be exactly `gpt-6-astra, gpt-6-sol, gpt-6-luna`. + +## Diff + +`src/config/subagent-models.ts` + +- `SUBAGENT_MODELS_VERSION` 1 → 2. +- `DEFAULT_SUBAGENT_MODELS = [NATIVE_GPT6_ASTRA_MODEL, NATIVE_GPT6_SOL_MODEL, NATIVE_GPT6_LUNA_MODEL]`. +- Keep the v1 list as a private `V1_DEFAULT_SUBAGENT_MODELS` constant. +- `migrateSubagentModels`: version < 1 runs the existing Astra upgrade (an unset + list receives the new defaults); then, for version < 2, a stored list that is + element-for-element equal to `V1_DEFAULT_SUBAGENT_MODELS` becomes the new + defaults. Any other list — reordered, trimmed, custom, empty — is untouched. + The version marker is written in both cases so the step runs once. + The comparison runs after the v1 step. A version-1 list is compared as stored, + so a reordered version-1 roster is untouched. An unversioned list that the v1 + step itself turns into the exact old default was produced by migration (the + pre-Astra generated default is the known case), so it continues to the trio. + +Consumers (`proxy-env.ts` fresh config, `hub-state.ts`, `claude/agents-inject.ts`, +management routes) read the constant and need no edit. +`rewriteLegacyOpenAiModelList` only rewrites `openai-multi/` ids, so it cannot +disturb the exact-equality check on bare native ids. + +Tests. `tests/server/config.test.ts` sits exactly at its file-size cap, so the +"Astra-first subagent upgrade" describe moves into +`tests/routing/subagent-roster-migration.test.ts` (registered in +`scripts/test-layout/layout.json` explicit and +`tests/fixtures/test-layout-expected.json`); the moved block is then edited so +that: + +- fresh defaults are the trio at version 2; +- legacy (unversioned) cases keep their v1 results except where the v1 result is + the old generated default, which continues to the trio; +- a version-1 config holding the exact old default migrates to the trio; a + version-1 config holding anything else keeps it and only gains version 2; +- the save/load preservation test uses versions 2 and 3. +- "repair does not invent migration version" expects a version-1 config to gain + version 2 without touching its list. + +`tests/server/server-startup-reconcile-resilience.test.ts` asserts the +post-migration roster and version; update to the version-2 outcome. + +Docs: `docs-site/src/content/docs/{,fr,ja,ko,ru,tr,zh-cn,zh-tw}/reference/configuration/agents.md` +default cell → `gpt-6-astra`, `gpt-6-sol`, `gpt-6-luna`; English upgrade +section gains the version-2 rule. Locale pages link to the English anchor. + +## Verification + +`bun test tests/server/config.test.ts tests/routing/subagent-roster-migration.test.ts +tests/server/server-startup-reconcile-resilience.test.ts tests/routing/subagent-roster-retention.test.ts +tests/test-layout.test.ts tests/test-layout-tooling.test.ts`, the file-size ratchet test, +`bun run test:changed`, `bun run typecheck`, `bun run structure:check`. diff --git a/devlog/_plan/260923_subagent_roster_defaults_autosave/020_roster_autosave.md b/devlog/_plan/260923_subagent_roster_defaults_autosave/020_roster_autosave.md new file mode 100644 index 0000000000..b13e7e6978 --- /dev/null +++ b/devlog/_plan/260923_subagent_roster_defaults_autosave/020_roster_autosave.md @@ -0,0 +1,45 @@ +# 020 — Subagents page roster autosave + +## Problem + +On the Subagents dashboard page, adding, removing, or reordering a featured model +only changes local state; nothing persists until the user clicks Save. + +## Diff + +`gui/src/pages/Subagents.tsx` + +- Replace `save()` with `persistRoster(next)`: `toggle` and `move` compute the next + list from the current `chosen`, set it optimistically, and persist it. +- Serialize writes: while a PUT is in flight, keep only the newest requested list in + a ref and send it when the current request settles, so rapid edits end at the + last list the user made. +- `committed` always advances to the last successful write, including a write + that a queued list has already superseded, because the server holds it. + Only the newest write's `applied` list is copied into `chosen`. +- A failed write with a queued successor skips the restore and still sends the + successor. A failed final write restores `committed.chosen` and shows the error. +- `saveInFlight` stays true across the whole drain, so a roster refresh cannot + slip into the gap between one response and the next send. +- Controls stay enabled during a write (no `busy` lockout per click). + +`gui/src/components/subagents-workspace/SubagentsWorkspace.tsx` + +- Remove the Save button row and the `onSave`/`busy` props it needed. +- Update the header comment ("reorder + save" → autosaved). + +Fallback list and delegation sections keep their existing controls. + +GUI tests: `gui/tests/subagents-classic.test.tsx` (PUT now follows a toggle, no +Save click), `gui/tests/subagents-busy-race.test.tsx` (rewrite around +latest-write-wins serialization and failure restore), and +`gui/tests/subagents-fallback.test.tsx` (`.swi-save-row` queries and +"no roster PUT" assertions). + +## Verification + +`cd gui && bun test --isolate tests`, `bun run typecheck`, `bun run lint:gui`, +`bun run build:gui`, a browser +check against a source proxy on a scratch port and `OPENCODEX_HOME` that +toggles a model and confirms the PUT and the persisted config, plus a screenshot +for the PR description. diff --git a/devlog/_plan/260923_subagent_roster_defaults_autosave/030_delivery.md b/devlog/_plan/260923_subagent_roster_defaults_autosave/030_delivery.md new file mode 100644 index 0000000000..71a78e5902 --- /dev/null +++ b/devlog/_plan/260923_subagent_roster_defaults_autosave/030_delivery.md @@ -0,0 +1,10 @@ +# 030 — Delivery + +1. Push both branches; open PR1 (base `dev`) and PR2 (base PR1 branch) with the + repository template. PR2 carries a screenshot hosted on `pr-assets`. +2. Inspect exact-head required CI for each head; queued, skipped, or cancelled + checks are not passes. +3. Squash-merge PR1 into `dev`, retarget PR2 to `dev`, confirm its checks at the + final head, squash-merge PR2. No rebase (user instruction). +4. Record the post-merge `dev` SHA and move this unit to `devlog/_fin/` in a + later docs pass. diff --git a/devlog/_plan/260923_subagent_roster_defaults_autosave/040_legacy_roster_cleanup.md b/devlog/_plan/260923_subagent_roster_defaults_autosave/040_legacy_roster_cleanup.md new file mode 100644 index 0000000000..bcf00f406e --- /dev/null +++ b/devlog/_plan/260923_subagent_roster_defaults_autosave/040_legacy_roster_cleanup.md @@ -0,0 +1,39 @@ +# 040 — Legacy 5.x cleanup in the version-2 roster upgrade + +User request (2026-09-23, after 010 shipped to PR #5640): when the roster upgrades +automatically, drop every gpt-5.5 / gpt-5.6 entry, and replace Sol and Luna with +their GPT-6 successors. + +## Rule (replaces 010's exact-match rule) + +In `migrateSubagentModels`, for version < 2, after the existing Astra step: + +- map each stored id in order: `gpt-5.6-sol` → `gpt-6-sol`, `gpt-5.6-luna` → + `gpt-6-luna`; drop any other bare id matching `/^gpt-5\.[56](?:-|$)/` + (5.5, 5.5-pro, 5.6-terra, and the rest of both families); +- keep the first occurrence of each id; +- only bare ids (no `/`) are rewritten. Routed `provider/model` ids and + account-qualified `<selector>/<model>` ids keep their exact spelling, because a + routed id with a 5.x suffix names a different provider's model; +- a list that was non-empty and becomes empty receives the GPT-6 defaults; an + explicitly empty list stays empty. + +The old exact default `[astra, 5.6-sol, 5.6-terra, 5.6-luna, 5.5]` becomes +`[astra, 6-sol, 6-luna]` under this rule, so the exact-match special case goes away. + +## Tests (tests/routing/subagent-roster-migration.test.ts) + +- replace "an edited version-1 roster is kept" with a table of version-1 inputs and + their cleaned results (reorder kept, 5.x mapped/dropped, routed ids untouched, + all-5.x list → defaults, empty stays empty); +- legacy table: `[one, astra, astra, 5.5, two]` → `[astra, one, two]`, + `[pool/gpt-6-astra, 5.5]` → `[astra, pool/gpt-6-astra]`; +- startup rebasing test: disk `[new, gpt-5.5]` → `[astra, new]`. + +Docs: English `agents.md` upgrade paragraph and `structure/subagents.md`. + +## Delivery + +Commit on `codex/subagent-roster-gpt6-defaults` (PR #5640), then merge that branch +into `codex/subagent-roster-autosave` (PR #5644) with a merge commit, no rebase, +so the stack's squash merges do not conflict. diff --git a/devlog/_plan/260923_usage_timeline_pool_merge/000_plan.md b/devlog/_plan/260923_usage_timeline_pool_merge/000_plan.md new file mode 100644 index 0000000000..b72e9ab042 --- /dev/null +++ b/devlog/_plan/260923_usage_timeline_pool_merge/000_plan.md @@ -0,0 +1,19 @@ +# Usage timeline: one series per provider/model across pool accounts + +## Problem + +The companion usage chart (native macOS tray, WidgetKit snapshot, and the GUI companion panel) +reads `GET /api/usage/timeline`. `src/usage/timeline.ts` keyed each series on the raw logged +provider, and ChatGPT/OpenAI pool accounts log as `openai-p<hex6>` (older rows as `openai-main` +or `chatgpt`). One model therefore drew one line per account: `openai-p6bc633/gpt-6-astra`, +`openai-pe2d42f/gpt-6-astra`, `openai/gpt-6-astra`, and so on, which also pushed real models into +the folded `other` row. The usage summary already folds these through `baseProviderLabel`. + +## Decision + +The timeline uses the same label as the summary. The account split stays available through the +existing `modelAccount` grouping. + +## Phases + +- [010_phase1.md](./010_phase1.md) — timeline normalization and regression test. diff --git a/devlog/_plan/260923_usage_timeline_pool_merge/010_phase1.md b/devlog/_plan/260923_usage_timeline_pool_merge/010_phase1.md new file mode 100644 index 0000000000..e1c109233b --- /dev/null +++ b/devlog/_plan/260923_usage_timeline_pool_merge/010_phase1.md @@ -0,0 +1,41 @@ +# Phase 1 — normalize timeline providers + +## Diff + +`src/usage/timeline.ts` + +- import `baseProviderLabel` from `../providers/label`. +- `timelineModelId(provider, model)` returns `${baseProviderLabel(provider)}/${model}`; series id, + series `provider`, and `availableModels` use it. +- `normalizeTimelineModelId` maps a saved `models` selection such as `openai-p6bc633/gpt-6-astra` + onto the merged id, so an old selection keeps selecting the (whole) merged row. + `appliedFilters.models` still echoes the raw request, which the Rust host and Swift client + compare against their settings. +- `hiddenProviders` hides an attribution when either the raw or the base provider is listed. +- `modelAccount` grouping: account = explicit `accountLogLabel`, else the `main`/`p<hex6>` + suffix of a provider, else `unknown`. + +`tests/usage/usage-timeline.test.ts` + +- one case feeding `openai-p6bc633`, `openai-pe2d42f`, `openai`, `openai-main`, `chatgpt` rows of + one model: merged row total, legacy model filter, both hidden-provider forms, account grouping. + +`src/companion/settings.ts` (audit finding) + +- The GUI panel (`companionTimelineProjection`), Swift `UsageTimeline.projected`, and the Rust + `timeline_rows`/`selected` projector all re-filter timeline rows against `settings.models` + verbatim. A saved `openai-p6bc633/...` selection would drop the merged `openai/...` row in every + client even though the server matched it. `applyCompanionSettingsPatch` (used by load and PUT) + now maps `models` through `normalizeTimelineModelId` and dedupes, so every client receives, + sends, and compares the merged ids from one server-side place. + +`tests/server/companion-settings.test.ts` + +- a saved per-account selection loads as merged ids, and a PUT with duplicate account forms dedupes. + +## Verification + +- `bun install --frozen-lockfile`: passed (104 packages installed). +- `bun test tests/usage/usage-timeline.test.ts tests/server/companion-settings.test.ts tests/providers/provider-registry-parity.test.ts tests/ci-workflows/structure-ssot.test.ts`: 129 pass, 0 fail after merging current `origin/dev`. This covers the timeline and persisted-settings behavior, the registry conflict resolution, and the owned structure contract. +- `bun run typecheck`: passed. +- `bun run test:changed` and the full local suite were not run under this repair lane's focused-validation limit. Exact-head hosted CI after the merge remains to be checked. diff --git a/devlog/_plan/260924_anthropic_fast_opt_in/010_plan.md b/devlog/_plan/260924_anthropic_fast_opt_in/010_plan.md new file mode 100644 index 0000000000..58a8a63a87 --- /dev/null +++ b/devlog/_plan/260924_anthropic_fast_opt_in/010_plan.md @@ -0,0 +1,68 @@ +# Anthropic Fast opt-in (default off) — plan (wp1) + +## Loop spec (HOTL wp1) + +- Request (2026-09-24): Anthropic fast mode spends usage credits, so keep it off by default; give the + Anthropic provider card on the dashboard Models page one row that turns it on and off; open a PR and + merge it. Push, PR, and merge are authorized for this change. +- Previous unit: devlog/_plan/260923_anthropic_fast_speed (#5604) made `claude-opus-5-5`, `claude-opus-5`, + `claude-opus-4-8` Fast-eligible on `anthropic` and `anthropic-apikey` with no switch other than the + global `fastMode`. Its residual already named the cost: every turn on an account without credits pays a + refused round trip, and an account with credits is billed 2x without a per-provider choice. +- Branch `codex/anthropic-fast-default-off` from `origin/dev` 6b7a91f575. + +## Design + +- D1 Registry: `ProviderRegistryEntry.fastOptIn?: boolean`. `true` means the provider's Fast lane is billed + beyond the plan and stays off until the operator enables it. Set on `anthropic` and `anthropic-apikey`. + The FastWire, model map, and tier description stay as they are, so enabling restores #5604 exactly. +- D2 Config: `OcxProviderConfig.fastEnabled?: boolean`. `false` turns Fast off for any provider; + `true` satisfies an opt-in registry entry; absent means "registry default" (off for opt-in entries, + unchanged elsewhere). Zod provider schema accepts a boolean; auth-cors field policy classifies it `editor`. +- D3 Policy: `buildFastPolicyAuthority` (src/providers/service-tier.ts) resolves the switch from the configured + provider, then the enriched provider, then `getProviderRegistryEntry(name)?.fastOptIn` (looked up by name + regardless of transport match, because it can only turn Fast off). An off switch sets the provider + capability to `false`, which `resolveFastPolicy` already treats as a global denial: eligibility becomes + `capability-unsupported`, catalog Fast toggles and `--fast` rows disappear, `decideTier` drops, and the + adapter never emits `speed`. No new policy branch in fastwire.ts. +- D4 Management API: PATCH /api/providers accepts `fastEnabled` (boolean, or null to clear). GET /api/providers + adds `fastOptIn: { enabled }` only for opt-in registry entries so the dashboard knows where to draw the row. +- D5 Dashboard: a small `ProviderFastRow` component (own file, Models.tsx is 8 lines under its ratchet cap) + with the same Off/On segmented control as the new-model policy row, label "Fast mode" and a hint that it + uses usage credits at 2x price. Rendered in the provider body for providers whose summary carries + `fastOptIn`. PATCH then reload. i18n keys in all ten locales. +- D6 Docs/SoT: docs-site providers reference (Anthropic Fast section + field table row), structure + providers-and-adapters note. + +## Tests + +- New `tests/providers/anthropic-fast-opt-in.test.ts`: registry default ineligible for both entries; + `fastEnabled: true` restores eligible; `fastEnabled: false` denies an ordinary service-tier provider; + `catalogFastRowEligible` false by default; PATCH round-trip sets/clears the field and GET exposes + `fastOptIn`. +- Rewrite fixtures that assumed default-on (responses-anthropic-fast-downgrade, fast pricing, fastwire roster + if affected) to set `fastEnabled: true` deliberately. + +## Acceptance + +- C1 default ineligible / opt-in eligible (c-1). C2 Models row renders and PATCHes (c-2, screenshot). +- C3 typecheck, focused tests, test:changed, ratchet/layout, structure:check, privacy:scan, lint:gui, build:gui. +- C4 PR to dev with template, exact-head CI green, merged (c-3). + +## Residuals + +- Claude Messages native passthrough forwards a caller's own `speed` field (Claude Code /fast); that is the + caller's explicit choice and stays outside this switch. +## Audit fold (A, reviewer NEAR-PASS) + +- B1 folded: one helper `providerFastSwitchOff(name, provider)` (src/providers/fast-opt-in.ts) is applied in + three places: the FastPolicy authority (service-tier.ts), `resolveModelPolicy` (static supportsServiceTier + and fastTierDescription), and router registry enrichment, which writes `supportsServiceTier: false` on the + resolved runtime provider so nameless `fastPolicyForModel` callers also see the denial. +- B2 folded by the same enrichment write. +- B3: PATCH already clears the provider model cache and runs `convergeCodexCatalog` for any non-pacing + field; the policy test covers the PATCH round trip. Claude listings compute per request. +- B4: flipped fixtures set `fastEnabled: true` deliberately; load-degrade's inherited-fastWire warning skips + providers whose switch is off. +- Residual accepted: native Claude Messages passthrough forwards the caller's own `speed`; documented. +- Scope confirmed by user: Cursor unchanged; only anthropic and anthropic-apikey default off. diff --git a/devlog/_plan/260924_anthropic_forced_tool_choice/010_plan.md b/devlog/_plan/260924_anthropic_forced_tool_choice/010_plan.md new file mode 100644 index 0000000000..d66a0b19b3 --- /dev/null +++ b/devlog/_plan/260924_anthropic_forced_tool_choice/010_plan.md @@ -0,0 +1,43 @@ +# Anthropic forced tool choice and Opus 5.5 + +## Scope +One implementation unit: inspect adapter-generated body and live status for required/named/auto tool choices on Opus 5.5 and a control, with and without explicit reasoning. A current token is read only; no proxy starts and no token/body is printed. + +## Hypotheses +H1: adaptive thinking and forced any/tool conflict; falsifier is a 2xx live response for the same adapter-produced body. +H2: Opus 5.5 rejects forced any/tool regardless of thinking; falsifier is a 2xx response with thinking omitted or disabled. +H3: proxy/router rather than upstream is responsible; falsifier is the same upstream error from the adapter-produced request. + +## Diff-level plan +If live evidence confirms H1, change only src/adapters/anthropic.ts around its reasoning and tool_choice construction, retaining forced any/tool and suppressing or disabling thinking only when live evidence proves the wire shape works. Add a focused test in tests/adapters/anthropic/ proving required/named semantics and unaffected auto/none cases. Update owning structure document if required; test the focused file, typecheck, structure:check, privacy:scan. Obtain gpt-6-sol adversarial review, push one branch and open a template PR to dev. Do not run the full suite, merge, or touch main/preview. + +## Audit and outcome + +Live adapter-generated probe, 2026-09-24, using the active Anthropic OAuth access token read +in memory from `~/.opencodex/auth.json` (token and response bodies were never printed). The +probe sent 12 small Messages requests through `createAnthropicAdapter(...).buildRequest()`; +the pre-fix wire fields and status were: + +| Model | Effort | Choice | Wire choice | Thinking | Status | +|---|---|---|---|---|---:| +| claude-opus-5-5 | medium / omitted | required | any | adaptive / omitted | 400 | +| claude-opus-5-5 | medium / omitted | named | tool | adaptive / omitted | 400 | +| claude-opus-5-5 | medium / omitted | auto | auto | adaptive / omitted | 200 | +| claude-opus-5 | medium / omitted | required | any | adaptive / omitted | 200 | +| claude-opus-5 | medium / omitted | named | tool | adaptive / omitted | 200 | +| claude-opus-5 | medium / omitted | auto | auto | adaptive / omitted | 200 | + +After the fix, the same 12 requests returned 200. Opus 5.5 required/named choices are sent +as `auto`; Opus 5 keeps `any`/`tool`. This agrees with Anthropic's published migration guide: +https://platform.claude.com/docs/en/models/opus-5-5/whats-new-opus-5-5 (forced tool use is +unsupported for Opus 5.5; `auto` and `none` are supported). The compatibility downgrade +preserves a named or allowed tool's candidate set and preserves `disable_parallel_tool_use`, +but it cannot preserve the upstream forced-call guarantee. + +For completeness, two supplementary requests manually overrode the adapter body to send +`thinking:{type:"disabled"}` together with `tool_choice` `any` and `tool`. Both returned +HTTP 400. The published guide also states that Opus 5.5 rejects disabled thinking, so there +is no thinking-off wire shape that can retain forced tool use for this model. + +## Plan reflection after vendor source check +Official Opus 5.5 change guide (https://platform.claude.com/docs/en/models/opus-5-5/whats-new-opus-5-5, read 2026-09-24) says thinking cannot be disabled and any/tool forced choices return 400; auto/none are supported. This falsifies a semantics-preserving repair for Opus 5.5. If the adapter-generated live probe agrees, change only Opus 5.5 forced choices to auto and explicitly document that required/named callers lose the guarantee; preserve the choice where upstream accepts it. The guide says the same restriction applies to Fable 5.1, but that model needs separate live or test evidence before expansion. diff --git a/devlog/_plan/260924_claude_desktop_picker_mode/000_plan.md b/devlog/_plan/260924_claude_desktop_picker_mode/000_plan.md new file mode 100644 index 0000000000..0d7b3b0054 --- /dev/null +++ b/devlog/_plan/260924_claude_desktop_picker_mode/000_plan.md @@ -0,0 +1,170 @@ +# Claude Desktop: gateway by default, first-party with a risk warning, and a picker that lists opencodex models + +Claude Desktop reaches opencodex in two ways. Gateway (3P) switches the whole app to the local +gateway and shows every opencodex model by name. First-party (1P) keeps the app on claude.ai and +routes only the Code tab's Claude Code through a local interception proxy; until now its model +picker could only show Anthropic's models, because claude.ai builds that list. This unit makes +gateway the default for new installs, warns that first-party sends a Claude subscription through +a local interception proxy and can get the account suspended, and adds a first-party **picker +mode**, on by default, that shows opencodex routes by their real names in the Code tab picker. +Picker mode points Desktop's own traffic at the opencodex CONNECT proxy through the +`egressProxyUrl` config-library key (supported in 1P since Desktop 1.44121.1), trusts a local CA +whose name constraints permit only `claude.ai` and its subdomains, and adds entries to the Code surface of the claude.ai +bootstrap. Evidence is in [001_research.md](001_research.md). The threat model stays in scratch +space (`.tmp/260924_claude_desktop_picker_mode/threat_model.md`, untracked) per AGENTS.md until the +change ships; its controls are restated in the decade docs. + +## Loop spec + +- Loop archetype: satisfy-spec, five work-phases (docs-first, then four dependency-ordered cycles). +- Trigger: the user asked for gateway as the default, a first-party account-suspension warning, + and first-party picker mode on by default, delivered through a merged PR. +- Goal: a fresh install applies gateway; choosing first-party shows the risk everywhere it can be + chosen or seen; with first-party applied on macOS the Desktop Code tab picker lists opencodex + routes by name and a turn with one of them is served by opencodex. +- Non-goals: Chat tab and other non-CLI surfaces; picker trust on Windows and Linux (reported as + unsupported); OS-wide proxy settings; changing `api.anthropic.com` interception; mentioning or + copying any third-party project. +- Verifier: per-phase focused suites named in each decade doc, `bun run typecheck`, + `bun run structure:check`, `bun run skill:surface:check`, `bun run privacy:scan`, + `bun run lint:gui`, `bun run build:gui`, `bun run test:changed` on the merge result, then the + live Desktop proof in [040](040_wp5_live_proof_pr_merge.md). +- Stop condition: PR merged to `dev` by squash after exact-head CI, merged tree verified. +- Memory artifact: this unit; goalplan + `.codexclaw/goalplans/opencodex-claude-desktop-default-to-gateway-3p-w/`. +- Expected terminal outcomes: DONE when every goalplan criterion has fresh evidence; NEEDS_HUMAN + only for the macOS keychain password dialog during the live proof; BLOCKED on repeated external + CI blockage; UNSAFE if picker mode could leave Desktop without connectivity. +- Escalation condition: any change to the operator's live service beyond the documented restart, + any need to disable upstream TLS verification, or a live bootstrap shape that contradicts 001. +- Resource bounds: writes limited to this worktree, `/private/tmp` scratch, the operator's + Desktop config library and login keychain during the live proof; pushes limited to the PR + branch and `pr-assets`; gpt-6-sol subagents for discovery, bounded slices and review; no token or + time budget was set by the user. + +## Work-phase map + +| Work-phase | Doc | Closes with | +| --- | --- | --- | +| wp1 docs-first | 000, 001 and the decade docs | roadmap locked, no code | +| wp2 gateway default + 1P warning | [010](010_wp2_gateway_default_and_warning.md) | resolver/CLI/API/sync tests, typed locales, docs parity | +| wp3 picker core | [020](020_wp3_picker_core.md) | CA/trust, CONNECT decision, claude.ai relay, bootstrap rewrite, route snapshot tests | +| wp4 picker activation | [030](030_wp4_picker_activation.md) | egress profile, CLI/API/GUI controls, docs, default-on in 1P | +| wp5 live proof + PR + merge | [040](040_wp5_live_proof_pr_merge.md) | Desktop screenshots, usage.jsonl proof, CI green, squash merge | + +The phases follow build order: the mode contract (wp2) is consumed by activation (wp4); the relay +and CA (wp3) must exist before anything selects the egress profile (wp4); nothing is shown to a +real Desktop before wp5. One branch, one PR, ordered commits. + +## Decisions + +Decision IDs come from the architect proposal; dispositions are main's. + +| ID | Decision | Disposition | +| --- | --- | --- | +| D1 | Default `gateway`; resolver takes observations (owned first-party settings applied/stale → legacy first-party) with precedence explicit → observed gateway → gateway fingerprint → owned first-party settings → gateway. Implicit applies persist the preserved mode. `/api/sync` stops writing a gateway profile when the resolved mode is first-party. | Accepted. Status stays read-only. Reflections r1/r2: the selected owned gateway row joins the observation; the intercept-disabled fallback is removed, so an observed first-party install keeps its mode and an apply with the intercept disabled is refused with `intercept_disabled`. | +| D2 | One owner for the risk text (`src/claude/desktop-risk.ts`), `riskWarning` in status, CLI apply/status, native toggle response, dashboard selector + active card, 10 locales, 8 guides; default badge moves to gateway. | Accepted. | +| D3 | Separate picker CA under `<configDir>/claude-picker/`, critical nameConstraints permitting `claude.ai` and excluding every IP address, leaf SAN `claude.ai` only; macOS `security add-trusted-cert -r trustRoot -p ssl -k <login keychain>` (the first build passed `-s claude.ai`; #5731 dropped it because Chromium skips host-scoped trust), trust checked with `security verify-cert -q -L -c <leaf> -p ssl -n claude.ai`, removal with `security remove-trusted-cert`. | Accepted, amended: tests use a fake command runner; no real keychain in CI. Reflection r1 gap 4 folded: "trusted" also requires the login keychain to hold a certificate whose SHA-1 equals the current picker CA (`security find-certificate -a -Z -c <CN> <login keychain>`), and `verify-cert` searches that keychain (`-k`). Audit r1 blocker 1 folded: the claim is narrowed to "`claude.ai` and its subdomains" (an RFC 5280 dNSName subtree cannot be exact), and it is only claimed for verifiers shown to enforce it: a Bun/BoringSSL test rejects an off-host leaf issued by the picker CA, and wp5 runs Apple `verify-cert` on an ephemeral off-host leaf after trust; whatever that shows is what the PR states. The primary control stays the 0600 key that never leaves the machine. | +| D4 | CONNECT decides per connection: `messages` (api.anthropic.com), `picker` (claude.ai, only when desired + first-party effective + listener ready + cached trust matches the CA fingerprint), else `blind`. Trust cache invalidated on toggle/rotation, rechecked on a bounded interval. | Accepted. | +| D5 | Dedicated `node:https` HTTP/1.1 terminator for claude.ai with `request` and `upgrade` handlers, fixed upstream `claude.ai:443`, verified TLS, raw headers and bodies streamed. | Accepted after a spike on Bun 1.4.0 (`/private/tmp/ocx-picker-spike/spike.ts`): gzip bytes, two `Set-Cookie` headers and a WebSocket upgrade passed through unchanged. | +| D6 | Rewrite only GET bootstrap responses (`/edge-api/bootstrap`, `/edge-api/bootstrap/{org}/app_start`, `/api/bootstrap…`) with a `code` surface; clone a selectable Claude entry per route; decode gzip/br/deflate under compressed and decompressed caps; fail open with the original bytes. | Accepted, amended: the bootstrap request's `accept-encoding` is narrowed to `gzip, deflate, br` so zstd never arrives. Reflection r1 gap 6 partly folded: clones drop `fast_mode` and every version-gate key (`/version/i`) with the presentation fields; `thinking` and `capabilities` stay because the intercept translates effort and handles images for routed models. | +| D7 | Picker routes mirror the gateway's rendered Desktop profile, ids minted with `aliasForRoute`/`claudeCodeNativeAlias`, served from a snapshot so bootstrap never waits on provider discovery. | Accepted, amended: snapshot built at server start (when picker is desired), on picker on/apply, and refreshed stale-while-revalidate after 10 minutes. | +| D8 | Picker egress profile is an owned standard row containing only `egressProxyUrl`; previous selection recorded in opencodex state (never in `_meta.json`); off/removal/gateway switch: stop terminating, reselect previous, delete owned row, untrust CA; partial cleanup reported. | Accepted. Config field is `claudeCode.intercept.picker?: boolean` (absent = on in 1P on macOS). Reflections r1–r3: every disable first calls the running runtime's `disarm()` (new claude.ai CONNECTs go blind at once, independent of the preference and of the cached mode), then removes the profile, then untrusts. Only an explicit `picker off` persists `false`; mode-transition cleanup leaves the preference unset so first-party turns the picker back on. | +| D9 | `ocx claude desktop picker on|off|status|trust`, `GET/PUT /api/claude-desktop/picker` (standard management auth: CLI admin token and dashboard session), `firstParty.picker` in status with desired/effective/reason/hint/residual. | Accepted; reshaped by D11: `on`/`off` go through the server when one runs; `trust` is the only CLI-local mutation; `off` and transition cleanup run locally only when no server runs. | +| D10 | Live mode propagation: the runtime's `refresh()` decides from a fresh persisted read (resolved mode, Desktop intent, picker preference); `selectTunnel` reads only the cached decision; a disarm latch that only a completed, verified enable clears; server paths adopt committed `claudeCode` for status. | Accepted with the architect's amendments; the POST disarm/refresh routes of earlier revisions are dropped under D11. | +| D11 | While a server runs, every picker mutation (enable, disable, transition cleanup) runs in one server-side `DesktopPickerController`, serialized by one lock; CLI first-party apply delegates to `POST /api/claude-desktop/apply` like gateway apply already does; enable re-reads persisted state, commits an explicit preference first, checks mode/intent/preference/bound proxy before and after trust, and compensates trust it added on any later failure; startup never trusts or writes a profile. | Added at the second P re-entry after audit round 5 (ordering and race findings from rounds 4–5 all came from two mutation sites); architect reflection below. | + +## Field chains (PLAN-FIELD-CHAIN-01) + +| Field | Creation | Serialization | Deserialization | Consumers | +| --- | --- | --- | --- | --- | +| `claudeCode.intercept.picker?: boolean` | `DesktopPickerController.enable({ persist: true })` / `.disable({ persist: true })` (CLI `picker on|off` and the dashboard toggle through `PUT /api/claude-desktop/picker`); offline CLI `picker off` writes `false` locally before `removeDesktopPickerArtifacts`; first-party apply and transition cleanup never write it (absent = on) | `config.json` through the field-scoped config writer; `src/types/config.ts:151` type; `src/config/schema/config-schema.ts` boolean validation | `loadConfig` → `OcxClaudeCodeConfig.intercept.picker`; a non-boolean fails schema validation | `pickerDesired` (picker-runtime `refresh()`), `DesktopPickerController.enable` checks, `DesktopPickerController.status`, status payload | +| `riskWarning: { code, message } \| null` | status builder (agent-settings-routes), first-party apply response | JSON response | GUI `DesktopStatus` (`gui/src/pages/ClaudeDesktop.tsx:62`); CLI `status` prints every key | GUI callout (localized by `claudeDesktop.mode.firstPartyRisk`, gated on presence), CLI status | +| `firstParty.picker: DesktopPickerStatus` | `DesktopPickerController.status()` in the status builder (with no controller running: a static "proxy_unavailable" status); `GET/PUT /api/claude-desktop/picker` | JSON response | GUI `DesktopFirstPartyStatus` (`ClaudeDesktop.tsx:49`); CLI `picker status` | `ClaudeDesktopPicker` card, CLI output | +| config-library row name `opencodex-picker` | `applyDesktopPickerProfile` | `_meta.json` `entries[].name` (Desktop's schema; no opencodex keys added) | `parseMetadata` (desktop-3p-library) | `isOwnedDesktopEntry` (owned, inspection kind `standard` because the profile has no `inferenceProvider`), `removeDesktopPickerProfile`; `isOwnedDesktopGatewayEntry` stays `name === "opencodex"`, so gateway writes and gateway cleanup never pick it | +| `PUT /api/claude-desktop/picker` body `{ enabled, persist, trustedLocally?, callerAddedTrust? }` | CLI `picker on|off|trust`, dashboard toggle, CLI transition helpers | JSON request | route handler validation (booleans only; unknown keys rejected) | `DesktopPickerController.enable/disable`; `trustedLocally` only selects the trust-outcome wording (`trust_declined` vs `trust_pending`) and never skips a check; `callerAddedTrust` makes the server compensate that trust inside its lock on refusal or failure | +| `ClaudeDesktopModeObservation` | `observeClaudeDesktopMode` | N/A — in-process value | N/A | `resolveClaudeDesktopMode`, `resolveClaudeDesktopApplyMode` callers listed in 010 | + +## Guard bypasses (PLAN-BYPASS-NAMED-01) + +The trust gate on claude.ai termination is a safety guard, not enforcement. + +- Tier: runtime check in process (no OS or build gate). +- Executing surface: `PickerRuntime.selectTunnel` on every new CONNECT to `claude.ai:443`. +- Known bypass: trust removed outside opencodex (Keychain Access) stays cached as trusted for up to + the 60 s refresh interval; connections opened in that window fail TLS in Desktop until the next + refresh. An operator who edits the config library by hand can point Desktop elsewhere. While a + server runs, enable and disable are serialized by the picker controller's lock, and the disarm + latch is cleared only at the end of an enable whose checks all passed; with no server running, + nothing can terminate claude.ai. A process that edits opencodex's config or the keychain + directly, outside these paths, is not constrained. +- Residual risk: Desktop has no network while its pinned egress proxy is down; stated in CLI, GUI + and docs. +- Wording downgrade: described as a guard everywhere; no document calls it enforcement. +- Final layer: none. + +## Architect consultation + +- Handle: `01a0d104-9116-7d22-b92d-81bc155eab50` (gpt-6-sol, CXC-ROLE architect, read-only). +- Proposal D1–D9 above; dispositions recorded in the table. +- Reflection on revision r1: MISALIGNED with six gaps. Gap 1 (security working note tracked in + devlog) folded: the threat model moved to scratch. Gap 2 partly folded, gaps 3–5 folded, gap 6 + partly folded; dispositions are in the decision table. +- Reflection r2: MISALIGNED (3 gaps: intercept-disabled fallback, picker mode from config alone, + transition cleanup persisting false) — all folded into 010/020/030. +- Reflection r3: MISALIGNED (cleanup could keep terminating while the cached mode was still + first-party; stale table wording) — folded: `disarm()` first, table updated. +- Reflection r4: MISALIGNED (no server path to disarm without changing the preference; in-flight + `ensureStarted()` could re-arm) — folded: `POST /api/claude-desktop/picker/disarm`, arm + generation guard, route and race tests. +- Reflection r5: **ALIGNED**, no remaining material gap. +- After audit round 1 and round 2 amendments: rechecks MISALIGNED (overflow chunk, failed compensation; + grok-lifecycle source assertions) → folded → **ALIGNED**. +- P re-entry after audit round 3 (LOOP-REPAIR-01): D10 proposed by main, amended by the architect + (disarm latch, integration intent, native adoption, OFF→ON test), reflection MISALIGNED once + (integration intent not visible to the runtime) → decision source changed to a fresh persisted + read → **ALIGNED**. +- Second P re-entry after audit round 5: D11 (single server-side controller) proposed by main; + the architect amended it three times (transition lease, CLI trust compensation, preference commit + after independent checks; then restart_required and lost-response handling) → **ALIGNED**. + +## Audit record + +- Reviewer `01a0d114-541c-7d13-8277-ed9f711dad59` (gpt-6-sol, CXC-ROLE reviewer). +- Round 1: FAIL (7 blockers: CA boundary claim, /api/sync race, durable-OFF cleanup, oversize + fail-open handoff, signature mismatches, trust compensation, trust_pending test) → all folded. +- Round 2: FAIL (5: overflow triggering chunk, async ensure caller, stale mode passed to enable, + missing cap seam, hint/declined contract) → all folded. +- Round 3: FAIL (1: committed mode not reaching the running runtime) → returned to P, D10 added. +- Round 4: FAIL (4: stranding Desktop without a bound proxy, failed mode write, latch cleared by a + caller claim, CLI/route auth mismatch) → folded. +- Round 5: FAIL (4: off→on refused by its own guard, stale intent in CLI apply, concurrent enable + during cleanup, second-guard compensation) → second P re-entry, D11. +- Round 6: FAIL (4: /api/sync outside the lock, busy guard vs in-lock rearm, lost CLI response + racing server enable, live-proof rollback leaving picker state) → folded; architect rechecks added + early-refusal compensation. +- Round 7: FAIL (2: startup refresh missing, config-routes.ts absent from the wp4 inventory) → + folded; architect recheck added the bounded first-CONNECT wait and the persisted route snapshot. +- Round 8: **GO-WITH-FIXES (blockers=1)** — field-chain rows still named pre-D11 functions → + folded. Main's judgment: near-pass; no High/Critical blocker remains. + +## wp1 close (D) + +Roadmap locked at D1–D11; the next cycle is wp2 (gateway default and the first-party risk warning). + +- What changed: the unit's plan, research and four diff-level decade docs; the threat model stays in + scratch. +- Hypotheses that died: "no local change can add a row to the first-party picker" (true only for + Desktop builds older than 1.44121.1, which lack `egressProxyUrl` in 1P); "Desktop filters + non-Anthropic model ids" (only the custom-3P provider does; 1P returns `{ok:true}`); "Cloudflare + in front of claude.ai rejects a re-originated TLS client" — a Bun `node:https` GET of + `/edge-api/bootstrap` and `/api/bootstrap` returned 200 JSON (brotli) with no `cf-mitigated` + header (`/private/tmp/ocx-picker-spike/cf.ts`, 2026-09-24). +- What did not improve: the audit needed eight rounds and two returns to P; every late finding was + about ordering between two mutation sites, which D11 removed. The activation surface is still the + largest part of the change. +- Evidence that would show the direction is wrong: a logged-in bootstrap without a `code` surface + in `model_selector_config` (the logged-out bootstrap has no `model_selector_config` at all, so + this is only checkable in wp5); Desktop's Chromium refusing a login-keychain-trusted root with + name constraints; the launchd service never able to raise the keychain dialog (then + `trust_pending` + `picker trust` is the only path, which the plan already supports). diff --git a/devlog/_plan/260924_claude_desktop_picker_mode/001_research.md b/devlog/_plan/260924_claude_desktop_picker_mode/001_research.md new file mode 100644 index 0000000000..1c3e3b2f88 --- /dev/null +++ b/devlog/_plan/260924_claude_desktop_picker_mode/001_research.md @@ -0,0 +1,83 @@ +# 001 — Research: how the Desktop Code tab picker can list opencodex models (2026-09-24) + +Evidence was read from Claude.app 2.7032.0 (`/Applications/Claude.app/Contents/Resources/app.asar`, +byte offsets below), from claude.ai renderer bundles saved during the 2026-09-23 probe (kept outside +the repository under `/tmp/ocx-claude-probe/web/`; they may be older than the live renderer), and +from this repository at `37f93da6ac`. Nothing here was observed on the wire yet; the items marked +**live** are verified in wp5. + +## Desktop app facts + +- `egressProxyUrl` is a config-library key supported in both deployment scopes: + `support:{enabled:{scopes:["3p","1p"],availableInVersion:"1.44121.1"}}`, `appBehaviorOnly:!0` + (app.asar ≈11009657). It is read once at launch and applied as Chromium `--proxy-server` with a + bypass list for loopback and `*.local`; PAC (`egressProxyPacUrl`) takes precedence + (≈20277302–20278073). Startup logs `[egress-proxy] pinned …; OS proxy settings ignored` + (≈20329028). Claude Code processes the app spawns receive `HTTPS_PROXY`/`HTTP_PROXY`/`NO_PROXY`. + The 1.18286.0 build probed for the previous unit predates this key, which is why that unit could + not reach the picker. +- On macOS the app reads its config library from the user-data directory with a `-3p` suffix + (`~/Library/Application Support/Claude-3p/configLibrary/`) in both scopes; `_meta.json` + `appliedId` selects `<id>.json` (≈11136500–11138403, ≈11158102). opencodex already owns this + writer (`src/claude/desktop-3p-library.ts`, `src/claude/desktop-3p-paths.ts:47`). +- The model-list keys `modelCatalogUrl`, `modelCatalogEnabled` and `modelDiscoveryEnabled` are + 3P-only (`scopes:["3p"]`). No config key sets an app-level trusted CA, so Desktop's renderer + relies on the operating system trust store for claude.ai. +- The first-party provider class (`hasClaudeAiProductFeatures(){return!0}`, + `managesProviderRouting(){return!1}`) implements `validateSessionModel(e,t){return{ok:!0}}` + (≈12961260). The non-Anthropic model regex (`…|gpt|grok|kimi|…`, ≈10697300) applies only to the + custom-3P provider's `validateSessionModel` (≈12952304). In 1P, a picked id reaches Claude Code + unchanged through `query.setModel`. + +## claude.ai renderer facts (saved bundles) + +- The picker catalog comes from the bootstrap response field `model_selector_config`: an array of + surfaces `{ id, models[], description?, presets?, featured?, auto_compact_window? }`. The Desktop + bridge takes the selectable models of the surface `"code"` and sends their ids to the app with + `setAvailableCodeModels`. `"cowork"` has its own catalog; `"ccr"`/`"ccd"` share Code-like + selection persistence but are not shown to feed this bridge. +- A model entry carries `id`, `name`, `section`, `description`, `badge`, `tooltip`, + `disabled`, `disabled_reason`, `context_window`, `thinking`, `fast_mode`, `capabilities` and + optional version gates. It is listed when `section` is `"main"` or `"overflow"` and selectable when + it is not disabled, has no `disabled_reason` and is not `"deprecated"`. The default selection is a + separate `model_selector_state`; `contextWindowByModel` is derived from each entry's + `context_window`. +- The bootstrap is fetched with credentials from `/edge-api/bootstrap/{org}/app_start` or + `/edge-api/bootstrap` with `statsig_hashing_algorithm=djb2&growthbook_format=sdk&cache_bust=1` + (another provider defaults to an `/api` prefix) and parsed with `response.json()`; no body + signature or integrity check exists on that path. A `bootstrap_push_revision` guard keeps a newer + catalog revision when an older network result arrives. +- claude.ai also carries WebSockets (`/v1/sessions/ws/{id}/subscribe`, Code terminal + `/v1/code/sessions/{id}/terminal`, `/api/ws/…` voice) and streaming responses + (`/v1/code/sessions/{id}/events/stream`, `/v1/code/sessions/watch`, SSE chat, an MCP + `EventSource`). A terminating proxy has to pass all of them through. + +## opencodex facts + +- Mode: `DEFAULT_CLAUDE_DESKTOP_MODE = "first-party"` (src/claude/desktop-first-party.ts:36); + `resolveClaudeDesktopMode` returns an explicit `desktopMode`, then `gateway` for an applied + gateway fingerprint, then the default (:49–53). Every apply path persists the chosen mode through + `recordClaudeDesktopMode` (:62). An install that applied first-party before the field existed + would silently resolve to gateway if only the constant changed, and the next implicit apply would + retire its env. Five tests in tests/claude-integration/claude-desktop-first-party.test.ts assert + the current default (:64, :78, :86, :138, :182). +- Intercept: the CONNECT proxy splices only `api.anthropic.com` (a startup snapshot, + src/claude/intercept/connect-proxy.ts:15, :121) and blind-tunnels the rest; loopback targets get + 403 and non-CONNECT requests 405. The TLS listener is one `Bun.serve` with one certificate + (listener.ts:106) and relays non-Messages paths through `fetch`, which decodes bodies and drops + `Upgrade` (listener.ts:19–29, :62–88), so it cannot carry claude.ai as is. +- CA: `local-ca.ts` builds P-256 certificates with local DER helpers (`tlv`, `contextTag`, + `objectIdentifier`, `extension`, :48–109) and keeps the CA out of OS trust (:9); a + nameConstraints extension (OID 2.5.29.30) is expressible with the same helpers. +- Routed ids: `aliasForRoute` mints `ocx-claude-<provider>--<model>` (src/claude/alias.ts:94) and the + Messages path already resolves it, so a picker entry with that id routes without a binding. +- File-size ratchet: none of the touched intercept/desktop files has a cap; `src/server/index.ts` is + at 883 of 893, so no new wiring lands there. + +## Open questions verified live in wp5 + +1. The live bootstrap still carries `model_selector_config` with a `"code"` surface, and a cloned + entry renders and is selectable. +2. Chromium in Desktop accepts a claude.ai leaf chained to a login-keychain-trusted root that + carries nameConstraints. +3. Desktop behaves with an HTTP/1.1-only terminator (ALPN) for claude.ai, including WebSockets. diff --git a/devlog/_plan/260924_claude_desktop_picker_mode/010_wp2_gateway_default_and_warning.md b/devlog/_plan/260924_claude_desktop_picker_mode/010_wp2_gateway_default_and_warning.md new file mode 100644 index 0000000000..950a89e840 --- /dev/null +++ b/devlog/_plan/260924_claude_desktop_picker_mode/010_wp2_gateway_default_and_warning.md @@ -0,0 +1,285 @@ +# 010 — wp2: gateway by default, first-party account-risk warning + +Consumes: D1, D2 in [000](000_plan.md). Produces the mode contract wp4 builds on. + +## Files + +| Path | Change | +| --- | --- | +| `src/claude/desktop-first-party.ts` | MODIFY: default constant, observation-aware resolver, observation helper, header comment | +| `src/claude/desktop-risk.ts` | NEW: the single owner of the first-party account-risk text | +| `src/cli/claude-desktop.ts` | MODIFY: help text, `defaultDesktopApplyMode` observes settings, apply prints the risk | +| `src/cli/ensure-desired-integrations.ts` | MODIFY: resolve with observation | +| `src/server/management/agent-settings-routes.ts` | MODIFY: apply default + status `riskWarning` | +| `src/server/management/native-integration-routes.ts` | MODIFY: status/enable observe; enable message carries the risk | +| `src/server/management/config-routes.ts` | MODIFY: `/api/sync` skips the gateway writer when the resolved mode is first-party | +| `src/server/management/agent-settings-routes.ts` (also) | MODIFY: `autoApplyDesktopBestEffort` (:212, the roster-update gateway writer) returns early when the resolved mode is first-party, both before and after its model discovery await | +| `src/types/config.ts` | MODIFY: the `desktopMode` doc comment (:183) names gateway as the default and first-party's risk | +| `tests/claude-integration/claude-desktop-mode-explanation.test.ts` | MODIFY: explanation cases for the new default | +| `structure/gui-and-management-api.md` | MODIFY: the Desktop apply default (:182) is gateway; `riskWarning` in the status payload | +| `gui/src/styles/claude-desktop-mode-picker.css` | MODIFY: header comment (:1) — gateway is the default, first-party is the opt-in with the risk callout (plus the callout style if it fits here) | +| `gui/src/pages/ClaudeDesktop.tsx` (also) | MODIFY: stale comments at :240 and :351 that call first-party the default | +| `gui/src/pages/ClaudeDesktop.tsx` | MODIFY: default badge + fallback mode = gateway; first-party risk callout | +| `gui/src/pages/ClaudeDesktop.tsx` (Desktop status type lives here) | MODIFY: `riskWarning` in the status type | +| `gui/src/i18n/{en,de,fr,ko,zh,zh-TW,ru,ja,tr,vi}.ts` | MODIFY: `claudeDesktop.mode.firstPartyRisk`; hints no longer call first-party the default | +| `docs-site/src/content/docs/{,fr/,ja/,ko/,ru/,tr/,zh-cn/,zh-tw/}guides/claude-code.md` | MODIFY: gateway is the default; caution block in the first-party section | +| `structure/clients/claude-desktop.md` | MODIFY: mode contract (default, legacy observation, risk warning, sync guard) | +| `tests/claude-integration/claude-desktop-first-party.test.ts` | MODIFY: the five default assertions + new cases | + +## Diff + +`src/claude/desktop-first-party.ts` + +```diff +- * - `first-party` (default): the app keeps its ordinary claude.ai login, … ++ * - `gateway` (default): the third-party deployment profile (src/claude/desktop-3p.ts) … ++ * - `first-party`: the app keeps its claude.ai login … Carries an account-risk warning ++ * (src/claude/desktop-risk.ts). +-export const DEFAULT_CLAUDE_DESKTOP_MODE: ClaudeDesktopMode = "first-party"; ++export const DEFAULT_CLAUDE_DESKTOP_MODE: ClaudeDesktopMode = "gateway"; ++ ++/** What the resolver may learn from disk. Only owned rows and owned settings count. */ ++export interface ClaudeDesktopModeObservation { ++ /** Desktop's selected config-library row is our gateway (current or drifted). */ ++ ownedGatewaySelected?: boolean; ++ ownedFirstPartySettings?: boolean; ++} ++ ++/** Observe owned first-party settings (applied or stale). Never throws; unreadable = none. */ ++export function observeClaudeDesktopMode( ++ config: Pick<OcxConfig, "claudeCode" | "port" | "runtimeRole">, ++ options: DesktopFirstPartyOptions = {}, ++): ClaudeDesktopModeObservation { ++ const observed: ClaudeDesktopModeObservation = {}; ++ try { ++ const library = inspectDesktop3pConfigLibrary({ appliedFingerprint: config.claudeCode?.desktopProfile?.appliedFingerprint ?? null }); ++ observed.ownedGatewaySelected = library.kind === "gateway_ours" || library.kind === "gateway_drifted"; ++ } catch { /* unreadable library: no gateway evidence */ } ++ try { ++ const kind = inspectDesktopFirstParty(config, options).settings.kind; ++ observed.ownedFirstPartySettings = kind === "applied" || kind === "stale"; ++ } catch { /* unreadable settings: no first-party evidence */ } ++ return observed; ++} +-export function resolveClaudeDesktopMode(config: DesktopModeConfig): ClaudeDesktopMode { ++export function resolveClaudeDesktopMode( ++ config: DesktopModeConfig, ++ observed: ClaudeDesktopModeObservation = {}, ++): ClaudeDesktopMode { + const explicit = config.claudeCode?.desktopMode; + if (isClaudeDesktopMode(explicit)) return explicit; ++ if (observed.ownedGatewaySelected) return "gateway"; + if (config.claudeCode?.desktopProfile?.appliedFingerprint) return "gateway"; ++ // An install that applied first-party before the mode was persisted keeps first-party. ++ if (observed.ownedFirstPartySettings) return "first-party"; + return DEFAULT_CLAUDE_DESKTOP_MODE; + } + export function resolveClaudeDesktopApplyMode( + config: Pick<OcxConfig, "claudeCode" | "runtimeRole">, ++ observed: ClaudeDesktopModeObservation = {}, +): ClaudeDesktopMode { +- const resolved = resolveClaudeDesktopMode(config); ++ const resolved = resolveClaudeDesktopMode(config, observed); +- if (resolved === "gateway" || isClaudeDesktopMode(config.claudeCode?.desktopMode)) return resolved; +- return claudeInterceptEnabled(config) ? "first-party" : "gateway"; ++ return resolved; +``` + +With gateway as the default, an implied first-party can only come from observed legacy settings, +so the intercept-disabled fallback branch is removed: an observed first-party install keeps its +mode, and an apply with the intercept disabled is refused with `intercept_disabled` exactly like an +explicit first-party (reflection r2 gap 1). `resolveClaudeDesktopApplyMode` stays as the named +entry point callers use; `claudeInterceptEnabled` is no longer imported by it. +`observeClaudeDesktopMode` is placed after `inspectDesktopFirstParty` (it calls it) and reaches +`inspectDesktop3pConfigLibrary` without a new static import cycle (confirm the import graph at P). +Resolver stays pure; callers that decide an apply or a write pass `observeClaudeDesktopMode(config)`. + +`src/claude/desktop-risk.ts` (NEW) + +```ts +/** Account-risk notice every first-party surface shows. One owner so the wording cannot drift. */ +export const FIRST_PARTY_ACCOUNT_RISK = { + code: "first_party_account_suspension_risk", + message: "First-party mode sends Claude subscription traffic through a local interception proxy. " + + "Anthropic may treat this as a violation of its terms and suspend the account. Use it at your own risk; " + + "gateway mode is the default.", +} as const; +export type FirstPartyAccountRisk = typeof FIRST_PARTY_ACCOUNT_RISK; +``` + +`src/cli/claude-desktop.ts` + +```diff +- --first-party (default) keep Desktop on claude.ai; route only the Code tab's Claude Code +- through the local intercept proxy via ~/.claude/settings.json env +- --gateway install the third-party gateway profile for the whole app ++ --gateway (default) install the third-party gateway profile for the whole app ++ --first-party keep Desktop on claude.ai; route only the Code tab's Claude Code through the ++ local intercept proxy. Risk: Anthropic may suspend the account. +-export function defaultDesktopApplyMode( +- config: Pick<OcxConfig, "claudeCode" | "runtimeRole">, ++export function defaultDesktopApplyMode( ++ config: Pick<OcxConfig, "claudeCode" | "port" | "runtimeRole">, +- const resolved = resolveClaudeDesktopApplyMode(config); ++ const resolved = resolveClaudeDesktopApplyMode(config, observeClaudeDesktopMode(config)); + if (target.kind === "first-party") { + console.log(`Claude Desktop first-party 설정을 적용했습니다: ${result.path}`); + console.log("Desktop 앱 설정은 그대로이며, Code 탭의 Claude Code만 로컬 프록시를 거칩니다."); ++ console.warn(`⚠️ ${FIRST_PARTY_ACCOUNT_RISK.message}`); +``` + +`gatewayModeExplanation` (src/cli/claude-desktop.ts:212–247) is rewritten for the new default: when gateway +was applied without an explicit flag and first-party can run here (not a connected client, intercept +enabled), it prints "Gateway is the default.", the first-party alternative +(`ocx claude desktop apply --first-party`) and `FIRST_PARTY_ACCOUNT_RISK.message`; explicit gateway +requests and machines that cannot run first-party get nothing. Its doc comment drops the "help calls +first-party the default" premise. + +The bindings gateway warning (`resolveClaudeDesktopMode(config) === "gateway"`) also passes the +observation. `status` prints the payload, which now carries `riskWarning`. `parseDesktopApplyArgs` +(the only caller of `defaultDesktopApplyMode`) widens its config type the same way; its callers +already pass a loaded `OcxConfig`. The gateway-mode explanation that suggests +`ocx claude desktop apply --first-party` (`gatewayModeExplanation`, src/cli/claude-desktop.ts:243) +appends `FIRST_PARTY_ACCOUNT_RISK.message` under the suggestion. + +`src/cli/ensure-desired-integrations.ts` + +```diff +- if (resolveClaudeDesktopMode(config) !== "first-party") return; ++ if (resolveClaudeDesktopMode(config, (deps.observeClaudeDesktopMode ?? observeClaudeDesktopMode)(config)) !== "first-party") return; +``` + +`src/server/management/agent-settings-routes.ts` + +```diff +- let desktopMode: "first-party" | "gateway" = resolveClaudeDesktopApplyMode(config); ++ let desktopMode: "first-party" | "gateway" = resolveClaudeDesktopApplyMode(config, observeClaudeDesktopMode(config)); + …status… +- const mode = gatewayApplied ? "gateway" : resolveClaudeDesktopApplyMode(persisted); ++ const mode = gatewayApplied ? "gateway" : resolveClaudeDesktopApplyMode(persisted, observeClaudeDesktopMode(persisted)); ++ const { FIRST_PARTY_ACCOUNT_RISK } = await import("../../claude/desktop-risk"); ++ const riskWarning = mode === "first-party" || firstPartySeen.applied || firstPartySeen.stale ++ ? { ...FIRST_PARTY_ACCOUNT_RISK } : null; + return jsonResponse({ + desiredEnabled, + mode, ++ riskWarning, + firstParty, +``` + +The first-party apply response adds `riskWarning: { ...FIRST_PARTY_ACCOUNT_RISK }`. + +`src/server/management/native-integration-routes.ts`: `desktopStatus` and the enable branch call +`resolveClaudeDesktopApplyMode(config, observeClaudeDesktopMode(config))`; the first-party enable +message appends `FIRST_PARTY_ACCOUNT_RISK.message`. + +`src/server/management/config-routes.ts` (`/api/sync` client integrations) + +```diff +- if (claudeDesktopIntegrationEnabled(config)) { ++ if (claudeDesktopIntegrationEnabled(config) ++ && resolveClaudeDesktopMode(config, observeClaudeDesktopMode(config)) !== "first-party") { + … + const latest = loadConfig(); +- if (claudeDesktopIntegrationEnabled(latest)) { ++ // Discovery awaited: the mode may have changed meanwhile. Re-resolve on the fresh read, ++ // immediately before the writer, so a first-party switch during fetchAllModels still wins. ++ if (claudeDesktopIntegrationEnabled(latest) ++ && resolveClaudeDesktopMode(latest, observeClaudeDesktopMode(latest)) !== "first-party") { +``` + +wp4 moves this post-discovery block (the re-read, the re-resolve and `writeDesktop3pConfig`) +inside the picker controller's `transition` when a controller exists (030, D11), so a sync that +resumes while a first-party apply is between gateway cleanup and its mode commit waits for the +lock and then sees first-party. wp2 lands the re-resolve; wp4 adds the lock. + +A first-party Desktop no longer gets a gateway profile written and selected by a catalog sync. + +`autoApplyDesktopBestEffort` (agent-settings-routes.ts:212) gets the same guard twice: after +`loadConfig()` into `admitted` and again after the `fetchAllModels` await on `current`: + +```diff + if (!claudeDesktopIntegrationEnabled(admitted)) return; ++ if (resolveClaudeDesktopMode(admitted, observeClaudeDesktopMode(admitted)) === "first-party") return; + … + if (!claudeDesktopIntegrationEnabled(current)) return; ++ if (resolveClaudeDesktopMode(current, observeClaudeDesktopMode(current)) === "first-party") return; +``` + +`gui/src/pages/ClaudeDesktop.tsx` + +```diff +- const effectiveMode: DesktopMode = status?.mode ?? "first-party"; ++ const effectiveMode: DesktopMode = status?.mode ?? "gateway"; +- {mode === "first-party" && <span className="claude-mode-default">{t("claudeDesktop.mode.defaultBadge")}</span>} ++ {mode === "gateway" && <span className="claude-mode-default">{t("claudeDesktop.mode.defaultBadge")}</span>} + …after the mode options… ++ {selectedMode === "first-party" && ( ++ <p className="claude-mode-risk" role="note">{t("claudeDesktop.mode.firstPartyRisk")}</p> ++ )} +``` + +The callout also renders under the status bar when `status.riskWarning` is set and the picker +fieldset is not showing first-party (so an applied first-party install sees it after reload). +Styling goes in the existing Claude Desktop stylesheet only if it has room under the ratchet; +otherwise in a new `gui/src/styles/claude-desktop-risk.css` imported from `gui/src/main.tsx`. + +i18n: English source + +```ts +"claudeDesktop.mode.firstPartyRisk": "Account risk: first-party sends your Claude subscription traffic through a local interception proxy. Anthropic may treat this as a terms violation and suspend the account. Gateway is the default.", +``` + +plus the nine translations with the same three facts (proxy, possible suspension, gateway default). + +Docs (all eight guides): the mode section states gateway is the default; the first-party +subsection opens with + +```md +:::caution[Account risk] +First-party mode sends your Claude subscription traffic through a local interception proxy. +Anthropic may treat this as a violation of its terms and suspend the account. Gateway is the +default; choose first-party only if you accept that risk. +::: +``` + +and the dashboard recap line (`claude-code.md:760` in English) names gateway as the default. + +## Tests + +`tests/claude-integration/claude-desktop-first-party.test.ts` + +1. "mode resolution: explicit wins, applied gateway fingerprint keeps gateway, owned first-party + settings keep first-party, otherwise gateway" (replaces :64). +2. "implied apply mode is gateway; a legacy first-party install keeps first-party unless the + intercept is disabled" (replaces :78) → renamed "implied apply mode is gateway; a legacy + first-party install keeps first-party, and with the intercept disabled the apply is refused + with intercept_disabled instead of switching to gateway". +3. "CLI apply flags: default gateway, --first-party explicit, legacy shape flags imply gateway, + conflicts rejected" (replaces :86). +4. "POST /api/claude-desktop/apply defaults to gateway, first-party on request, and returns the + risk warning" (replaces :138). +5. "native toggle: enable applies gateway by default; a legacy first-party install keeps + first-party and its message carries the risk" (replaces :182). +6. NEW "status carries riskWarning for first-party and null for gateway". +7. NEW "/api/sync does not write a gateway profile while first-party is resolved" (fake + `writeDesktop3pConfig` dep asserts it is not called). Activation: config with explicit + `desktopMode: "first-party"` and `claudeDesktop` integration enabled. +7b. NEW "/api/sync re-resolves after discovery": a fake `fetchAllModels` resolves only after the + test persists `desktopMode: "first-party"`; the writer must not be called. +8. NEW "a foreign HTTPS_PROXY in settings.json is not first-party evidence" (settings kind + `foreign` → gateway). +9. NEW "a selected owned gateway row outranks legacy first-party settings" (both observed → + gateway). +10. NEW "a roster update does not write a gateway profile on an explicit first-party install" and + "… nor when the mode switches to first-party while its discovery is pending" (fake + `fetchAllModels` resolves after the switch; fake `writeDesktop3pConfig` must not be called). + +`tests/claude-integration/claude-desktop-mode-explanation.test.ts`: implicit gateway where +first-party can run → the default line, the first-party command and the risk text; explicit +`--gateway` → nothing; connected client or disabled intercept → nothing. + +Verifier (run at P of wp2, before writing it into the plan as proof): +`bun test tests/claude-integration/claude-desktop-first-party.test.ts tests/claude-integration/claude-desktop-cli.test.ts tests/claude-integration/claude-desktop-mode-explanation.test.ts`, +`bun run typecheck` (locale records are `Record<TKey,string>`, so a missing key fails), +`cd gui && bun test tests/locale-parity.test.ts`, `bun run structure:check`. diff --git a/devlog/_plan/260924_claude_desktop_picker_mode/020_wp3_picker_core.md b/devlog/_plan/260924_claude_desktop_picker_mode/020_wp3_picker_core.md new file mode 100644 index 0000000000..2affdd4c9d --- /dev/null +++ b/devlog/_plan/260924_claude_desktop_picker_mode/020_wp3_picker_core.md @@ -0,0 +1,362 @@ +# 020 — wp3: picker core (CA, trust, CONNECT decision, claude.ai relay, bootstrap rewrite, routes) + +Consumes: D3–D7 in [000](000_plan.md) and facts in [001](001_research.md). Controls this phase +carries: only claude.ai is terminated and only while trusted; the upstream leg always verifies +certificates; exactly one response (the bootstrap) is rewritten and the rewrite fails open; no +body, cookie or token is logged; every listener is loopback. Nothing here selects the Desktop +egress profile; with no profile applied, none of this code sees Desktop traffic. wp4 turns it on. + +## Files + +| Path | Change | +| --- | --- | +| `src/claude/intercept/local-ca.ts` | MODIFY: export a generic authority/leaf issuer; add the nameConstraints OID and encoder; existing intercept CA unchanged | +| `src/claude/intercept/picker-ca.ts` | NEW: picker CA (`<configDir>/claude-picker/`), claude.ai leaf, persisted leaf PEM for trust checks | +| `src/claude/intercept/picker-trust.ts` | NEW: macOS login-keychain trust/verify/untrust through an injectable `security` runner | +| `src/claude/intercept/picker-bootstrap.ts` | NEW: bootstrap request match, accept-encoding narrowing, decode, JSON injection, header rewrite | +| `src/claude/intercept/picker-models.ts` | NEW: picker entries from the rendered Desktop profile + snapshot holder | +| `src/claude/intercept/picker-listener.ts` | NEW: `node:https` HTTP/1.1 terminator for claude.ai with request + upgrade relay | +| `src/claude/intercept/picker-runtime.ts` | NEW: desired/trust/listener state, tunnel decision, lazy start/stop, trust cache | +| `src/claude/intercept/connect-proxy.ts` | MODIFY: per-connection `selectTunnel`; header comment | +| `src/claude/intercept/runtime.ts` | MODIFY: create the picker runtime, pass `selectTunnel`, stop it; expose picker state | +| `src/server/index/claude-intercept-lifecycle.ts` | MODIFY: pass the picker route loader (dynamic imports; `src/server/index.ts` untouched) | +| `src/claude/desktop-3p.ts` | MODIFY: export `displayModelId` (label parity) | +| `src/types/config.ts`, `src/config/schema/config-schema.ts` | MODIFY: `claudeCode.intercept.picker?: boolean` | +| `structure/runtime.md` | MODIFY: intercept pair gains the picker terminator; invariants | +| tests (below) + `scripts/test-layout/layout.json`, `tests/fixtures/test-layout-expected.json` | NEW files registered in both | + +## local-ca.ts + +```diff + const OID = { ++ nameConstraints: "2.5.29.30", + … ++export interface AuthorityOptions { ++ commonName: string; ++ permittedDnsNames?: readonly string[]; ++ excludeAllIpAddresses?: boolean; // default true (#5731) ++} ++ ++/** iPAddress bases (address + mask, all zero) covering every IPv4 and every IPv6 address. */ ++export const ALL_IP_ADDRESS_BASES: readonly Uint8Array[] = [new Uint8Array(8), new Uint8Array(32)]; ++ ++/** RFC 5280 NameConstraints: permittedSubtrees of dNSName bases, excludedSubtrees of every IP. */ ++function nameConstraints(permitted: readonly string[], excludeAllIpAddresses: boolean): Uint8Array { ++ const subtrees = permitted.map(name => sequence(contextTag(2, new TextEncoder().encode(name), false))); ++ const excluded = ALL_IP_ADDRESS_BASES.map(base => sequence(contextTag(7, base, false))); ++ return sequence( ++ contextTag(0, concat(...subtrees)), ++ ...(excludeAllIpAddresses ? [contextTag(1, concat(...excluded))] : []), ++ ); ++} ++ ++export function createCertificateAuthority(options: AuthorityOptions): LocalInterceptCa { …same body as ++ createLocalInterceptCa, with commonName from options and, when permittedDnsNames is non-empty, ++ extension(OID.nameConstraints, true, nameConstraints(options.permittedDnsNames, ++ options.excludeAllIpAddresses !== false)) … } ++export function issueServerLeaf(ca: LocalInterceptCa, issuerCommonName: string, hosts: readonly string[]): PemKeyPair +-export function createLocalInterceptCa(): LocalInterceptCa { … ++export function createLocalInterceptCa(): LocalInterceptCa { ++ return createCertificateAuthority({ commonName: CLAUDE_INTERCEPT_CA_COMMON_NAME }); ++} +-export function issueLocalInterceptLeaf(ca, hosts) { … ++export function issueLocalInterceptLeaf(ca: LocalInterceptCa, hosts: readonly string[]): PemKeyPair { ++ return issueServerLeaf(ca, CLAUDE_INTERCEPT_CA_COMMON_NAME, hosts); ++} +``` + +The persistence helpers become parameterized by directory and common name +(`ensurePersistedAuthority(dir, options, lockName)`) so the picker CA reuses loading, validation, +atomic 0600 key writes and the lifecycle lease without copying them. + +## picker-ca.ts + +```ts +export const PICKER_HOST = "claude.ai"; +export const PICKER_CA_COMMON_NAME = "opencodex Claude Desktop Picker CA"; +export const PICKER_STATE_DIR = "claude-picker"; +export function pickerStateDir(configDir: string): string; // <configDir>/claude-picker +export function pickerCaCertPath(configDir: string): string; // …/ca.pem +export function pickerLeafCertPath(configDir: string): string; // …/leaf.pem (public) +export interface PickerCa extends LocalInterceptCa { fingerprint: string } // sha256 of the CA DER +export function ensurePickerCa(configDir: string): PickerCa; // permittedDnsNames: [PICKER_HOST] +export function issuePickerLeaf(ca: PickerCa, configDir: string): PemKeyPair; // SAN claude.ai; writes leaf.pem 0644 +``` + +Reload validation (audit wp3 r1, High). The shared loader only checks CA status, key pairing and +self-signature, so `ensurePersistedAuthority` gains `accept?: (cert: X509Certificate) => boolean`, +and `ensurePickerCa` passes one that requires subject CN `PICKER_CA_COMMON_NAME` and a **critical** +nameConstraints extension whose permittedSubtrees hold exactly one dNSName, `claude.ai`, and +whose excludedSubtrees hold exactly two iPAddress bases, all-zero IPv4 (8 bytes) and all-zero IPv6 +(32 bytes), so no IP-address leaf chains to it (PR #5731 review: a DNS-only permitted list leaves the +iPAddress form unconstrained). A persisted CA that fails it (for example a valid, key-matching CA +without constraints, or the first claude.ai-only format) is regenerated under the lease. The new +fingerprint makes trust `untrusted` until the operator trusts it again, so an unconstrained root is +never loaded and trusted as the picker CA. + +## picker-trust.ts + +```ts +export type PickerTrustState = "trusted" | "untrusted" | "unsupported" | "unknown"; +export interface SecurityResult { code: number | null; stdout: string; stderr: string } +export type SecurityRunner = (args: readonly string[]) => Promise<SecurityResult>; +export const defaultSecurityRunner: SecurityRunner; // Bun.spawn(["/usr/bin/security", …]) +export function loginKeychainPath(home?: string): string; // ~/Library/Keychains/login.keychain-db +export async function inspectPickerTrust(leafPath: string, caSha1: string, run?: SecurityRunner, platform?: NodeJS.Platform): Promise<PickerTrustState>; +// darwin only; "trusted" needs both: +// 1. ["find-certificate", "-a", "-Z", "-c", PICKER_CA_COMMON_NAME, loginKeychainPath()] lists caSha1 (the current CA) +// 2. ["verify-cert", "-q", "-L", "-c", leafPath, "-p", "ssl", "-n", "claude.ai", "-k", loginKeychainPath()] exits 0 +// 3. ["trust-settings-export", <temp plist>] does not show kSecTrustSettingsPolicyString in the +// caSha1 entry (a host-scoped setting from an earlier build; Chromium skips it, so it counts +// as untrusted and the trust step replaces it); an unreadable export → unknown, so the picker +// never arms on a setting it could not inspect +// missing record or exit 1 → untrusted; a runner failure → unknown +export async function trustPickerCa(caPath: string, run?: SecurityRunner, platform?: NodeJS.Platform): Promise<{ ok: boolean; reason?: "unsupported" | "declined_or_failed" }>; +// ["add-trusted-cert", "-r", "trustRoot", "-p", "ssl", "-k", loginKeychainPath(), caPath] +// (no "-s claude.ai": found in the live proof, Chromium skips host-scoped trust settings and +// Desktop failed with ERR_CERT_AUTHORITY_INVALID; the name constraints do the scoping) +export async function untrustPickerCa(caPath: string, fingerprintSha1: string, run?: SecurityRunner, platform?: NodeJS.Platform): Promise<{ ok: boolean }>; +// ["remove-trusted-cert", caPath] then ["delete-certificate", "-Z", fingerprintSha1, loginKeychainPath()] +``` + +Only stdout/stderr lengths are logged, never content. + +## picker-bootstrap.ts + +```ts +export const BOOTSTRAP_MAX_ENCODED_BYTES = 4 * 1024 * 1024; +export const BOOTSTRAP_MAX_DECODED_BYTES = 16 * 1024 * 1024; +export const PICKER_SURFACE_ID = "code"; +const BOOTSTRAP_PATH = /^\/(?:edge-api|api)\/bootstrap(?:\/[A-Za-z0-9-]+\/app_start)?\/?$/; +export function isPickerBootstrapRequest(method: string, pathname: string): boolean; // GET/HEAD-free: GET only +export function narrowBootstrapAcceptEncoding(): string; // "gzip, deflate, br" +export interface PickerModelEntry { id: string; name: string; contextWindow?: number } +export function injectPickerModels(bootstrap: unknown, models: readonly PickerModelEntry[]): number; +export function rewriteBootstrapBody(encoded: Buffer, contentEncoding: string | undefined, models: readonly PickerModelEntry[]): Buffer | null; +export function rewrittenHeaders(raw: readonly string[], bodyLength: number): string[]; +``` + +`injectPickerModels`: find `model_selector_config` (array) → surface `id === "code"` with array +`models`; template = first entry whose id starts with `claude-`, not `disabled`, no +`disabled_reason`, section not `deprecated`; for each model whose id is not already present: +`structuredClone(template)`, set `id`, `name`, `section: "main"`, set or delete +`context_window`, delete `disabled`, `disabled_reason`, `badge`, `tooltip`, `description`, +`fast_mode` and every key matching `/version/i` (version gates); keep `thinking` and +`capabilities`, which the intercept serves for routed models (effort translation, image +handling); push. Returns the number added (0 = leave the body untouched). `rewriteBootstrapBody` decodes +`gzip`/`x-gzip`/`deflate`/`br`/identity with `maxOutputLength` = the decoded cap, rejects +anything else or oversize, parses JSON, injects, and returns identity-encoded UTF-8 or `null`. +`rewrittenHeaders` drops `content-encoding`, `content-length`, `etag`, `digest`, `content-md5`, +`transfer-encoding` and sets `content-length`. + +## picker-models.ts + +```ts +export interface PickerRouteInput { nativeSlugs: string[]; routedModels: Desktop3pRoutedModel[]; profile?: OcxClaudeDesktopProfile; nativeContextCap?: NativeContextLimitsInput } +export function buildPickerModels(input: PickerRouteInput): PickerModelEntry[]; +export interface PickerModelSnapshot { current(): { models: PickerModelEntry[]; builtAt: number } | null; refresh(): Promise<void>; refreshIfStale(maxAgeMs: number): void } +export function createPickerModelSnapshot(load: () => Promise<PickerRouteInput>, persistPath?: string): PickerModelSnapshot; +// persistPath (<configDir>/claude-picker/models.json, 0600): written after each successful build and +// read synchronously at construction, so a bootstrap served right after a restart already injects +// the last known routes while discovery refreshes in the background. +``` + +With a profile: `reconcileDesktopProfile` + `renderDesktopProfile` (src/claude/desktop-profile.ts:230, +:296) in the gateway's order and labels; without one: every candidate, label +`${displayModelId(id)} (${provider})`. Id: `native/<slug>` → `claudeCodeNativeAlias(slug)`, +otherwise `aliasForRoute(provider, id)`; routes whose alias is `null` and real +`anthropic/claude-*` routes are skipped (Anthropic's own rows already exist). Context window from +the candidate. Refresh errors keep the previous snapshot. + +## picker-listener.ts + +```ts +export interface PickerListenerOptions { + leaf: PemKeyPair; + models: () => readonly PickerModelEntry[]; + upstream?: { host: string; port: number; servername: string; ca?: string }; // test seam; default claude.ai:443, system roots + /** Test seam: encoded bootstrap cap; production uses BOOTSTRAP_MAX_ENCODED_BYTES. */ + maxEncodedBytes?: number; + log?: (line: string) => void; +} +export interface PickerListenerHandle { port: number; close(): Promise<void> } +export function startPickerListener(options: PickerListenerOptions): Promise<PickerListenerHandle>; +``` + +`https.createServer({ cert, key, ALPNProtocols: ["http/1.1"] })` on 127.0.0.1:0. +- `request`: build upstream headers from `req.rawHeaders` minus hop-by-hop + (`connection`, `keep-alive`, `proxy-connection`, `proxy-authorization`, `te`, `trailer`, + `transfer-encoding`, `upgrade`); `https.request` with `servername`, + `rejectUnauthorized: true`, `agent: false` or a dedicated keep-alive agent. Non-bootstrap: + `res.writeHead(status, filteredRawHeaders)` and `upRes.pipe(res)`; `req.pipe(upReq)`. + Bootstrap (`isPickerBootstrapRequest`, status 200, JSON content type): set accept-encoding to + `narrowBootstrapAcceptEncoding()` and hold the response head. Collect upstream chunks in order + while the running total stays within the encoded cap. When a chunk would push the total past + the cap, stop collecting: `writeHead` with the original filtered headers, write every collected + chunk and then that triggering chunk in order (each byte exactly once), and only then + `upRes.pipe(res)` for later data (the complete original body, unmodified). The collection + listener is removed before piping, and `upRes` is paused while the head and collected bytes are + written so no chunk is emitted between the switch; `pipe` then owns backpressure. If the + upstream ends within the cap, call `rewriteBootstrapBody`; on `null` or zero injections + `writeHead` the original headers and write the collected bytes; otherwise `writeHead` with + `rewrittenHeaders` and the rewritten body. An upstream error before `writeHead` → 502; after + it → destroy the client socket (the client sees a truncated response, never a spliced one). +- `upgrade`: `tls.connect` to upstream with the same verification, write the request line and + raw headers, then `head`, and pipe both ways; destroy both on either error/close. +- Errors: 502 with an empty body; log `picker <method> <bootstrap|other> <status>` only. + +## picker-runtime.ts + +```ts +export type TunnelChoice = { kind: "intercept"; port: number } | { kind: "blind" }; +export interface PickerRuntime { + /** null → default behaviour. For claude.ai while the first refresh is pending, returns a promise that + * settles on `ready` or after PICKER_STARTUP_WAIT_MS (3 s), whichever is first; timeout or failure → blind. */ + selectTunnel(host: string, port: number): TunnelChoice | null | Promise<TunnelChoice | null>; + refreshTrust(): Promise<PickerTrustState>; + /** + * Re-read the persisted config (options.readConfig, default loadConfig), recompute the mode + * (resolveClaudeDesktopMode(fresh, observeClaudeDesktopMode(fresh))) and pickerDesired(fresh, mode), + * refresh trust, then ensureStarted() when armable, else stop terminating. Never clears the + * disarm latch. The 60 s interval calls this. selectTunnel only reads the cached result. + */ + refresh(): Promise<void>; + /** + * Stop terminating claude.ai now, set the disarm latch and bump the arm generation. While latched, + * refresh() and ensureStarted() never arm. An ensureStarted() already in flight captured the old + * generation and must not arm when it completes. + */ + disarm(): void; + /** + * Owner-only: called by the picker controller while it holds its lock, at the end of an enable + * whose checks passed. Clears the latch and recomputes the decision with the isBusy() check + * bypassed, so the next CONNECT is intercepted even before the lock is released. + */ + rearm(): Promise<void>; + ensureStarted(): Promise<void>; // CA, leaf, listener, snapshot + /** First refresh() now, then the refresh interval; resolves with that first refresh. Called once after the CONNECT proxy binds. */ + start(): Promise<void>; + readonly ready: Promise<void>; + status(): PickerRuntimeStatus; // desired, supported, trust, listenerReady, effective, reason, models, snapshotAt + stop(): Promise<void>; +} +export function createPickerRuntime(options: { config: OcxConfig; readConfig?: () => OcxConfig; isBusy?: () => boolean; configDir: string; loadRoutes: () => Promise<PickerRouteInput>; security?: SecurityRunner; platform?: NodeJS.Platform; now?: () => number; trustTtlMs?: number }): PickerRuntime; +export function pickerDesired(config: Pick<OcxConfig, "claudeCode" | "clientIntegrations">, mode: ClaudeDesktopMode, platform?: NodeJS.Platform): boolean; +// darwin && claudeDesktopIntegrationEnabled(config) (src/codex/desired-state.ts:214) && mode === "first-party" +// && config.claudeCode?.intercept?.picker !== false +// `mode` is the observation-aware resolved mode (010): the runtime computes +// resolveClaudeDesktopMode(config, observeClaudeDesktopMode(config)) at start and on every trust +// refresh and caches it; status, apply wiring and the CLI pass the same resolved mode. +``` + +`selectTunnel` returns `{ kind: "intercept", port: listener.port }` for `claude.ai:443` only when +the decision cached by the last `refresh()` says armed (desired from the persisted config, not +latched), the listener +is up, and the cached trust is `trusted` for the current CA fingerprint; `{ kind: "blind" }` for +`claude.ai` otherwise; `null` for every other host. Trust is refreshed on `ensureStarted`, on +`refreshTrust` and on a 60 s interval while desired; the interval never blocks the event loop. + +## connect-proxy.ts + +```diff + export interface ConnectProxyOptions { + interceptPort: number; + interceptHosts?: readonly string[]; ++ /** Per-connection override, consulted before `interceptHosts`; `null` keeps the default. */ ++ selectTunnel?: (host: string, port: number) => TunnelDecision | null | Promise<TunnelDecision | null>; ++ // TunnelDecision = { kind: "intercept"; port: number } | { kind: "blind" } + dialUpstream?: (host: string, port: number) => Socket; + } +- const intercept = target.port === 443 && options.interceptHosts.includes(target.host); +- const upstream = intercept +- ? connect({ host: "127.0.0.1", port: options.interceptPort }) +- : options.dialUpstream(target.host, target.port); ++ // The client socket is already paused; an async decision (only claude.ai during the picker's ++ // first refresh, bounded by the runtime) keeps it paused until it settles; a rejection → blind. ++ const selected = await Promise.resolve(options.selectTunnel?.(target.host, target.port) ?? null) ++ .catch(() => ({ kind: "blind" as const })); ++ const choice = selected ++ ?? (target.port === 443 && options.interceptHosts.includes(target.host) ++ ? { kind: "intercept" as const, port: options.interceptPort } ++ : { kind: "blind" as const }); ++ const upstream = choice.kind === "intercept" ++ ? connect({ host: "127.0.0.1", port: choice.port }) ++ : options.dialUpstream(target.host, target.port); +``` + +Loopback 403 and non-CONNECT 405 stay before the choice. The header comment names claude.ai as the +only host that picker mode may terminate. + +Callback shape (audit wp3 r1, High). The diff above is the decision logic, not the literal code: +`onData` stays synchronous. After the 405/403 checks it calls `dialFor(choice)` in the same tick +when `selectTunnel` is absent or returns a non-promise, so the default path is unchanged; otherwise +it runs `void Promise.resolve(decision).catch(() => blind).then(dialFor)`. `dialFor` returns +without dialing when `socket.destroyed` (the client left while the decision was pending) and then +runs today's dial, connect-timeout, error and splice block unchanged. `handleConnection` takes a +`ResolvedConnectProxyOptions` type that adds the optional `selectTunnel`, and `startConnectProxy` +copies `options.selectTunnel` into the resolved object. + +## runtime.ts / lifecycle + +`startClaudeIntercept` gains `loadPickerRoutes?: () => Promise<PickerRouteInput>`; it creates the +picker runtime (`createPickerRuntime({ config: options.config, … })`), passes +`selectTunnel: picker.selectTunnel` to `startConnectProxy`, and, once the CONNECT proxy has bound, +starts the picker with `picker.start()`: an immediate `refresh()` (which calls `ensureStarted()` +when armable and fills the cached decision) followed by the 60 s refresh interval. `start()` +returns the first refresh's promise (`picker.ready`) so tests and status can await it. It stops the +runtime in `stop()`. `getClaudePickerRuntime()` mirrors +`getClaudeInterceptState()` for status and the wp4 routes. `claude-intercept-lifecycle.ts` passes a +loader built from `fetchAllModels`, `filterCatalogVisibleModels`, `desktopVisibleNativeSlugs` and +`nativeContextLimits` through dynamic imports (the same inputs `/api/sync` uses, +src/server/management/config-routes.ts:213–231). `startServer` stays synchronous; no line is added +to `src/server/index.ts`. + +Startup settlement (audit wp3 r1, High). `refresh()` catches its own failures (trust runner, CA, +listener bind, snapshot load), records them as `status().reason` and leaves the decision blind, so +`start()` resolves. `startClaudeIntercept` awaits `picker.start()` inside the same `try` that +guards `startConnectProxy`. The picker handle is nullable (`let picker: PickerRuntime | null = null`). If creating the picker +or `start()` still throws or rejects, it awaits `picker?.stop()` (picker listener and interval, only +when construction returned), then always `proxy.close()` and `listener.stop(true)`, before +rethrowing, so the lifecycle's catch never leaves a bound socket without a handle. `stop()` closes +the picker, then the proxy, then the listener. A `createPicker` option on +`StartClaudeInterceptOptions` is the test seam. + +## Tests (NEW unless noted; every new file registered in layout.json and test-layout-expected.json) + +| File | Cases (activation → observable) | +| --- | --- | +| `tests/claude-integration/claude-picker-ca.test.ts` | CA has critical nameConstraints permitting only claude.ai (parse extension bytes); leaf SAN is exactly claude.ai and chains (`X509Certificate.verify`); a TLS handshake through Bun (BoringSSL) with the picker CA as the only root accepts the claude.ai leaf and rejects a test-only leaf for `example.com` issued by the same CA; key file 0600; corrupt key regenerates; intercept CA has no nameConstraints (unchanged); a valid key-matching CA without the claude.ai constraint in the picker directory is regenerated with a new fingerprint | +| `tests/claude-integration/claude-picker-trust.test.ts` | fake runner receives the exact argv for find/verify/trust/untrust; a verified leaf whose SHA-1 record is missing or different → untrusted; exit 0/1/other → trusted/untrusted/unknown; non-darwin → unsupported without spawning | +| `tests/claude-integration/claude-picker-bootstrap.test.ts` | path matcher (both prefixes, org app_start, rejects others and POST); injection clones template, skips existing ids, drops fast_mode and version gates, leaves cowork and model_selector_state; gzip/br/deflate/identity round trip; malformed JSON, missing surface, unknown encoding and oversize → `null`; headers rewritten | +| `tests/claude-integration/claude-picker-models.test.ts` | routed alias `ocx-claude-xai--grok-4.7` and native alias; profile order/labels match the gateway render; anthropic/claude routes skipped; snapshot keeps last good on loader failure | +| `tests/claude-integration/claude-picker-listener.test.ts` | local https upstream (picker CA-issued fixture via the upstream seam): gzip body and two Set-Cookie headers pass byte-identical; SSE chunks arrive before the upstream ends; WebSocket upgrade echoes; bootstrap gets the injected entry with identity encoding; a bootstrap larger than the encoded cap (`maxEncodedBytes` seam set low, with the cap crossed in the middle of an upstream chunk) arrives byte-identical with its original headers; malformed bootstrap JSON arrives byte-identical; upstream with an untrusted cert → 502 | +| `tests/claude-integration/claude-picker-runtime.test.ts` | startup: with a pre-existing selected picker profile, trusted current CA, persisted first-party and intent on, the first CONNECT to claude.ai after `start()` resolves (`await picker.ready`) is `intercept`, with no timer tick; | +| (same file, continued) | a claude.ai CONNECT arriving while the first refresh is pending waits and is intercepted once `ready` resolves; with `ready` held past the 3 s bound it is blind; a snapshot persisted to `models.json` is injected into the first bootstrap after a restart before discovery completes; | +| (same file, continued) | selectTunnel: claude.ai blind until desired+trusted+listening, intercept after; trust loss flips back on refresh; non-claude hosts → null; non-darwin never intercepts | +| (same file, continued) | legacy install: owned first-party env in Claude Code settings, no saved `desktopMode`, picker intent unset → `pickerDesired` is true, so picker mode stays on by default across the upgrade; a `createPicker` that throws, and one whose `start()` rejects, each make `startClaudeIntercept` reject, and the proxy port binds again at once | +| `tests/claude-integration/claude-intercept-proxy.test.ts` (MODIFY) | a selectTunnel override is consulted per connection; an async decision keeps the client socket paused and pipelined bytes are delivered after it settles; a rejected decision is blind; loopback/405 refusals unchanged; a client that closes while the decision is pending causes no upstream dial | + +Verifier: `bun test` on the files above plus `tests/claude-integration/claude-intercept*.test.ts`, +`tests/server/claude-intercept-integration.test.ts`, `tests/lab/core-lab-boundary.test.ts`, +`tests/test-layout.test.ts`, `tests/test-layout-tooling.test.ts`, +`tests/ci-workflows/file-size-ratchet.test.ts`; `bun run typecheck`; `bun run structure:check`. + + +## Desktop egress proxy (B-phase amendment) + +Found while wiring: one CONNECT proxy cannot serve both clients. The Code tab's Claude Code trusts +only the intercept CA (`NODE_EXTRA_CA_CERTS`), so a claude.ai connection it opens through that proxy +would fail against the picker terminator; Desktop trusts only the login keychain, so its own +api.anthropic.com connections would fail against the intercept listener. Picker mode therefore gets +its own CONNECT proxy on `claudePickerProxyPort` (the intercept proxy port + 1, or − 1 at 65535), +started by `startClaudeIntercept` only when `loadPickerRoutes` is given (the server lifecycle always +passes `loadPickerRoutes`), with `interceptHosts: []` and `selectTunnel` from the picker runtime. On +it every host is blind except claude.ai while armed. The Claude Code proxy keeps its exact current +behaviour and gets no `selectTunnel`. `ClaudeInterceptState.pickerProxyPort` (null when unwired or +unbound) is what wp4 writes into `egressProxyUrl`. A bind failure on that port logs a warning, stops +the picker and leaves the intercept pair running. Tests: the Claude Code proxy never consults the +picker; the egress proxy blind-tunnels api.anthropic.com. + +## Audit record + +- wp3 round 1 (reviewer, FAIL, 3 High): persisted picker CA reload lacked constraint validation; the CONNECT diff awaited inside a synchronous callback and missed the options plumbing; picker startup failure could leave bound sockets. All three folded above. Round 2 GO-WITH-FIXES (1 High): a picker construction failure before assignment; folded as a nullable handle with a createPicker-throws test. Architect reflection ALIGNED, with the legacy first-party upgrade case added to the runtime tests. diff --git a/devlog/_plan/260924_claude_desktop_picker_mode/030_wp4_picker_activation.md b/devlog/_plan/260924_claude_desktop_picker_mode/030_wp4_picker_activation.md new file mode 100644 index 0000000000..bf25035e4f --- /dev/null +++ b/devlog/_plan/260924_claude_desktop_picker_mode/030_wp4_picker_activation.md @@ -0,0 +1,194 @@ +# 030 — wp4: picker activation (egress profile, controls, default-on in first-party) + +Consumes: D8–D11 in [000](000_plan.md) and the wp3 runtime. After this phase a first-party apply +on macOS turns picker mode on unless `claudeCode.intercept.picker === false`. + +The organising rule (D11): **while an opencodex server runs, every picker mutation — enable, +disable, transition cleanup — runs inside that server, in one controller, serialized by one +lock.** The CLI and the dashboard call its routes. The only thing a CLI does locally while a +server runs is the keychain trust step, because the password dialog belongs to the operator's +terminal session. With no server running, nothing can terminate claude.ai, so the CLI may remove +picker artifacts locally, and enabling is refused. + +> B-phase amendment from wp3 (see 020, "Desktop egress proxy"): Desktop's `egressProxyUrl` names +> `getClaudeInterceptState().pickerProxyPort`, the dedicated picker CONNECT proxy, never the Claude +> Code proxy port. Every "proxy bound" check below means `pickerProxyPort !== null`, and +> `applyDesktopPickerProfile({ proxyPort })` receives that port. + +## Files + +| Path | Change | +| --- | --- | +| `src/claude/desktop-3p-library.ts` | MODIFY: `DESKTOP_PICKER_ENTRY_NAME = "opencodex-picker"`; `isOwnedDesktopEntry` accepts it; gateway predicate does not | +| `src/claude/desktop-picker-profile.ts` | NEW: apply/remove/inspect the owned egress profile; state in `<configDir>/claude-picker/profile-state.json` | +| `src/claude/desktop-picker.ts` | NEW: `DesktopPickerController` (server) with `enable`/`disable`/`status` under one async lock; `removeDesktopPickerArtifacts` (local cleanup when no server runs) | +| `src/claude/desktop-first-party.ts` | MODIFY: nothing picker-specific beyond exports used by the controller | +| `src/claude/intercept/runtime.ts` | MODIFY (audit wp4 pre-audit, Medium 5): create the controller next to the picker runtime (`isBusy: () => controller?.busy() ?? false`), expose `getClaudePickerController()`, clear it on stop and on a failed start | +| `src/cli/claude-desktop.ts` | MODIFY: first-party apply delegates to the server when one runs; gateway apply and removal ask the server to clean up; `picker on|off|status|trust` subcommand; help | +| `src/cli/ensure-desired-integrations.ts` | MODIFY: `ensureClaudeDesktopMatchesDesired` async and awaited by reconcile; durable-OFF picker cleanup | +| `tests/providers/xai/grok-lifecycle.test.ts` | MODIFY: source-boundary assertions (:61, :79) expect the async declaration and (:86) the awaited call | +| `src/server/management/agent-settings-routes.ts` | MODIFY: `GET/PUT /api/claude-desktop/picker`; status `firstParty.picker`; first-party apply/remove and gateway apply call the controller in process | +| `src/server/management/native-integration-routes.ts` | MODIFY: first-party enable/disable call the controller; `persistDesktopModeMarker` returns the committed subtree and callers adopt it | +| `src/server/management/config-routes.ts` | MODIFY: `/api/sync` Claude Desktop writer runs its post-discovery re-read, re-resolve and write inside `controller.transition` when a controller exists; race test in `claude-desktop-picker.test.ts` | +| `src/server/management/route-registry.ts` | MODIFY: register GET and PUT `/api/claude-desktop/picker` as standard entries (like `/api/claude-desktop/first-party-bindings`, :155) | +| `src/cli/capabilities.ts`, `skills/ocx/references/01_management_surface.md` (generated) | MODIFY: `claude-desktop.picker` capability for both routes; regenerate with `bun run skill:surface` | +| `gui/src/components/ClaudeDesktopPicker.tsx` + `gui/src/styles/claude-desktop-picker.css` | NEW: picker card (toggle, state, offline and restart notes) mounted in first-party mode | +| `gui/src/pages/ClaudeDesktop.tsx`, `gui/src/main.tsx` | MODIFY: mount the card; import CSS | +| `gui/src/i18n/*.ts` (10) | MODIFY: `claudeDesktop.picker.*` keys | +| `docs-site/src/content/docs/**/guides/claude-code.md` (8) | MODIFY: picker mode subsection (what it does, keychain prompt, offline dependency, how to turn off) | +| `structure/clients/claude-desktop.md`, `structure/gui-and-management-api.md` | MODIFY: picker contract, controller, routes, card | + +## desktop-picker-profile.ts + +```ts +export interface DesktopPickerProfileState { entryId: string; previousAppliedId: string | null } +export type DesktopPickerProfileInspection = + | { kind: "absent" } | { kind: "applied"; entryId: string; proxyUrl: string } + | { kind: "not_selected"; entryId: string } | { kind: "unsafe"; reason: string }; +export function pickerEgressUrl(proxyPort: number): string; // http://127.0.0.1:<port> +export function applyDesktopPickerProfile(options: { proxyPort: number; configDir?: string } & Desktop3pConfigLibraryOptions): + { ok: true; changed: boolean; path: string } | { ok: false; reason: "gateway_selected" | "foreign_unreadable" | "write_failed" }; +export function removeDesktopPickerProfile(options: { configDir?: string } & Desktop3pConfigLibraryOptions): + { ok: true; changed: boolean } | { ok: false; reason: string; residualPaths?: string[] }; +export function inspectDesktopPickerProfile(options?: Desktop3pConfigLibraryOptions & { configDir?: string }): DesktopPickerProfileInspection; +``` + +Apply: refuse while an owned gateway row is selected; reuse the existing picker row or create one +(`randomUUID`, name `opencodex-picker`); write exactly `{"egressProxyUrl":"http://127.0.0.1:<port>"}\n` +atomically; write `profile-state.json` with the current `appliedId` (unless it already is the picker +row); then set `appliedId` to the picker row. Remove: if the picker row is selected, reselect +`previousAppliedId` when that row still exists, else the owned standard row (created as `{}` like +`removeDesktop3pStandardPivot`); delete the picker profile and its `.bak`; drop the metadata row; +delete `profile-state.json`. Foreign rows and `_meta.json` keys other than `appliedId`/`entries` are +preserved; `_meta.json` never carries opencodex keys. + +## desktop-picker.ts + +```ts +export type DesktopPickerReason = "active" | "restart_required" | "unsupported_platform" | "not_first_party" + | "integration_off" | "disabled" | "proxy_unavailable" | "mode_not_committed" | "trust_pending" + | "trust_declined" | "profile_failed"; +export interface DesktopPickerStatus { desired: boolean; supported: boolean; trust: PickerTrustState; + profile: DesktopPickerProfileInspection["kind"]; listenerReady: boolean; effective: boolean; + reason: DesktopPickerReason; models: number; snapshotAt: number | null; lastBootstrapAt: number | null; + hint?: string; residual?: string[] } +export interface DesktopPickerController { + enable(options: { persist: boolean; context: "cli-trusted" | "server"; callerAddedTrust?: boolean }): Promise<DesktopPickerStatus>; + disable(options: { persist: boolean }): Promise<DesktopPickerStatus>; + /** + * Run a whole Desktop mode transition under the controller lock: callers stage slow work + * (model discovery) first, then inside `fn` do cleanup → mode/profile commit → optional enable + * with the lock-free inner helpers `ops.disableLocked` / `ops.enableLocked`. + */ + transition<T>(fn: (ops: { disableLocked(o: { persist: boolean }): Promise<DesktopPickerStatus>; + enableLocked(o: { persist: boolean; context: "cli-trusted" | "server"; callerAddedTrust?: boolean }): Promise<DesktopPickerStatus> }) => Promise<T>): Promise<T>; + status(): Promise<DesktopPickerStatus>; + busy(): boolean; +} +export function createDesktopPickerController(deps: { runtime: PickerRuntime; readConfig: () => OcxConfig; + persistPreference: (value: boolean) => boolean; proxyPort: () => number | null; configDir: string; security?: SecurityRunner; + platform?: NodeJS.Platform }): DesktopPickerController; +export function removeDesktopPickerArtifacts(options: { configDir?: string; security?: SecurityRunner }): Promise<{ ok: boolean; residual?: string[] }>; +``` + +One promise-chain lock serializes `enable`, `disable` and `transition`. The runtime's periodic and +startup `refresh()` calls `isBusy()` and does not arm while the lock is held; the lock owner arms +through `runtime.rearm()`, which is owner-only and bypasses that check (it is only ever called from +inside the lock). Wiring in `src/claude/intercept/runtime.ts`: `let controller: DesktopPickerController +| null = null; const picker = createPickerRuntime({ …, isBusy: () => controller?.busy() ?? false }); +controller = createDesktopPickerController({ runtime: picker, … });` and `getClaudePickerController()` +exposes it to the management routes. + +**enable({ persist, context })**, inside the lock: +1. Re-read the persisted config (`readConfig`) and check the conditions that do not depend on the + preference: macOS, persisted first-party mode, Desktop intent on, proxy bound. A failure returns + its reason and writes nothing. Then, if `persist`, commit `claudeCode.intercept.picker = true` + (an explicit `picker on` from a false preference is allowed). +2. Re-read the persisted config again and check everything: macOS; persisted resolved mode is + first-party (`resolveClaudeDesktopMode(fresh, observeClaudeDesktopMode(fresh))`); Desktop + integration intent on; preference not false; the intercept proxy is bound + (`getClaudeInterceptState() !== null`). Failures → `unsupported_platform` / `mode_not_committed` / + `integration_off` / `disabled` / `proxy_unavailable`, nothing written. +3. `ensurePickerCa`, `issuePickerLeaf`, `inspectPickerTrust`. If not trusted: `context: "server"` tries + `trustPickerCa` once and re-inspects; still untrusted → `trust_pending` with + `hint: "ocx claude desktop picker trust"`. `context: "cli-trusted"` means the CLI already ran the + trust step; untrusted then → `trust_declined`. Record whether this attempt added trust. +4. Re-run the step-2 checks. +5. `applyDesktopPickerProfile({ proxyPort })`. +6. `runtime.rearm()` (clears the latch, refreshes, arms when everything holds) → `restart_required` + until a bootstrap has been served, then `active`. +Any failure after step 3 that follows trust added by this attempt calls `untrustPickerCa`. For a request with `callerAddedTrust: true`, every refusal or failure at any step (including the step-1 and step-2 checks) compensates, but only when the trusted certificate is the current picker CA (SHA-1 match) and the owned picker profile is not selected — a selected profile means an earlier successful enable still depends on that trust; if that fails too the +status carries `residual: ["trust"]`, `effective: false` and `hint: "ocx claude desktop picker off"`. + +**disable({ persist })**, inside the lock: `runtime.disarm()` (latch + generation bump) → if +`persist`, commit `claudeCode.intercept.picker = false` → `removeDesktopPickerProfile` → +`untrustPickerCa`. Failed steps are listed in `residual`; status is never "off" while the profile is +selected or trust remains. `persist: true` only for an explicit `picker off`; mode-transition +cleanup passes `false` and leaves the preference unset, so returning to first-party turns the +picker back on. + +## Callers + +Every server-side Desktop mode change goes through `runDesktopTransition(fn)` (audit wp4 pre-audit, High 2): with a controller it is `controller.transition(fn)`; with none (intercept disabled, client role, failed intercept or picker proxy bind) it calls `fn` with offline ops, where `disableLocked` runs `removeDesktopPickerArtifacts` (no runtime exists, so nothing can terminate claude.ai) and `enableLocked` returns `proxy_unavailable` without writing anything. Gateway apply, first-party removal, native enable and disable, and `/api/sync` therefore keep today's behaviour when no controller exists, plus leftover-artifact cleanup. Tests: with the intercept disabled, gateway apply and native disable succeed and remove a leftover picker row; with the picker proxy unbound, first-party apply succeeds and reports the picker as `proxy_unavailable`. + +| Caller | Runs where | Picker call | +| --- | --- | --- | +| Management first-party apply (`POST /api/claude-desktop/apply`) | server | the whole transition inside `runDesktopTransition`, in today's order (audit wp4 pre-audit, High 1): env write with its rollback, then gateway cleanup with today's partial-cleanup reporting, then the committed and adopted mode, then `enableLocked({ persist: false, context: "server" })` when the preference is not false; a partial apply whose mode write failed does not enable | +| `/api/sync` Claude Desktop writer (src/server/management/config-routes.ts:211–237) | server | discovery (`fetchAllModels`) staged first; then inside `runDesktopTransition`: re-read, re-resolve (010), and `writeDesktop3pConfig` only when the resolved mode is not first-party; no controller (intercept not running) → unchanged behaviour | +| Management first-party removal / gateway apply | server | model discovery staged first; then inside `runDesktopTransition`: `disableLocked({ persist: false })`, the gateway write or env removal, and the mode commit | +| Native enable (first-party branch) / native disable | server | inside `runDesktopTransition`: enable after `persistDesktopModeMarker` returns the committed subtree and it is adopted; disable before OFF cleanup and the intent commit | +| `GET/PUT /api/claude-desktop/picker` | server | `PUT { enabled, persist }` → `enable({ persist, context })` with `context: "cli-trusted"` when the request carries `trustedLocally: true` (sent only by the CLI after its trust step), else `"server"`; or `disable({ persist })` | +| CLI `ocx claude desktop apply --first-party` | CLI | on the local hub path with a live proxy, the same branch where gateway apply already delegates (src/cli/claude-desktop.ts:326), delegate to `POST /api/claude-desktop/apply { mode: "first-party" }` (audit wp4 pre-audit, High 3: the connected-client branch before it stays unchanged and never touches the picker); without one: apply locally as today and report the picker as `proxy_unavailable` | +| CLI gateway apply / first-party removal | CLI | with a live proxy: the delegated server apply performs the disable; without one: `removeDesktopPickerArtifacts` locally | +| CLI `picker on` | CLI | requires a live proxy (else `proxy_unavailable`); `PUT { enabled: true, persist: true }`; if the answer is `trust_pending`, run `picker trust` below and repeat the PUT with `trustedLocally: true` | +| CLI `picker trust` | CLI | local `ensurePickerCa` read + `trustPickerCa` (operator's dialog), recording whether this run added trust; then `PUT { enabled: true, persist: false, trustedLocally: true, callerAddedTrust }`. Compensation for trust the CLI added is done by the server inside the lock: enable treats `callerAddedTrust: true` like trust added by the attempt itself, so any refusal or failure after its trust check untrusts it (residual reported if that fails), and success keeps it. The CLI compensates locally only when the PUT could not be delivered at all (connection refused: no server, so nothing can race). A timeout or lost response is ambiguous: the CLI does not touch trust and prints "state unknown — run `ocx claude desktop picker status`" | +| CLI `picker off` | CLI | with a live proxy: `PUT { enabled: false, persist: true }`; without: persist false locally, then `removeDesktopPickerArtifacts`; prints "Fully quit and reopen Claude Desktop" | +| `ensureClaudeDesktopMatchesDesired` durable OFF | CLI / update hook | becomes async and is awaited by `reconcileEnsureDesiredIntegrations` (:188, :199); with a live proxy `PUT { enabled: false, persist: false }`, else `removeDesktopPickerArtifacts` | + +Auth: GET and PUT `/api/claude-desktop/picker` are standard registry entries like +`/api/claude-desktop/first-party-bindings` (route-registry.ts:155), so both the CLI admin token +(`runtimeRequest`, src/cli/runtime-api.ts:140) and the dashboard gui-session are accepted; the +`claude-desktop.picker` capability names both routes, as `tests/server/management-route-registry.test.ts` +requires. `trustedLocally` is only a wording hint for the trust outcome; it never skips a check. + +Server startup: the runtime's first `refresh()` arms only when every piece already exists +(persisted first-party, intent on, preference not false, trust for the current CA, profile selected +with the current proxy URL, listener up). It never trusts or writes a profile; status reports what is +missing with the `ocx claude desktop picker on` hint. + +Native persistence: `persistDesktopModeMarker` (native-integration-routes.ts:656) returns +`{ ok: true; claudeCode } | { ok: false }`; callers adopt with `adoptPersistedClaudeCode` +(src/config/live-reconcile.ts:123), fixing today's unadopted marker. The runtime itself decides from +persisted reads, so adoption is for status and later whole-config saves. + +## GUI + +`ClaudeDesktopPicker` card (first-party only): title, one-line explanation, toggle +(`PUT { enabled, persist: true }`), state line from `reason` (active / restart Desktop / waiting for +the keychain step with the `hint` command / declined / proxy not running / unsupported on this OS), +model count, and a fixed note: "While picker mode is on, Claude Desktop reaches the network through +OpenCodex. If OpenCodex stops, Desktop is offline until it restarts or picker mode is turned off." +Keys `claudeDesktop.picker.{title,hint,toggle,state.active,state.restart,state.trustPending, +state.trustDeclined,state.proxyUnavailable,state.unsupported,state.notFirstParty,state.profileFailed, +models,offlineNote}` in all ten catalogs. + +## Tests (NEW files registered in layout.json + test-layout-expected.json) + +| File | Cases | +| --- | --- | +| `tests/claude-integration/claude-desktop-picker-profile.test.ts` | apply creates the row, writes only egressProxyUrl, records previousAppliedId, selects it; re-apply idempotent; remove restores the previous selection, falls back to a standard row when it vanished, keeps foreign rows; gateway selected → refused; metadata write failure rolls back; gateway removal (`removeDesktop3pStandardPivot`) leaves the picker row alone | +| `tests/claude-integration/claude-desktop-picker.test.ts` | enable order CA → trust → recheck → profile → rearm (recorded); preference false → `enable({ persist: true })` → armed; `enable({ persist: true })` with a failed independent condition (mode, intent, proxy, platform) leaves the preference unchanged; each failed precondition (mode not committed, intent off, proxy unbound, non-darwin) → its reason, nothing written; server-context trust failure → trust_pending + hint, no profile; cli-trusted but untrusted → trust_declined; newly added trust removed when the step-4 recheck fails or the profile write fails, pre-existing trust kept, failed untrust → residual ["trust"] and not effective; disable order disarm → persist (only when asked) → remove → untrust; **serialization**: an enable held pending, then a disable queued → after both, the runtime is disarmed and the profile absent; a disable held pending, then an enable queued → the enable runs after cleanup and re-arms only if its checks pass; an enable requested while a gateway `transition` is between cleanup and mode commit waits and then fails `mode_not_committed`; after a successful enable the CONNECT decision is `intercept` both while the lock is still held (owner `rearm()`) and after release, while a periodic `refresh()` during a pending disable does not arm; a `/api/sync` whose discovery resolves while a first-party `transition` holds the lock waits and then writes nothing; a server enable with `callerAddedTrust: true` that fails its recheck untrusts; one refused at the step-1 checks (for example `integration_off`) also untrusts; one refused while the owned profile is already selected from an earlier enable keeps the trust | +| `tests/claude-integration/claude-picker-runtime.test.ts` (wp3 file, extended) | `disarm()` makes the next claude.ai CONNECT blind while the cached mode is still first-party; periodic `refresh()` never clears the latch and skips arming while the controller is busy; a disarm during an in-flight `ensureStarted()` stays disarmed; startup refresh arms only with every piece present | +| `tests/claude-integration/claude-desktop-picker-routes.test.ts` | `PUT { enabled:false, persist:true }` disarms and persists false; `PUT { enabled:true, persist:true }` from a false preference re-arms; both routes accept the admin token; CONNECT decision (`selectTunnel("claude.ai", 443)`, trust and listener faked) is `intercept` after management and native first-party apply from an explicit `desktopMode: "gateway"` marker and after native OFF→ON on a server whose runtime started with the integration OFF, and `blind` after native OFF and after management gateway apply even when a periodic refresh runs | +| `tests/claude-integration/claude-desktop-cli.test.ts` (MODIFY) | `picker on|off|status|trust` parsing and usage errors; `picker trust` sends `callerAddedTrust` truthfully; a refused PUT leaves compensation to the server (fake server untrusts under its lock); connection refused before sending → local untrust of trust this run added; a PUT timeout while the server enable is held before profile selection → the CLI leaves trust alone and the later successful enable keeps it; `on` without a live proxy → proxy_unavailable; `off` offline removes artifacts locally; CLI first-party apply with a live proxy delegates to `POST /api/claude-desktop/apply { mode: "first-party" }` (fake runtimeRequest), also from a durable-OFF start | +| `tests/claude-integration/claude-desktop-first-party.test.ts` (MODIFY) | first-party apply (management, native) enables the picker by default and not when `intercept.picker === false` or when the mode write failed; gateway apply disables it without writing the preference; `ensureClaudeDesktopMatchesDesired` is awaited and its OFF branch removes a selected picker row (live: PUT; offline: local) | +| `tests/server/management-route-registry.test.ts` (MODIFY) | picker routes registered as standard entries and named by the capability | +| `tests/ci-workflows/skill-ocx.test.ts` | surface map current | + +Verifier: the files above plus `tests/providers/xai/grok-lifecycle.test.ts`, `bun run typecheck`, +`bun run skill:surface:check`, `bun run structure:check`, `bun run lint:gui`, `bun run build:gui`, +`cd gui && bun test --isolate tests`. + +## Audit record + +- wp4 pre-audit (reviewer, FAIL: 3 High, 2 Medium) folded: first-party apply keeps its env-first order inside the transition; runDesktopTransition defines the no-controller path; CLI delegation is limited to the local hub branch; callerAddedTrust is in the enable signatures and forwarded by the route (test in claude-desktop-picker-routes); runtime.ts is in the file inventory. Round 2 GO-WITH-FIXES (2): the controller gets a late-bound proxyPort accessor; DesktopPickerStatus carries lastBootstrapAt. diff --git a/devlog/_plan/260924_claude_desktop_picker_mode/040_wp5_live_proof_pr_merge.md b/devlog/_plan/260924_claude_desktop_picker_mode/040_wp5_live_proof_pr_merge.md new file mode 100644 index 0000000000..ec35fd1ed3 --- /dev/null +++ b/devlog/_plan/260924_claude_desktop_picker_mode/040_wp5_live_proof_pr_merge.md @@ -0,0 +1,61 @@ +# 040 — wp5: live proof, PR, CI, squash merge + +## Live proof (macOS, this machine) + +The operator's service runs the `dev` checkout. The proof runs this branch as that service only +for the proof window, then returns it to `dev` if the PR does not merge. + +1. Commit everything; record the branch head. +2. In `/Users/jun/Developer/new/700_projects/opencodex` (clean `dev`): `git fetch origin + codex/claude-desktop-picker-mode` and `git switch --detach FETCH_HEAD`; `bun run build:gui`; + `ocx service restart`; verify `/healthz`, PID path, listener. +3. `ocx claude desktop apply --first-party` (the operator's saved mode is already first-party): + the CLI delegates to the running service; expect the risk warning and a picker status. If the + service could not raise the keychain dialog the status is `trust_pending`: run + `ocx claude desktop picker trust` in the terminal, where the macOS password dialog is the + operator's step (NEEDS_HUMAN). Record which path showed the dialog (service or CLI). Then + `ocx claude desktop picker status` → `active` or `restart_required`. +3b. Name-constraint check on the Apple path: issue an ephemeral leaf for `example.com` from the + picker CA into `/private/tmp` (never into the config directory) and run + `security verify-cert -q -L -c <leaf> -p ssl -n example.com -k <login keychain>`; record the + exit code. Non-zero → the PR may state that macOS enforces the constraint; zero → the PR states + that only the key's confidentiality protects other names on this OS. Delete the ephemeral leaf. +4. Quit and reopen Claude Desktop (Computer Use). Check `main.log` for the egress pin line pointing + at the picker proxy port (`pickerProxyPort`, the intercept port + 1, never the Claude Code proxy + port), record the applied profile's exact `egressProxyUrl`, and find the picker log line + `picker GET bootstrap 200`. +5. Code tab → model picker: screenshot showing opencodex models by name next to Anthropic's. +6. Pick one (e.g. the xai Grok route), send "Reply with exactly: OCX-PICKER-PROBE. Do not use any + tools." Screenshot the reply; `usage.jsonl` must show the routed provider on the `messages` + ingress within the minute. +7. Connectivity: Chat tab still loads (screenshot), a claude.ai WebSocket session (Code session + list refresh) still works. +8. Dashboard screenshots: mode selector with the gateway default badge and the first-party risk + callout; picker card active. +9. If the operator declines the dialog: record NEEDS_HUMAN for criterion c-8, keep the rest. +10. After the proof, if the PR has not merged, roll back in this order while the branch service is + still running: `ocx claude desktop picker off` (branch CLI), verify the `opencodex-picker` row is + gone from `_meta.json`, `security find-certificate -a -Z -c "opencodex Claude Desktop Picker CA"` + finds nothing in the login keychain, and `claudeCode.intercept.picker` is false; fully quit and + reopen Desktop and confirm its log shows no egress pin; only then return the service checkout to + `dev` (`git switch dev`, rebuild GUI, restart). Record each result as rollback proof. If the PR + has merged, fast-forward `dev` and restart instead; picker stays under the new code's control. + +## PR + +- Title: `feat(claude): gateway by default, first-party risk warning, and Desktop picker mode`. +- Body per `.github/PULL_REQUEST_TEMPLATE.md`: Summary (problem, behavior before/after), screenshots + uploaded to the `pr-assets` branch and linked by commit SHA, Verification (commands + results, + what was not run locally and why), Checklist. No mention of third-party projects. +- Security review: an independent gpt-6-sol reviewer reads the full diff against the untracked + threat model at `.tmp/260924_claude_desktop_picker_mode/threat_model.md`; findings are folded + before merge and summarized in the PR. + +## CI and merge + +- Exact-head check-runs for the PR head (aggregate `ci` and its producers) must be completed and + successful; skipped path-gated jobs are listed as skipped, not as passed. +- Merge-result preflight: `git merge-tree --write-tree origin/dev HEAD`, file-size preflight on that + tree (offenders 0), typecheck and focused suites on a worktree of the merge result. +- `gh pr merge <n> --squash --admin --match-head-commit <head>` (user authorized merge). +- Verify `origin/dev` tip is the squash commit whose tree equals the verified merge-result tree. diff --git a/devlog/_plan/260924_cursor_fast_pricing/000_plan.md b/devlog/_plan/260924_cursor_fast_pricing/000_plan.md new file mode 100644 index 0000000000..f72d87d59d --- /dev/null +++ b/devlog/_plan/260924_cursor_fast_pricing/000_plan.md @@ -0,0 +1,30 @@ +# Cursor Fast pricing correction + +## Objective + +Price Cursor Claude Opus Fast usage at Cursor's published Fast rates for Opus 4.8, 5 and 5.5. + +## Scope + +- Modify `src/usage/expected-prices.ts` to recognize explicit Cursor Claude `-fast` IDs and register Cursor Fast multipliers for base model selections. +- Modify `src/usage/cost.ts` to apply the explicit-ID rate after user overlays have won and before the final expected-price result is returned. +- Add focused usage tests covering explicit fast IDs, persisted `tierOutcome` from `computeEntryCost`, standard turns, and user overlay precedence. +- Do not modify request logging: `src/adapters/cursor.ts:138-148` creates a `cursor-variant` tier outcome, `src/server/request-log.ts:767-790` stores it, and `src/usage/summary.ts:270-292` consumes it. + +## Rates + +Cursor's official model pages ([Opus 4.8](https://cursor.com/docs/models/claude-opus-4-8), +[Opus 5](https://cursor.com/docs/models/claude-opus-5), [Opus 5.5](https://cursor.com/docs/models/claude-opus-5-5)) publish: + +- Opus 4.8: standard 5/25/0.5/6.25; Fast 10/50/1/12.5. +- Opus 5: standard 5/25/0.5/6.25; Fast 10/50/1/12.5. +- Opus 5.5: standard 4/20/0.2/5; Fast 8/40/0.4/10. + +Opus 4.7 Fast is excluded because Anthropic says Fast requests error for that model. + +## Acceptance + +- Explicit Cursor Claude `-fast` IDs resolve to an official Fast tuple through the cost resolver. +- A persisted Cursor `tierOutcome` with `canonical:"priority", wireKind:"cursor-variant", wireValue:"fast", fastOutcome:"applied"` doubles the base rate for the three supported models. +- Standard and user configured prices remain correct. +- Focused usage tests and typecheck pass. diff --git a/devlog/_plan/260924_cursor_fast_pricing/010_implementation.md b/devlog/_plan/260924_cursor_fast_pricing/010_implementation.md new file mode 100644 index 0000000000..a2d5ed9f16 --- /dev/null +++ b/devlog/_plan/260924_cursor_fast_pricing/010_implementation.md @@ -0,0 +1,8 @@ +# Implementation + +1. Add a Cursor Claude Fast multiplier helper in `src/usage/expected-prices.ts` using the existing `normalizeCursorClaudeId` parser. It returns 2 only for fast Opus 4.8, 5 and 5.5 IDs. +2. Add Cursor priority pricing rules for the canonical base IDs. These rules do not require a response echo because the Cursor adapter's variant serialization is the observed wire evidence. +3. In `resolveMatchedPriceExact`, return user overlays first as today. For compiled Cursor expected or model-level prices, multiply only explicit fast IDs. Preserve `source`, `sourceRef`, and user overlay behavior. + Unsupported Fast spellings are fail-closed; supported explicit Fast rows carry Cursor's + direct source URL and `verified` provenance. +4. Add a new usage test file in the usage domain and register it in both layout registries. diff --git a/devlog/_plan/260924_cursor_fast_pricing/020_verification.md b/devlog/_plan/260924_cursor_fast_pricing/020_verification.md new file mode 100644 index 0000000000..2901bb16da --- /dev/null +++ b/devlog/_plan/260924_cursor_fast_pricing/020_verification.md @@ -0,0 +1,8 @@ +# Verification + +- Run the new Cursor Fast usage test and the existing usage cost test. +- Run `bun run typecheck`. +- Run `bun run structure:check` after staging the plan files. +- Run `bun run privacy:scan` because usage and pricing metadata are changed. +- Run a gpt-6-sol read-only adversarial review of the final diff. +- Push `codex/260924-cursor-fast-pricing` and open one PR to `dev`; the coordinator merges it. diff --git a/devlog/_plan/260924_l3_provider_adapters/010_plan.md b/devlog/_plan/260924_l3_provider_adapters/010_plan.md new file mode 100644 index 0000000000..960e278485 --- /dev/null +++ b/devlog/_plan/260924_l3_provider_adapters/010_plan.md @@ -0,0 +1,122 @@ +# L3 provider adapters — diff-level plan (wp1) + +Lane L3 bundles seven independent provider-adapter fixes into one PR against `dev` +(branch `codex/260924-l3-provider-adapters`, base `be0b5294e5`). Each item has its own +writer scope, and the lane lead registers every new test file in +`scripts/test-layout/layout.json` `explicit` and `tests/fixtures/test-layout-expected.json`. + +## Items and diffs + +### 1. #5692 DeepSeek quota currency symbol +- `src/providers/quota/vendor-probes-key.ts` `fetchDeepSeekQuota`: read `preferred.currency`, + map USD→`$`, CNY→`¥`, anything else → `"<CODE> "` prefix (trimmed, upper-cased; empty/missing → `$` + keeps legacy behaviour only when the row has no currency). Both label branches use it. +- Test: new sibling `tests/providers/deepseek-quota-currency.test.ts` (provider-quota.test.ts sits at + its 3763-line cap): CNY-only row → `API balance (¥76.88)`; USD row → `$`; CNY with granted → both + amounts use `¥`; unknown currency (e.g. EUR) → code prefix. + +### 2. #5689 Google array without items +- `src/adapters/google-tool-schema.ts` `sanitizeSchema`: after the `items` block, when + `out.type === "array"` and `out.items` is absent (source had no items, tuple items dropped, invalid + items widened, or budget ran out), set `out.items = { type: "string" }` and count a loss + (`invalid-schema-widened`) only when the source had no usable items. Valid `items` untouched; + nullable arrays keep `nullable`. Also covers `anyOf`-normalized arrays (apply after anyOf merge). +- Tests in `tests/adapters/google/google-tool-schema.test.ts` (486 lines, uncapped): issue repro + `{type:object, required:[values], properties:{values:{type:array}}}`; nested array; tuple items; + existing valid items byte-identical; non-array unaffected. + +### 3. #5695 mimo token-plan capacity facts +- `src/providers/registry/entries-extended.ts` `mimo` entry: add + `modelContextWindows` (all four ids 1_048_576), `modelMaxOutputTokens` (all four 131_072), + `modelInputModalities` (v2.6-pro, v2.6-flash, v2.5: `["text","image"]`; v2.5-pro: `["text"]`). + Source: mimo.mi.com/models/en-US/<id> (fetched 2026-09-24: 1M context, 128K output; v2.6-pro/flash + and v2.5 input Text/Image/Video/Audio, v2.5-pro Text). Video/audio are not representable in the + catalog's modality vocabulary, so only text/image are claimed. Keep `noVisionModels` and + `preserveCustomDestination`; update the comment. +- Test: registry/catalog assertion in a sibling test file (e.g. `tests/providers/mimo-token-plan-capacity.test.ts`). + +### 4. Carry #5693 (Vadevious) on #5725 +- `src/adapters/openai-chat/serialized-tool-call-content.ts`: add `repeatedCallIn` built on the + current `callsIn`/`blockAt`; in `duplicatedSerializedToolCallRanges` suppress the adjacent identical + pair only when exactly one structured call matches (compare with `freeformBody`); in + `repairArgumentsDuplicatedBesideSerializedCall` reduce a doubled `input` (direct or newline joined) + when `input` is the only key. +- Tests: port the PR's tests into `tests/adapters/openai/openai-chat-serialized-tool-call-content.test.ts` + and `tests/responses/responses-chat-tool-call-content.test.ts`; docs: adapters.md bullet, + `structure/providers/chat-compat.md` paragraph, ADR-5548 consequences line. +- Commit trailer: `Co-authored-by: Vadevious <Vadevious@users.noreply.github.com>`. + +### 5. #5698 command-code filter (marciodps) +- `src/adapters/command-code-tool-text.ts` per the reporter's final patch, with fixes: + mid-prose marker split in `textDelta` (not when the prefix is only whitespace on a probing block + with an empty probe — the existing probe already holds `"\n<tool_call>"`); shared + `probeBlockText` / `queueProseDelta` helpers; `breakOpenBlocks` skips held blocks; + `isLooseEnvelope` (null-safe `exec` result) used in `matchNative` and `settle` to drop malformed + envelopes naming a declared tool. +- Tests: new sibling `tests/providers/command-code-tool-text-prose-split.test.ts`: prose+markup in + one delta with native duplicate (dropped); prose+markup clean finish (restored); marker at index 0 + after streaming prose; leading-whitespace markup still held; interleaved reasoning keeps held + block; captured malformed `<parameter=` envelope with native duplicate (dropped, one call) and with + clean finish (dropped, no restore); `<tool_call>junk</tool_call>` does not throw; never-closing + partial markup released as text. +- Commit trailer: `Co-authored-by: marciodps <marciodps@users.noreply.github.com>` (or their commit email if public). + +### 6. #5096 remainder: `ocx effort model` slug resolution +- `src/cli/effort.ts` `inspectModelEffort`: after splitting provider/model, when the model is not a + known id, decode it with `decodeRoutedModelId(model, knownModelIdsForProvider(provider, prov, config))` + (router.ts / slug-codec.ts). Report `model` as the resolved native id and add `requestedModel` when it + differs. Unresolvable ids keep today's behaviour. +- Tests: new sibling `tests/cli/cli-effort-slug.test.ts`: `command-code/deepseek-deepseek-v4.1-flash` + and `command-code/deepseek/deepseek-v4.1-flash` report the same ladder as + `COMMAND_CODE_MODEL_REASONING_EFFORTS["deepseek/deepseek-v4.1-flash"]`; same for GLM 5.3 FlashX and + Gemini 3.8 Flash (the ids named in the latest issue comment). Ladders are read from the SSOT, not + restated, so no tier is invented. + +### 7. #5576 grok-4.7-build-fast +- `src/providers/registry/entries-core.ts` xAI entry: add `grok-4.7-build-fast` to + `modelContextWindows` (500_000), `modelReasoningEfforts` (low..xhigh), `modelDefaultReasoningEfforts` + (high), `modelInputModalities` (text,image). Not added to `XAI_MODELS`: xAI documents Grok 4.7 Fast as + "the same model served on faster infrastructure… not available on the public xAI API" + (docs.x.ai/developers/grok-4-7, fetched 2026-09-24). No service-tier claim. +- Test: new sibling `tests/providers/xai/grok-47-build-fast-metadata.test.ts` asserting the four facts + equal grok-4.7's. + +## Out of scope +#5421; the "per-model maps lost on restart" half of #5576; the opencode-go `openai-chat` wiring from +#5698 (reported against #5499 in the PR body). + +## Verification +`bun run typecheck`; each focused test file above plus existing neighbours +(`tests/providers/command-code-tool-text.test.ts`, `tests/providers/provider-quota.test.ts`, +`tests/adapters/google/google-tool-schema*.test.ts`, `tests/cli/cli-effort.test.ts`, xAI and catalog +parity suites); `bun run test:changed`; `bun run privacy:scan`; `bun run structure:check`. + + +## Audit fold (A, round 1 verdict FAIL → amendments) + +1. Item 2 materializes `items: {type:"string"}` without adding a loss category (representation fix, keeps + `lossy:false` contracts). Budget-exhausted return stays untouched (`google-tool-schema.test.ts:479-481`). + Existing tests that pin an array output without items (contract test ~578-587, tuple case ~303-322) update + their expected `parameters` only; category sets stay. `structure/providers/google.md` gains one sentence. +2. Item 1 also rewrites `tests/providers/provider-quota.test.ts:1131-1136` (CNY row) to `¥`, line-neutral + (file at its 3763 cap). +3. Item 7: add `grok-4.7-build-fast` to `modelContextWindows`, `modelReasoningEfforts`, + `modelDefaultReasoningEfforts`, `modelInputModalities`, plus the reasoning-model parameter lists xAI documents + for reasoning models (`noStopModels`, `noPenaltyModels`, `preserveReasoningContentModels`). Not added: + `modelWireDefaults` and `modelSupportsServiceTier` (live-probed on grok-4.7 only), `XAI_MODELS`. Update exact + literals: `provider-registry-parity.test.ts:1319`, `xai-no-stop.test.ts:47`, `xai-transport.test.ts:634,844`. + `structure/providers/xai-grok.md` gains a line. +4. Item 6: the decoded id replaces `modelId` before `modelInList` / `configuredReasoningEfforts` / + `reasoningEffortMapFor`. +5. Item 4: compare with `freeformBody` on both sides, tail must be exactly one repetition, add a leading-newline + case. Rebuttal: the newline-joined doubled-input repair stays — #5693 commit 7854ac8 added it with its own + regression test and the PR body documents it. +6. Item 5 source is marciodps' third follow-up comment on #5698 (2026-09-23T19:33Z), full patch vs 2.64.0. + Preserve the marker-free fast path (`command-code-tool-text.test.ts:388`), the queue-visit bound (`:324`), + and whitespace-probe salvage. +7. Lead registers every new test file in both layout maps. + + +Round 2 verdict PASS. Residuals: item 2 drops the original "budget ran out" clause, and the two contract +tests assert only loss reports, so no expectation needs editing there; item 7 keeps grok-4.7-build-fast on +the provider-default wire, to be re-checked on first live discovery. diff --git a/devlog/_plan/260924_l4_codex_cli_service/000_roadmap.md b/devlog/_plan/260924_l4_codex_cli_service/000_roadmap.md new file mode 100644 index 0000000000..ec3ed89728 --- /dev/null +++ b/devlog/_plan/260924_l4_codex_cli_service/000_roadmap.md @@ -0,0 +1,14 @@ +# L4 roadmap — Codex integration, CLI and service + +Branch codex/260924-l4-codex-cli-service from origin/dev be0b5294e5. One PR to dev. Merge is the coordinator's. + +| Unit | Doc | Items | Method | +|---|---|---|---| +| wp1 | 010 | #5713, #5703, #5548 slice | squash-diff apply per PR, one commit each with Co-authored-by | +| wp2 | 020 | #5221 | rebuild on dev by DeepSeek writer; sibling test files | +| wp3 | 030 | #5009 | squash-diff apply, review sender/admission checks | +| wp4 | 040 | #5694 | DeepSeek writers: default-on 98% hard lock | +| wp5 | 050 | publish | rebase, validate, push, PR, checks | + +Ratchet: tests/fixtures/file-size-baseline.json caps only move down; every commit re-runs tests/test-layout.test.ts and the file-size test. + diff --git a/devlog/_plan/260924_l4_codex_cli_service/010_carry_ready_prs.md b/devlog/_plan/260924_l4_codex_cli_service/010_carry_ready_prs.md new file mode 100644 index 0000000000..ca9be0de84 --- /dev/null +++ b/devlog/_plan/260924_l4_codex_cli_service/010_carry_ready_prs.md @@ -0,0 +1,20 @@ +# 010 — carry #5713, #5703, #5548 slice + +## #5713 (fixes #5699) — author 정우철 <oocheol@naver.com> +Files: src/client/connect.ts, src/client/state.ts (pending-connect fingerprint marker client-connect-pending), src/service/cli.ts (removeServiceTokenAfterUninstall under client lifecycle + config mutation locks: removed|absent|retained|unverified), structure/clients/claude-desktop.md, structure/runtime.md, docs-site guides/remote-hub.md (en+ko), tests/clients/client-connect.test.ts, tests/service/service-secrets.test.ts. +Method: git diff merge-base..carry-5713 | git apply -3. Check other docs-site locales of remote-hub.md for consistency (PR touched en and ko only). +Tests: bun test tests/service/service-secrets.test.ts tests/clients/client-connect.test.ts. +Security: credential deletion boundary — uninstall deletes the service token only when client state is disconnected and no pending marker owns the fingerprint. + +## #5703 (fixes #5701) — Konstantinos <37538071+konstantinosbotonakis@users.noreply.github.com> +Files: src/codex/native-residue.ts, structure/config.md, tests/codex-integration/codex-native-residue.test.ts. Check file-size caps for the test file. + +## #5548 slice — Vadevious <Vadevious@users.noreply.github.com> +Only src/codex/home.ts (import expandUserPath from ../config/paths), structure/codex-home.md line, tests/codex-integration/codex-home-wsl.test.ts (new: register in layout.json explicit + test-layout-expected.json if not matched by a seed). Excluded: tests/cli/cli-help.test.ts, tests/service/service-probe-docker.test.ts, tests/service/service.test.ts. +Audit fold: codex-home-wsl.test.ts already exists on dev (from #5720) and is registered; carry only the PR's added fresh-process case into it. home.ts:4 currently imports from ../config (barrel) — the fix switches to ../config/paths. + +## wp1 P (executable) +- All three squash diffs pass git apply --check -3 on 34fb6d649c (/tmp/l4-5713.diff, /tmp/l4-5703.diff, /tmp/l4-5548.diff limited to 3 files). +- Commit order: #5713, #5703, #5548 slice; each commit carries Co-authored-by for the PR author. +- #5713 docs: en/ko remote-hub.md updated by the PR; DeepSeek writer adds the same paragraph to fr, ja, ru, tr, zh-cn, zh-tw remote-hub.md next to the service-api-token paragraph. +- Focused tests: tests/service/service-secrets.test.ts tests/clients/client-connect.test.ts tests/codex-integration/codex-native-residue.test.ts tests/codex-integration/codex-home-wsl.test.ts tests/test-layout.test.ts; plus the file-size ratchet test. diff --git a/devlog/_plan/260924_l4_codex_cli_service/020_subagent_identity_5221.md b/devlog/_plan/260924_l4_codex_cli_service/020_subagent_identity_5221.md new file mode 100644 index 0000000000..feed2618b9 --- /dev/null +++ b/devlog/_plan/260924_l4_codex_cli_service/020_subagent_identity_5221.md @@ -0,0 +1,14 @@ +# 020 — rebuild #5221 (fixes #5217) — sbc1-code <207095575+sbc1-code@users.noreply.github.com> + +Change: src/adapters/identity.ts gains ROUTED_IDENTITY_RE, NEUTRAL_IDENTITY_RE, hasRoutedIdentity, repairRoutedIdentity, repairIdentityInResponsesBody, stripRoutedIdentity; identifyRoutedModel also rewrites neutral + routed lines. Catalog (src/codex/catalog/derive-entry.ts, metadata.ts) writes neutralizeIdentity instead of identifyRoutedModel. src/responses/parser.ts repairs developer items to data.model. src/adapters/openai-responses/passthrough.ts repairs raw body (routed rewrite, forward strip). +Rebuild on current dev (221 commits drift): re-read current versions of those files; keep dev behaviour elsewhere. +Tests: new tests/adapters/identity-subagent.test.ts (from PR). Catalog expectation edits: codex-catalog.test.ts is at cap, so changed expectations must not grow it; new catalog cases go to a sibling file (e.g. tests/codex-integration/codex-catalog-identity-neutral.test.ts) registered in both manifests. +Check files at cap: parser.ts, passthrough.ts, identity.ts in file-size-baseline. +Audit fold: neutralizeIdentity and NEUTRAL_IDENTITY_LINE already exist in identity.ts; the six routed-identity helpers do not. parser/passthrough/identity are uncapped (<2000 lines). codex-catalog.test.ts cap 7985 vs 7974 actual: expectation edits may not add net lines beyond headroom; new cases still go to a sibling file. + +## wp2 P (executable) +- /tmp/l4-5221.diff (src, 5 files) passes git apply --check -3 on current HEAD. Current request-time callers of identifyRoutedModel: anthropic.ts:772, google.ts:300, kiro/payload.ts:145, command-code.ts:607, openai-chat/messages.ts:144 — all receive the extended rewrite (GPT line, neutral line, routed line), so a neutral catalog still names the wire model. +- Writer applies the src diff, adds tests/adapters/identity-subagent.test.ts from carry-5221, and rewrites the catalog expectations the PR changed (codex-catalog.test.ts, codex-catalog-sync-hardening.test.ts) so codex-catalog.test.ts grows by 0 net lines; any added catalog assertion goes to new tests/codex-integration/codex-catalog-identity-neutral.test.ts registered in scripts/test-layout/layout.json explicit and tests/fixtures/test-layout-expected.json. +- Writer also checks every other test that asserted a baked model id in base_instructions (rg 'powered by the' tests) and updates it. +- structure: add a line to the structure doc owning src/adapters/identity.ts (find via structure/manifest.json) describing request-time identity repair. +- Focused: bun test tests/adapters/identity-subagent.test.ts tests/adapters/identity-neutralize.test.ts tests/codex-integration/codex-catalog.test.ts tests/codex-integration/codex-catalog-sync-hardening.test.ts plus new sibling, tests/test-layout.test.ts, tests/ci-workflows/file-size-ratchet.test.ts. diff --git a/devlog/_plan/260924_l4_codex_cli_service/030_agent_message_recovery_5009.md b/devlog/_plan/260924_l4_codex_cli_service/030_agent_message_recovery_5009.md new file mode 100644 index 0000000000..62632fcbb5 --- /dev/null +++ b/devlog/_plan/260924_l4_codex_cli_service/030_agent_message_recovery_5009.md @@ -0,0 +1,10 @@ +# 030 — carry #5009 — Zhaofeng Li <lzfxxx@gmail.com> + +Files: src/server/responses/agent-task-recovery.ts (FOLLOWUP_TASK, FINAL_ANSWER with optional Task name; author===sender check kept; recipient cross-check when task name present; JSON tuple cache key including recipient; foreign-family echo rejected), src/server/responses/encrypted-payload.ts (guard regex covers four types), structure/subagents.md, docs-site subagent-v1-default.md, configuration/agents.md, providers.md (check locales), tests/server/agent-task-recovery.test.ts, server-agent-task-recovery-replay.test.ts, v2-agent-message-failfast.test.ts, tests/helpers/agent-task-recovery.ts. +Method: squash diff from merge-base e9643875f0 applied with -3; check caps on test files (agent-task-recovery.test.ts +161). +Keep: recoveryAdmission before cache; agentTaskRecovery.enabled default-off. + +## wp3 P (executable) +- /tmp/l4-5009.diff (squash from merge-base e9643875f0, 10 files) passes git apply --check -3 on HEAD af0a3713ea. Test files are below the 2000-line ratchet threshold (1128/438/907 before +163/+33/+48). +- DeepSeek writer updates item 4 of guides/subagent-v1-default.md in fr, ja, ko, ru, tr, zh-cn, zh-tw: they still say recovery loses message-type follow-ups, which contradicts the carried English text. +- Focused: bun test tests/server/agent-task-recovery.test.ts tests/server/server-agent-task-recovery-replay.test.ts tests/server/v2-agent-message-failfast.test.ts tests/responses/*encrypted* (rg for encrypted-payload tests). diff --git a/devlog/_plan/260924_l4_codex_cli_service/040_main_hard_lock_default_5694.md b/devlog/_plan/260924_l4_codex_cli_service/040_main_hard_lock_default_5694.md new file mode 100644 index 0000000000..a35ecb810e --- /dev/null +++ b/devlog/_plan/260924_l4_codex_cli_service/040_main_hard_lock_default_5694.md @@ -0,0 +1,35 @@ +# 040 — #5694 98% main-account hard lock by default + +Existing mechanism: codexMainAccountHardLock (src/types/config.ts, schema .catch(false)), MAIN_ACCOUNT_HARD_LOCK_PERCENT=99 in src/codex/quota-types.ts, getMainAccountHardLockStatus in src/codex/main-account-hard-lock.ts, consumers gate on === true in src/codex/auth-context.ts, src/codex/native-profile-startup.ts, src/server/management/config-routes.ts, gui/src/components/MainAccountHardLockSetting.tsx. +Change: +- MAIN_ACCOUNT_HARD_LOCK_PERCENT = 98. +- New resolver isMainAccountHardLockEnabled(config) = config.codexMainAccountHardLock !== false; replace every === true gate. +- Schema: invalid value -> undefined (default on), explicit false persists opt-out; management PUT false stores false (not delete), true deletes key or stores true — decide by reading config-routes semantics. +- GUI: toggle shows on when unset; copy 99% -> 98% in all locales; confirmation dialog still shown when enabling. +- Docs: providers-accounts.md (en, ko, others), configuration/providers.md, guides/providers.md, ru; structure/providers/openai-tiers.md. +- Tests: default-on status, explicit false off, 98 threshold (97.9 ready, 98 blocked), config-route round trip. +Trade-off for PR: default lock can keep main-account Luna Reserve from activating (Reserve needs exhausted normal window); opt-out by setting false. + +## Audit folds (wp0 A) +- src/codex/main-account-hard-lock.ts:26 gates on !== true: switch to resolver. +- config-schema.ts:152 .catch(false) -> .catch(undefined) so malformed values fall back to the default (on). +- config-routes.ts:598 deletes the key on PUT false: must store false; PUT true deletes the key (default on). Projections at :351 and :706 become resolver-based, else GUI invariant at MainAccountHardLockSetting.tsx:14 fails. +- quota.ts:262 and :388 also use MAIN_ACCOUNT_HARD_LOCK_PERCENT (blocking-evidence retention); they follow the constant. +- auth-context.ts:512 hardcoded "99%" message -> derive from constant. + +## wp4 P (executable) +Semantics: codexMainAccountHardLock absent or true = on; explicit false = off (persisted). Threshold 98 (MAIN_ACCOUNT_HARD_LOCK_PERCENT). Same convention as fastRows in config-routes.ts. +Writer A (core, write scope: src/**, tests/**, structure/**, scripts/test-layout/layout.json): +- src/codex/quota-types.ts: MAIN_ACCOUNT_HARD_LOCK_PERCENT = 98 (#5694). +- src/codex/main-account-hard-lock.ts: export isMainAccountHardLockEnabled(config) = config.codexMainAccountHardLock !== false; getMainAccountHardLockStatus uses it. +- Replace every === true gate: auth-context.ts 960/973/1017/1545/1623, native-profile-startup.ts 191/228/239/389/708, config-routes.ts projections 351/706. +- config-routes.ts PUT: true deletes key, false stores false (mirror fastRows); rollback path unchanged. +- config-schema.ts: .catch(undefined) so malformed values mean default-on. +- types/config.ts doc comment; auth-context.ts:512 message uses the constant. +- Tests: update existing hard-lock tests; add tests/codex-integration/main-account-hard-lock-default.test.ts (absent=on, false=off, malformed=on, 97.9 ready, 98 blocked, settings GET/PUT round trip) registered in layout.json explicit + test-layout-expected.json. codex-auth-api.test.ts cap 6549 (6514 now): no net growth beyond cap. +- structure/providers/openai-tiers.md:291-315 rewrite (on by default, 98%, opt-out persists false, Reserve trade-off). +Writer B (gui/src/i18n/*.ts only): 99 -> 98 in mainHardLock* strings in all locales; desc adds "On by default." +Writer C (docs-site/** only): providers-accounts.md en + ko (and other locales mentioning it) -> 98%, on by default, opt-out; check configuration/providers.md:71 and guides/providers.md:377. +Previous opt-outs deleted the key, so they cannot be distinguished; PR notes it. +Focused: main-account-hard-lock-*.test.ts, settings-main-account-hard-lock, main-quota-*, reserve-*, reserve-claude-policy, codex-quota-auto-refresh-main-admission, codex-auth-api, codex-account-threshold-api; bun run lint:gui; GUI i18n tests. + diff --git a/devlog/_plan/260924_l4_codex_cli_service/050_publish.md b/devlog/_plan/260924_l4_codex_cli_service/050_publish.md new file mode 100644 index 0000000000..89e1484903 --- /dev/null +++ b/devlog/_plan/260924_l4_codex_cli_service/050_publish.md @@ -0,0 +1,9 @@ +# 050 — publish +Rebase on origin/dev; bun run typecheck; focused tests for every item; bun run test:changed; bun run privacy:scan; bun run structure:check; bun run lint:gui; push branch; gh pr create --base dev with template + Security review + Carries/Closes; watch automatic checks; fix failures on head. + +## wp5 P (executable) +- origin/dev moved to df61bcecd5 (#5737, #5738). Overlapping files: docs-site reference/configuration/providers.md, scripts/test-layout/layout.json, structure/config.md, structure/subagents.md, tests/fixtures/test-layout-expected.json. Resolve by keeping both sides' entries. +- Before rebase: git mv devlog/_plan/260924_l4_codex_cli_service -> devlog/_fin/ is deferred until after merge (unit stays open while the PR is open). +- Validation after rebase: bun run typecheck; focused union of all item tests; bun run test:changed (worktree under /Users/jun/.codex trips the test-home guard for some temp-dir tests — compare any failure with a pristine git archive copy); bun run privacy:scan; bun run structure:check; bun run lint:gui. +- GUI screenshot: run the dashboard (bun run src/cli/index.ts start on a spare port with a throwaway OPENCODEX_HOME), open Codex settings > Multi-auth > Advanced, capture the 98% card to .tmp/. Hosting: pr-assets push only if the user authorizes; otherwise the coordinator uploads it. +- PR body: Summary per item with Carries/Closes lines, Security review for #5713, Verification with commands, Checklist, coordinator decisions (Reserve trade-off, previous opt-outs re-enabled, stale connect marker retention, 64 MiB indeterminate). diff --git a/devlog/_plan/260924_protocol_first_class/000_plan.md b/devlog/_plan/260924_protocol_first_class/000_plan.md new file mode 100644 index 0000000000..4c98f28292 --- /dev/null +++ b/devlog/_plan/260924_protocol_first_class/000_plan.md @@ -0,0 +1,88 @@ +# 260924 Protocol first class — plan + +Responses stays the first-class feature surface. What changes is that Chat Completions and +Anthropic Messages stop needing the public Responses JSON/SSE as a mandatory intermediate to +reach the shared execution policy. Same-wire requests keep their source representation; +cross-wire requests convert through the adapter-neutral IR; one execution owner keeps account +selection, affinity, send budget, retry, cancellation and logging. + +## Outcome + +```text +Responses / Chat / Messages request + -> source body kept + lazily parsed intent + -> shared admission, routing, execution policy + -> final provider / model / credential settled + -> per-attempt protocol plan + same wire : native builder from the source body + different wire : codec -> IR -> target builder + not migrated yet : legacy bridge (internal Responses), labelled as such + not expressible : refused before any send (when policy = reject) + -> upstream + +same wire : upstream -> safe relay + observation -> client wire +different wire: upstream -> AdapterEvent -> client encoder +``` + +## Non-goals + +- Files, Batches and Responses CRUD APIs; every beta feature; lossless behavior for arbitrary + custom providers; emulating every Responses-only feature on Chat or Messages. +- A second or third execution engine. Native lanes reuse the shared attempt, budget, cancel + and log owners; they do not copy them. +- Renaming the Responses schema and calling it a neutral IR. +- Forwarding arbitrary headers or body fields to every provider unchecked. +- Guessing native support from a provider name in the GUI. +- A shadow mode that sends two inferences. Shadow compares plans only. +- Presenting one successful connection as protocol verification. + +## Invariants every work packet keeps + +1. The planner never selects a provider. It consumes the route the router settled + (`ResolvedModelPolicy` precedence: hard-pin, explicit override, ingress-scoped registry + default, provider default) and decides only the wire within that route. +2. No native lane may bypass admission scope, send budget, affinity, key failover, cancellation, + request logging or spend accounting. A native lane that lands before its safety wiring is + not acceptable in any order. +3. Every fallback candidate builds its request from the source envelope. An earlier candidate's + deleted fields or injected headers are never the next candidate's input. +4. Plan and trace records carry only fixed vocabulary (`src/protocols/contract.ts`) and + identifiers the server already exposes. No prompt, tool argument, token, signature or key. +5. `native` (a delivery mode) and `VERIFIED` (a Lab evidence verdict) are different axes and + are never merged into one badge. +6. Every rollout switch defaults off and changes nothing while off + (`resolveProtocolSettings`, `src/protocols/settings.ts`). +7. The core request path stays free of Lab imports + (`tests/lab/core-lab-boundary.test.ts`). +8. Files at their file-size cap (`tests/fixtures/file-size-baseline.json`) do not grow; code + moves out first. `src/server/request-log.ts` sits at 1996 of a 2000-line threshold. + +## Work packets and stack order + +Each packet is one pull request, stacked on the previous one. PF numbers are work ids, not +GitHub numbers. + +| Packet | Branch | Scope | Doc | +|---|---|---|---| +| PF-01 | `feat/pf01-protocol-contract` | vocabulary, feature dispositions, 18-cell baseline, plan/trace DTOs, settings keys | [010](010_contract_and_baseline.md) | +| PF-02 | `feat/pf02-protocol-trace` | observed path trace on request/attempt rows, persisted; Logs badge, detail, filter | [030](030_gui_and_management_api.md#pf-02-observed-path-trace) | +| PF-03 | `feat/pf03-protocol-plan` | pure planner, `GET /api/protocols`, `POST /api/protocols/plan`, API page preview | [030](030_gui_and_management_api.md#pf-03-planner-and-preview) | +| PF-05 | `feat/pf05-inference-primitives` | shared execution context, attempt and delivery primitives, client-wire marker | [020](020_engine_and_codecs.md#pf-05-shared-inference-primitives) | +| PF-04 | `feat/pf04-api-surfaces` | Messages exposure split from Claude integration, settings PATCH, API cards | [030](030_gui_and_management_api.md#pf-04-api-surface-settings) | +| PF-06 | `feat/pf06-source-envelope` | source envelope, codecs, unrepresentable guard | [020](020_engine_and_codecs.md#pf-06-source-envelope-and-guard) | +| PF-09 | `feat/pf09-direct-encoders` | AdapterEvent to Chat/Messages encoders behind `directEncoders` | [020](020_engine_and_codecs.md#pf-09-direct-client-encoders) | +| PF-07 | `feat/pf07-native-chat-combos` | eligible Chat candidates in combos/policy send natively | [020](020_engine_and_codecs.md#pf-07-native-chat-candidates-in-combos) | +| PF-08 | `feat/pf08-managed-messages-native` | key-auth Anthropic targets receive `/v1/messages` natively | [020](020_engine_and_codecs.md#pf-08-managed-native-messages) | +| PF-11 | `feat/pf11-protocol-evidence-gui` | provider protocol panel, compatibility pair filters, combo guarantees, deep links | [030](030_gui_and_management_api.md#pf-11-evidence-combo-and-provider-views) | +| PF-10 | `feat/pf10-auth-opaque-state` | beta allowlist, OAuth native Messages, cross-domain opaque state guard | [020](020_engine_and_codecs.md#pf-10-auth-and-opaque-state) | +| PF-12 | `feat/pf12-protocol-rollout` | shadow plan comparison, docs, not-migrated inventory | [040](040_acceptance_and_rollout.md) | + +PF-05 lands before PF-04 because PF-06 through PF-09 build on its primitives and PF-04 does +not; the dependency order, not the id order, decides the stack. + +## Verification policy for this unit + +Each pull request records exactly what ran. Unit tests are written beside each change and +registered in the test layout; whether they were executed is stated per PR, never implied. +Default flips for rollout switches are out of scope until the acceptance scenarios in +[040](040_acceptance_and_rollout.md) have recorded evidence. diff --git a/devlog/_plan/260924_protocol_first_class/010_contract_and_baseline.md b/devlog/_plan/260924_protocol_first_class/010_contract_and_baseline.md new file mode 100644 index 0000000000..fcd3f833d4 --- /dev/null +++ b/devlog/_plan/260924_protocol_first_class/010_contract_and_baseline.md @@ -0,0 +1,104 @@ +# 010 — PF-01 contract and baseline + +## Modules + +| File | Role | Import rule | +|---|---|---| +| `src/protocols/contract.ts` | protocols, upstream wires, hops, delivery modes, reason codes, name mappings | leaf | +| `src/protocols/features.ts` | feature keys, sources, cross-wire hop dispositions, body extraction, path effects | leaf (+ type from `src/compatibility/manifest.ts`) | +| `src/protocols/baseline.ts` | 18-cell current/target matrix | leaf | +| `src/protocols/dto.ts` | `ProtocolPlanV1`, `ProtocolTraceV1`, validators, limits | leaf | +| `src/protocols/settings.ts` | only reader of `apiSurfaces` and `protocols` config | type-only config import | + +"Leaf" means importable from `gui/src/*`: no server, router, provider or Lab imports, not even +as types. `tests/responses/protocol-contract.test.ts` enforces it by reading import specifiers. + +## Vocabulary + +- **Protocol**: `responses | chat | messages` — the public API a client speaks. +- **Upstream wire**: a protocol or `other` (Gemini, Kiro, Cursor, ...). Nothing is claimed about + `other`; absent feature dispositions there mean unknown. +- **Hop**: a wire name, `ir` (`OcxParsedRequest` / `AdapterEvent`), or `responses-internal` + (Responses JSON/SSE produced only as an internal bridge). +- **Delivery mode**: `native` (same wire, source body is the wire source), `translated` + (cross-wire through the IR or the target wire only), `legacy-bridge` (path contains + `responses-internal`), `blocked` (refused before any send). +- **Fidelity**: `preserved`, `degraded`, `unknown`. + +Internal spellings keep their names and map explicitly: routing `InboundWire` "anthropic" is +`messages`; adapters `openai-responses`/`openai-chat`/`anthropic` are the three protocol wires; +Lab identities `openai-responses`/`openai-chat`/`anthropic-messages` map one to one. Persisted +rows are not rewritten. + +## Feature dispositions and their evidence + +Same-wire hops are passthrough by definition. Cross-wire claims about current code: + +| Claim | Evidence | +|---|---| +| Chat `n`, `logprobs`/`top_logprobs`, `logit_bias`, `seed`, `audio`/`modalities`, `prediction` are unsupported on `chat>responses` | `chatCompletionsToResponsesBody` in `src/chat/inbound.ts` builds the body from an explicit field list without them | +| Chat tools, sampling, stop, user, parallel tool calls, service tier, prompt cache key, metadata, reasoning effort, response format are translated on `chat>responses` | same function, explicit field copies | +| Messages `top_k` is unsupported on `messages>responses` | `src/claude/inbound.ts` header: accepted and dropped | +| Messages `thinking.budget_tokens` is degraded on `messages>responses` | maps to an effort tier, never forwarded raw | +| Messages `cache_control` is degraded on `messages>responses` | block-level cache hints are not carried into the Responses body; caching is re-derived downstream | +| Responses hosted tools are degraded on `responses>chat` and `responses>messages` | only web search and image generation have sidecar bridges | +| Responses `previous_response_id` and compaction are translated off-wire | expanded from proxy-side state before the adapter runs | +| Responses `store` is degraded and `background` unsupported off-wire | only proxy-side state exists for non-Responses upstreams | + +The `chat>messages` and `messages>chat` rows describe the target direct codecs (PF-06). Today +those pairs travel through `responses-internal`, so their effective disposition is the +composition of the two Responses hops, which `featureEffectsForPath` computes from the path. + +## Baseline (eligible single-provider route) + +| inbound → upstream | current | target | +|---|---|---| +| responses → responses | native `responses,responses` | native | +| responses → chat / messages | translated `responses,ir,X` | translated | +| chat → chat | native `chat,chat` (JSON when `stream:false`) | native | +| chat → responses | translated `chat,responses` | translated | +| chat → messages | legacy-bridge `chat,responses-internal,ir,messages` | translated `chat,ir,messages` | +| messages → messages | legacy-bridge for managed keys (native only for caller-forwarded Anthropic credentials) | native | +| messages → responses | translated `messages,responses` | translated | +| messages → chat | legacy-bridge | translated `messages,ir,chat` | + +Every routed (non-native) Chat and Messages path streams internally and folds for a +non-streaming client (`sse-folded`). The stream axis doubles the table to 18 cells in +`src/protocols/baseline.ts`. + +Out of scope for the baseline, and described per request by the planner instead: combos and +policy routes (legacy bridge today), OAuth/forward credentials, synthetic effort rows, +vision-preprocessed images, tool-result images, and Responses-only features on Chat. + +## DTOs + +`ProtocolTraceV1` is the observed record for one request (`v: 1`), persisted with the request +log row; `attempts[]` records each physical attempt's upstream, mode and request path. A blocked +request has empty paths. `ProtocolPlanV1` is the predicted record (`schemaVersion: 1`) with one +candidate per route target, `guaranteedFeatures` (preserved by every eligible candidate) and +`partialFeatures`. Both validators reject anything that is not exactly v1, oversize, or outside +the vocabulary. Limits: 6 hops, 8 reason codes, 24 feature effects, 16 candidates, 16 attempts, +200-character identifiers without control characters. + +## Settings + +```jsonc +{ + "apiSurfaces": { "messages": { "enabled": false } }, // absent => inherit claudeCode.enabled + "protocols": { + "unrepresentable": "legacy", // or "reject" + "rollout": { + "nativeChatCombos": false, + "managedMessagesNative": false, + "managedMessagesNativeOAuth": false, // effective only with the key-auth switch + "directEncoders": false, + "shadowPlan": false + } + } +} +``` + +`apiSurfaces` is kept raw by the config schema and parsed fail-closed: a present malformed value +closes the surface rather than inheriting. `protocols` is schema-validated and drops to defaults +when malformed, because every default is the conservative one. PF-01 adds the keys and the +resolver only; no request path reads them until PF-04. diff --git a/devlog/_plan/260924_protocol_first_class/020_engine_and_codecs.md b/devlog/_plan/260924_protocol_first_class/020_engine_and_codecs.md new file mode 100644 index 0000000000..4639664e6b --- /dev/null +++ b/devlog/_plan/260924_protocol_first_class/020_engine_and_codecs.md @@ -0,0 +1,136 @@ +# 020 — execution, codecs and encoders (PF-05 to PF-10) + +The execution owner stays single. Native lanes and the Responses pipeline share the same +primitives for budgets, attempts, cancellation, final logging and client-wire identity; the +Responses-only state (`previous_response_id`, compaction, response-state retention, Codex +specifics) stays in `src/responses/*` and `src/server/responses/*` and is reached through hooks, +not moved. + +## PF-05 shared inference primitives + +New directory `src/server/inference/` (inside the already-documented `src/server/` area). + +| File | Exports | Replaces | +|---|---|---| +| `context.ts` | `createInferenceSendBudget(req, logCtx)` — the one construction of `createRequestExecutionBudget(undefined, undefined, attachRequestSpendTracker(req, logCtx))` | the inline construction in `src/server/responses/core.ts` `handleResponses` and the native Chat equivalent | +| `final-log.ts` | `createFinalRequestLog(logIds, logCtx)` returning `{ finish(status, meta), finished() }`; finish-once | the `finalizeNativeLog` / `finishLog` / `nativeLogged` closures repeated in `chat-completions.ts`, `chat-native.ts`, `claude-messages.ts` | +| `attempt.ts` | `beginInferenceAttempt(logCtx, { provider, model, adapter })` → `{ attempt, seal(accountLabel?), finish(status, usage?) }`; wraps `beginRequestAttempt`, `activeAttempt`, `activeAttemptStartedAt`, `attempts.push`, `sealRequestAttemptIdentity`, `finishRequestAttempt` | the hand-rolled attempt opening in `chat-native.ts` and the combo child | +| `client-wire.ts` | `markClientWire(response, protocol)`, `clientWireOf(response)` (WeakMap on `Response`) | nothing yet; PF-07 and PF-09 use it so an ingress can tell a body already in its client wire from a Responses body | + +Rules: + +- Behavior-preserving. Every moved statement keeps its order relative to the sends it guards. +- `src/server/responses/core.ts` must shrink (it is at its file-size cap); PF-07 needs a few + lines of headroom there. +- Split `handleNativeChatCompletions` in `src/server/chat-native.ts` into + `runNativeChatAttempt(execution, attemptHandle)` — send loop, key failover, 429 replay, + relay, usage — and the existing wrapper that opens the attempt and owns the final log. The + split is what lets a combo child run a native attempt whose final log belongs to the parent. +- No new behavior, no new config reads. + +## PF-06 source envelope and guard + +| File | Role | +|---|---| +| `src/protocols/envelope.ts` | `createProtocolEnvelope({ inbound, body, translatorBudget })` → `{ inbound, features(), freshBody() }`. The source body is retained for the request lifetime only; `features()` is computed once lazily; `freshBody()` returns a structured clone charged to the translator budget (`request_copies`). Server-side module (may import `src/lib/translator-budget`). | +| `src/protocols/codecs/chat.ts`, `codecs/messages.ts`, `codecs/responses.ts` | thin, named entry points over the existing translators (`chatCompletionsToResponsesBody`, `anthropicToResponsesTranslation`, the Responses parser) so ingress code calls one codec surface. No translator behavior changes. | +| `src/protocols/guard.ts` | `checkRepresentable({ inbound, requestPath, features, policy })` → `{ ok: true } \| { ok: false, features, reasonCodes }` using `featureEffectsForPath` and `unrepresentableFeatures`. Pure. | + +Wiring (only when `resolveProtocolSettings(config).unrepresentable === "reject"`): + +- `src/server/chat-completions.ts`: after the route settles and before the Responses projection + is built, compute the path the request will take (native Chat path when native-eligible; + otherwise the bridge path for the settled route's adapter). Unknown-adapter hops never block. + A refusal returns HTTP 400 in Chat error shape, `type: "invalid_request_error"`, + `code: "unsupported_feature"`, message naming the feature keys only, marks the trace blocked + (`feature-unrepresentable`) and logs the request with no upstream send. +- `src/server/claude-messages.ts`: the same after route settlement, Anthropic error shape. +- Combo and policy routes are checked per candidate in PF-07; at ingress they are not refused. +- `legacy` policy: no behavior change; the would-be refusal is still reflected in the trace's + `featureEffects`. + +Native Chat's in-place effort normalization (`chatEffortSnapshots`) keeps working; PF-07 moves +combo children onto `freshBody()` so no candidate inherits another's rewrite. + +## PF-07 native Chat candidates in combos + +Behind `protocols.rollout.nativeChatCombos`. + +- `HandleResponsesOptions` (`src/server/responses/core-options.ts`) gains + `protocolSource?: { inbound: "chat"; envelope; dispatchNativeChild(input) }`, supplied only by + `src/server/chat-completions.ts` for combo routes when the switch is on. +- In `src/server/responses/core-combo.ts`, where each child is dispatched through + `requestDispatchers.handleResponses`, a child whose concrete route passes + `isNativeChatRouteEligible(targetRoute, envelope.freshBody(), config)` is dispatched through + `protocolSource.dispatchNativeChild` instead. That runs `runNativeChatAttempt` on the attempt the + combo already opened, with the combo's `targetSendBudget`, abort signal and turn lease. It + returns a Chat-wire `Response` marked with `markClientWire(response, "chat")`. +- A marked child response skips `preflightComboStreamResponse` (native Chat reports pre-stream + failures by status before any byte). Non-OK native responses go through the existing + `consumeComboFailure` path unchanged. +- `src/server/chat-completions.ts` returns a response whose `clientWireOf` is `chat` without the + Responses-to-Chat conversion, still wrapped by the deferred request log. +- Policy routes: migrate only if their child dispatch goes through the same combo loop; + otherwise record them as not migrated in [040](040_acceptance_and_rollout.md). +- `n > 1` is never emulated with multiple inferences. A candidate whose path cannot carry a + requested feature is skipped under `reject` policy with reason `feature-unrepresentable`. +- Each native child records its attempt path with `markAttemptProtocolPath` (PF-02) as native. + +## PF-08 managed native Messages + +Behind `protocols.rollout.managedMessagesNative`. + +| File | Role | +|---|---| +| `src/adapters/anthropic/passthrough.ts` | `buildAnthropicMessagesPassthroughRequest(provider, modelId, body, config)` → `{ url, headers, body }`. URL is the provider's Messages endpoint as the existing adapter (`src/adapters/anthropic.ts`) computes it; auth headers from the provider's key exactly as that adapter injects them; `anthropic-version` pinned as the adapter pins it. Body = the source Messages body with `model` replaced by the wire model and a field allowlist (`model, messages, system, max_tokens, metadata, stop_sequences, stream, temperature, top_p, top_k, tools, tool_choice, thinking, output_config, service_tier`). | +| `src/server/messages-native.ts` | `isNativeMessagesRouteEligible(route, body, config)` and `handleNativeMessages(...)`, modelled on the native Chat lane and built on PF-05 primitives: attempt, key failover and 429 replay (`src/providers/key-failover.ts`), spend reservation, SSE relay with the existing Anthropic log tap, JSON for non-streaming callers. | + +Eligibility: adapter `anthropic`, `authMode` key (OAuth is PF-10), not combo/policy, no vision +preprocessing required, no synthetic effort/fast row, switch on. + +Wiring in `src/server/claude-messages.ts`: after route settlement and after the managed-client +steps that already ran on the Anthropic body (alias/modelMap resolution, `ocx-route`, effort +directives), an eligible route goes native. The caller-forward passthrough (the caller's own +Anthropic credential) stays a separate branch with separate authority. +`handleClaudeCountTokens` counts the body the native lane would send when the route is eligible. + +## PF-09 direct client encoders + +Behind `protocols.rollout.directEncoders`. + +| File | Role | +|---|---| +| `src/protocols/encoders/chat.ts` | `encodeChatCompletionSse(events, opts)` and `foldChatCompletion(events, opts)` from `AdapterEvent` | +| `src/protocols/encoders/messages.ts` | `encodeAnthropicMessageSse(events, opts)` and `foldAnthropicMessage(events, opts)` from `AdapterEvent` | + +- `HandleResponsesOptions` gains `clientEncoder?: { protocol: "chat" \| "messages"; stream: boolean }`, + set by the two ingresses when the switch is on. +- In `src/server/responses/adapter-delivery.ts`, when `clientEncoder` is set, the guarded event + stream is encoded directly instead of through `bridgeToResponsesSSE`. Effects keep parity: the + events are also collected (charged to the translator budget) and, at the terminal, folded with + `buildResponseJSON` so `rememberResponseState`, `notifyResponseComplete`, + `commitReasoningReplayServingRoute` and the key-usage binding run exactly as before. +- The returned response is marked with `markClientWire`; the ingress passes it through. +- Encoders preserve chunk indexes, one role frame, finish/stop reasons, tool-call identity and + argument streaming, usage, error frames and cancellation. Passthrough (Responses upstream) + and native lanes are unaffected. +- The attempt's `responsePath` becomes `[upstream, "ir", client]`; the request path is still the + bridge until the codecs decode to IR directly, and the trace says so. + +## PF-10 auth and opaque state + +- `src/adapters/anthropic/beta-allowlist.ts`: the `anthropic-beta` values a managed native + Messages request may forward, per provider class (first-party vs Anthropic-compatible). Others + are dropped and recorded as a degraded feature, never forwarded blind. +- OAuth native Messages behind `managedMessagesNativeOAuth`: eligibility extends to Anthropic + OAuth accounts, credentials resolved through the existing OAuth account selection; the account + chosen is the one the router/pool already chose. No refresh or selection happens in planning. +- `src/protocols/opaque-state.ts`: thinking signatures and `redacted_thinking` blocks are + forwarded only to first-party Anthropic destinations on the native lane; for any other + destination they are removed from the fresh body and recorded as a degraded feature (blocked + under `reject`). A fallback to a different provider or credential domain rebuilds from the + envelope and applies the same rule. + +## Not migrated after this unit + +Recorded and maintained in [040](040_acceptance_and_rollout.md#not-migrated-inventory). diff --git a/devlog/_plan/260924_protocol_first_class/030_gui_and_management_api.md b/devlog/_plan/260924_protocol_first_class/030_gui_and_management_api.md new file mode 100644 index 0000000000..f512914ef6 --- /dev/null +++ b/devlog/_plan/260924_protocol_first_class/030_gui_and_management_api.md @@ -0,0 +1,143 @@ +# 030 — management API and dashboard (PF-02, PF-03, PF-04, PF-11) + +No new top-level page. Each existing screen answers one question: + +| Screen | Hash | Question | +|---|---|---| +| Integrations → API / Keys | `#integrations/keys` | How do I connect, and which path would this request take? | +| Providers → detail / settings | `#providers` | Which wire does this provider receive, and who decided that? | +| Models → Compatibility | `#models/compatibility` | Which protocol pair and feature is backed by which Lab evidence? | +| Models → Combos / Routing | `#models/combos`, `#models/routing` | What does each candidate do, and what do all candidates guarantee? | +| Logs | `#logs` | What did this request actually do, attempt by attempt? | + +The GUI never re-implements protocol policy. It imports the leaf contract +(`src/protocols/contract.ts`, `features.ts`, `dto.ts`) for vocabulary and validation, and gets +every decision from the server. + +## Management routes + +| Route | Packet | Mutates | CLI | +|---|---|---|---| +| `GET /api/protocols` | PF-03 | no | `ocx api protocols`, bare `ocx api policy` (PF-12) | +| `POST /api/protocols/plan` | PF-03 | no | `ocx api explain` (PF-12) | +| `PATCH /api/protocols/settings` | PF-04 | yes | `ocx api policy` with a setting flag (PF-12) | + +All three live in `src/server/management/protocol-routes.ts`, are mounted lazily from +`src/server/management-api.ts` under the `/api/protocols` namespace, and are declared in +`src/server/management/route-registry.ts`. They carried a `deferred-verb` exemption owned by PF-12 +until PF-12 declared the three `api` capabilities in `src/cli/capabilities.ts` +(`src/cli/api-protocols.ts`) and removed it. Existing `/api/providers`, `/api/logs`, `/api/request-history` and `/api/lab/*` are +reused, not duplicated. + +## PF-02 observed path trace + +Server: + +- `src/protocols/trace.ts` (server side, no GUI import): entry marks and attempt marks kept in + WeakMaps keyed by the request log context and attempt objects, so `RequestLogContext` does not + grow. `markProtocolEntry(logCtx, { inbound, lane, reasonCodes, features })` with lane + `native | bridge`; `markProtocolBlocked(logCtx, { inbound, reasonCodes, features })`; + `markAttemptProtocolPath(attempt, { mode, requestPath, responsePath? })`; + `protocolTraceForRequest(logCtx, attempts)` derives `ProtocolTraceV1` at finalize. +- Derivation without an explicit attempt mark: Responses inbound → `responses,responses` + (adapter `openai-responses`) or `responses,ir,<wire>`; Chat/Messages lane `native` → + `<in>,<in>`; lane `bridge` → `<in>,responses` for a Responses upstream, otherwise + `<in>,responses-internal,ir,<wire>`. No attempt and no native/blocked mark → no trace. +- Entry marks: `src/server/chat-completions.ts` (native vs bridge; features from the Chat body; + reason codes for why the native lane declined), `src/server/claude-messages.ts` (caller-forward + native passthrough, bridge, disabled surface and compatibility reject as blocked). The Responses + ingress needs no mark. +- `src/server/request-log.ts` is 4 lines under the 2000-line threshold. First move + `filterRequestLogs` and `filteredRequestLogCount` byte-for-byte into + `src/server/request-log-filter.ts` (re-exported from `request-log.ts`), then add + `protocolTrace?: ProtocolTraceV1` to `RequestLogEntry`, compute it in `addFinalRequestLog`, + persist it through the usage row (`src/usage/log.ts` `PersistedUsageEntry`, validated with + `parseProtocolTraceV1` on read) and hydrate it back in `requestLogEntryFromPersistedUsage`. +- `/api/logs` spreads the entry, so the DTO carries the trace with no route change. Add a + `protocolMode` query filter (`native | translated | legacy-bridge | blocked | none`) in + `request-log-filter.ts`. + +Dashboard: + +- `gui/src/components/protocols/ProtocolBadge.tsx`: compact `Chat → Chat` style path label with + the mode as text, not colour alone. Rows without a trace show nothing in the list. +- `gui/src/components/protocols/ProtocolTracePanel.tsx`: a section in the Logs detail dialog — + inbound, final mode, request/response path (internal Responses hop labelled as internal), + reason codes, feature effects, per-attempt paths. A row without a trace says "no path data" + instead of guessing. +- Logs filter: protocol mode, client-side in `gui/src/pages/logs-filter.ts` beside the existing + filters. + +## PF-03 planner and preview + +Server: + +- `src/protocols/plan.ts` (leaf, pure): `planProtocol(input): ProtocolPlanV1`. Input is a + snapshot: `inbound`, `requestedModel`, `routeKind`, `candidates[]` (provider, model, adapter, + `nativeEligible`, `declineReasons`), `features`, `surfaces`, `settings`, `policyRevision`, + `basis`. It computes each candidate's path with the same rules PF-02 uses for observed paths, + its feature effects, eligibility under the unrepresentable policy, and the + guaranteed/partial feature sets. A disabled surface yields `blocked` with `surface-disabled`. +- `src/protocols/plan-snapshot.ts` (server side): builds the snapshot from config without side + effects — `routeModel` (synchronous; no fetch, refresh or write), `captureRouteStaticPolicy`, + `resolveWireProtocolOverride` with the inbound's wire spelling, `routeConcreteModel` for each + combo target, and `isNativeChatRouteEligible` for Chat candidates. Messages caller-forward + passthrough depends on the caller's credential and is reported with + `caller-credential-required`, never assumed. Unknown models produce a plan with + `routeKind: "unknown"`, no candidates and `unknown-model`. +- `GET /api/protocols` → `{ schemaVersion: 1, contractVersion, policyRevision, surfaces, settings, features: PROTOCOL_FEATURES }`. +- `POST /api/protocols/plan` body `{ model: string, inbound: Protocol, features?: ProtocolFeature[] }` + (bounded: model ≤ 200 chars, at most 24 features, unknown keys rejected with 400) → + `ProtocolPlanV1` with `basis: "preview"`. Never reads a request body sample, never logs input. + +Dashboard (`#integrations/keys`): + +- `gui/src/protocol-api.ts`: fetch + validate with `isProtocolPlanV1`; cache key is + `apiBase + model + inbound + sorted features + policyRevision`; an older server that answers + 404 disables the panel quietly. +- `gui/src/components/protocols/ProtocolPlanPanel.tsx` and `FeatureDispositionList.tsx`: a + "Request path preview" section in `ApiKeysWorkspace` (new section anchor after the endpoints + section): model picker from the existing model list, inbound selector, feature toggles, and a + Preview button. It states that preview sends nothing and costs nothing, shows per-candidate + path, mode, fidelity, feature effects, reasons and the policy revision, and separates + "guaranteed by all candidates" from "some candidates only". +- The existing per-protocol "Test" button stays the explicit live test; its result keeps saying + that one success is a connection test, not verification. + +## PF-04 API surface settings + +- `resolveApiSurfaceSettings` becomes the only reader: `claudeInboundDisabled` in + `src/server/claude-messages.ts` (both `/v1/messages` and `/v1/messages/count_tokens`), + `buildApiAccessEndpoints` in `src/server/management/api-access.ts` (adds + `surfaces: { responses, chat, messages }` with `enabled` and `source`; keeps + `claudeCodeEnabled` for older dashboards), and the dashboard. +- `PATCH /api/protocols/settings` body `{ messagesEnabled?: boolean, unrepresentable?: "legacy" | "reject", rollout?: Partial<...> }` + through the existing config mutation path (locked, atomic). Disabling Messages writes + `apiSurfaces.messages.enabled = false` **and** `claudeCode.enabled = false` in the same save, so + an older binary after rollback cannot reopen the endpoint. Enabling writes only + `apiSurfaces.messages.enabled = true`. The `claudeCode` subtree is written through the same + helper the Claude settings route uses, respecting its hand-edit protection. +- Dashboard: the endpoints panel becomes three API cards (Responses, Chat Completions, Messages) + with state, endpoint, source ("explicit", "inherited from Claude settings", "invalid value — + closed") and, for Messages, a toggle plus a link to the Claude page. A disabled Messages card + stays visible. +- Upgrade/rollback matrix to record: absent → inherit; explicit false + old binary → closed; + explicit true + `claudeCode.enabled=false` + old binary → closed (safe direction). + +## PF-11 evidence, combo and provider views + +- `GET /api/protocols?provider=<name>` adds `provider: { name, adapter, adapterSource, authMode, upstream, modelOverrides: [{ model, adapter, source }] }` + from the provider's resolved static policy (`adapterSource` from `ResolvedModelPolicy` + provenance; `hard-pin | operator | registry | provider-default`). Bounded to 64 overrides. +- `gui/src/components/provider-workspace/ProviderProtocolPanel.tsx` in provider settings: + labels the adapter as "upstream wire this provider receives", shows the decision source and + model overrides, and saves only through the existing `onUpdateProvider` → `PATCH /api/providers`. + It never looks like an API exposure switch. +- Compatibility matrix: inbound and upstream protocol filters in + `gui/src/pages/compatibility-matrix-shared.ts` / `CompatibilityMatrix.tsx`, mapping Lab + identities with `protocolFromLabProtocol`. Absent Lab data reads "unverified", never "failed" + or "unsupported". +- Combo detail (`gui/src/components/combo-workspace-detail-panel.tsx`): per-candidate path and + the guaranteed/partial feature split from `POST /api/protocols/plan`. +- Deep links through the existing hash route helpers: plan panel → provider settings and + compatibility; Logs trace → compatibility for that pair. diff --git a/devlog/_plan/260924_protocol_first_class/040_acceptance_and_rollout.md b/devlog/_plan/260924_protocol_first_class/040_acceptance_and_rollout.md new file mode 100644 index 0000000000..d4615fb503 --- /dev/null +++ b/devlog/_plan/260924_protocol_first_class/040_acceptance_and_rollout.md @@ -0,0 +1,70 @@ +# 040 — acceptance and rollout (PF-12) + +## Rollout order + +1. Existing execution is the default. Every `protocols.rollout.*` switch is off. +2. `shadowPlan` on: the Chat and Messages ingresses record the dispatch-basis plan input at their + entry mark (`src/protocols/shadow-plan.ts`); at finalize the pure planner runs on it and the + plan for the settled route is compared with the observed trace (`src/protocols/shadow.ts`); a + disagreement sets `planMismatch: true` on the trace. No second request is ever sent. +3. Per-target opt-in: `nativeChatCombos`, `managedMessagesNative`, `directEncoders`, then + `managedMessagesNativeOAuth`. +4. A switch's default flips only after its scenarios below have recorded evidence on a merged + head, in a separate reviewed change. +5. `unrepresentable: "reject"` is an operator policy, not a rollout step; it stays opt-in. + +## Acceptance scenarios + +Status vocabulary, per row: **implemented behind switch** (code on this stack, reachable only with +the named switch on), **default path** (what runs with every switch off, unchanged by this unit), +**not implemented** (the target shape does not exist yet), **test written (not run)** (unit tests +on this stack name the behavior; none were executed by the packet that wrote them unless its PR +says so), **pending live evidence** (no fixture or live run has been recorded against a merged +head). No row has recorded evidence. Test files are named so a reviewer can run them; naming a +file is not a claim that it passed. + +| Scenario | Accepted when | Status | +|---|---|---| +| Chat → Chat, `n=2` / `logprobs` | every choice and logprobs survive; direct and combo give the same upstream body; every choice terminal is honoured | direct: default path (native Chat lane). Combo: implemented behind `nativeChatCombos`. Test written (not run): `tests/responses/chat-native-combo.test.ts`. Pending live evidence | +| Chat → Responses, `n=2` | under `reject`, refused before any send with `unsupported_feature`; never reduced to one choice and never emulated with extra calls | implemented behind `unrepresentable: "reject"` for direct routes; combos only with `nativeChatCombos` on. Test written (not run): `tests/responses/protocol-ingress-guard.test.ts`, `tests/responses/protocol-guard.test.ts`. Pending live evidence | +| Messages → Messages (managed key) | source blocks and declared options (`top_k`, `cache_control`, `thinking`) survive; caller-forward, managed key and OAuth stay separate authority cases | managed key: implemented behind `managedMessagesNative`. OAuth: not implemented on this stack (PF-10). Test written (not run): `tests/claude-integration/messages-native.test.ts`, `tests/adapters/anthropic/anthropic-messages-passthrough.test.ts`, `tests/responses/messages-native-eligibility.test.ts`. Pending live evidence | +| Messages → Chat | role order and tool pairing preserved; unsupported fields follow policy | default path only (legacy bridge through the existing translators); the direct `messages,ir,chat` codec is not implemented. Pending live evidence | +| Chat → Messages | function definitions/results map to content blocks; stop reason and usage mapped | default path only (legacy bridge); the direct `chat,ir,messages` codec is not implemented. Response side: implemented behind `directEncoders`, test written (not run): `tests/responses/protocol-direct-encoders-messages.test.ts`. Pending live evidence | +| Responses → Chat / Messages | continuation and compaction unchanged | default path; this unit does not change the Responses ingress. Pending live evidence | +| Mixed combo failover | every attempt built from the source envelope; send budget shared; affinity kept; ineligible candidates skipped with a recorded reason; no resend after partial stream output | implemented behind `nativeChatCombos` (Chat only). Known gap: a native child's zero-output in-band failure does not hop (inventory below). Test written (not run): `tests/responses/chat-native-combo.test.ts`. Pending live evidence | +| `stream: false` / `true` | correct envelope, error frames, terminal, chunk boundaries, backpressure, cancellation, timeout, memory budget | native lanes: default path (Chat) and behind `managedMessagesNative` (Messages). Encoders: implemented behind `directEncoders`. Test written (not run): `tests/responses/protocol-direct-encoders-chat.test.ts`, `tests/responses/protocol-direct-encoders-messages.test.ts`, `tests/server/inference-client-encoder-delivery.test.ts`. Pending live evidence | +| Unknown extension or media | never silently dropped on a translated path without a recorded effect | declared features only: effects recorded on the trace (default path) and refused under `reject`. Undeclared fields have no feature name and record no effect (managed Messages allowlist). Test written (not run): `tests/responses/protocol-trace.test.ts`, `tests/responses/protocol-features.test.ts`. Pending live evidence | +| Dashboard and remote runtime | stale/unknown/unsupported distinguished; per-request trace; policy-revision mismatch visible; hash state survives Back/Forward | per-request trace, preview with its policy revision, and deep links: implemented (PF-02, PF-03, PF-11). `planMismatch` is recorded but not rendered by the dashboard. Remote runtime: not exercised. Test written (not run): `tests/usage/request-log-protocol-trace.test.ts` and the PF-11 GUI tests. Pending live evidence | +| API disable migration | `/v1/messages` and `count_tokens` agree; upgrade, old-UI writes and rollback never reopen a closed surface | implemented (PF-04, default path). Test written (not run): `tests/claude-integration/messages-surface-matrix.test.ts`, `tests/server/protocol-settings-route.test.ts`. Pending live evidence | +| Shadow plan | disagreement between the dispatch plan and the trace is marked; no second request; switch off changes nothing; a failing comparison never affects the log row | implemented behind `shadowPlan` for the Chat and Messages ingresses. Test written (not run): `tests/responses/protocol-shadow-plan.test.ts`. Pending live evidence | +| CLI parity | every protocol management route has a CLI verb; `ocx api policy` writes only when a setting flag is given | implemented (`ocx api protocols`, `explain`, `policy`). Test written (not run): `tests/cli/cli-api-protocols.test.ts`, `tests/cli/cli-capabilities.test.ts` | + +Verification runs in isolated fixtures with no access to a user's home, credentials or services. +Live provider probes happen only with a consenting operator's keys and budget and are recorded +separately from fixture results. + +## Not-migrated inventory + +Kept current by each packet that migrates something. Checked against the code on the PF-12 +branch (`feat/pf12-protocol-rollout`); PF-10 is developed in parallel and is not on this stack. + +| Path | State after this unit | +|---|---| +| Chat/Messages request decode | still produces a Responses-shaped body before the IR (`responses-internal`) on every non-native path; `src/protocols/codecs/*` are named entry points over the existing translators (`chatCompletionsToResponsesBody`, `anthropicToResponsesTranslation`), not direct codecs | +| Chat ↔ Messages direct codecs | not implemented: `chat,ir,messages` and `messages,ir,chat` are baseline targets only; both pairs travel `responses-internal` | +| Chat/Messages response encode (PF-09) | migrated behind `directEncoders` only where `directEncodersApply` and `clientEncoderForDelivery` both agree: one concrete, non-Responses-wire route in the streaming adapter delivery, giving response path `[upstream, ir, client]` while the request path stays the bridge path. Still through `responses-internal`: combo and policy children (`comboAttempt`, `routeKind` combo/policy), routed compaction, run-turn adapters (Cursor, Devin, coding-agent CLIs, CodeBuddy), sidecar turns, the buffered `parseResponse` branch (unused by these ingresses, which always stream internally), and every route while the switch is off. Responses-wire upstreams keep their existing codec path | +| Policy-route children | not migrated (PF-07): `routeModel` evaluates the policy and returns one concrete candidate, so a policy request never reaches the combo child loop and keeps the Chat bridge | +| Chat combos with `nativeChatCombos` off | bridge for every candidate, and not judged per candidate under `reject` (the ingress guard also skips combos) | +| Chat combos reached through an effort row | bridge (PF-07): the row's effort lives only on the Responses body, so the native source is not supplied | +| Native Chat combo child, streamed, zero-output in-band failure | no hop (PF-07): the child's 200 is committed without `preflightComboStreamResponse`, so a failure frame before any output reaches the client instead of the next target; the bridge child would have hopped | +| Messages combos and policies | bridge: `nativeMessagesDeclineReason` returns `combo-or-policy-route`; not judged per candidate under `reject` | +| Sidecars (web search, vision, image generation) | Responses pipeline only | +| Responses-only features on Chat/Messages | `previous_response_id`, `store`, `background`, compaction stay on the bridge | +| Non-public-wire adapters (`other`) | translated through the IR; no feature claims | +| OAuth native Chat | not planned in this unit | +| Messages → key-auth Anthropic | native behind `managedMessagesNative` (PF-08); bridge while off. Caller `anthropic-beta` passes only through the PF-10 allowlist (`interleaved-thinking-2025-05-14` to `api.anthropic.com`, nothing to a compatible host); a dropped value is traced as `anthropic-beta-dropped`, never by value. Top-level fields outside the allowlist are dropped with no feature effect | +| Messages → Anthropic OAuth | native behind `managedMessagesNativeOAuth` (PF-10) for the unpooled `anthropic` provider on `api.anthropic.com`; bridge while off | +| Messages → pooled Anthropic OAuth | not migrated (PF-10): `anthropicAccountPool.enabled` or two usable stored accounts decline with `oauth-account-pool`, because rotation, session affinity and quota ranking live in the Responses transport | +| Opaque thinking state on the native lane | signatures and `redacted_thinking` reach `api.anthropic.com` only; elsewhere removed and traced as `opaque-state-stripped` (reason code, not a feature effect: a same-wire hop has no degraded disposition), refused before any send under `reject` | +| Messages native lane, translated-only steps | a pinned route effort, blocked-skill elision, the web-search sidecar and vision preprocessing keep the request on the bridge (`bridge-only-policy` / `vision-preprocessing`); `stabilizePromptCache` is a recorded gap — the native lane does not apply it | +| Shadow plan coverage | Chat and Messages ingresses only; the Responses ingress records no shadow input. The response path is not compared (the planner does not model `directEncoders`); caller-forward Messages passthrough and compatibility rejects are not compared. The planner cannot see body-dependent Messages decline rules (skill elision, web search), so those surface as `planMismatch` rather than being predicted. The Claude fast-selector decode used by the ingress is not applied by the snapshot. The dashboard does not render `planMismatch` | diff --git a/devlog/_plan/260924_regression_risk_fixes/000_overview.md b/devlog/_plan/260924_regression_risk_fixes/000_overview.md new file mode 100644 index 0000000000..f5ce3c3fd8 --- /dev/null +++ b/devlog/_plan/260924_regression_risk_fixes/000_overview.md @@ -0,0 +1,14 @@ +# 260924 regression-risk fixes + +The 260923 bundle round landed eight lane PRs (#5672-#5685). A post-merge review found four behaviour changes that can silently hurt existing users. The owner asked to fix them directly, merge to dev without waiting for PR CI, and then drive ci.yml lane=all on the dev tip to success. + +- 010_wsl_home.md — WSL Codex home switch (#5441 carry). +- 020_windows_mise_node.md — Windows npm global under mise-managed Node refused (#5316 carry). +- 030_echo_filter.md — tool-envelope echo filter truncates ordinary answers (#5098 carry). +- 040_tool_call_hold.md — unbounded hold of an unmatched bare <tool_call> block (#5548 carry). +- 050_delivery.md — branch, merge and dev CI. + +Accepted and out of scope: Claude Code <=2.1.222 picker filter (/^(claude|anthropic)/i) drops ocx-claude-* ids; other residual risks are documented in the round's landing log only. + + +Cycle note: wp1's first close attempt failed on a malformed receipt command and the cycle was re-walked with the same artifacts. diff --git a/devlog/_plan/260924_regression_risk_fixes/010_wsl_home.md b/devlog/_plan/260924_regression_risk_fixes/010_wsl_home.md new file mode 100644 index 0000000000..09b49b6f7a --- /dev/null +++ b/devlog/_plan/260924_regression_risk_fixes/010_wsl_home.md @@ -0,0 +1,20 @@ +# WSL Codex home + +Defect: src/codex/home.ts defaultCodexHome now returns the local ~/.codex whenever it is a directory. Before #5441 a local home without config.toml let WSL discovery pick the Windows Codex home. A WSL user whose ~/.codex exists but holds no Codex state, and who ran against the Windows home, is moved to an empty local home on upgrade: auth and sessions disappear and sync writes a new local config. + +#5441's case is a fresh local install that Codex itself is using before config.toml exists. Codex writes auth.json on login and sessions/ plus history.jsonl on first use. + +Change (src/codex/home.ts): + +- keep localCodexHomeIsDirectory (stat, ENOENT/ENOTDIR = absent, other errors = present). +- add localCodexHomeInUse(home, deps): true when any of config.toml, auth.json, sessions, history.jsonl is present by the same stat rule (unexpected stat error counts as present, never switch on doubt). +- defaultCodexHome: not a directory -> discovery ?? local (unchanged); directory and in use -> local; directory with no Codex state -> findWslWindowsCodexHome ?? local (the pre-#5441 behaviour). + +Tests (new tests/codex-integration/codex-home-wsl-local-state.test.ts, registered in layout.json and test-layout-expected.json): bare local dir + Windows home -> Windows; local dir with auth.json -> local; local dir with sessions -> local; unexpected stat error on markers -> local; non-WSL unaffected. The existing codex-home-wsl.test.ts fresh-home case keeps passing (its statSync mock reports every path present). + +Docs: structure/codex-home.md and the Codex integration guide sentence that says directory presence decides. + + +## Build note + +A local ~/.codex that exists but is empty, with a discoverable Windows home, resolves to the Windows home (the pre-#5441 behaviour). That is the accepted direction: moving an existing user off the Windows home loses their auth and sessions, while a fresh user who has not run Codex locally yet loses nothing and CODEX_HOME overrides. The existing fresh-home test now models a fresh install honestly (auth.json present, config.toml absent) instead of a mock that reported every path present. diff --git a/devlog/_plan/260924_regression_risk_fixes/020_windows_mise_node.md b/devlog/_plan/260924_regression_risk_fixes/020_windows_mise_node.md new file mode 100644 index 0000000000..2b4ef6e5a0 --- /dev/null +++ b/devlog/_plan/260924_regression_risk_fixes/020_windows_mise_node.md @@ -0,0 +1,19 @@ +# Windows npm global under mise-managed Node + +Defect: src/update/install-detection.mjs detectMiseOwner walks /node_modules/ markers and reads <toolRoot>/.mise.backend.toml. On Windows, npm -g under a mise-managed Node installs to <mise>/installs/node/<ver>/node_modules/@bitkyc08/opencodex, so toolRoot is the Node tool root and its metadata (short = "node", full = "core:node") is read as contradictory OpenCodex ownership: ocx update is refused with metadata_inconsistent. POSIX is unaffected (lib/node_modules puts toolRoot one level deeper, where no metadata exists). + +Change: after parsing, if the metadata names a Node runtime (short is node or nodejs) and its backend is not an npm: backend, the package sits in that runtime's global node_modules and mise did not install OpenCodex: return { recognized: false }, so ordinary npm detection applies. Every other mismatch, including npm:some-other-package, stays metadata_inconsistent (fail closed). + +Tests (new tests/update/update-mise-node-runtime.test.ts): Windows and POSIX-shaped paths under a core:node tool root detect as npm with no mise error; short = "node" with an npm: backend stays inconsistent; the existing contradictory-metadata tests keep passing. + +Docs: structure owner of src/update (install detection section). + + +## Audit fold (round 1 FAIL) + +The exemption is narrowed: only metadata that is exactly the mise core Node runtime (short = "node", full = "core:node") AND a package path that is that tool's <version>/node_modules/<package> (the Windows npm global layout, installPath directly under toolRoot) is classified as not mise-owned. Every other backend for short node/nodejs, unreadable metadata and every OpenCodex mismatch stay fail-closed. + + +## Residual + +detectMiseOwner treats a backslash UNC path (\\server\share) as Windows but a slash-form //server/share path as POSIX, as before this change; on such a path a case difference in the node tool directory misses the exemption and keeps the old metadata_inconsistent refusal. diff --git a/devlog/_plan/260924_regression_risk_fixes/030_echo_filter.md b/devlog/_plan/260924_regression_risk_fixes/030_echo_filter.md new file mode 100644 index 0000000000..d5142c4e03 --- /dev/null +++ b/devlog/_plan/260924_regression_risk_fixes/030_echo_filter.md @@ -0,0 +1,30 @@ +# Tool-envelope echo filter + +Defect: src/lib/tool-envelope-echo-filter.ts ToolEnvelopeEchoFilter.feed matches as soon as a line starts with a marker (probe === marker) and drops that line and everything after it. The filter arms on almost every Codex agentic turn (input carrying tool calls/outputs or previous_response_id), so a legitimate line such as "[Tool Result] shows the build passed." silently truncates the answer. The replayed envelope OpenCodex builds is always a marker alone on its line ("[Tool Result]\n<payload>", protobuf-request.ts), and the replay-side stripper in cursor/envelope-echo.ts already requires whole-line markers. + +Change (tool-envelope-echo-filter.ts only): + +- feed: a line that equals a marker is kept pending (candidate) instead of matching immediately; trailing spaces/CR after a marker stay candidates; a "[Tool call:" line stays pending up to a bounded length (MAX_CALL_LINE = 4096) and is released as prose beyond it. +- completeLine decides: whole-line marker (exact after trimEnd) or a "[Tool call:" line ending with "]" is an echo (outside a fence: match; inside: hold as today). Anything else is prose. +- finish applies the same whole-line rule to an unterminated last line; a truncated marker prefix ("[Tool Res") or an unterminated "[Tool call:" line at stream end still counts as an echo, as today. + +Tests (new tests/lib/tool-envelope-echo-whole-line.test.ts): prose starting with a marker survives whole and char-by-char; marker alone mid-answer still drops the tail; "[Tool call: x (call_id: 1) with args: {}]" line drops; final unterminated "[Tool Result] header text" survives; existing passthrough-grok-upstream-envelope-echo and cursor-envelope-echo-retry tests keep passing. + + +## Audit fold (round 1 FAIL) + +Two Cursor paths share the marker vocabulary: + +- The replay-side stripper isEchoMarkerLine (src/adapters/cursor/envelope-echo.ts) was whole-line at 685321e297 (exact marker) and #5676 widened it to prefix matching, so assistant history prose such as "[Tool Result] shows ..." is now stripped with its following non-blank lines. Restore whole-line semantics: trimmed line equals a marker (with or without the closing bracket for the truncated forms), or a "[Tool call:" line that ends with "]". Test it. +- The first-bytes prefix sniffer CursorEnvelopeEchoSniffer is unchanged since before the round (probe.startsWith(marker) at 685321e297) and triggers a remint retry rather than truncation. Not a regression of the round; left as is and recorded. + +finish() replaces the startsWith(truncated marker) rule with: exact truncated prefix of a marker, whole-line marker, or an unterminated "[Tool call:" line. Tests cover CRLF, trailing whitespace, fenced markers, streaming char-by-char and the buffered Responses JSON path (stripGrokUpstreamEnvelopeEchoFromResponsesJson). + + +## Audit round 2 rebuttal + +The Cursor first-bytes sniffer stays out of scope: it predates the round (685321e297) and its false positive costs a remint retry, not a truncated answer. Narrowing a pre-existing echo guard is a separate decision; recorded as a residual. + +## Build note + +The "[Tool call:" line keeps its prefix rule instead of the planned completed-line rule. An existing Cursor replay test showed why: a call echo wraps when its arguments do ("[Tool call: Glob" then "args"), so a completed-line rule leaks it. Prose that opens with "[Tool call:" is rare, while "[Tool Result] ..." prose is common, so only the result/error markers move to the whole-line rule. The diff review flagged the prefix rule; this is the recorded reason for keeping it. diff --git a/devlog/_plan/260924_regression_risk_fixes/040_tool_call_hold.md b/devlog/_plan/260924_regression_risk_fixes/040_tool_call_hold.md new file mode 100644 index 0000000000..9432cc1da2 --- /dev/null +++ b/devlog/_plan/260924_regression_risk_fixes/040_tool_call_hold.md @@ -0,0 +1,23 @@ +# Bound the serialized tool-call hold + +Defect: src/adapters/openai-chat/serialized-tool-call-content.ts SerializedToolCallContentBuffer.ingest appends everything after a bare <tool_call><function=...> opening to the held text until the stream ends. An unmatched block followed by a long answer is delivered only at the end of the turn and can approach the translator budget. + +Change: + +- MAX_HELD_CHARS = 64 KiB: when the held text exceeds it, the buffer releases everything it holds (text and queued events, in order, nothing suppressed) and resumes scanning from the carried context. +- MAX_TRAILING_CHARS = 4 KiB: once the held text contains a closed </tool_call> and the non-whitespace text after the last closer exceeds this, release as above: a duplicated block is the tail of the content, not the head of a long answer. +- The adapter (openai-chat.ts emitContent) drains the released events through a new buffer method so ordering stays intact; no other adapter change (openai-chat.ts is ratchet-capped, so the change stays inside the helper plus a one-line call). + +Tests (new tests/adapters/openai-chat-serialized-tool-call-hold-bound.test.ts): a bare block followed by >4 KiB of prose emits text before the stream ends and suppresses nothing; a block exceeding 64 KiB is released; a real duplicate (block then matching structured call) is still suppressed. + + +## Audit fold (round 1 FAIL) + +Streaming and buffered paths are separated. ingest() keeps its current unbounded behaviour because reconcileSerializedToolCallEvents (buffered responses, structured calls already known) uses it. A new ingestEvents(delta) is used only by the streaming emitContent in openai-chat.ts (two lines replaced by two lines, inside the 822-line cap). + +Overflow policy, stated: past the bound the stream prefers showing text over suppressing a possible duplicate. A duplicate block larger than the bound, or one followed by more than MAX_TRAILING_CHARS of prose before its structured call, reaches the client as raw markup, which is the behaviour before #5548. Bounds: MAX_HELD_BYTES = 64 KiB measured on this.bytes (held text plus queued events) and checked BEFORE retaining the next delta, so one large delta cannot push retention past the bound; MAX_TRAILING_CHARS = 8 KiB of non-whitespace text after the last closed </tool_call>. + + +## Audit fold (round 2) + +MAX_HELD_BYTES rises to 4 MiB: it is a runaway guard only, well under the translator budget, so a large duplicated write inside the block itself is still reconciled. The latency case is handled by the 8 KiB trailing-prose rule alone. A duplicated block is the tail of the content (#5548's observed shape), so a closed block followed by more than 8 KiB of prose before any structured call is treated as prose; that narrow residual is accepted. diff --git a/devlog/_plan/260924_regression_risk_fixes/050_delivery.md b/devlog/_plan/260924_regression_risk_fixes/050_delivery.md new file mode 100644 index 0000000000..d4b1012dee --- /dev/null +++ b/devlog/_plan/260924_regression_risk_fixes/050_delivery.md @@ -0,0 +1,4 @@ +# Delivery + +One branch codex/260924-regression-risk-fixes from origin/dev with one commit per fix plus this plan. Local gate: bun run typecheck, bun run structure:check, focused tests for each touched area, test-layout and file-size-ratchet tests. Open one PR to dev with the repository template and merge it immediately with gh pr merge --admin --squash (owner instruction: no PR CI wait). Then dispatch gh workflow run ci.yml --ref dev -F lane=all on the merged tip and require every job completed success; a multi-file timeout is rerun once, a repeat is a defect fixed by another PR merged the same way, followed by a fresh lane=all on the new tip. + diff --git a/devlog/_plan/260924_update_indicator/000_plan.md b/devlog/_plan/260924_update_indicator/000_plan.md new file mode 100644 index 0000000000..97afd57d38 --- /dev/null +++ b/devlog/_plan/260924_update_indicator/000_plan.md @@ -0,0 +1,110 @@ +# 260924 update indicator + +Service installs of opencodex never learn that a new version exists: the sidebar's blue update orb reads `~/.opencodex/version.json`, and that cache is refreshed only by an interactive `ocx start`. The desktop app has the opposite gap: its Tauri updater knows about updates, but the GUI inside the app asks the compiled sidecar, which classifies itself as a source build and never lights the orb. This unit makes the package cache refresh itself inside the running proxy without blocking requests, lets the desktop app publish its own updater state to the GUI it hosts, and puts a blue dot on the tray icon itself on the macOS menu bar, the Windows and Linux Tauri trays, and the npm Windows PowerShell tray. Users on any install see one consistent signal, and in the desktop app the update button installs the app update instead of the npm package. + +Research input: [001_sol_plan_draft.md](001_sol_plan_draft.md) (first planning draft plus coordinator review). External survey of Codex, Claude Code, gemini-cli, Ollama, cloudflared, Tauri apps and UI badges was done with Aside and is summarised in the PR, not copied here. + +## Loop spec + +- **Loop archetype:** satisfy-spec, multi-cycle HOTL (one work-phase per PABCD cycle). +- **Trigger:** the owner asked for tray and menu-bar blue update indicators on top of the update-signalling fixes, planned by sol and shipped as one PR ("그 가정으로 cxc-loop 해서 pr 날려놔"). +- **Goal:** the four behaviours in the work-phase map below, landed as ordered commits on `codex/update-indicator` with one PR to `dev`. +- **Non-goals:** automatic installation policy; npx, Volta, Yarn and Homebrew install detection; a light/dark redesign of the Windows Tauri base tray glyph; merging, releasing or deploying; unrelated refactors. +- **Verifier:** focused Bun tests named in each decade doc, `bun run typecheck`, `bun run structure:check`, `bun run privacy:scan`, `cd gui && bun run lint && bun run build` for GUI commits, `cargo test --manifest-path desktop/src-tauri/Cargo.toml` for desktop commits, then exact-head hosted CI on the PR. Visual tray rendering is human review (see Stop condition). +- **Stop condition:** PR open against `dev` with every template section filled and exact-head CI inspected, branch-caused failures fixed. macOS menu-bar rendering is checked locally; Windows and Linux tray rendering cannot be observed on this Mac and is reported as NEEDS_HUMAN, not claimed. +- **Memory artifact:** this unit directory, the codexclaw goalplan `ship-end-to-end-update-available-signalling-for`, and the PR description. +- **Expected terminal outcomes:** DONE (PR open, gates green or branch-caused failures fixed); BLOCKED (push or PR creation rejected, CI unavailable after investigation); UNSAFE (a change would weaken management auth or updater signature verification); NEEDS_HUMAN (cross-platform visual QA). +- **Escalation condition:** any change to management authentication semantics, updater signature checks, or the Lab boundary files stops for the owner. A worker that fails the same packet twice is reclaimed by main after its work is stopped (DISPATCH-RETIRE-01); moving a slice to a worker mid-B needs a P amendment. +- **Resource bounds:** tools are the local shell, git, gh (push and PR creation authorised, merge not), Bun and Cargo; write scope is this worktree; gpt-6-sol subagents are authorised without a count limit; no token or wall-clock budget was set by the owner. + +## Work-phase map + +Dependency order follows the build order: the package cache is the foundation the badge and the npm tray read; the desktop snapshot and icons need the badge route shape from commit 1; the desktop update page consumes the desktop state from commit 2; the npm Windows tray only needs commit 1 and is last because it is independent. + +| Work-phase | Commit | Decade doc | +| --- | --- | --- | +| wp0 | none (this roadmap) | 000_plan.md | +| wp1 | 1. package cache freshness and async checks | [010_phase1_package_cache.md](010_phase1_package_cache.md) | +| wp2 | 2. desktop update snapshot and tray blue dot | [020_phase2_desktop_state_icons.md](020_phase2_desktop_state_icons.md) | +| wp3 | 3. desktop GUI update page | [030_phase3_desktop_update_page.md](030_phase3_desktop_update_page.md) | +| wp4 | 4. npm Windows tray indicator | [040_phase4_windows_npm_tray.md](040_phase4_windows_npm_tray.md) | +| wp5 | none (verification, push, PR, CI) | recorded in 050_delivery.md at wp5 | + +## Decisions + +Architect proposal from McClintock (gpt-6-sol, read-only, agent 01a0d34a-4b14-7aa1-aca3-b42310f6b3d3), decisions D1-D7. Main dispositions: + +| ID | Proposal | Disposition | +| --- | --- | --- | +| D1 macOS | Keep the template glyph; draw the dot in a transparent NSView on the NSStatusBarButton via the Swift bridge (app/Sources/NativeTray), positioned against the image, removed when cleared. tray-icon 0.24.2 `set_icon` does not remove subviews. | **Accepted.** Keeps system tinting and highlight correct without guessing the menu-bar appearance. | +| D2 Windows/Linux | Swap PNG variants with a haloed blue dot; also add light/dark Windows base glyphs. | **Amended.** Dotted variants accepted. The light/dark base glyph is out of scope (owner-fixed non-goal); recorded as a follow-up. | +| D3 snapshot | `POST /api/update/desktop-snapshot`, admin-token principal only, bounded scalar payload keyed by a random desktop session id, in-memory with bounded entries, 60 s heartbeat, ~3 min expiry; `GET /api/update/badge?surface=desktop&session=<id>` returns `installer: "desktop"` or `unknown: true`. Session id travels in the embedded dashboard URL. | **Accepted.** No installation authority, no persistence, no logging of the session id. | +| D4 page | Bundled `desktop/ui/update.html` on the app origin with narrow Tauri commands; tray and page share one install function with a compare-and-swap claim. | **Accepted.** No updater IPC for the loopback dashboard. | +| D5 refresh | Scheduler worker in `src/update/`, tiny start/stop hook in `src/server/index.ts`; async `child_process.spawn` through `registrySpawnTarget`, 12 s kill, bounded output; write-through; coalesce per channel; backoff; 40 h max cache age reports unknown; skipped for source and mise installs; `/api/update/run` awaits the async check and passes the result into `startUpdateJob`. | **Accepted.** Keeps the Lab boundary and the synchronous `startServer` window intact. | +| D6 npm tray | Hidden read-only CLI command printing `readUpdateBadge()` JSON, probed every 60 s, bounded and non-overlapping; six icons (online, warning, offline × normal, dotted); update menu item opens the dashboard. | **Accepted.** Probe cost on Windows is measured in CI only; cadence stays 60 s with a timeout. | +| D7 commits | Four ordered commits with the file scopes and docs owners listed. | **Accepted.** | + +Reflection by the same architect: recorded below after the decade docs are written. + +## Source-of-truth sync + +- `structure/runtime.md` and `structure/ops/service-and-sidecars.md` own `src/update/` and `src/tray/` (commits 1 and 4). +- `structure/gui-and-management-api.md` owns the management routes (commits 1-3). +- `structure/desktop-shell.md` and `structure/companion.md` own `desktop/` (commits 2-3). +- docs-site: `reference/management-api.md`, `guides/desktop-app.md`, `guides/macos-menu-bar.md`, `guides/web-dashboard.md`, `reference/cli/lifecycle.md`, each with its existing locale siblings. + +### Main decisions raised by decade writers + +- 010: no update-check opt-out exists today. **Accepted** `OCX_DISABLE_UPDATE_CHECK=1` as a documented opt-out for automatic checks (the scheduler); explicit `/api/update/check` still runs. The Bun test preload must set it so test-started servers never spawn a registry lookup. +- 040: the PowerShell source assertion stays in `tests/windows/tray-proxy.test.ts` next to the existing probe assertions. +- 020: `structure/runtime.md` is at its 600-line budget. **Amendment for 010 and 040:** runtime.md edits must be net-zero (replace sentences, do not add); the new scheduler/probe prose goes to `structure/ops/service-and-sidecars.md`. Each B re-checks the structure line budgets with `bun run structure:check` before committing. +- 020: `cargo test` needs the sidecar placeholder `desktop/src-tauri/binaries/ocx-<target-triple>`. wp2 B reuses whatever CI uses to satisfy `externalBin` for Rust tests (checked at wp2 P), and never commits a built binary. +- 030 cites 020 line numbers that moved while 020 was written; symbol names are authoritative, line numbers are refreshed at each cycle's P. + +## Architect reflection + +McClintock (same agent) returned **MISALIGNED** with eight material and three minor gaps against revision 1 of the decade docs. Main accepted all eleven and assigned the fixes: 010 gets net-zero runtime.md edits, explicit-check write-through independent of a joined automatic flight (join-then-stop regression), a non-source injected route regression proving the event loop stays responsive, and an explicit one-server-per-process scheduler invariant; 020 separates heartbeat receipt age from `checkedAtMs`, adds a native check generation so only the latest completed check applies, and limits the post-title redraw to macOS; 030 refreshes phase-2 anchors by heading and consumes the phase-2 check generation; 040 makes the general-suite icon test renderer-independent (structural checks; byte-exact generator `--check` stays a local/desktop tool), caps probe output bytes, replaces the synchronous `WaitForExit` on the UI tick with a request-and-reap across ticks, and makes runtime.md edits net-zero. Revision 2 is sent back to the same architect before audit. + +Re-reflection on revision 2: all eleven gaps CLOSED; two new material gaps in the desktop check/install flow (install claim outside the check-application critical section; a superseded page check reporting "up to date"). Main accepted both; one writer revises 020 and 030 together into revision 3. + +Re-reflection on revision 3: both gaps CLOSED; one new material deadlock risk (tray setters that wait on the AppKit thread were called while the gate mutex was held, and a synchronous status command could wait on that mutex from the main thread). Main accepted it; revision 4 moves UI projection outside the gate with a generation re-check and makes gate-reading commands async. + +Re-reflection on revision 4: deadlock gap and both rev-3 gaps CLOSED; one new material gap (a no-pending install click committed the install flag and epoch before discovering there was nothing to install, orphaning an in-flight check). Main fixed it directly in 030 (revision 5): `claim_install` returns `InstallClaim::{Claimed,Busy,NoPending}`, reads the pending version under the gate before committing anything, and a new regression `install_click_without_pending_leaves_in_flight_check_valid` covers it; the tray install arm uses the same call. + +Re-reflection on revision 5: **ALIGNED** (McClintock). No material gap remains; all reported gaps are closed with regressions named in the decade docs. This completes architect consultation for wp0; the independent A audit follows. + +## Audit + +Independent reviewer Bohr (gpt-6-sol, agent 01a0d39a-32c8-7553-8d80-9b2fff2ef419). Round 1: FAIL with three High test-specification blockers (inherited commit-2 claim_install test breaks in commit 3; registry child failure paths untested; Windows stale-dot expiry untested) and two non-blocking notes; all accepted and folded (030 by main, 010 and 040 by fixers, 020 note by main). Round 2: FAIL with one High blocker (the new failure tests expected the retry timer after eight microtask turns; Bohr measured sixteen); main added a bounded `settle()` fixture wait in 010. +Round 3: **PASS** — the reviewer re-executed the planned scheduler and lookup in memory with the settle helper; all five failure scenarios reach the retry timer and recover. No High or Critical blocker remains. + +## wp1 Check review + +Implementation reviewer Noether (gpt-6-sol, agent 01a0d3be-6826-71c1-9ceb-331b2db6b176) on commit 1: +- **Medium, accepted and fixed:** an automatic tick after stop/start joined the stopped listener's pending flight, whose write the old generation suppressed, pushing the next check an hour out. Flights now carry their epoch and automatic ticks only join a same-generation flight; regression `restart does not join the stopped listener's pending lookup` fails on the pre-fix scheduler and passes after (red-green verified). Folded into commit 1. +- **High, rebutted for this unit:** on Windows, `POST /api/update/run` still reaches `spawnGuiUpdateWorker`'s 15-second-bounded `spawnSync` of PowerShell `Start-Process` (src/update/job.ts). It predates this unit, runs once per user-initiated install immediately before the proxy restarts, and 010 scoped the detached installer worker OUT. Making it asynchronous changes `startUpdateJob`'s lock-then-launch contract in a file with seven lines of headroom. Recorded as a follow-up in the PR; this unit's claim is limited to the registry check path. +- Round 2 (Noether): rebuttal accepted; GO-WITH-FIXES with 0 blockers. Medium (an older explicit flight could overwrite a newer generation's write after stop/start) fixed with per-channel flight sequencing and a red-green regression; Low (010's scheduler block no longer matched) fixed with an authoritative amendment note above that block. + +## wp1 Done + +Commit 1 (`feat(update): refresh the package update cache inside the running proxy`) shipped the scheduler, async registry lookup, write-through and 40 h unknown badge. Evidence: receipt 425 pass / 0 fail with typecheck, structure and privacy; reviewer PASS after two scheduler fixes (stop/start flight join, flight ordering) found in Check. What did not improve: Windows `/api/update/run` still launches the installer through a bounded synchronous PowerShell call (follow-up). What would show this direction wrong: a service install whose badge stays unknown for more than a day with network access. Next: wp2 per 020. + +## wp2 Check review + +Implementation reviewer Ohm (gpt-6-sol, agent 01a0d3d7-409a-7892-ad81-d0bdedc0fa41) round 1 on commit 2: FAIL. Accepted and being fixed in commit 2: (1) High, the snapshot POST accepted browser-origin requests — reject any `Origin`; (2) High, startup `wake()` could republish a stale snapshot — make notify atomic with publish; (4) Medium, the companion link lost `#/usage/companion` — preserve the validated fragment. (3) High, the tray install arm takes `PendingUpdate` outside `CheckGeneration::claim_install`: deferred to commit 3, whose plan (030) converts the tray arm to the gated three-way claim; wp3 Check must verify it is closed on the combined branch. + +macOS visual QA (commit 2): the real `UpdateDotView`/`UpdateDot` code from `app/Sources/NativeTray/Popover.swift` was compiled into a local harness, attached to real `NSStatusItem`s using `desktop/src-tauri/icons/tray/icon.png` as a template image, and rendered with `cacheDisplay` under aqua and darkAqua, with and without a title. The template glyph keeps its system tint in both appearances, the 7 pt blue dot sits at the glyph's lower right with a background-coloured halo, and a title does not overlap it. Limits: cached rendering does not reproduce the live menu-bar selection tint, and the live menu bar on this machine hid the harness items (crowded bar), so on-bar placement across displays remains human QA. +Round 1 fixes folded into commit 2: the snapshot POST now refuses any `Origin` (403, fixed body, no store write; red-green verified); `wake()` notifies without replacing the current snapshot and the companion link keeps its validated fragment (both red-green verified; cargo 139 pass, clippy and fmt clean). A test demanding that the outer management CORS wrapper omit `Access-Control-Allow-Origin` on this 403 was dropped: the wrapper is shared by every management route, the refusal body is a fixed string, and the boundary (refusal plus no write) holds without changing that shared layer. + +## wp2 Done + +Commit 2 (`feat(desktop): publish updater state to the GUI and dot the tray icon`) shipped the desktop snapshot route, per-session desktop badge, native check generation and UI projection worker, the macOS NSView dot and the dotted Windows/Linux PNG. Evidence: receipt 478 Bun + 4 GUI + 139 cargo pass with clippy, icons, typecheck, structure and privacy; reviewer GO-WITH-FIXES with zero blockers. Carried into wp3: the tray install arm must move onto the gated `claim_install` (reviewer High, deferred by plan). What would show this direction wrong: a dotted tray icon that stays after the update installs, or a GUI orb that disagrees with the tray. + +## wp3 Done + +Commit 3 (`feat(desktop): route the desktop update button to a bundled update page`) shipped `desktop/ui/update.html` with four app-origin Tauri commands, the shared `InstallClaim` gate for tray and page (closing the wp2 High), the page checking state, and desktop-only GUI routing with ten catalogs. Evidence: receipt 485 Bun + 5 GUI + 147 cargo pass with lint, i18n lint, clippy, typecheck, structure and privacy; reviewer Mill PASS with no findings. A pending-state screenshot of the page (headless Chrome with a stubbed invoke) is kept outside the repository for the PR. Packaged cross-origin navigation on Windows and Linux remains human QA. Next: wp4 per 040. + +## wp4 Check notes + +Commit 4 builds the dotted Windows icons from the shipped base frames; a local contact sheet at 16 and 32 px on light and dark backgrounds shows the status colours unchanged (online cyan, warning yellow, offline grey) with a white-haloed blue dot at the lower right, and all nine frame sizes present. PowerShell child-process scenarios cannot run on this Mac (no pwsh) and are left to the Windows CI legs. +Implementation reviewer Kierkegaard (gpt-6-sol, agent 01a0d417-79b9-7980-b94c-4d256aa5b24c) round 1: FAIL, one High — an install made before the dotted icons existed had only the three base icons, the ownership check required all six, so the tray was classified stale and the updater stopped it without reinstalling. Fixed in commit 4: `windowsTrayRequiredFilesPresent` requires only the base icons (the dotted ones stay in install/rollback/uninstall lists and the script falls back to the base icon); regression `an install from before the dotted icons still owns its registration` fails when all six are required and passes after. diff --git a/devlog/_plan/260924_update_indicator/001_sol_plan_draft.md b/devlog/_plan/260924_update_indicator/001_sol_plan_draft.md new file mode 100644 index 0000000000..a4e543665f --- /dev/null +++ b/devlog/_plan/260924_update_indicator/001_sol_plan_draft.md @@ -0,0 +1,64 @@ +# Update signalling plan + +This is a read-only plan. The current failure is a missing state producer: the sidebar reads `version.json`, but the refresh runs only after an interactive startup passes the TTY gate. A service startup skips it, so the badge can remain unknown indefinitely. The explicit check also runs a synchronous registry command in the request handler. [Badge reader](src/update/badge.ts:48), [refresh gate](src/update/notify.ts:238), [blocking lookup](src/update/index.ts:273), [check route](src/server/management/config-routes.ts:745). + +## Target behaviour + +| Installation | Update source and timing | Visible signal | Click path | +| --- | --- | --- | --- | +| npm, pnpm, Bun service on macOS/Linux | Registry result cached at startup and refreshed while the service runs; explicit check refreshes immediately | Existing blue sidebar orb and dot when newer | Sidebar opens the package update dialog; install uses the existing update job | +| npm, pnpm, Bun with Windows PowerShell tray | Same package cache; tray reads the same badge calculation every minute | Blue dot on the existing online, warning, or offline icon; safety colour remains visible | Tray double-click still opens the dashboard; a labelled menu item opens it for updating, then the sidebar opens the update dialog | +| Tauri desktop on macOS | Tauri updater check after 30 seconds, then every six hours; explicit check on demand | Blue dot on a colour-preserving menu bar icon, matching the sidebar’s blue signal | Icon retains the native usage popup; tray menu and the desktop update page offer **Install update** | +| Tauri desktop on Windows | Same Tauri state | Blue dot on the Tauri tray icon | Existing icon behaviour; tray menu or desktop update page installs | +| Tauri desktop on Linux with an AppIndicator host | Same Tauri state | Blue dot where the host renders the icon | Tray menu installs; the desktop update page also works | +| Tauri desktop on Linux without a tray host | Same Tauri state | Desktop dashboard orb; no claimed tray icon | Desktop update page installs | +| Source, mise, or unverified package ownership | No automatic package installation | No misleading “installable” signal; show manual guidance where available | No automatic install | + +The current sidebar polls `/api/update/badge` every ten minutes and paints the blue orb and 7 px dot. The desktop updater already owns `PendingUpdate`, the six-hour check, and the drain-before-install sequence. Linux can have no usable tray host. [Sidebar](gui/src/components/sidebar-github-row.tsx:70), [orb styling](gui/src/styles.css:370), [updater](desktop/src-tauri/src/updater.rs:81), [Linux tray contract](structure/desktop-shell.md:168). + +## Design decisions + +**A — macOS icon.** Keep the existing 44×44 template PNG for the normal state. A template image is recoloured by macOS, so a blue pixel cannot survive in it. Pre-render two 44×44 RGBA update variants from the existing tray SVG: a dark glyph for a light menu bar and a light glyph for a dark menu bar, each with a blue dot and a contrasting halo. Select by window theme and update on `ThemeChanged`; restore the template image when the update clears. Use `set_icon_with_as_template(image, false/true)` from the updater task, after releasing state locks. The pinned Tauri API performs this switch atomically; Linux/Windows use its ordinary icon setter. This avoids a new runtime image dependency and avoids the flicker from separate icon/template calls. A blue ring is an alternative, but it consumes more of the 22 pt mark; title-only or menu-only signalling misses the requested icon signal. The native macOS popup borrows the existing status item, so switching that item’s image does not change popup ownership. [Current icon](desktop/src-tauri/src/tray.rs:104), [popup status item](desktop/src-tauri/src/native_tray.rs:63), [Tauri API](https://docs.rs/tauri/latest/tauri/tray/struct.TrayIcon.html). Test normal and highlighted menu bars at 1× and Retina 2×. + +**B — Windows and Linux icons.** Tauri uses its `PendingUpdate` state to switch a pre-rendered PNG: 32 px for Windows taskbar scaling and 44 px for Linux AppIndicator scaling, with the same blue dot. Check Windows light/dark taskbars and Linux hosts for resampling or icon caching. The separate npm PowerShell tray uses `online-update.ico`, `warning-update.ico`, and `offline-update.ico`, each containing the existing nine sizes (16 through 256 px). It polls a bounded, read-only internal CLI badge command every 60 seconds, never on its three-second health tick; this calls the same `readUpdateBadge` logic as HTTP without copying semver rules into PowerShell or reading the admin token. Keep the existing health/safety icon as the base. Test that this hidden command performs no startup repair or cache refresh. The current PowerShell icon selection and owned asset list are at [windows-tray.ps1](src/tray/windows-tray.ps1:367) and [windows.ts](src/tray/windows.ts:18). + +**C — one authority per installation.** Package installs use `version.json`; desktop installs use Tauri’s signed updater result, never the compiled sidecar’s path-based `"source"` verdict. Tauri posts a scalar desktop snapshot—app version, latest version, availability, check time, and session ID—through its identity-bound `ProxyClient` to a management route. The proxy keeps it **in memory** with a short host heartbeat expiry. `/api/update/badge?surface=desktop` returns that snapshot; the default endpoint continues to represent the package installation. This works even when the app is a guest attached to an npm service: the embedded desktop dashboard asks for the desktop surface, while an ordinary browser sees the service’s package state. No desktop snapshot grants installation authority. The existing proxy client verifies PID, port, and binding generation before sending its management token. [Install detection](src/update/install-detection.mjs:248), [proxy binding](desktop/src-tauri/src/proxy.rs:191). + +In desktop mode, either GUI update button navigates the **main WebView** to a bundled `desktop/ui/update.html`. That app-origin page reads the native updater snapshot, checks again on request, and calls one shared native install function also used by the tray menu. It offers a return-to-dashboard action. The proxy-served dashboard receives no updater IPC permission; its current remote capability grants only zoom. On Linux without a tray host, this page supplies the missing install path. The GUI’s present update dialog calls the package `/api/update/check` and `/api/update/run`, so both sidebar and dashboard update entry points must make this desktop-specific navigation before entering that flow. [Current dashboard flow](gui/src/pages/use-dashboard-data.ts:812), [capability](desktop/src-tauri/capabilities/dashboard-zoom.json:1), [allowed app origin](desktop/src-tauri/src/window.rs:31). + +**D — cache freshness and event-loop safety.** Start one cancellable refresh scheduler after the proxy binds; stop it with the server. It attempts an immediate refresh if the package cache is missing/stale, then checks staleness hourly against the existing 20-hour interval. Coalesce concurrent background and explicit checks, bound the registry process and output, and back off after failures. Keep badge GET strictly read-only. Make `/api/update/check` await an asynchronous registry lookup and write a successful answer through to `version.json`; make `/api/update/run` await a fresh lookup before handing that checked result to the existing job starter. Keep synchronous installer work in the detached worker, outside the server event loop. An old cache beyond a defined expiry reports `unknown`, rather than silently asserting “current.” Preserve dismissal only for the exact dismissed version. The current job starter already accepts an injected check function, allowing the request path to change without growing its nearly 2,000-line file. [Job seam](src/update/job.ts:619), [cache writer](src/update/notify.ts:57). + +**E — automatic installation.** Defer `off/notify/download/auto` policy. This unit adds reliable notification and an explicit install path. Automatic installation would require a separate idle definition, in-flight request drain budget, service ownership checks, and recovery policy across npm and desktop. Existing desktop install already coordinates a drain after a user action. [Desktop install ordering](desktop/src-tauri/src/updater.rs:50). + +## Ordered commits in one branch and one PR to `dev` + +1. **Package cache and asynchronous checks.** Change `src/update/index.ts`, `notify.ts`, `badge.ts`, `src/server/management/config-routes.ts`, and `src/server/index.ts`; add `src/update/refresh-scheduler.ts` and `tests/update/update-refresh.test.ts`. Extend `tests/update/update-notify.test.ts`, `update-badge.test.ts`, and `update-job.test.ts`; register the new test in `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`. Update `structure/runtime.md`, `structure/ops/service-and-sidecars.md`, `structure/gui-and-management-api.md`, and the English plus existing translated `reference/management-api.md` pages. Verify `bun test` on these focused files, `bun run typecheck`, `bun run structure:check`, and `bun run privacy:scan`. + +2. **Desktop update state and icon.** Add `src/update/desktop-badge.ts`, `tests/update/update-desktop-badge.test.ts`, `desktop/src-tauri/src/update_status.rs`, `desktop/src-tauri/src/tray_icons.rs`, and generated update PNGs/SVG sources under `desktop/src-tauri/icons/tray/`. Change `src/server/management/sidebar-routes.ts`, `route-registry.ts`, `desktop/src-tauri/src/proxy.rs`, `updater.rs`, `tray.rs`, `lib.rs`, and `desktop/scripts/generate-icons.ts`. The new POST route accepts only bounded scalar state under normal management authentication; the registry declares it. Register the new test in both layout files. Update `structure/desktop-shell.md`, `companion.md`, `gui-and-management-api.md`, and the desktop guide and management API pages in English and their existing locale siblings. Test `tests/update/update-desktop-badge.test.ts`, `tests/server/sidebar-routes.test.ts`, `tests/server/management-route-registry.test.ts`, `tests/ci-workflows/build-desktop-icon-set.test.ts`, `bun --cwd desktop run icons:check`, and `cargo test --manifest-path desktop/src-tauri/Cargo.toml`. + +3. **Desktop GUI install route.** Add `desktop/ui/update.html` and `gui/src/lib/desktop-update.ts`; change `desktop/src-tauri/src/lib.rs`, `updater.rs`, `tray.rs`, `gui/src/App.tsx`, `gui/src/components/sidebar-github-row.tsx`, and `gui/src/pages/use-dashboard-data.ts`. Reuse the native updater’s pending/install lock and retry behaviour for tray and page actions. Add focused desktop-page and GUI tests under `tests/clients/` and `gui/tests/`, registering any new Bun test in both layout files. Add visible copy to all ten `gui/src/i18n/{en,de,fr,ja,ko,ru,tr,vi,zh,zh-TW}.ts` catalogs. Update `structure/desktop-shell.md`, `gui-and-management-api.md`, and the desktop guides in English and their existing locale siblings. Run focused tests, `bun run typecheck`, `cd gui && bun run lint:i18n && bun run build`, and desktop `cargo test`. Attach a desktop GUI screenshot to the PR description. + +4. **Windows npm tray indicator.** Add three update `.ico` files under `src/tray/assets/` and a deterministic generator/check script under `scripts/` for their nine-size contents. Change `src/tray/windows.ts`, `windows-tray.ps1`, `src/lib/config-ownership.ts`, `src/cli/registry.ts`, and `src/cli/dispatch.ts`; extend `tests/windows/windows-tray.test.ts`, `tests/windows/tray-proxy.test.ts`, and `tests/update/update-badge.test.ts`. Update `structure/runtime.md`, `structure/ops/service-and-sidecars.md`, and the English plus translated CLI lifecycle pages. Test the hidden command’s zero-side-effect path, icon asset install/rollback ownership, probe timeout and no-overlap behaviour, and online/warning/offline icon precedence. Run focused Bun tests, `bun run typecheck`, `bun run skill:surface:check`, `bun run structure:check`, and `bun run privacy:scan`. + +The current ratchet permits no growth in `gui/src/styles.css` (2,958/2,958); reuse its existing orb CSS. `src/server/index.ts` has **10 lines** of headroom (883/893), so only a small scheduler start/stop hook belongs there. `src/update/job.ts` has **6 lines** before the new-file 2,000-line threshold (1,993 lines), and `src/cli/index.ts` has **5** (1,994); the plan avoids growing either. Other planned TypeScript headroom at this HEAD: `notify.ts` +1,729, `badge.ts` +1,924, `index.ts` +1,116, `config-routes.ts` +892, `sidebar-routes.ts` +1,893, `route-registry.ts` +1,604, `windows.ts` +1,214, `config-ownership.ts` +1,615, CLI `registry.ts` +1,362 and `dispatch.ts` +890, GUI `App.tsx` +1,472, `sidebar-github-row.tsx` +1,838, and `use-dashboard-data.ts` +1,048. New TypeScript files must remain below 2,000 lines. Rust and `.ps1` files are outside this ratchet; current `tray.rs` is 620 lines, `updater.rs` 144, and `windows-tray.ps1` 522. Recalculate against the shared tree immediately before each commit. [Ratchet rule](scripts/file-size-ratchet.ts:114). + +Before review readiness, run the repository’s required broad gates or document the permitted focused-test exception in the PR template. Run `bun run structure:check`, `bun run privacy:scan`, `bun run typecheck`, GUI lint/build, desktop Cargo tests, and the relevant macOS and Windows CI legs at the **PR’s exact head**. The PR targets `dev`; the host retains merge authority. + +## Visual QA and failure cases + +Manually compare no-update, pending, check-failed, installing, install-failed, and post-restart states. On macOS check light/dark and highlighted menu bars at 1×/2×, icon switching without flicker, and the native popup/menu. On Windows check light/dark taskbars and 100%/200% scaling for both Tauri and npm trays, including safety-warning precedence. On Linux check at least one GNOME AppIndicator host, one KDE host, and a session without a host; verify the desktop update page remains reachable in the last case. Confirm a guest desktop app shows the **desktop** update in its embedded GUI while a normal browser attached to the same service shows the **package** update. + +The main risks are stale host state, process overlap during refresh, a tiny dot disappearing under tray scaling, and a check/install race. Expire the in-memory desktop heartbeat, coalesce registry work, verify the smallest icon sizes visually, and recheck availability in the shared native install action. A failed desktop download or drain keeps `PendingUpdate` and restores the indicator. Rollback is the single PR’s ordered commits; no persisted desktop update schema or automatic-install setting is introduced. + +**Open questions:** None blocks this design. Broader notify-only support for npx/bunx/Volta/Yarn/Homebrew and automatic installation are separate product decisions. + +--- + +## Coordinator review (2026-09-24) + +Verified at dd7cb695a9: tauri 2.11.6 has `TrayIcon::set_icon_with_as_template` (tray/mod.rs:577); `gui/src/styles.css` 2958/2958 and `src/server/index.ts` 883/893 match the ratchet baseline. + +1. **macOS theme source.** The menu bar appearance is not the app window theme: the bar can be dark while the app is light (wallpaper tint, "Reduce transparency", a second display). Picking the light/dark variant from `ThemeChanged` can paint a dark glyph on a dark bar. Before committing to pre-rendered variants, evaluate a native overlay: keep the template image and add a small blue sublayer to the `NSStatusBarButton` through the existing ObjC bridge in `native_tray.rs`. That keeps automatic tinting and the highlighted state. If variants are kept, read `statusItem.button.effectiveAppearance` natively instead of the window theme. +2. **Windows Tauri tray contrast.** `tray.rs` ships the same black template PNG on Windows, where `icon_as_template` is ignored. On a dark taskbar the base icon is already low contrast. Commit 2 should add light/dark base variants for Windows, not only the update variant. +3. **Indicator shape** (dot vs ring) is pending the maintainer's answer; the plan assumes a dot. +4. **Scope.** Four commits touch Rust, TS, PowerShell, ten locales, structure docs and docs-site. Commit 4 (npm Windows tray) is independent and can split out if review load is high. diff --git a/devlog/_plan/260924_update_indicator/002_architect_proposal.md b/devlog/_plan/260924_update_indicator/002_architect_proposal.md new file mode 100644 index 0000000000..544b1c698d --- /dev/null +++ b/devlog/_plan/260924_update_indicator/002_architect_proposal.md @@ -0,0 +1,86 @@ +# Architect proposal (McClintock, gpt-6-sol) + +Read-only design proposal for this unit. Dispositions live in 000_plan.md. + +# D1 — macOS menu bar dot + +**Decision and recommendation.** Keep the current tray glyph as a macOS template image and draw the blue dot in a small transparent `NSView` attached to the `NSStatusBarButton`. Add a main-thread bridge call that obtains the status item through the tray handle each time the pending state changes. Position the dot relative to the image as the button lays out, because the tray also changes its title. Remove the view when no update is pending. + +**Alternatives.** Pre-rendered non-template RGBA icons can preserve blue, but their glyph color must follow the *status button’s* `effectiveAppearance`, including display and highlighted-state changes; the window theme is insufficient. Toggling `set_icon_with_as_template` alone cannot preserve a blue pixel while template mode is on. It is useful only for an icon-variant implementation. + +**Evidence.** The tray is created with `.icon_as_template(true)` at [tray.rs:104](desktop/src-tauri/src/tray.rs:104). The existing bridge already borrows `NSStatusItem` via `tray.ns_status_item()` at [native_tray.rs:63](desktop/src-tauri/src/native_tray.rs:63), and Swift obtains `item.button` at [Popover.swift:24](app/Sources/NativeTray/Popover.swift:24). There is no `.m`/`.h` bridge here: [build.rs:11](desktop/src-tauri/build.rs:11) compiles the Swift sources. The lockfile pins Tauri **2.11.6** and `tray-icon` **0.24.2** at [Cargo.lock:3796](desktop/src-tauri/Cargo.lock:3796) and [Cargo.lock:4459](desktop/src-tauri/Cargo.lock:4459). In the pinned crate, `set_icon` calls `button.setImage` and `setImagePosition`; it does not remove button subviews or layers ([tray-icon macOS source:283](~/.cargo/registry/src/<index>/tray-icon-0.24.2/src/platform_impl/macos/mod.rs:283)). Tauri’s combined setter avoids a two-render flicker but merely falls back to `set_icon` off macOS ([Tauri source:570](~/.cargo/registry/src/<index>/tauri-2.11.6/src/tray/mod.rs:570)). + +**Risk.** The overlay’s alignment under changing titles, highlighted menus, 1×/2× scaling, and status-item recreation needs native visual QA. Use a drawing view rather than making the button layer-backed solely for the dot. + +# D2 — Windows and Linux Tauri icons + +**Decision and recommendation.** Swap Tauri tray PNGs when pending changes. Generate normal and dotted variants from SVG sources under `desktop/src-tauri/icons/tray/`. Provide contrasting light and dark Windows base glyphs as well as their dotted variants; the present black template PNG receives no Windows template tint. Use a blue dot with a narrow contrasting halo and verify the result at the host’s rendered sizes. Linux should use a normal/dotted icon pair only when an AppIndicator host actually displays the tray. + +**Alternatives.** A title or menu label alone misses the required icon signal. Runtime compositing removes committed variants but introduces rendering and platform-scaling complexity. + +**Evidence.** The same `tray/icon.png` is currently loaded on every platform ([tray.rs:463](desktop/src-tauri/src/tray.rs:463)). The existing generator treats the menu bar asset as a 44 px SVG-derived image and checks generated output ([generate-icons.ts:62](desktop/scripts/generate-icons.ts:62)); [build-desktop-icon-set.test.ts:184](tests/ci-workflows/build-desktop-icon-set.test.ts:184) checks its size and generator wiring. Linux can have no registered tray host ([tray_availability.rs:60](desktop/src-tauri/src/tray_availability.rs:60)). + +**Risk.** Windows taskbar and Linux hosts resample or cache icons differently. Extend the icon generator and its existing test to check every declared variant, then inspect 16/20/24/32 px presentation on Windows and at least one AppIndicator host on Linux. + +# D3 — Desktop updater snapshot and badge + +**Decision and recommendation.** Let Tauri remain the desktop update authority. Add `POST /api/update/desktop-snapshot` to the management registry as a mutation, accepted only from the existing `admin-token` principal. The payload should contain bounded scalar fields: a random desktop session ID, app/current version, latest version or `null`, availability, and check time. Store it only in proxy memory, keyed by session ID, with a bounded entry count. Post after each updater state change and heartbeat about every 60 seconds; expire entries after roughly three minutes. `GET /api/update/badge?surface=desktop&session=<id>` should return the badge shape with `installer: "desktop"` and `unknown: true` if the entry is absent or expired. The default badge URL retains package semantics. Carry the session ID in the embedded dashboard URL and preserve it on subsequent in-app navigation; a normal browser without it reads the package badge. + +**Alternatives.** Treating the compiled sidecar’s `source` result as a desktop result leaves the dot dark. One global “last desktop snapshot” would let two shells attached to the same service show each other’s update. Persisting desktop state in `version.json` would mix authorities and survive after the app exits. + +**Evidence.** Install detection returns `"source"` for a path without `node_modules` ([install-detection.mjs:248](src/update/install-detection.mjs:248)), and the package badge suppresses source updates ([badge.ts:48](src/update/badge.ts:48)). Tauri already owns `PendingUpdate` and six-hour checks ([updater.rs:6](desktop/src-tauri/src/updater.rs:6), [updater.rs:81](desktop/src-tauri/src/updater.rs:81)). Existing routes declare method, owner and mutation status in [route-registry.ts:81](src/server/management/route-registry.ts:81). `ProxyClient` checks the bound PID, port and generation before releasing the management token, and disables redirects and system proxies ([proxy.rs:82](desktop/src-tauri/src/proxy.rs:82), [proxy.rs:191](desktop/src-tauri/src/proxy.rs:191)). The auth model distinguishes raw admin tokens from GUI sessions ([management-auth.ts:310](src/server/management-auth.ts:310)). + +**Risk.** A local holder of the admin token can forge display state; that token already authorizes management operations. The snapshot must grant **no installation authority**. Require the principal explicitly, validate the bounded payload at ingress, avoid logging the session ID or raw request, and return `unknown` on expiry rather than falling back to a package update. + +# D4 — Desktop update page and install path + +**Decision and recommendation.** Add a bundled `desktop/ui/update.html` app-origin page. The embedded dashboard’s sidebar action and dashboard update entry should navigate the main WebView there when `isDesktopShell()` is true. The page reads native updater status and offers check, install, retry and return-to-dashboard actions through narrowly named Tauri commands. Both page and tray must call one native install function that atomically claims the install lock, takes `PendingUpdate`, downloads and verifies, drains the owned runtime, and restores pending state on failure. + +**Alternatives.** Granting updater IPC to the loopback dashboard would widen remote-origin permissions. Sending desktop users through `/api/update/run` invokes the package updater. A tray-only install path fails on Linux sessions without a tray host. + +**Evidence.** The dashboard currently sends its update action to `#dashboard/update` ([App.tsx:454](gui/src/App.tsx:454)), whose handler calls package `/api/update/check` ([use-dashboard-data.ts:812](gui/src/pages/use-dashboard-data.ts:812)). The sidebar already paints its blue dot from the badge GET ([sidebar-github-row.tsx:70](gui/src/components/sidebar-github-row.tsx:70), [sidebar-github-row.tsx:148](gui/src/components/sidebar-github-row.tsx:148)). The main window permits app-origin and bound `127.0.0.1` navigation ([window.rs:31](desktop/src-tauri/src/window.rs:31)); its current loopback capability grants only zoom ([dashboard-zoom.json:5](desktop/src-tauri/capabilities/dashboard-zoom.json:5)). The bundled app origin is served from `desktop/ui` ([tauri.conf.json:6](desktop/src-tauri/tauri.conf.json:6)). The tray’s current install arm takes pending state and calls `updater::install` ([tray.rs:198](desktop/src-tauri/src/tray.rs:198)); its Boolean installing flag uses `store`, so two new callers need one compare-and-swap claim ([tray.rs:303](desktop/src-tauri/src/tray.rs:303)). The updater’s download-before-drain order is already established ([updater.rs:50](desktop/src-tauri/src/updater.rs:50)). + +**Risk.** Cross-origin navigation from the dashboard to the bundled page needs packaged macOS, Windows and Linux verification. Test return navigation and reload, including a no-tray Linux session. Keep the updater command available only on the app-origin page. + +# D5 — Package cache refresh and nonblocking checks + +**Decision and recommendation.** Put a small scheduler start/stop hook in `src/server/index.ts`, with the worker in `src/update/`. It should run for eligible package installs on service and interactive startup, check immediately when the cache is missing or stale, recheck staleness hourly against the existing 20-hour interval, coalesce by channel, and use bounded retry backoff after failure. Do not start it for the desktop sidecar classified as `source` or for externally managed `mise`. Keep badge GET read-only. Use asynchronous `node:child_process.spawn` with the existing `registrySpawnTarget` (including pnpm owner binding and Windows argument options), a 12-second kill deadline, bounded stdout/stderr and no raw output in logs. On successful lookup, write through to `version.json`, preserving dismissal only for the same version. Treat a failed lookup as unknown after a defined maximum cache age; a suggested policy is 40 hours. + +Both `/api/update/check` **and** `/api/update/run` need the async path. For run, await a fresh result, then pass that already validated result into `startUpdateJob` through its existing synchronous check dependency, so its lock and worker creation remain together. The detached worker may retain synchronous checks because it does not run on the server event loop. + +**Alternatives.** `Bun.spawn` is viable, but adapting the current Node-style invocation options would be extra work. A detached `__refresh-version` helper handles interactive startup today but does not give the server an awaited, coalesced answer. + +**Evidence.** The package lookup calls `spawnSync` with a 12-second timeout ([update/index.ts:272](src/update/index.ts:272)); `/api/update/check` invokes it in the request handler ([config-routes.ts:750](src/server/management/config-routes.ts:750)). `/api/update/run` calls `startUpdateJob`, which invokes `checkForUpdateFn` synchronously ([config-routes.ts:759](src/server/management/config-routes.ts:759), [job.ts:619](src/update/job.ts:619)). The existing target preserves pnpm ownership and Windows options ([update/index.ts:188](src/update/index.ts:188), [update/index.ts:217](src/update/index.ts:217)). Cache refresh currently occurs only after the interactive gate ([notify.ts:125](src/update/notify.ts:125), [notify.ts:238](src/update/notify.ts:238)); successful writes are atomic and failures do not advance the timestamp ([notify.ts:57](src/update/notify.ts:57), [notify.ts:193](src/update/notify.ts:193)). `src/server/index.ts` is 883 lines against a 893-line cap ([index.ts:883](src/server/index.ts:883), [file-size-baseline.json:34](tests/fixtures/file-size-baseline.json:34)); its stop path already owns shutdown callbacks ([index.ts:749](src/server/index.ts:749)). + +**Risk.** A successful explicit preview check can temporarily replace the single-file latest-channel cache. Serialize writes and let the default-channel scheduler refresh on channel mismatch. Keep the scheduler out of `router.ts`, `server/lifecycle.ts` and `server/responses/core.ts` to preserve the optional Lab boundary; do not add an `await` between `Bun.serve` and synchronous Lab activation ([index.ts:858](src/server/index.ts:858)). + +# D6 — npm Windows PowerShell tray + +**Decision and recommendation.** Add a read-only hidden CLI command that prints the existing `readUpdateBadge()` scalar JSON, then probe it asynchronously from PowerShell about every 60 seconds. Bound its duration and output, prohibit overlap, and keep the last result only until its own expiry. Select among six icons: online, warning and offline, each with normal and blue-dot variants. Safety color has precedence as the base; the update dot overlays it. Keep the existing double-click dashboard action and add a labelled update menu action that opens the dashboard. + +**Alternatives.** Calling HTTP needs the admin token, fails while the proxy is offline, and couples the tray to listener availability. Reading `version.json` directly is cheaper but duplicates channel/version comparison and cache validity rules in PowerShell. + +**Evidence.** The tray currently loads three `.ico` assets ([windows-tray.ps1:30](src/tray/windows-tray.ps1:30)) and chooses warning before online, then offline when disconnected ([windows-tray.ps1:366](src/tray/windows-tray.ps1:366)). Its UI timer runs every three seconds ([windows-tray.ps1:473](src/tray/windows-tray.ps1:473)); the existing startup-health process probe shows a bounded, non-overlapping pattern ([windows-tray.ps1:254](src/tray/windows-tray.ps1:254)). Hidden commands already have registry and dispatch entries ([registry.ts:590](src/cli/registry.ts:590), [dispatch.ts:725](src/cli/dispatch.ts:725)). The installed asset list and owned config paths are explicit ([windows.ts:18](src/tray/windows.ts:18), [config-ownership.ts:63](src/lib/config-ownership.ts:63)). + +**Risk.** Prove the hidden command does not start/repair a service, refresh the registry, or write the cache. Extend the Windows tray ownership and icon-size tests; ensure timeout and shutdown dispose the badge probe. + +# D7 — ordered commits and proof + +The four-commit order in the draft is sound once D1, D4 and D5 are corrected. The structure ownership table assigns `src/update/` to `runtime.md` and `ops/service-and-sidecars.md`, `src/tray/` to `runtime.md`, and `desktop/` to `desktop-shell.md` and `companion.md` ([structure/INDEX.md:151](structure/INDEX.md:151), [structure/INDEX.md:105](structure/INDEX.md:105)). + +| Commit | File scope and required proof | +|---|---| +| **1. Package cache** | `src/update/{index,notify,badge,job}.ts`, new `refresh-scheduler.ts`, `src/server/management/config-routes.ts`, a small `src/server/index.ts` hook; extend `tests/update/{update-notify,update-badge,update-job}.test.ts`, add `tests/update/update-refresh.test.ts`. Update `structure/{runtime,gui-and-management-api}.md`, `structure/ops/service-and-sidecars.md`, and `docs-site/src/content/docs/reference/management-api.md` plus its existing `fr,ja,ko,ru,tr,zh-cn,zh-tw` siblings. Run focused Bun tests, `bun run typecheck`, `bun run structure:check`, `bun run privacy:scan`. | +| **2. Desktop state and icons** | `desktop/src-tauri/src/{updater,tray,proxy,lib}.rs`, `app/Sources/NativeTray/Popover.swift`, `src/update/desktop-badge.ts`, `src/server/management/{sidebar-routes,route-registry}.ts`, tray SVG/PNG variants and `desktop/scripts/generate-icons.ts`; extend `tests/server/{sidebar-routes,management-route-registry}.test.ts` and `tests/ci-workflows/build-desktop-icon-set.test.ts`, add `tests/update/update-desktop-badge.test.ts`. Update `structure/{desktop-shell,gui-and-management-api}.md` and the management API and desktop guides. Run those focused tests, icon check and Cargo tests. | +| **3. Desktop GUI route** | `desktop/ui/update.html`, `desktop/src-tauri/src/{lib,updater,tray,window,startup}.rs`, `gui/src/{App.tsx,components/sidebar-github-row.tsx,lib/desktop-shell.ts,pages/use-dashboard-data.ts}` and a small navigation helper; extend `gui/tests/desktop-shell.test.ts`, add focused desktop page tests such as `tests/clients/desktop-update-surface.test.ts`. Update the ten GUI i18n catalogs, `structure/desktop-shell.md`, `structure/gui-and-management-api.md`, and the desktop guides. Run focused tests, GUI lint/build, Cargo tests, and packaged navigation QA. | +| **4. npm Windows tray** | Three dotted `.ico` assets, deterministic generator/check, `src/tray/{windows.ts,windows-tray.ps1}`, `src/lib/config-ownership.ts`, `src/cli/{registry,dispatch}.ts`; extend `tests/windows/{windows-tray,tray-proxy}.test.ts`. Update `structure/{runtime,ops/service-and-sidecars}.md` and `docs-site/src/content/docs/reference/cli/lifecycle.md` plus its existing `fr,ja,ko,ru,tr,zh-cn,zh-tw` siblings. Run focused Bun tests, `bun run skill:surface:check`, typecheck, structure and privacy gates. | + +Every **new Bun test** must be entered in both [scripts/test-layout/layout.json:1](scripts/test-layout/layout.json:1) and [tests/fixtures/test-layout-expected.json:1](tests/fixtures/test-layout-expected.json:1). Recalculate line headroom before edits: the draft’s `src/server/index.ts` figure is confirmed above, while the shared tree can move. The existing management, desktop guide and CLI lifecycle paths listed here exist under [docs-site/src/content/docs](docs-site/src/content/docs). Run the repository’s broad gate or record the allowed focused-test exception before review readiness; this read-only task ran no tests. + +# Unresolved assumptions + +- A transparent child `NSView` remains correctly positioned when AppKit changes status-button layout or highlighted appearance. The pinned crate does not remove it on `set_icon`, but packaged visual QA must settle alignment. +- The loopback dashboard can navigate into `tauri://localhost/update.html` on macOS/Linux and `http://tauri.localhost/update.html` on Windows under each packaged WebView. The allowlist permits both origins, but that alone does not prove the page transition. +- The desktop snapshot heartbeat interval and expiry (proposed 60 seconds and three minutes), package maximum cache age (proposed 40 hours), and exact dot dimensions need implementation tests and visual QA. +- A hidden read-only CLI probe may still pay substantial CLI import cost on Windows. Measure its duration before fixing the 60-second cadence; keep the three-second UI timer independent. +- Auto-install policy remains outside this unit, as specified. diff --git a/devlog/_plan/260924_update_indicator/010_phase1_package_cache.md b/devlog/_plan/260924_update_indicator/010_phase1_package_cache.md new file mode 100644 index 0000000000..b477e73fca --- /dev/null +++ b/devlog/_plan/260924_update_indicator/010_phase1_package_cache.md @@ -0,0 +1,1038 @@ +# 010 — Package cache freshness and asynchronous checks + +## Commit contract + +Goal: a service-owned package install refreshes `version.json` after bind and while running; explicit management checks do not block the Bun request loop; a cache older than 40 hours cannot claim an available update. This is commit 1 / work-phase wp1 of `000_plan.md`, decision D5. It has no preceding implementation commit. Commits 2–4 may consume the unchanged package badge shape and its read-only GET. + +IN: npm, pnpm and Bun package installs; registry discovery; cache and dismissal; scheduler lifetime; `/api/update/check`, `/api/update/run`, package badge; focused tests; structure and public management API docs. OUT: desktop snapshot, native tray, GUI navigation, npm Windows tray, auto-install policy, Lab boundary modules, authentication semantics, updater signature/integrity checks, and change to detached installer worker. No GUI i18n key is introduced. + +The proposed opt-out is **new** `OCX_DISABLE_UPDATE_CHECK=1`, applying only to automatic scheduler startup. Search at this HEAD found no existing `OCX_DISABLE_UPDATE_CHECK`, `NO_UPDATE`, or `checkForUpdates` symbol under `src/config*`, `src/update`, or `src/server/index.ts`; `OCX_SERVICE` currently gates interactive prompting, not update checks (`src/update/notify.ts:125-132`). Explicit checks remain user-initiated. The source/mise guard is required regardless of the env var. Do not reinterpret `OCX_SERVICE` as an opt-out. + +## Existing contracts to preserve + +`registrySpawnTarget` rejects unowned pnpm queries and carries owner environment plus Windows invocation options (`src/update/index.ts:188-228`); `latestVersion` currently uses `spawnSync` and a 12-second timeout (`src/update/index.ts:272-290`). `resolveCurrentPnpmGlobalOwner` performs synchronous manager probes (`src/update/index.ts:97-123`), so an async registry child alone is insufficient on pnpm. `checkForUpdate` is the existing response builder, including source/mise guidance (`src/update/job.ts:486-524`); its injectable `latestVersion` dependency is declared in `src/update/check-types.ts:3-9`. `startUpdateJob` checks synchronously before creating a job (`src/update/job.ts:619-663`), and an injected `checkForUpdateFn` lets the request supply an already checked answer. The badge reader currently has no refresh dependency (`src/update/badge.ts:32-75`), and its GET only calls that reader (`src/server/management/sidebar-routes.ts:100-103`). The interactive prompt spawns `__refresh-version` (`src/update/notify.ts:174-206,238-247`), while the CLI invokes the prompt before server bind (`src/cli/index.ts:489`; test: `tests/update/update-notify.test.ts:127-139`). Server stop delegates cleanup through `runListenerShutdown` (`src/server/index.ts:749-793`); the synchronous Lab activation point is `src/server/index.ts:858-866`. Its specific structure owner is `structure/adapters/compatibility-lab.md:3-18`; this commit leaves that contract text accurate, so it is reviewed without editing it. + +Other existing symbols used by the proposed blocks: package/channel/install detection at `src/update/index.ts:58-80,173-185`; `registrySpawnTarget` at `src/update/index.ts:217-228`; pnpm owner lookup and running shim at `src/update/index.ts:87-123`; `unprivilegedOwnershipMutationEnvironment` import at `src/update/index.ts:14`; `VersionCache`, `readVersionCache`, `writeVersionCache`, and `isSourceBuildVersion` at `src/update/notify.ts:19-63,121-123`; `detectInstallOwnership` and `miseUpdateCommand` at `src/update/index.ts:70-80`; `UpdateCheckResult`, `checkForUpdate`, and `startUpdateJob` at `src/update/job.ts:82-92,486-524,619-663`; `normalizeUpdateChannel` at `src/update/job.ts:256`; `readUpdateBadge` at `src/update/badge.ts:48-75`; the GUI package consumers at `gui/src/pages/use-dashboard-data.ts:823,907`; the route handler at `src/server/management/config-routes.ts:290,750-777`. Runtime standard-library `spawn`, `Worker`, timers, and `Request` are platform APIs, not repository symbols. + +## File change map and executable edits + +| Path | Action | Edit anchor | +| --- | --- | --- | +| `src/update/index.ts` | MODIFY | Export the existing registry target builder/type only | +| `src/update/pnpm-owner-worker.ts` | NEW | Complete worker source below | +| `tests/fixtures/pnpm-owner-stall-worker.ts` | NEW | Worker deadline regression fixture below | +| `src/update/async-check.ts` | NEW | Complete async lookup source below | +| `src/update/notify.ts` | MODIFY | Export interval, add write-through, stop prompt launch | +| `src/update/refresh-scheduler.ts` | NEW | Complete coordinator source below | +| `src/update/badge.ts` | MODIFY | 40-hour validity and read-only state | +| `src/server/management/config-routes.ts` | MODIFY | Await coordinated checks for check/run | +| `src/server/management/context.ts` | MODIFY | Inject one delayed check at the management route boundary for regression proof | +| `src/server/index.ts` | MODIFY | Paired, idempotent listener start/stop hook below | +| `tests/update/update-refresh.test.ts` | NEW | Complete source below | +| `tests/server/update-async-routes.test.ts` | NEW | Complete source below | +| `tests/update/update-notify.test.ts`, `tests/update/update-badge.test.ts`, `tests/update/update-job.test.ts`, `tests/server/sidebar-routes.test.ts` | MODIFY | Cases and exact assertions below | +| `tests/preload.ts` | MODIFY | Suppress automatic registry checks in all Bun test-started servers | +| `scripts/test-layout/layout.json`, `tests/fixtures/test-layout-expected.json` | MODIFY | Two explicit registrations below | +| `structure/runtime.md`, `structure/ops/service-and-sidecars.md`, `structure/gui-and-management-api.md` | MODIFY | Exact contract text below | +| `docs-site/src/content/docs/reference/management-api.md` and existing `fr`, `ja`, `ko`, `ru`, `tr`, `zh-cn`, `zh-tw` siblings | MODIFY | Row replacements and locale instruction below | + +No DELETE path. No source, test, config, or other doc file is modified in this docs-only pass; this table is the later implementation commit's map. + +### `src/update/index.ts` — MODIFY + +Change the declaration `function registrySpawnTarget(` at line 217 to `export function registrySpawnTarget(`. Keep lines 218–228 byte-for-byte, as well as the synchronous `latestVersion` for the detached worker. Export its return type for the async module: + +```ts +export type RegistrySpawnTarget = SpawnTarget; +``` + +Place that alias after `SpawnTarget` at line 193. The async module below uses exactly the same `registrySpawnTarget`, including `pnpmOwnerInvocation` (`src/update/index.ts:201-228`) and `unprivilegedOwnershipMutationEnvironment` (`src/update/index.ts:14,286`). Do not duplicate target construction. No other existing export changes. + +Preserve pnpm's running-shim proof across the Worker boundary. Replace the first line of `runningPnpmShimPath` (`src/update/index.ts:87-95`) and the signature/call in `resolveCurrentPnpmGlobalOwner` (`src/update/index.ts:114-123`) as follows; all other function lines stay unchanged: + +```ts +function runningPnpmShimPath(invoked = process.argv[1]): string | undefined { + if (!invoked) return undefined; + const name = invoked.replaceAll("\\", "/").split("/").at(-1)?.toLowerCase(); + if (!new Set(["ocx", "opencodex", "ocx.cmd", "opencodex.cmd", "ocx.ps1", "opencodex.ps1"]).has(name ?? "")) { + return undefined; + } + return resolve(invoked); +} + +export function resolveCurrentPnpmGlobalOwner(invoked = process.argv[1]): PnpmGlobalOwnerResult { + return resolvePnpmGlobalOwner({ + packageName: PKG, + packagePath: packageRoot(), + commandPaths: resolvePnpmCommands(), + runningShimPath: runningPnpmShimPath(invoked), + runPnpm: runPnpmCandidate, + }); +} +``` + +### `src/update/pnpm-owner-worker.ts` — NEW; complete file + +The worker is only used for pnpm. It runs the existing ownership proof off the request thread; it never serializes an owner to disk or a response. Bun's module worker loads this TypeScript source at runtime. A worker error/timeout yields null; it must not fall back to unowned `pnpm` on PATH. + +```ts +import { resolveCurrentPnpmGlobalOwner } from "./index"; + +onmessage = event => { + try { + const invoked = typeof event.data === "string" ? event.data : ""; + if (!invoked) { postMessage(null); return; } + const result = resolveCurrentPnpmGlobalOwner(invoked); + postMessage(result.ok ? result.owner : null); + } catch { + postMessage(null); + } +}; +``` + +### `src/update/async-check.ts` — NEW; complete file + +This module owns bounded asynchronous package-manager lookup. Its `pnpmOwner` worker has an independent 12-second deadline because owner discovery may run several synchronous manager probes. The registry child has its own 12-second deadline. `spawn`'s `error` and `close` can both fire; `finish` settles once. Output is limited while streaming, so a noisy manager cannot grow memory indefinitely. No stderr/stdout is logged or returned. The worker is terminated on completion. Tests inject `spawnFn` and `ownerFn`, avoiding an installed registry or pnpm owner. The optional worker URL/deadline/invocation parameters below permit a real Bun Worker startup/deadline test without invoking a package manager; production retains the fixed source URL and 12-second deadline. + +```ts +import { spawn, type ChildProcessWithoutNullStreams } from "node:child_process"; +import { unprivilegedOwnershipMutationEnvironment } from "../service/ownership-mutation-lease.mjs"; +import { PKG, registrySpawnTarget, type Channel, type Installer } from "./index"; +import type { PnpmGlobalOwner } from "./pnpm-global-install.mjs"; + +export const REGISTRY_DEADLINE_MS = 12_000; +export const REGISTRY_OUTPUT_LIMIT = 4_096; + +export interface PnpmOwnerDeps { + workerUrl?: URL; + deadlineMs?: number; + invoked?: string; +} + +export async function pnpmOwner(deps: PnpmOwnerDeps = {}): Promise<PnpmGlobalOwner | null> { + return new Promise(resolve => { + let worker: Worker; + try { + worker = new Worker((deps.workerUrl ?? new URL("./pnpm-owner-worker.ts", import.meta.url)).href); + } catch { + resolve(null); + return; + } + let done = false; + const finish = (owner: PnpmGlobalOwner | null) => { + if (done) return; + done = true; + clearTimeout(timer); + void worker.terminate(); + resolve(owner); + }; + const timer = setTimeout(() => finish(null), deps.deadlineMs ?? REGISTRY_DEADLINE_MS); + worker.onmessage = event => finish(event.data as PnpmGlobalOwner | null); + worker.onerror = () => finish(null); + try { worker.postMessage(deps.invoked ?? process.argv[1] ?? ""); } + catch { finish(null); } + }); +} + +export interface AsyncLookupDeps { + ownerFn: () => Promise<PnpmGlobalOwner | null>; + spawnFn: typeof spawn; + deadlineMs?: number; +} + +const defaultDeps: AsyncLookupDeps = { ownerFn: pnpmOwner, spawnFn: spawn }; + +export async function latestVersionAsync( + channel: Channel, + installer: Installer, + deps: AsyncLookupDeps = defaultDeps, +): Promise<string | null> { + if (installer === "source" || installer === "mise") return null; + let owner: PnpmGlobalOwner | null | undefined; + try { owner = installer === "pnpm" ? await deps.ownerFn() : undefined; } + catch { return null; } + if (installer === "pnpm" && !owner) return null; + const target = registrySpawnTarget(installer, ["view", `${PKG}@${channel}`, "version"], owner); + if (!target) return null; + + return new Promise(resolve => { + let child: ChildProcessWithoutNullStreams; + try { + child = deps.spawnFn(target.bin, target.args, { + stdio: ["pipe", "pipe", "pipe"], + windowsHide: true, + env: unprivilegedOwnershipMutationEnvironment(target.env ?? process.env), + ...target.options, + }) as ChildProcessWithoutNullStreams; + } catch { + resolve(null); + return; + } + child.stdin.end(); + let done = false; + let bytes = 0; + let stdout = ""; + let failed = false; + const finish = (version: string | null) => { + if (done) return; + done = true; + clearTimeout(timer); + resolve(version); + }; + const accept = (chunk: Buffer, capture: boolean) => { + bytes += chunk.length; + if (bytes > REGISTRY_OUTPUT_LIMIT) { + failed = true; + child.kill(); + } else if (capture) { + stdout += chunk.toString("utf8"); + } + }; + child.stdout.on("data", (chunk: Buffer) => accept(chunk, true)); + child.stderr.on("data", (chunk: Buffer) => accept(chunk, false)); + child.once("error", () => finish(null)); + child.once("close", code => { + const value = stdout.trim(); + finish(!failed && code === 0 && (channel === "latest" + ? /^\d+\.\d+\.\d+$/.test(value) + : /^\d+\.\d+\.\d+(?:-preview\.\d+)?$/.test(value)) + ? value : null); + }); + const timer = setTimeout(() => { + child.kill(); + finish(null); + }, deps.deadlineMs ?? REGISTRY_DEADLINE_MS); + }); +} +``` + +Implementation check: `ChildProcessWithoutNullStreams` is valid only with all three pipes; keep `stdio` as shown. If Bun's Worker constructor cannot load the TypeScript URL on a supported platform, the focused pnpm test must fail and B must replace it with a bundled Bun worker launch before shipping; never fall back to synchronous owner discovery in the handler. + +`tests/fixtures/pnpm-owner-stall-worker.ts` — NEW; complete file. It deliberately receives a message and never replies. It is loaded only by the deadline test; the empty-invocation production-worker test below proves the actual `pnpm-owner-worker.ts` can start and exchange a message without launching pnpm. + +```ts +onmessage = () => {}; +``` + +### `src/update/notify.ts` — MODIFY + +Export the existing interval (`src/update/notify.ts:20`) for the scheduler and add a single write-through function after `writeVersionCache` (`src/update/notify.ts:57-63`). The read occurs at commit time, after lookup completion. A dismissal follows only the exact same latest version; a different channel has no inherited dismissal. Atomic `writeVersionCache` remains the sole file writer. Since all new writes are synchronous and occur in one Bun event loop, their completion order is serialized; channel mismatch is rechecked by the next scheduler tick. No new persisted field is needed. + +```ts +export const REFRESH_INTERVAL_MS = 20 * 60 * 60 * 1000; + +export function writeFreshVersionCache(channel: Channel, latest: string, nowMs = Date.now()): void { + const previous = readVersionCache(channel); + writeVersionCache({ + latest_version: latest, + last_checked_at: new Date(nowMs).toISOString(), + dismissed_version: previous?.latest_version === latest && previous.dismissed_version === latest + ? latest : undefined, + tag: channel, + }); +} +``` + +Remove the original non-exported interval declaration, leaving exactly one. Replace lines 197–207 with the following compatibility body for the existing hidden CLI command, which is already detached and is not on the request loop: + +```ts +export async function refreshVersionCache(channel: Channel): Promise<void> { + const latest = latestVersion(channel); + if (latest) writeFreshVersionCache(channel, latest); +} +``` + +Remove only `triggerBackgroundRefreshIfStale(channel, cache);` at line 245. Keep `triggerBackgroundRefreshIfStale`, its imports, and the hidden `__refresh-version` dispatch exported for compatibility, but `maybeShowUpdatePrompt` no longer launches it. A normal interactive `ocx start` gets the same scheduler after bind. This prevents two local writers on each new startup. Leave the pre-bind update prompt and its dismissal untouched. + +### `src/update/refresh-scheduler.ts` — NEW; complete file + +> **wp1 Check amendments (authoritative over the block below):** in-flight entries carry their `epoch`, and an automatic tick joins only a same-generation flight (a stopped listener's flight cannot write for the new one); every flight takes a per-channel sequence number and writes only when it is newer than the last flight that wrote that channel, so an explicit caller that kept an old flight writable across stop/start cannot overwrite a newer result. Regressions: `restart does not join the stopped listener's pending lookup` and `an older explicit flight cannot overwrite a newer generation's write` in `tests/update/update-refresh.test.ts`. The shipped file is `src/update/refresh-scheduler.ts` at commit 1. + +The same coordinator serves background and explicit checks. `check` coalesces per channel. Explicit checks bypass staleness/backoff, but join an in-flight lookup; a failed lookup returns the existing `latest_unavailable` response via `checkForUpdate`. The flight records explicit interest before awaiting. Its successful completion writes through when any caller explicitly requested the result, even if `stop()` changed the generation meanwhile; only a purely automatic late result is suppressed. This keeps one cache write for concurrent explicit callers. `start` is a synchronous timer registration and queues the first due lookup in a microtask; it does no filesystem or registry work between `Bun.serve` and Lab activation. The module singleton uses paired start/stop references: each successful `startServer` call registers one reference, and stopping one listener cannot disarm another. The scheduler has injected clock/timer/lookup dependencies for deterministic tests. + +```ts +import { checkForUpdate, type UpdateCheckResult } from "./job"; +import { currentVersion, defaultUpdateTag, detectInstall, detectInstallOwnership, miseUpdateCommand, type Channel, type Installer } from "./index"; +import { isSourceBuildVersion, readVersionCache, REFRESH_INTERVAL_MS, writeFreshVersionCache } from "./notify"; +import { latestVersionAsync } from "./async-check"; + +export const STALENESS_TICK_MS = 60 * 60 * 1000; +export const RETRY_BASE_MS = 60_000; +export const RETRY_CAP_MS = STALENESS_TICK_MS; + +export interface RefreshDeps { + now: () => number; + lookup: (channel: Channel, installer: Installer) => Promise<string | null>; + current: () => string; + install: () => Installer; + ownership: typeof detectInstallOwnership; + guidance: typeof miseUpdateCommand; + read: typeof readVersionCache; + write: typeof writeFreshVersionCache; + setTimer: typeof setTimeout; + clearTimer: typeof clearTimeout; + disabled: () => boolean; +} + +const defaults: RefreshDeps = { + now: Date.now, + lookup: latestVersionAsync, + current: currentVersion, + install: detectInstall, + ownership: detectInstallOwnership, + guidance: miseUpdateCommand, + read: readVersionCache, + write: writeFreshVersionCache, + setTimer: setTimeout, + clearTimer: clearTimeout, + disabled: () => process.env.OCX_DISABLE_UPDATE_CHECK === "1", +}; + +export function createRefreshScheduler(deps: RefreshDeps = defaults) { + const inFlight = new Map<Channel, { task: Promise<string | null>; markExplicit: () => void }>(); + let timer: ReturnType<typeof setTimeout> | undefined; + let running = false; + let starts = 0; + let failures = 0; + let retryAt = 0; + let generation = 0; + + const eligible = () => { + const version = deps.current(); + const installer = deps.install(); + return !deps.disabled() && version !== "?" && !isSourceBuildVersion(version) + && installer !== "source" && installer !== "mise"; + }; + const channel = () => defaultUpdateTag(deps.current()); + const stale = (tag: Channel) => { + const checked = Date.parse(deps.read(tag)?.last_checked_at ?? ""); + return !Number.isFinite(checked) || checked > deps.now() || deps.now() - checked >= REFRESH_INTERVAL_MS; + }; + const schedule = (delay: number) => { + if (!running) return; + if (timer) deps.clearTimer(timer); + timer = deps.setTimer(() => { timer = undefined; void tick(); }, delay); + timer.unref?.(); + }; + const lookup = (tag: Channel, automatic: boolean): Promise<string | null> => { + const existing = inFlight.get(tag); + if (existing) { + if (!automatic) existing.markExplicit(); + return existing.task; + } + const epoch = generation; + let explicitInterest = !automatic; + const task = Promise.resolve().then(() => deps.lookup(tag, deps.install())).then(latest => { + if (latest && (explicitInterest || (automatic && running && epoch === generation))) { + deps.write(tag, latest, deps.now()); + } + if (automatic && running && epoch === generation) { + failures = latest ? 0 : failures + 1; + retryAt = latest ? 0 : deps.now() + Math.min(RETRY_CAP_MS, RETRY_BASE_MS * 2 ** Math.min(failures - 1, 6)); + } + return latest; + }).catch(() => { + if (automatic && running && epoch === generation) { + failures += 1; + retryAt = deps.now() + Math.min(RETRY_CAP_MS, RETRY_BASE_MS * 2 ** Math.min(failures - 1, 6)); + } + return null; + }).finally(() => { if (inFlight.get(tag)?.task === task) inFlight.delete(tag); }); + inFlight.set(tag, { task, markExplicit: () => { explicitInterest = true; } }); + return task; + }; + const tick = async () => { + if (!running || !eligible()) return; + const tag = channel(); + if (stale(tag) && deps.now() >= retryAt) await lookup(tag, true); + schedule(retryAt > deps.now() ? Math.min(STALENESS_TICK_MS, retryAt - deps.now()) : STALENESS_TICK_MS); + }; + return { + start() { + starts += 1; + if (starts !== 1 || !eligible()) return; + running = true; + schedule(0); + }, + stop() { + if (starts === 0) return; + starts -= 1; + if (starts !== 0) return; + running = false; + generation += 1; + if (timer) deps.clearTimer(timer); + timer = undefined; + }, + async check(tag: Channel): Promise<UpdateCheckResult> { + const installer = deps.install(); + const latest = installer === "source" || installer === "mise" ? null : await lookup(tag, false); + return checkForUpdate(tag, { + currentVersion: deps.current, + detectInstall: deps.install, + detectInstallOwnership: deps.ownership, + miseUpdateCommand: deps.guidance, + latestVersion: () => latest, + }); + }, + }; +} + +export const packageRefresh = createRefreshScheduler(); +``` + +The `detectInstallOwnership`/`miseUpdateCommand` inputs preserve the existing mise guidance (`src/update/index.ts:70-80`). An explicit caller marks an automatic flight before it settles, so the flight writes exactly once even after the final listener stops. A purely automatic flight retains the generation guard. `CACHE_MAX_AGE_MS` is defined once in `notify.ts` below; the badge imports it. + +### `src/update/badge.ts` — MODIFY + +Replace the stale top comment (`src/update/badge.ts:1-14`) with: “The badge reads the package cache only. The server scheduler and explicit checks produce it; GET never launches a lookup. A missing, wrong-channel or 40-hour-old cache reports unknown.” Define the constant in `notify.ts` beside `REFRESH_INTERVAL_MS` and import it from `./notify` to avoid a badge→scheduler→job→notify→badge runtime cycle. The final declaration is: + +```ts +export const CACHE_MAX_AGE_MS = 40 * 60 * 60 * 1000; +``` + +Add optional `now?: () => number` to `UpdateBadgeDeps` (existing callers such as `tests/update/update-mise.test.ts:324-332` remain valid), `now: Date.now` to `defaultDeps`, and update the badge test fixture. Immediately after `if (!cache) return base;` insert: + +```ts + const checked = Date.parse(cache.last_checked_at); + const now = deps.now?.() ?? Date.now(); + if (!Number.isFinite(checked) || checked > now || now - checked >= CACHE_MAX_AGE_MS) return base; +``` + +Import `CACHE_MAX_AGE_MS` from `./notify` and return `base` so stale `latestVersion` is not offered as installable. No route or JSON shape changes. `tests/update/update-badge.test.ts:58-64` pins the fixture's existing dependency keys; keep those keys unchanged because `now` is optional and defaults at the read site. + +### `src/server/management/config-routes.ts` — MODIFY + +In `src/server/management/context.ts`, add the two imports immediately after its existing `OcxConfig` type import (`:1`): + +```ts +import type { Channel } from "../../update/index"; +import type { UpdateCheckResult } from "../../update/job"; +``` + +Inside `ManagementApiDeps`, immediately after `requestMetrics?` (`:51`), add: + +```ts +checkPackageUpdate?: (channel: Channel) => Promise<UpdateCheckResult>; +``` + +Production passes no seam and always uses the coordinator; a route test injects a delayed package result without source-checkout detection or a real registry child. In the check branch (`src/server/management/config-routes.ts:750-757`), keep current invalid-tag validation and channel normalization, replace `checkForUpdate` import and final return: + +```ts +const { normalizeUpdateChannel } = await import("../../update/job"); +const { packageRefresh } = await import("../../update/refresh-scheduler"); +// existing rawTag validation unchanged +return jsonResponse(await (deps.checkPackageUpdate ?? packageRefresh.check)(normalizeUpdateChannel(rawTag))); +``` + +In the run branch (`src/server/management/config-routes.ts:759-777`), keep body parsing, validation and existing `UpdateJobError` catch. Inside its existing `try`, replace the return with: + +```ts +const channel = normalizeUpdateChannel(body.tag as string | undefined); +const { packageRefresh } = await import("../../update/refresh-scheduler"); +const checked = await (deps.checkPackageUpdate ?? packageRefresh.check)(channel); +return jsonResponse({ ok: true, job: startUpdateJob(channel, body.restart !== false, { + checkForUpdateFn: () => checked, +}) }); +``` + +The checked result is already built by `checkForUpdate`, so `startUpdateJob` still owns its existing 409 lock and worker creation. The detached worker continues its own synchronous integrity/update logic (`src/update/job.ts:619-680`). A registry failure gives `latest_unavailable`; source and mise preserve their existing `UpdateJobError` codes. An explicit preview check may replace the single-file cache; the scheduler's default-channel `readVersionCache` mismatch causes a new lookup on its next hourly tick. Document that maximum mismatch window; if immediate repair is required, B may queue a default-channel tick after an explicit opposite-channel write without changing the badge GET. + +### `src/server/index.ts` — MODIFY + +Add one import near the existing startup lifecycle imports (`src/server/index.ts:63-64`): + +```ts +import { packageRefresh } from "../update/refresh-scheduler"; +``` + +Each `startServer` owns exactly one start reference. The existing `server.stop` wrapper can be called twice, so release that reference once. Immediately before `Object.defineProperty(server, "stop", {` (`src/server/index.ts:749`), insert: + +```ts +let packageRefreshStopped = false; +``` + +Inside that wrapper before `packageTreeIntegrity.dispose()` (`src/server/index.ts:757`), insert: + +```ts +if (!packageRefreshStopped) { + packageRefreshStopped = true; + packageRefresh.stop(); +} +``` + +Immediately before `return server;` (`src/server/index.ts:877`), after Lab activation and reset-credit activation, add `packageRefresh.start();`. Both methods are synchronous. Net growth is 7 lines including the import, below the 893-line source cap from the current 883 lines (10 lines headroom). Avoid any await between `Bun.serve` (`src/server/index.ts:690`) and Lab activation (`src/server/index.ts:864`); `tests/lab/core-lab-boundary.test.ts:1052` checks this boundary. On a bind or activation failure, no scheduler reference was registered; the return immediately follows the start call. + +### Tests and fixture map — MODIFY/NEW + +`tests/update/update-refresh.test.ts` — NEW; complete file. This virtual clock exercises startup, staleness, coalescing, backoff, cancellation, and the opt-out without a live registry. Windows invocation behavior is already covered by `tests/update/update-pnpm.test.ts:84-108` and its owner-binding tests; B also checks the same `registrySpawnTarget` is used. + +```ts +import { describe, expect, test } from "bun:test"; +import { EventEmitter } from "node:events"; +import { PassThrough } from "node:stream"; +import { createRefreshScheduler, RETRY_BASE_MS, STALENESS_TICK_MS, type RefreshDeps } from "../../src/update/refresh-scheduler"; +import { latestVersionAsync, pnpmOwner, REGISTRY_DEADLINE_MS, REGISTRY_OUTPUT_LIMIT } from "../../src/update/async-check"; +import type { VersionCache } from "../../src/update/notify"; +import type { Channel, Installer } from "../../src/update/index"; + +function fixture(installer: Installer = "npm", disabled = false, lookupFn?: RefreshDeps["lookup"]) { + let now = 1_700_000_000_000; + let nextId = 0; + const timers = new Map<number, { at: number; run: () => void }>(); + const cache = new Map<Channel, VersionCache>(); + const calls: Channel[] = []; + const replies: Array<Promise<string | null>> = []; + const writes: VersionCache[] = []; + const deps: RefreshDeps = { + now: () => now, + current: () => "2.7.43", + install: () => installer, + ownership: () => ({ installer, owner: null }) as ReturnType<RefreshDeps["ownership"]>, + guidance: () => null, + disabled: () => disabled, + read: tag => cache.get(tag) ?? null, + write: (tag, latest, at) => { + const previous = cache.get(tag); + const value: VersionCache = { + tag, latest_version: latest, last_checked_at: new Date(at).toISOString(), + dismissed_version: previous?.latest_version === latest && previous.dismissed_version === latest + ? latest : undefined, + }; + cache.set(tag, value); + writes.push(value); + }, + lookup: async (tag, activeInstaller) => { + calls.push(tag); + return lookupFn ? lookupFn(tag, activeInstaller) : replies.length ? await replies.shift()! : "2.7.44"; + }, + setTimer: ((run: () => void, delay: number) => { + const id = ++nextId; + timers.set(id, { at: now + delay, run }); + return { id, unref() {} } as ReturnType<typeof setTimeout>; + }) as typeof setTimeout, + clearTimer: ((timer: ReturnType<typeof setTimeout>) => { + timers.delete((timer as unknown as { id: number }).id); + }) as typeof clearTimeout, + }; + async function advance(ms: number) { + now += ms; + for (const [id, timer] of [...timers]) { + if (timer.at <= now) { timers.delete(id); timer.run(); } + } + for (let index = 0; index < 8; index++) await Promise.resolve(); + } + // Bounded condition wait for chains longer than advance()'s eight fixed turns (the real async + // lookup's fake child settles over ~16 microtask turns). Fails loudly instead of hanging. + async function settle(done: () => boolean, turns = 64) { + for (let index = 0; index < turns && !done(); index++) await Promise.resolve(); + if (!done()) throw new Error(`condition not reached within ${turns} microtask turns`); + } + return { scheduler: createRefreshScheduler(deps), calls, replies, writes, cache, timers, advance, settle, now: () => now }; +} + +describe("package cache refresh", () => { + test("missing cache refreshes immediately and writes through", async () => { + const f = fixture(); + f.scheduler.start(); + await f.advance(0); + expect(f.calls).toEqual(["latest"]); + expect(f.cache.get("latest")?.latest_version).toBe("2.7.44"); + f.scheduler.stop(); + }); + + test("fresh cache is checked hourly and refreshed at 20 hours", async () => { + const f = fixture(); + f.cache.set("latest", { tag: "latest", latest_version: "2.7.43", last_checked_at: new Date(f.now()).toISOString() }); + f.scheduler.start(); + await f.advance(0); + for (let hour = 0; hour < 19; hour++) await f.advance(STALENESS_TICK_MS); + expect(f.calls).toHaveLength(0); + await f.advance(STALENESS_TICK_MS); + expect(f.calls).toEqual(["latest"]); + f.scheduler.stop(); + }); + + test("explicit checks coalesce per channel", async () => { + const f = fixture(); + let release!: (value: string | null) => void; + f.replies.push(new Promise(resolve => { release = resolve; })); + const a = f.scheduler.check("latest"); + const b = f.scheduler.check("latest"); + await Promise.resolve(); + expect(f.calls).toEqual(["latest"]); + release("2.7.44"); + expect((await a).latestVersion).toBe("2.7.44"); + expect((await b).latestVersion).toBe("2.7.44"); + expect(f.writes).toHaveLength(1); + }); + + test("background and explicit checks join the same channel flight", async () => { + const f = fixture(); + let release!: (value: string | null) => void; + f.replies.push(new Promise(resolve => { release = resolve; })); + f.scheduler.start(); + await f.advance(0); + const explicit = f.scheduler.check("latest"); + expect(f.calls).toEqual(["latest"]); + release("2.7.44"); + expect((await explicit).latestVersion).toBe("2.7.44"); + expect(f.writes).toHaveLength(1); + f.scheduler.stop(); + }); + + test("explicit interest writes a joined automatic result after stop", async () => { + const f = fixture(); + let release!: (value: string | null) => void; + f.replies.push(new Promise(resolve => { release = resolve; })); + f.scheduler.start(); + await f.advance(0); + const explicit = f.scheduler.check("latest"); + expect(f.calls).toEqual(["latest"]); + f.scheduler.stop(); + release("2.7.44"); + expect((await explicit).latestVersion).toBe("2.7.44"); + expect(f.cache.get("latest")?.latest_version).toBe("2.7.44"); + expect(f.writes).toHaveLength(1); + expect(f.timers.size).toBe(0); + }); + + test("one listener stopping leaves the other listener's refresh active", async () => { + const f = fixture(); + f.scheduler.start(); + f.scheduler.start(); + f.scheduler.stop(); + await f.advance(0); + expect(f.calls).toEqual(["latest"]); + expect(f.cache.get("latest")?.latest_version).toBe("2.7.44"); + expect(f.timers.size).toBe(1); + f.scheduler.stop(); + expect(f.timers.size).toBe(0); + }); + + test("different channels have separate flights", async () => { + const f = fixture(); + await Promise.all([f.scheduler.check("latest"), f.scheduler.check("preview")]); + expect(f.calls).toEqual(["latest", "preview"]); + expect(f.writes.map(value => value.tag)).toEqual(["latest", "preview"]); + }); + + test("failed lookup does not stamp and retries with exponential delay", async () => { + const f = fixture(); + f.replies.push(Promise.resolve(null), Promise.resolve(null), Promise.resolve("2.7.44")); + f.scheduler.start(); + await f.advance(0); + expect(f.cache.size).toBe(0); + await f.advance(RETRY_BASE_MS); + expect(f.calls).toHaveLength(2); + await f.advance(RETRY_BASE_MS * 2); + expect(f.calls).toHaveLength(3); + expect(f.cache.get("latest")?.latest_version).toBe("2.7.44"); + f.scheduler.stop(); + }); + + test("repeated failure caps retry at one hour", async () => { + const f = fixture(); + for (let n = 0; n < 9; n++) f.replies.push(Promise.resolve(null)); + f.scheduler.start(); + await f.advance(0); + for (const minutes of [1, 2, 4, 8, 16, 32, 60]) await f.advance(minutes * 60_000); + expect(f.calls).toHaveLength(8); + await f.advance(59 * 60_000); + expect(f.calls).toHaveLength(8); + await f.advance(60_000); + expect(f.calls).toHaveLength(9); + f.scheduler.stop(); + }); + + test("invalid and wrong-channel caches refresh immediately", async () => { + const invalid = fixture(); + invalid.cache.set("latest", { tag: "latest", latest_version: "2.7.44", last_checked_at: "bad" }); + invalid.scheduler.start(); + await invalid.advance(0); + expect(invalid.calls).toEqual(["latest"]); + invalid.scheduler.stop(); + const mismatch = fixture(); + mismatch.cache.set("preview", { tag: "preview", latest_version: "2.8.0-preview.1", last_checked_at: new Date(mismatch.now()).toISOString() }); + mismatch.scheduler.start(); + await mismatch.advance(0); + expect(mismatch.calls).toEqual(["latest"]); + mismatch.scheduler.stop(); + }); + + test("stop cancels timer and suppresses a late automatic write", async () => { + const f = fixture(); + let release!: (value: string | null) => void; + f.replies.push(new Promise(resolve => { release = resolve; })); + f.scheduler.start(); + await f.advance(0); + f.scheduler.stop(); + release("2.7.44"); + await f.advance(STALENESS_TICK_MS); + expect(f.writes).toHaveLength(0); + expect(f.timers.size).toBe(0); + }); + + test.each(["source", "mise"] as Installer[])("%s never starts automatic lookup", async installer => { + const f = fixture(installer); + f.scheduler.start(); + await f.advance(0); + expect(f.calls).toHaveLength(0); + expect(f.timers.size).toBe(0); + }); + + test("opt-out stops only automatic checks", async () => { + const f = fixture("npm", true); + f.scheduler.start(); + await f.advance(0); + expect(f.calls).toHaveLength(0); + expect((await f.scheduler.check("latest")).latestVersion).toBe("2.7.44"); + }); +}); + +test("an absent pnpm owner cannot reach the registry child", async () => { + let spawned = false; + expect(await latestVersionAsync("latest", "pnpm", { + ownerFn: async () => null, + spawnFn: (() => { spawned = true; throw new Error("unexpected child"); }) as never, + })).toBeNull(); + expect(spawned).toBe(false); +}); + +test("pnpm owner resolution failure is unavailable, not an unowned PATH lookup", async () => { + let spawned = false; + expect(await latestVersionAsync("latest", "pnpm", { + ownerFn: async () => { throw new Error("owner probe failed"); }, + spawnFn: (() => { spawned = true; throw new Error("unexpected child"); }) as never, + })).toBeNull(); + expect(spawned).toBe(false); +}); + +const CAN_RUN_BUN_WORKER = ["darwin", "linux", "win32"].includes(process.platform) + && typeof Worker === "function"; + +test.skipIf(!CAN_RUN_BUN_WORKER)("production pnpm worker starts and answers an empty invocation", async () => { + const worker = new Worker(new URL("../../src/update/pnpm-owner-worker.ts", import.meta.url).href); + let timeout: ReturnType<typeof setTimeout> | undefined; + try { + const answer = new Promise<unknown>((resolve, reject) => { + worker.onmessage = event => resolve(event.data); + worker.onerror = reject; + }); + worker.postMessage(""); + expect(await Promise.race([ + answer, + new Promise((_, reject) => { timeout = setTimeout(() => reject(new Error("pnpm worker did not reply")), 2_000); }), + ])).toBeNull(); + } finally { + if (timeout) clearTimeout(timeout); + await worker.terminate(); + } +}); + +test.skipIf(!CAN_RUN_BUN_WORKER)("pnpm worker deadline terminates an unresponsive worker", async () => { + const started = performance.now(); + const result = await pnpmOwner({ + workerUrl: new URL("../fixtures/pnpm-owner-stall-worker.ts", import.meta.url), + deadlineMs: 25, + invoked: "held-shim", + }); + expect(result).toBeNull(); + expect(performance.now() - started).toBeLessThan(2_000); +}); + +function fakeChild() { + return Object.assign(new EventEmitter(), { + stdin: new PassThrough(), stdout: new PassThrough(), stderr: new PassThrough(), + killed: false, + kill() { this.killed = true; this.emit("close", null); return true; }, + }); +} + +// Each named failure enters through the scheduler's actual lookup dependency. +// The injected write seam is the version.json writer; empty writes/cache mean no stamp. +for (const scenario of [ + { name: "child error event", output: "", exitCode: 0, error: true }, + { name: "nonzero child exit", output: "2.7.44\n", exitCode: 1, error: false }, + { name: "malformed zero-exit output", output: "not-a-version\n", exitCode: 0, error: false }, + { name: "empty zero-exit output", output: "", exitCode: 0, error: false }, + { name: "multiline zero-exit output", output: "2.7.44\n2.7.45\n", exitCode: 0, error: false }, +]) { + test(`registry ${scenario.name} returns null and retries after backoff`, async () => { + const observed: Array<string | null> = []; + let spawned = 0; + const f = fixture("npm", false, async (tag, installer) => { + const result = await latestVersionAsync(tag, installer, { + ownerFn: async () => null, + spawnFn: (() => { + const child = fakeChild(); + const attempt = ++spawned; + queueMicrotask(() => { + if (attempt === 1) { + if (scenario.output) child.stdout.write(scenario.output); + if (scenario.error) child.emit("error", new Error("registry child failed")); + child.emit("close", scenario.exitCode); + } else { + child.stdout.write("2.7.44\n"); + child.emit("close", 0); + } + }); + return child; + }) as never, + }); + observed.push(result); + return result; + }); + f.scheduler.start(); + await f.advance(0); + await f.settle(() => f.timers.size === 1); + expect(observed).toEqual([null]); + expect(f.calls).toEqual(["latest"]); + expect(f.writes).toHaveLength(0); + expect(f.cache.size).toBe(0); + expect(f.timers.size).toBe(1); + expect([...f.timers.values()].map(timer => timer.at)).toEqual([f.now() + RETRY_BASE_MS]); + await f.advance(RETRY_BASE_MS - 1); + expect(f.calls).toHaveLength(1); + await f.advance(1); + await f.settle(() => f.writes.length === 1); + expect(f.calls).toEqual(["latest", "latest"]); + expect(observed).toEqual([null, "2.7.44"]); + expect(f.writes).toHaveLength(1); + expect(f.cache.get("latest")?.latest_version).toBe("2.7.44"); + f.scheduler.stop(); + }); +} + +test("registry output is bounded and killed without leaking text", async () => { + const child = fakeChild(); + const result = latestVersionAsync("latest", "npm", { + ownerFn: async () => null, + spawnFn: (() => child) as never, + }); + child.stderr.write(Buffer.alloc(REGISTRY_OUTPUT_LIMIT + 1)); + expect(await result).toBeNull(); + expect(child.killed).toBe(true); +}); + +test("registry deadline kills an unresponsive child", async () => { + expect(REGISTRY_DEADLINE_MS).toBe(12_000); + const child = fakeChild(); + const result = latestVersionAsync("latest", "npm", { + ownerFn: async () => null, + spawnFn: (() => child) as never, + deadlineMs: 1, + }); + expect(await result).toBeNull(); + expect(child.killed).toBe(true); +}); +``` + +`tests/update/update-notify.test.ts` — MODIFY: add a check that the pre-bind prompt still reads cache and no longer invokes `triggerBackgroundRefreshIfStale`, and add same-version/new-version/channel-mismatch dismissal persistence assertions. + +`tests/update/update-badge.test.ts` — MODIFY: add injected `now` to helper (`tests/update/update-badge.test.ts:6-20`), test 39h59m known, 40h unknown, invalid/future timestamp unknown, and assert repeated read has no lookup hook. The final guard is: + +```ts +const now = deps.now?.() ?? Date.now(); +if (!Number.isFinite(checked) || checked > now || now - checked >= CACHE_MAX_AGE_MS) return base; +``` + +`tests/update/update-job.test.ts` — MODIFY: add `startUpdateJob uses injected prechecked result without a second registry call` using the existing `StartUpdateJobDeps` seam (`src/update/job.ts:126-130,619-642`). `tests/server/sidebar-routes.test.ts` — MODIFY: add repeated GET with an injected/controlled cache and assert no cache timestamp change; existing badge route test is `tests/server/sidebar-routes.test.ts:99-112`. + +`tests/preload.ts` — MODIFY: after `process.env.OCX_TEST_PRELOAD_PID = String(process.pid);` (`tests/preload.ts:82`), add exactly: + +```ts +process.env.OCX_DISABLE_UPDATE_CHECK = "1"; +``` + +This is the main plan's test-started-server guard. It must run before the lock and before tests start listeners. Tests of scheduler eligibility inject `disabled` into `createRefreshScheduler` rather than mutating the preload environment. + +`tests/server/update-async-routes.test.ts` — NEW; complete file. The delayed test traverses real management dispatch with an injected npm result, waits for the lookup to be entered, then proves a timer completes while that request is unresolved. The injected seam is the only source of the answer, so a regression that bypasses it or synchronously calls `checkForUpdate` fails. It proves request-loop responsiveness at the management dispatch boundary; it does not claim a live listener `/healthz` result. The source-checkout cases preserve existing guidance. + +```ts +import { describe, expect, test } from "bun:test"; +import { handleManagementAPI } from "../../src/server/management-api"; +import type { ManagementApiDeps } from "../../src/server/management/context"; +import type { OcxConfig } from "../../src/types"; + +const config = { port: 10100, defaultProvider: "openai", providers: {} } as OcxConfig; + +async function call(method: string, path: string, body?: object, deps: ManagementApiDeps = {}) { + const url = new URL(`http://127.0.0.1:10100${path}`); + const request = new Request(url, { + method, + headers: { host: "127.0.0.1:10100", ...(body ? { "content-type": "application/json" } : {}) }, + ...(body ? { body: JSON.stringify(body) } : {}), + }); + return handleManagementAPI(request, url, config, deps, "admin-token"); +} + +describe("asynchronous package update routes", () => { + test("non-source check leaves the event loop responsive while lookup is pending", async () => { + let lookupEntered!: () => void; + const entered = new Promise<void>(resolve => { lookupEntered = resolve; }); + let release!: () => void; + const gate = new Promise<void>(resolve => { release = resolve; }); + let settled = false; + const pending = call("GET", "/api/update/check?tag=latest", undefined, { + checkPackageUpdate: async channel => { + expect(channel).toBe("latest"); + lookupEntered(); + await gate; + return { + currentVersion: "2.7.43", latestVersion: "2.7.44", channel, + installer: "npm", updateAvailable: true, canUpdate: true, + command: "npm install -g @bitkyc08/opencodex@2.7.44", releaseNotesUrl: "https://example.test/releases", + }; + }, + }).then(response => { settled = true; return response; }); + try { + await entered; + await new Promise<void>(resolve => setTimeout(resolve, 0)); + expect(settled).toBe(false); + } finally { + release(); + } + const response = await pending; + expect(response?.status).toBe(200); + const body = await response!.json() as Record<string, unknown>; + expect(body.installer).toBe("npm"); + expect(body.latestVersion).toBe("2.7.44"); + }); + + test("source checkout check returns guidance without registry work", async () => { + const response = await call("GET", "/api/update/check?tag=latest"); + expect(response?.status).toBe(200); + const body = await response!.json() as Record<string, unknown>; + expect(body.installer).toBe("source"); + expect(body.canUpdate).toBe(false); + }); + + test("source checkout run rejects a worker", async () => { + const response = await call("POST", "/api/update/run", { tag: "latest", restart: false }); + expect(response?.status).toBe(409); + const body = await response!.json() as Record<string, unknown>; + expect(body.code).toBe("source_checkout"); + }); +}); +``` + +Register both new tests: + +```json +// scripts/test-layout/layout.json — explicit object, alphabetical neighbors +"update-async-routes.test.ts": "server", +"update-refresh.test.ts": "update", +``` + +```json +// tests/fixtures/test-layout-expected.json — top-level object, alphabetical neighbors +"update-async-routes.test.ts": "server", +"update-refresh.test.ts": "update", +``` + +The snippets are JSON member edits, not standalone JSON documents; preserve commas/indentation as adjacent entries. `tests/test-layout-tooling.test.ts` checks both registries. No existing test must be moved to break the file-size ratchet. + +### Structure and docs-site — MODIFY + +`structure/runtime.md` is exactly 600/600 lines (`structure/manifest.json` sets `sizeBudgetLines: 600`). Make **one line for one line** replacement at its support row, and no other insertion or deletion in this file. Before (current `structure/runtime.md:159`): + +```md +| Support | `src/lib/`, `src/storage/`, `src/usage/`, `src/update/`, `src/generated/` | +``` + +After (still one line; link the detailed owner contract rather than adding prose here): + +```md +| Support | `src/lib/`, `src/storage/`, `src/usage/`, `src/update/` ([package refresh](ops/service-and-sidecars.md#package-cache-refresh)), `src/generated/` | +``` + +`structure/ops/service-and-sidecars.md`: append the following section after the existing mise updater paragraph (`structure/ops/service-and-sidecars.md:248`), which is its current final paragraph. All new scheduler prose belongs here: + +```md +## Package cache refresh + +`src/update/refresh-scheduler.ts` owns the package cache timer and per-channel singleflight for the running proxy. Eligible npm, pnpm and Bun installs refresh missing or 20-hour-stale `version.json` after bind, check staleness hourly and retry failures with bounded backoff. Each server start owns one scheduler reference; the last matching stop disarms the timer. A stopped automatic lookup cannot write a late result, but an explicit check joining that lookup marks explicit interest and writes its successful result even if the last listener stops before it resolves. Source/mise installs and `OCX_DISABLE_UPDATE_CHECK=1` do not start automatic lookup; explicit requests remain available. + +`src/update/async-check.ts` uses the existing owner-bound registry target with a bounded asynchronous child; pnpm owner discovery runs in `src/update/pnpm-owner-worker.ts` off the request loop. `src/update/notify.ts` writes successful results atomically and preserves a dismissal only for the same channel and version. The interactive pre-bind prompt reads the cache and does not launch a second detached refresh. `src/update/badge.ts` only reads the cache and reports unknown at 40 hours. +``` + +Verify the runtime line delta is zero (`wc -l structure/runtime.md` remains 600), then run `bun run structure:check` after the implementation edits. wp4 may separately replace the `src/tray/` surfaces row, also one line for one line; it must preserve this support-row link. + +`structure/gui-and-management-api.md`: replace the Updates table cell (`structure/gui-and-management-api.md:168`) with its existing job/PID sentences plus: “`GET /api/update/check` and `POST /api/update/run` await one per-channel asynchronous registry lookup and write successful results through to the package cache; run passes that result to the job starter. `GET /api/update/badge` only reads the cache and reports unknown after 40 hours, on missing cache, or on channel mismatch.” Keep its existing `src/server/management/sidebar-routes.ts` owner row (`structure/gui-and-management-api.md:195`). These are present-tense changes for the implementation commit; do not land them ahead of source. + +`docs-site/src/content/docs/reference/management-api.md`: change the check row (`:268`) purpose to “Asynchronously check the `latest` or `preview` package channel and refresh the package cache on success”; change the run row (`:269`) purpose to “Asynchronously check a fresh package version, then start an update job, optionally followed by restart”; change the badge row (`:547`) purpose to “Read cached package badge state without a registry lookup; missing, wrong-channel or 40-hour-old cache returns `unknown: true`.” Add one short paragraph below that row: “The proxy checks an eligible package install after startup when its cache is missing or older than 20 hours, and checks freshness hourly. Set `OCX_DISABLE_UPDATE_CHECK=1` to disable automatic checks. Explicit check and run requests still work.” + +Apply semantically identical row and paragraph changes to the existing siblings `docs-site/src/content/docs/{fr,ja,ko,ru,tr,zh-cn,zh-tw}/reference/management-api.md`; translate “automatic checks only; explicit requests still work” and the 20-hour/40-hour values in each locale. Do not change absent de/vi siblings or any GUI catalog: no `gui/src/i18n/{en,de,fr,ja,ko,ru,tr,vi,zh,zh-TW}.ts` key is added. `docs-site/AGENTS.md` requires the docs build after implementation. + +## Field/value chains (PLAN-FIELD-CHAIN-01) + +| Value or state | Creation | Serialization | Deserialization | Every consumer | +| --- | --- | --- | --- | --- | +| `OCX_DISABLE_UPDATE_CHECK=1` | Operator environment; new guard in `src/update/refresh-scheduler.ts` | N/A, environment string | `process.env` equality in scheduler | Scheduler `start`/`tick`; explicit check intentionally ignores it | +| Per-channel flight `{ task, markExplicit }`, failures, retryAt, generation, start references | `createRefreshScheduler` | N/A, process-local only | N/A | `lookup` marks explicit interest before awaiting; successful flight writes once despite a later stop; `tick`/`stop` gate purely automatic writes; paired `start`/`stop` retain the timer until the final server stops | +| Fresh `latest_version`, `last_checked_at`, `dismissed_version`, `tag` | `writeFreshVersionCache` in `src/update/notify.ts` | Existing atomic JSON writer `src/update/notify.ts:57-63` | Existing `readVersionCache` at `src/update/notify.ts:40-55` | Prompt `src/update/notify.ts:152-160,244-247`; badge `src/update/badge.ts:66-75`; scheduler stale test; wp4 npm tray reads badge later | +| `unknown: true` after 40 hours | Badge guard in `src/update/badge.ts` | Existing `jsonResponse` via `src/server/management/sidebar-routes.ts:100-103` | Dashboard's existing badge JSON consumer; wp2 desktop badge extends shape later | Sidebar indicator; wp4 tray uses same badge scalar. No new enum value or response field | +| Async `UpdateCheckResult` | `checkForUpdate` with injected resolved latest in `src/update/refresh-scheduler.ts` | Existing check JSON and job-state JSON in `src/server/management/config-routes.ts:750-777`, `src/update/job.ts:649-663` | Existing management client/GUI request flow; no schema revision | `/api/update/check`, `startUpdateJob`, status polling, GUI package update dialog; wp3 desktop flow branches away | +| Injected `checkPackageUpdate` function | Test supplies `ManagementApiDeps` in `src/server/management/context.ts`; production leaves it absent | N/A, process-local function only | N/A | Both check and run branches in `src/server/management/config-routes.ts`; production fallback is `packageRefresh.check` | +| Worker pnpm owner | `resolveCurrentPnpmGlobalOwner` in `src/update/pnpm-owner-worker.ts` | Structured clone over worker message, no disk | `pnpmOwner` event handler in `src/update/async-check.ts` | `registrySpawnTarget` pnpm branch; no API response or persistent consumer | + +Search B must recheck all existing `VersionCache`, `UpdateBadge`, `UpdateCheckResult`, `latestVersion`, `readVersionCache`, and each channel literal consumer; the relevant current definitions are `src/update/notify.ts:23-29`, `src/update/badge.ts:19-36`, `src/update/job.ts:82-92`, and `src/update/index.ts:61-62`. + +## Conditional activation matrix (C-ACTIVATION-GROUNDING-01) + +| Guard/fallback | Test activation | Observable result | +| --- | --- | --- | +| source/mise/unknown version or opt-out | Inject each installer/version/env into scheduler | No timer or lookup; explicit check retains source/mise guidance | +| missing, invalid timestamp, wrong channel, ≥20h stale | Seed temp cache for each | Immediate queued lookup; otherwise hourly timer | +| concurrent same channel / different channels | Hold injected lookup promises | One call for same channel, two for distinct channels | +| child `error` event | `registry child error event returns null and retries after backoff` emits `error` then `close` | First lookup null, no `version.json` write, retry at `RETRY_BASE_MS` and success on the next child | +| nonzero child exit | `registry nonzero child exit returns null and retries after backoff` emits valid stdout then closes 1 | First lookup null, no `version.json` write, retry at `RETRY_BASE_MS` and success on the next child | +| malformed, empty, multiline zero-exit output | `registry malformed zero-exit output returns null and retries after backoff`, `registry empty zero-exit output returns null and retries after backoff`, and `registry multiline zero-exit output returns null and retries after backoff` each close 0 | Each first lookup null with no `version.json` write, then retries at `RETRY_BASE_MS` and succeeds | +| >4 KiB output / 12s deadline | `registry output is bounded and killed without leaking text`; `registry deadline kills an unresponsive child` | Null lookup; child killed on limit/deadline | +| pnpm ownership absent/worker error/timeout | Fake worker or run worker with invalid owner | Null result; no unowned PATH query | +| repeated failure / recovery | Resolve null 7 times then valid version | Delay 1,2,4,8,16,32,60m cap; success resets failure count | +| stop during timer or lookup | Stop before timer, then while held lookup settles | No new automatic lookup and no late automatic cache write | +| automatic lookup joined by explicit check, then stop | Start held automatic lookup, call explicit check, stop final listener, resolve version | Explicit result is returned and persisted once to `version.json` despite changed generation; no timer remains | +| two server start references in one process | Start scheduler twice, stop once, advance due timer; stop second | Refresh still runs after first stop; final stop disarms timer; repeated stop cannot underflow references | +| explicit lookup when scheduler stopped/opted out | Invoke `packageRefresh.check` with the scheduler fixture | User request succeeds and writes cache | +| non-source route with delayed lookup | Inject npm `checkPackageUpdate` through `ManagementApiDeps`, call `/api/update/check`, run a timer before releasing the lookup | Timer fires and route remains pending; after release HTTP 200 includes npm version; no registry process | +| Bun pnpm worker startup/deadline | On macOS/Linux/Windows with Worker, load production TS worker with empty invocation; load stall fixture with 25 ms deadline | Production worker answers null; stalled worker resolves null before 2 s and is terminated; unsupported platforms skip explicitly | +| same version dismissal / new version / channel switch | Seed dismissed cache and resolve each version/channel | Dismissal kept only for identical version/channel; other result clears it | +| fresh vs 40h badge boundary | Seed 39h59m and 40h timestamps | Known then `unknown: true`, `latestVersion: null`, no registry process | +| `/api/update/run` unavailable/already running | Resolve null or hold existing running job | Existing 409 code; no worker launched | +| no await before Lab activation | Run existing Lab source guard | `tests/lab/core-lab-boundary.test.ts` passes | + +## Ratchet and verification + +At this HEAD `src/server/index.ts` is 883/893 (`tests/fixtures/file-size-baseline.json:34`), leaving 10 lines; planned net +7 makes 890/893. Every other grown TS file is absent from the named baseline and uses the default 2,000-line ceiling: `src/update/index.ts` 883/2000, `notify.ts` 270/2000, `badge.ts` 75/2000, `src/server/management/config-routes.ts` 1112/2000, `src/server/management/context.ts` 153/2000, `tests/update/update-notify.test.ts` 165/2000, `update-badge.test.ts` 74/2000, `update-job.test.ts` 1943/2000, and `tests/server/sidebar-routes.test.ts` 304/2000. The new TS files start at 0/2000. `update-job.test.ts` has 57 lines to the default ceiling; if the new case exceeds it, place it in `update-refresh.test.ts`. The file-size ratchet scans `.json` and `.md` as well as TypeScript (`scripts/file-size-ratchet.ts:4-20`): `scripts/test-layout/layout.json` is 1813/2000 and `tests/fixtures/test-layout-expected.json` is 1620/2000, both absent from the named baseline and subject to the default ceiling. Structure and docs-site Markdown also remain subject to the ratchet and their separate checks; this `devlog/` plan is excluded from the ratchet. `structure/runtime.md` is 600/600 and must remain 600 after its one-line replacement; `structure/ops/service-and-sidecars.md` is 248/600 before its new section. Recount against the shared tree immediately before B. No cap is raised. `src/update/job.ts` remains unchanged at 1993 lines, preserving its six-line headroom. + +Run after B: `bun test tests/update/update-refresh.test.ts tests/update/update-notify.test.ts tests/update/update-badge.test.ts tests/update/update-job.test.ts tests/update/update-pnpm.test.ts tests/server/update-async-routes.test.ts tests/server/sidebar-routes.test.ts tests/lab/core-lab-boundary.test.ts tests/test-layout-tooling.test.ts`; `bun run typecheck`; `bun run structure:check`; `bun run privacy:scan`; `cd docs-site && bun install --frozen-lockfile && bun run build` (docs-site requirement). Test command directly names target tests; `typecheck` reads `src/**/*.ts` via `tsconfig.json`, not prose; structure checker reads `structure/**` through `scripts/structure-ssot.ts`; privacy scan reads tracked source and devlog; docs build reads docs-site pages. Exact-head hosted CI remains the PR gate. Do not claim unrun gates passed. + +Commands rerun during this docs-only revision, before B and after root dependencies were installed: + +| Command | Exit | Result and scope | +| --- | ---: | --- | +| `bun test tests/update/update-notify.test.ts tests/update/update-badge.test.ts tests/update/update-job.test.ts tests/update/update-pnpm.test.ts` | 0 | 124 pass, 0 fail; existing source behavior only | +| `bun test tests/server/sidebar-routes.test.ts tests/lab/core-lab-boundary.test.ts` | 0 | 37 pass, 0 fail; existing route and Lab guards | +| `bun test tests/test-layout-tooling.test.ts` | 0 | 16 pass, 0 fail; current test map only | +| `bun run typecheck` | 0 | Current source typechecks; proposed new code does not exist yet | +| `bun run structure:check` | 0 | Current structure files pass; this plan's future net-zero edit is not applied | +| `bun run privacy:scan` | 0 | Current tracked paths pass; this document is still untracked and must be scanned after staging by main | + +Additional verification for the fake-child and ratchet correction in this docs-only pass: + +| Verifier | Exit | Result and scope | +| --- | ---: | --- | +| `Bun.Transpiler({ loader: "ts" }).transformSync` on the fenced `tests/update/update-refresh.test.ts` block | 0 | Proposed 350-line test block parses; future source and behavior are not yet executable | +| `bun test tests/ci-workflows/file-size-ratchet.test.ts tests/test-layout-tooling.test.ts` | 0 | 25 pass, 0 fail; existing scanner and layout registries only, not this plan | +| `wc -l scripts/test-layout/layout.json tests/fixtures/test-layout-expected.json` | 0 | 1813 and 1620 lines, both below the default 2000-line ceiling | + +The earlier pre-install test load errors and docs-site `astro` failure are superseded by the fresh results above for root tests and typecheck. `docs-site/node_modules` is still absent, so the docs build remains deferred until the implementation cycle installs its dependencies; it would also write outside this leaf's one-document scope. Tests naming `update-refresh.test.ts` and `update-async-routes.test.ts` run only after B creates them. No Bun behavior test reads this Markdown plan; review its code blocks and source anchors directly. + +## Risks and rollback + +The worker API and `spawn` event behavior must be tested under Bun on macOS and Windows. A spawned child can outlive `kill()` briefly; the 12-second timer settles the request even if `close` is late. A 12-second pnpm owner proof plus 12-second registry query can take up to 24 seconds end to end, yet neither blocks the request loop. Explicit preview cache writes can hide the default channel until its next hourly tick; route responses remain correct per channel. File writes remain atomic but are best-effort (`src/update/notify.ts:57-63`); write failure must not be reported as a successful persistent refresh in verification. An opt-out environment variable is a new public behavior and needs docs. Rollback commit 1 as one unit; legacy `version.json` schema is unchanged, so no migration or data deletion is needed. Do not weaken management auth, updater integrity checks, or the synchronous Lab activation boundary to make this work. + +## wp1 P revalidation (2026-09-24) + +Previous cycle (wp0) concluded: roadmap locked at ff8db308e6 after architect ALIGNED and audit PASS; next direction is to execute this document unchanged. Stale check at the start of wp1: `git diff 6c171aa5a6..HEAD` touches only this unit's devlog files, so every source anchor above is current; line counts match (src/server/index.ts 883/893, src/update/index.ts 883, notify.ts 270, badge.ts 75, config-routes.ts 1112, tests/update/update-job.test.ts 1943, structure/runtime.md 600/600, structure/ops/service-and-sidecars.md 248). No amendment. Architect consultation for this unit (000_plan.md) covers this document at revision 5; no design decision changes in this cycle. diff --git a/devlog/_plan/260924_update_indicator/020_phase2_desktop_state_icons.md b/devlog/_plan/260924_update_indicator/020_phase2_desktop_state_icons.md new file mode 100644 index 0000000000..be8a608e57 --- /dev/null +++ b/devlog/_plan/260924_update_indicator/020_phase2_desktop_state_icons.md @@ -0,0 +1,1306 @@ +# Commit 2 — desktop snapshot and tray update dot + +Goal: make the Tauri updater the authority for a desktop-only badge and put its pending-update blue dot on each available native tray. This is work-phase wp2, ordered after commit 1 (the package badge/cache contract in 010_phase1_package_cache.md) and before commit 3 (the app-origin install page). The branch carries the four commits in 000_plan.md; this commit is not a separately releasable desktop update flow. + +**IN:** bounded admin-token-only POST; process-local session snapshot; desktop badge projection; 60-second native heartbeat; embedded dashboard session URL and GUI poll; macOS AppKit dot; generated dotted Windows/Linux Tauri PNG; focused Bun, Rust, and icon tests; structure and user docs. **OUT:** package cache logic owned by commit 1, install page/IPC and click routing owned by commit 3, npm PowerShell tray owned by commit 4, Windows glyph redesign, persistence, automatic install, signature policy changes, and Lab-boundary imports. + +This document specifies the diff to apply in B. All paths are repository-relative. Existing symbol anchors were checked against HEAD b429895f4a: the raw-token principal is assigned in src/server/management-auth.ts:541-564 and exposed in src/server/management/context.ts:136-151; route dispatch and origin/body gates are src/server/management-api.ts:187-205,276-306; the current badge route is src/server/management/sidebar-routes.ts:100-103 and registry row is src/server/management/route-registry.ts:350-353; package badge shape is src/update/badge.ts:6-67. On desktop, the bound credential gate is desktop/src-tauri/src/proxy.rs:181-228, updater transitions are desktop/src-tauri/src/updater.rs:81-114, tray transitions are desktop/src-tauri/src/tray.rs:198-222,284-325, dashboard construction is desktop/src-tauri/src/startup.rs:1407-1438, and the existing NSStatusItem borrow is desktop/src-tauri/src/native_tray.rs:58-75. The Swift status-button lookup is app/Sources/NativeTray/Popover.swift:24-32. These anchors name APIs that exist; the new symbols below are additions. + +Additional existing-symbol anchors used in patches: jsonResponse is src/server/auth-cors.ts:267; defaultUpdateTag is src/update/index.ts:176-178; detectInstall/Channel are src/update/index.ts:61-67; useKeyedClientResource is gui/src/client-resource.ts:616; the present sidebar poll is gui/src/components/sidebar-github-row.tsx:67-76; desktop-shell detection is gui/src/lib/desktop-shell.ts:7-13; the Tauri user agent is desktop/src-tauri/src/window.rs:4-14; app.package_info() is used at desktop/src-tauri/src/menu.rs:28; the Popup destination and matcher are desktop/src-tauri/src/popup.rs:256-263,313-320; and the icon generator's render/produced/check chain is desktop/scripts/generate-icons.ts:62-145,157-175. The pinned Tauri set_icon API is called through the existing tray handle at desktop/src-tauri/src/tray.rs:104-107,463-465; B also compiles it against the checked-in Cargo.lock. The AppKit NSButtonCell.imageRect(forBounds:) call was checked with the swift verifier below. + +## File change map and executable patches + +The following are the entire commit-2 file set. DELETE: none. Binary NEW output is generated, not hand-edited. MODIFY rows with an insertion block mean insert at the named existing anchor; replacement blocks name the exact old expression. Commit 1 changed src/update/badge.ts: its 40-hour cache-age guard and `unknown` result remain package-only. Change only the installer type line shown here; the desktop store supplies the same seven-field response independently. + +| Operation | Path | Change | +| --- | --- | --- | +| NEW | src/update/desktop-badge.ts | Strict snapshot DTO, bounded process store, desktop badge projection; full content below. | +| MODIFY | src/update/badge.ts | Widen only UpdateBadge.installer to include desktop. | +| MODIFY | src/server/management/sidebar-routes.ts | Desktop GET branch and admin-token POST with bounded stream read. | +| MODIFY | src/server/management/route-registry.ts | Declare POST and truthful desktop-internal parity exemption. | +| MODIFY | desktop/src-tauri/src/proxy.rs | Bound POST JSON through existing identity-checked client. | +| MODIFY | desktop/src-tauri/src/updater.rs | Random session, serial snapshot publisher, state transitions and 60 s heartbeat. | +| MODIFY | desktop/src-tauri/src/lib.rs | Manage and start the snapshot publisher. | +| MODIFY | desktop/src-tauri/src/startup.rs | Carry session in embedded dashboard URL; wake publisher on bind. | +| MODIFY | desktop/src-tauri/src/native_tray.rs | Borrow status button on main thread for dot bridge. | +| MODIFY | app/Sources/NativeTray/Popover.swift | NSView dot, @_cdecl show/hide bridge, image-relative drawing. | +| MODIFY | desktop/src-tauri/src/tray.rs | Apply native/PNG indicator on updater/tray state and title refresh. | +| MODIFY | desktop/src-tauri/src/popup.rs | Preserve the session when the web tray opens the main dashboard. | +| MODIFY | desktop/scripts/generate-icons.ts | Generate and check a dotted PNG from tray/icon.svg plus SVG halo/dot. | +| NEW generated | desktop/src-tauri/icons/tray/icon-update.png | 44×44 RGBA, output of the generator; no hand-authored binary. | +| MODIFY | gui/src/lib/desktop-shell.ts | Strict session query reader and badge URL helper. | +| MODIFY | gui/src/components/sidebar-github-row.tsx | Desktop-session poll URL and 60 s cadence. | +| NEW | tests/update/update-desktop-badge.test.ts | Full test content below. | +| MODIFY | tests/server/sidebar-routes.test.ts | Principal, malformed ingress, absent/expiry/isolation route tests. | +| MODIFY | tests/server/management-route-registry.test.ts | Registry row and exemption check. | +| MODIFY | tests/cli/cli-capabilities.test.ts | Assert that the desktop-only POST is deliberately not a CLI capability. | +| MODIFY | tests/ci-workflows/build-desktop-icon-set.test.ts | Generated dotted variant size/source/colour declaration. | +| MODIFY | gui/tests/desktop-shell.test.ts | Desktop URL, missing session, browser isolation. | +| MODIFY | scripts/test-layout/layout.json | Explicit owner for the new Bun test. | +| MODIFY | tests/fixtures/test-layout-expected.json | Same explicit owner. | +| MODIFY | structure/desktop-shell.md; structure/companion.md; structure/gui-and-management-api.md; structure/runtime.md; structure/ops/service-and-sidecars.md | Current-state contracts below; the last two own src/update/. | +| MODIFY | docs-site/src/content/docs/reference/management-api.md; docs-site/src/content/docs/guides/desktop-app.md and their existing fr, ja, ko, ru, tr, zh-cn, zh-tw siblings | User-facing contract below. | + +### New src/update/desktop-badge.ts — full content + +The current UpdateBadge shape has currentVersion, latestVersion, channel, installer, canUpdate, updateAvailable and unknown (src/update/badge.ts:6-18). The commit-1 package reader returns `unknown: true` for missing, wrong-channel, invalid-time, future-time or 40-hour-stale cache; the desktop store must use its receipt TTL instead and must not read that cache. Keep precisely that response shape; do not include sessionId or phase in GET output. A session id is a routing nonce, not an install credential. The map is bounded even if clients continually change ids. Expiry is measured from *receipt*, so a six-hour updater check remains displayable while the heartbeat continues. + +~~~ts +import type { UpdateBadge } from "./badge"; +import { defaultUpdateTag } from "./index"; + +export type DesktopPhase = + | "idle" | "checking" | "available" | "current" + | "error" | "installing" | "install-failed"; + +export interface DesktopSnapshot { + sessionId: string; + currentVersion: string; + latestVersion: string | null; + available: boolean; + checkedAtMs: number | null; + phase: DesktopPhase; +} + +const SESSION = /^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/; +const VERSION = /^[0-9A-Za-z][0-9A-Za-z.+_-]{0,63}$/; +const PHASES = new Set<DesktopPhase>([ + "idle", "checking", "available", "current", "error", "installing", "install-failed", +]); +export const DESKTOP_SNAPSHOT_TTL_MS = 180_000; +export const DESKTOP_SNAPSHOT_MAX_SESSIONS = 32; +const MIN_DESKTOP_CHECKED_AT_MS = 946_684_800_000; // 2000-01-01 UTC +const MAX_DESKTOP_CLOCK_SKEW_MS = 60_000; + +export function validDesktopSession(value: unknown): value is string { + return typeof value === "string" && SESSION.test(value); +} + +export function parseDesktopSnapshot(value: unknown, nowMs: number): DesktopSnapshot | null { + if (!value || typeof value !== "object" || Array.isArray(value)) return null; + const body = value as Record<string, unknown>; + const keys = Object.keys(body).sort(); + if (keys.join(",") !== "available,checkedAtMs,currentVersion,latestVersion,phase,sessionId") return null; + if (!validDesktopSession(body.sessionId) + || typeof body.currentVersion !== "string" || !VERSION.test(body.currentVersion) + || (body.latestVersion !== null && (typeof body.latestVersion !== "string" || !VERSION.test(body.latestVersion))) + || typeof body.available !== "boolean" + || typeof body.phase !== "string" || !PHASES.has(body.phase as DesktopPhase)) return null; + const checked = body.checkedAtMs; + if (checked !== null && (typeof checked !== "number" || !Number.isSafeInteger(checked) || checked < 0 + || checked < MIN_DESKTOP_CHECKED_AT_MS + || checked > nowMs + MAX_DESKTOP_CLOCK_SKEW_MS)) return null; + if (body.available !== (body.latestVersion !== null)) return null; + if (body.available && checked === null) return null; + if (body.phase === "available" && !body.available) return null; + if (body.phase === "current" && (body.available || checked === null)) return null; + if (body.phase === "idle" && (body.available || checked !== null)) return null; + if ((body.phase === "installing" || body.phase === "install-failed") && !body.available) return null; + return body as unknown as DesktopSnapshot; +} + +export class DesktopBadgeStore { + private readonly entries = new Map<string, { snapshot: DesktopSnapshot; receivedAtMs: number }>(); + constructor( + private readonly wallNowMs: () => number = Date.now, + private readonly elapsedNowMs: () => number = () => performance.now(), + ) {} + + put(value: unknown): boolean { + const snapshot = parseDesktopSnapshot(value, this.wallNowMs()); + if (!snapshot) return false; + const now = this.elapsedNowMs(); + this.prune(now); + this.entries.delete(snapshot.sessionId); + while (this.entries.size >= DESKTOP_SNAPSHOT_MAX_SESSIONS) { + this.entries.delete(this.entries.keys().next().value!); + } + this.entries.set(snapshot.sessionId, { snapshot, receivedAtMs: now }); + return true; + } + + read(sessionId: string | null): UpdateBadge { + const now = this.elapsedNowMs(); + this.prune(now); + const snapshot = sessionId && validDesktopSession(sessionId) + ? this.entries.get(sessionId)?.snapshot : undefined; + if (!snapshot) return { + updateAvailable: false, currentVersion: "?", latestVersion: null, + channel: "latest", installer: "desktop", canUpdate: true, unknown: true, + }; + return { + updateAvailable: snapshot.available, + currentVersion: snapshot.currentVersion, + latestVersion: snapshot.latestVersion, + channel: defaultUpdateTag(snapshot.currentVersion), + installer: "desktop", + canUpdate: true, + unknown: snapshot.phase === "idle" + || ((snapshot.phase === "checking" || snapshot.phase === "error") && !snapshot.available), + }; + } + + private prune(now: number): void { + for (const [key, entry] of this.entries) { + if (now - entry.receivedAtMs >= DESKTOP_SNAPSHOT_TTL_MS) this.entries.delete(key); + } + } +} + +export const desktopBadgeStore = new DesktopBadgeStore(); +~~~ + +Before this parser edit, the last guard was `checked < nowMs - 24 * 60 * 60_000`; after, the two named bounds in the block above enforce only pre-2000 and >60-second-future rejection. `checkedAtMs` records when the native updater last settled; it is not a freshness lease. A 25-hour-old successful check remains valid if the living shell continues to send the same pending snapshot. `DesktopBadgeStore.put()` stamps each accepted POST with a new monotonic receipt time, and `read()` expires that receipt after 180 seconds without a heartbeat. The absolute lower bound rejects nonsensical timestamps without coupling display lifetime to check cadence. A null timestamp remains valid only for phases permitted by the schema. + +MODIFY src/update/badge.ts:12: replace only the installer declaration. Exact current before: `installer: ReturnType<typeof detectInstall>;`. Keep `UpdateBadgeDeps.now`, the cache-age guard, and every other field unchanged. After: + +~~~ts +installer: ReturnType<typeof detectInstall> | "desktop"; +~~~ + +MODIFY src/server/management/sidebar-routes.ts:100-103. Keep the existing package GET exactly for an absent surface; add the POST and desktop branch *before* it. Management authentication happens before handleManagementAPI in the listener, but this route must additionally use ctx.principal, not request headers: src/server/management-auth.ts:555-563 gives the raw token the exact "admin-token" principal, while a GUI session yields "gui-session". Direct-dispatch tests with undefined principal also fail closed. The outer 2 MiB Content-Length check at src/server/management-api.ts:198-205 is insufficient for this 1 KiB schema, so the route measures bytes as it reads. The following is the exact block to insert immediately before the old badge GET: + +~~~ts + if (url.pathname === "/api/update/desktop-snapshot" && req.method === "POST") { + if (ctx.principal !== "admin-token") { + return jsonResponse({ error: "desktop snapshot requires admin token" }, 403, req, ctx.config); + } + if (req.headers.get("content-type")?.split(";")[0]?.trim().toLowerCase() !== "application/json") { + return jsonResponse({ error: "invalid desktop snapshot" }, 400, req, ctx.config); + } + const declared = Number(req.headers.get("content-length") ?? "0"); + if (Number.isFinite(declared) && declared > 1024) { + return jsonResponse({ error: "desktop snapshot too large" }, 413, req, ctx.config); + } + const reader = req.body?.getReader(); + if (!reader) return jsonResponse({ error: "invalid desktop snapshot" }, 400, req, ctx.config); + const bytes = new Uint8Array(1024); + let used = 0; + try { + while (true) { + const part = await reader.read(); + if (part.done) break; + if (used + part.value.length > bytes.length) { + await reader.cancel(); + return jsonResponse({ error: "desktop snapshot too large" }, 413, req, ctx.config); + } + bytes.set(part.value, used); + used += part.value.length; + } + const decoded = new TextDecoder("utf-8", { fatal: true }).decode(bytes.subarray(0, used)); + const { desktopBadgeStore } = await import("../../update/desktop-badge"); + if (!desktopBadgeStore.put(JSON.parse(decoded))) { + return jsonResponse({ error: "invalid desktop snapshot" }, 400, req, ctx.config); + } + } catch { + return jsonResponse({ error: "invalid desktop snapshot" }, 400, req, ctx.config); + } + return jsonResponse({ ok: true }, 200, req, ctx.config); + } + + if (url.pathname === "/api/update/badge" && req.method === "GET" + && url.searchParams.get("surface") === "desktop") { + const { desktopBadgeStore } = await import("../../update/desktop-badge"); + return jsonResponse(desktopBadgeStore.read(url.searchParams.get("session")), 200, req, ctx.config); + } +~~~ + +Immediately before the old package GET, also reject any other nonempty surface with a fixed 400; this prevents a misspelled desktop surface silently reading the package badge: + +~~~ts + if (url.pathname === "/api/update/badge" && req.method === "GET" + && url.searchParams.has("surface") && url.searchParams.get("surface") !== "desktop") { + return jsonResponse({ error: "invalid badge surface" }, 400, req, ctx.config); + } +~~~ + +Do not log request URLs, body, session id, or a parse exception. No body or session id is echoed. An authenticated ordinary browser can GET a known desktop session; the id is display routing, not an installation authorization. + +MODIFY src/server/management/route-registry.ts:28-60 and :350-353. Add to ExemptionReason, before "deferred-verb": + +~~~ts + /** Desktop shell's internal display-state POST; the CLI has no app updater state to publish. */ + | "desktop-internal" +~~~ + +Add after the existing GET badge row: + +~~~ts + { method: "POST", path: "/api/update/desktop-snapshot", + module: "server/management/sidebar-routes", mutates: true, + exempt: { reason: "desktop-internal", + why: "Only the Tauri shell has signed-updater state to publish; a CLI verb could only forge that state and would not create an operator action." } }, +~~~ + +The row is a process-state mutation, hence mutates: true. The CLI parity ratchet at tests/cli/cli-capabilities.test.ts:343-370 requires the exemption, and "local-transport" would be false because ProxyClient sends HTTP. The registry stays pure data. + +### Rust transport and state + +MODIFY desktop/src-tauri/src/proxy.rs:1-8,181-228. The new method goes immediately before request(). It uses the same authorised_token() identity/generation check (desktop/src-tauri/src/proxy.rs:191-209) and the same no-redirect/no-system-proxy client (desktop/src-tauri/src/proxy.rs:82-100). It never accepts a caller-supplied URL. + +~~~rust + pub async fn post_desktop_snapshot(&self, body: &Value) -> Result<(), ProxyError> { + let token = self.authorised_token().await?; + let response = self.client + .post(self.endpoint.url("/api/update/desktop-snapshot")) + .header("X-OpenCodex-API-Key", token) + .json(body) + .send().await.map_err(|error| { + if error.is_connect() { ProxyError::Unreachable } else { ProxyError::Decode(error) } + })?; + let _ = decode(response).await?; + Ok(()) + } +~~~ + +MODIFY desktop/src-tauri/src/updater.rs:1-6,81-114 and :116-144. Add imports and the following exact definitions above PendingUpdate. No new dependency: uuid v4 is already pinned in desktop/src-tauri/Cargo.toml:21 and used in desktop/src-tauri/src/identity.rs:19,48; serde_json and tokio are already used by desktop/src-tauri/src/proxy.rs:2-8. The publisher is a single serial task, so a late heartbeat cannot overwrite a later updater transition at the proxy. Native checks also need their own ordering gate: serialization of POSTs cannot correct a stale result already applied to PendingUpdate, tray, and the snapshot. + +~~~rust +use serde::Serialize; +use serde_json::to_value; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; +use tokio::sync::watch; +use uuid::Uuid; + +#[derive(Clone)] +pub enum UiProjection { Available(String), Current } + +#[derive(Clone)] +struct UiUpdate { revision: u64, projection: UiProjection } + +pub struct CheckGeneration { + latest_started: AtomicU64, + install_epoch: AtomicU64, + application: Mutex<()>, + latest_ui_revision: AtomicU64, + ui: watch::Sender<Option<UiUpdate>>, +} + +impl Default for CheckGeneration { + fn default() -> Self { + let (ui, _) = watch::channel(None); + Self { latest_started: AtomicU64::new(0), install_epoch: AtomicU64::new(0), + application: Mutex::new(()), latest_ui_revision: AtomicU64::new(0), ui } + } +} + +impl CheckGeneration { + pub fn begin_if_not_installing( + &self, installing: &std::sync::atomic::AtomicBool, publish_checking: impl FnOnce(), + ) -> Option<(u64, u64)> { + let _guard = self.application.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + if installing.load(Ordering::Acquire) { return None; } + let generation = self.latest_started.fetch_add(1, Ordering::AcqRel) + 1; + let epoch = self.install_epoch.load(Ordering::Acquire); + publish_checking(); + Some((generation, epoch)) + } + + // Commit 3 uses this for both the tray and page install paths. + pub fn claim_install(&self, installing: &std::sync::atomic::AtomicBool) -> bool { + let _guard = self.application.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + if installing.compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire).is_err() { + return false; + } + self.install_epoch.fetch_add(1, Ordering::AcqRel); + // Invalidate a queued check projection before the install can take PendingUpdate. + self.latest_ui_revision.fetch_add(1, Ordering::AcqRel); + true + } + + pub fn epoch_is_current(&self, epoch: u64) -> bool { + self.install_epoch.load(Ordering::Acquire) == epoch + } + + pub fn apply_if_current<T>(&self, generation: u64, apply: impl FnOnce() -> T) -> Option<T> { + let _guard = self.application.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + if self.latest_started.load(Ordering::Acquire) != generation { return None; } + Some(apply()) + } + + pub fn inspect<T>(&self, read: impl FnOnce() -> T) -> T { + let _guard = self.application.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + read() + } + + // Call only inside application/inspect. This is an in-memory send, never a Tauri setter. + fn queue_ui(&self, projection: UiProjection) { + let revision = self.latest_ui_revision.fetch_add(1, Ordering::AcqRel) + 1; + self.ui.send_replace(Some(UiUpdate { revision, projection })); + } + + fn apply_ui_projection_if_current( + &self, update: UiUpdate, apply: impl FnOnce(UiProjection), + ) -> bool { + // A short atomic check only. In particular, never take application here. + if update.revision != self.latest_ui_revision.load(Ordering::Acquire) { return false; } + apply(update.projection); + true + } +} + +pub fn start_ui_projection_worker(app: AppHandle) { + let mut receiver = app.state::<CheckGeneration>().ui.subscribe(); + tauri::async_runtime::spawn(async move { + while receiver.changed().await.is_ok() { + let Some(update) = receiver.borrow_and_update().clone() else { continue; }; + // One worker serializes all updater menu/icon calls. A newer transition wins; + // if one setter is already waiting on AppKit, the latest queued state follows it. + app.state::<CheckGeneration>().apply_ui_projection_if_current(update, |projection| { + match projection { + UiProjection::Available(version) => tray::show_update_available(&app, &version), + UiProjection::Current => tray::show_up_to_date(&app), + } + }); + } + }); +} + +#[derive(Clone, Debug, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct DesktopSnapshot { + session_id: String, + current_version: String, + latest_version: Option<String>, + available: bool, + checked_at_ms: Option<u64>, + phase: &'static str, +} + +pub struct DesktopUpdateState { + session_id: String, + tx: watch::Sender<DesktopSnapshot>, +} + +impl DesktopUpdateState { + pub fn new(current_version: String) -> Self { + let session_id = Uuid::new_v4().to_string(); + let (tx, _) = watch::channel(DesktopSnapshot { + session_id: session_id.clone(), current_version, latest_version: None, + available: false, checked_at_ms: None, phase: "idle", + }); + Self { session_id, tx } + } + + pub fn session_id(&self) -> &str { &self.session_id } + + pub fn publish(&self, phase: &'static str, latest: Option<String>, checked: Option<u64>) { + let previous = self.tx.borrow().clone(); + let next = DesktopSnapshot { + session_id: self.session_id.clone(), + current_version: previous.current_version, + available: latest.is_some(), + latest_version: latest, + checked_at_ms: checked, + phase, + }; + self.tx.send_replace(next); + } + + pub fn retain_phase(&self, phase: &'static str) { + let previous = self.tx.borrow().clone(); + self.publish(phase, previous.latest_version, previous.checked_at_ms); + } + + pub fn wake(&self) { + let snapshot = self.tx.borrow().clone(); + self.tx.send_replace(snapshot); + } +} + +fn now_ms() -> u64 { + SystemTime::now().duration_since(UNIX_EPOCH) + .unwrap_or_default().as_millis().min(u128::from(u64::MAX)) as u64 +} + +pub fn start_snapshot_publisher(app: AppHandle) { + let mut receiver = app.state::<DesktopUpdateState>().tx.subscribe(); + tauri::async_runtime::spawn(async move { + loop { + let snapshot = receiver.borrow_and_update().clone(); + if let Some(proxy) = app.try_state::<crate::AppState>().and_then(|state| state.proxy()) { + if let Ok(body) = to_value(&snapshot) { + let _ = proxy.post_desktop_snapshot(&body).await; + } + } + tokio::select! { + changed = receiver.changed() => if changed.is_err() { break; }, + _ = tokio::time::sleep(Duration::from_secs(60)) => {} + } + } + }); +} +~~~ + +The publisher swallows transport failures as best-effort display state. The next state change or 60 s tick retries. It does not log identifiers or errors. At most one POST is in flight from this shell. The 4 s client timeout is already set at desktop/src-tauri/src/proxy.rs:85-86. The sender lives for the app lifetime, so changed() normally stays open. + +Replace check_and_show() at desktop/src-tauri/src/updater.rs:91-114 with this complete body. Before, the function ran `match check(app).await` and applied each `Some`/`None` immediately; the complete after-body is below. `begin_if_not_installing` checks the install flag and publishes `"checking"` under the same short `application` mutex used by `claim_install` (commit 3) and `apply_if_current`. The result closure holds that mutex through the install guard, pending model, tray's pure `update_pending` flag, snapshot, and UI projection enqueue; an install claim cannot enter between its guard and those writes. The serialized projection worker applies menu, icon, and native overlay setters after the gate has been released. No mutex spans `check(app).await` or a Tauri setter. Every caller (the six-hour loop at updater.rs:81-89, tray manual action at tray.rs:192-196, and commit-3 page command) uses this one function. + +~~~rust +pub async fn check_and_show(app: &AppHandle) { + let gate = app.state::<CheckGeneration>(); + let state = app.state::<tray::TrayState>(); + let Some((generation, _epoch)) = gate.begin_if_not_installing(&state.installing, || { + app.state::<DesktopUpdateState>().retain_phase("checking"); + }) else { return; }; + let answer = check(app).await; + let applied_error = gate.apply_if_current(generation, || { + if tray::is_installing(app) { return None; } + match answer { + Ok(Some(update)) => { + let version = update.version.clone(); + if let Ok(mut pending) = app.state::<PendingUpdate>().0.lock() { + *pending = Some(update); + } + app.state::<DesktopUpdateState>() + .publish("available", Some(version.clone()), Some(now_ms())); + state.update_pending.store(true, Ordering::Release); + gate.queue_ui(UiProjection::Available(version)); + None + } + Ok(None) => { + if let Ok(mut pending) = app.state::<PendingUpdate>().0.lock() { + *pending = None; + } + app.state::<DesktopUpdateState>().publish("current", None, Some(now_ms())); + state.update_pending.store(false, Ordering::Release); + gate.queue_ui(UiProjection::Current); + None + } + Err(error) => { + app.state::<DesktopUpdateState>().retain_phase("error"); + Some(error) + } + } + }).flatten(); + if let Some(error) = applied_error { logging::log_once("updater check failed", &error); } +} +~~~ + +Commit 3 consumes `CheckGeneration::{begin_if_not_installing,claim_install,apply_if_current,epoch_is_current,inspect}` and the gate-private `queue_ui`. `begin_if_not_installing` returns the generation and epoch captured while it owns the gate, after rechecking the install flag and before publishing `"checking"`. `claim_install` takes the same gate for the flag CAS and epoch increment, invalidating queued check UI; commit 3 extends it to enqueue the install projection before release. `apply_if_current` returns `None` when another check started; `Some(T)` is the applied model result, and commit 3 compares the captured epoch inside its application closure before any state write. `inspect` lets the page read pending and checking status under the same gate. A stale result or error maps to `Ok(())` for the page without logging an obsolete failure. The UI worker checks `latest_ui_revision` immediately before each projection and serializes every updater menu/icon/overlay application; if a newer state arrives during a blocking setter, it applies afterward. The gate is managed once per app process immediately before the background checker can start. If a previous signed pending update exists, checking/error retains latestVersion and its blue dot; if not, error projects unknown. The Tauri updater still supplies the availability decision; the proxy never recomputes it. + +MODIFY desktop/src-tauri/src/lib.rs:1-35,223-270: immediately after managing PendingUpdate, insert: + +~~~rust + app.manage(updater::DesktopUpdateState::new(app.package_info().version.to_string())); + app.manage(updater::CheckGeneration::default()); + updater::start_ui_projection_worker(app.handle().clone()); + updater::start_snapshot_publisher(app.handle().clone()); +~~~ + +Tauri's package_info() use already exists at desktop/src-tauri/src/menu.rs:28. Start the publisher independently of the release-build updater-check gate at lib.rs:268-270 so a debug shell can still produce an unknown desktop badge. + +MODIFY desktop/src-tauri/src/startup.rs:1407-1438: replace the old endpoint.url("/#/usage") line at :1414 with: + +~~~rust + let path = format!( + "/?desktop_session={}#/usage", + app.state::<crate::updater::DesktopUpdateState>().session_id() + ); + let dashboard = endpoint.url(&path); +~~~ + +After the existing if !emit(app, progress, None) { ... return; } block at startup.rs:1417-1421, add: + +~~~rust + app.state::<crate::updater::DesktopUpdateState>().wake(); +~~~ + +The state is managed before startup::begin (desktop/src-tauri/src/lib.rs:223-266). This wake immediately retries only after a successful Ready publication; before bind, the publisher has no ProxyClient and sends nothing. Existing startup navigation retains the complete URL through progress.dashboard (startup.rs:1415-1435,1448-1487). + +MODIFY desktop/src-tauri/src/popup.rs:256-263: the incoming web-tray link is still matched against DASHBOARD_PATH at :16 and :313-320. Replace main.navigate(url.clone()) with the following destination, leaving the source matcher unchanged: + +~~~rust + let session = app.state::<crate::updater::DesktopUpdateState>().session_id().to_string(); + let destination = endpoint.url(&format!("/?desktop=open&desktop_session={session}#/usage")); + if let Ok(destination) = destination.parse() { let _ = main.navigate(destination); } +~~~ + +The popup's link stays at gui/src/pages/Tray.tsx:141,176, which already emits /?desktop=open#/usage, and popup.rs:313-320 recognizes it. Only the main webview destination gains the session. The external Open in Browser action at desktop/src-tauri/src/tray.rs:170-177 remains a package-view browser on purpose. + +### Native and generated icon changes + +MODIFY desktop/src-tauri/src/native_tray.rs:17-22: add the declaration to the existing extern block: + +~~~rust + fn ocx_native_tray_update_dot(item: *mut c_void, show: i32); +~~~ + +Add this function after present() at native_tray.rs:58-76. It gets a fresh status-item pointer each time; no borrowed pointer survives the closure. run_on_main_thread is already used in native_tray.rs:78-80. + +~~~rust +pub fn set_update_dot(app: &AppHandle, _show: bool) { + let app = app.clone(); + let target = app.clone(); + let _ = target.run_on_main_thread(move || { + let Some(tray) = app.tray_by_id("main") else { return; }; + let pending = app.try_state::<crate::tray::TrayState>() + .is_some_and(|state| state.update_pending.load(Ordering::Acquire)); + let _ = tray.with_inner_tray_icon(|inner| { + if let Some(item) = inner.ns_status_item() { + let pointer = (&*item as *const _ as *mut c_void).cast(); + unsafe { ocx_native_tray_update_dot(pointer, i32::from(pending)); } + } + }); + }); +} +~~~ + +The main-thread closure reads the latest atomic pending value when it executes. A queued title refresh cannot re-show an already cleared dot with an older captured Boolean. + +Also replace the static path selection in native_event() at native_tray.rs:95-102 so opening the main dashboard from the macOS panel retains the session: + +~~~rust + let session = app.state::<crate::updater::DesktopUpdateState>().session_id().to_string(); + let path = if event == 4 { + format!("/?desktop=open&desktop_session={session}#/usage/companion") + } else { + format!("/?desktop=open&desktop_session={session}#/usage") + }; + if let Ok(url) = proxy.endpoint().url(&path).parse() { + let _ = main.navigate(url); + window::show(&main); + } +~~~ + +MODIFY app/Sources/NativeTray/Popover.swift:1-68. Insert this complete NSView implementation after the NativeTrayPopover class, before its existing @_cdecl exports: + +~~~swift +@MainActor +private final class UpdateDotView: NSView { + weak var statusButton: NSStatusBarButton? + + init(button: NSStatusBarButton) { + statusButton = button + super.init(frame: button.bounds) + autoresizingMask = [.width, .height] + // AppKit keeps the template image and its highlighted tint. This view draws only + // the independent accent, without making the status button layer-backed. + wantsLayer = false + } + + required init?(coder: NSCoder) { nil } + override var isOpaque: Bool { false } + override func hitTest(_ point: NSPoint) -> NSView? { nil } + + override func layout() { + super.layout() + needsDisplay = true + } + + override func draw(_ dirtyRect: NSRect) { + guard let button = statusButton else { return } + let imageRect = button.cell?.imageRect(forBounds: button.bounds) ?? button.bounds + let image = imageRect.isEmpty ? button.bounds : imageRect + let diameter: CGFloat = 7 + let dot = NSRect(x: min(bounds.maxX - diameter, image.maxX - 4), + y: max(bounds.minY, image.minY + 1), + width: diameter, height: diameter) + NSColor.windowBackgroundColor.setFill() + NSBezierPath(ovalIn: dot.insetBy(dx: -1.25, dy: -1.25)).fill() + NSColor(calibratedRed: 0.18, green: 0.48, blue: 0.97, alpha: 1).setFill() + NSBezierPath(ovalIn: dot).fill() + } +} + +@MainActor +private enum UpdateDot { + static weak var button: NSStatusBarButton? + static var view: UpdateDotView? + + static func set(_ item: NSStatusItem, visible: Bool) { + guard let next = item.button else { return } + if button !== next { + view?.removeFromSuperview() + view = nil + button = next + } + guard visible else { + view?.removeFromSuperview() + view = nil + return + } + if view == nil { + let overlay = UpdateDotView(button: next) + next.addSubview(overlay) + view = overlay + } + view?.frame = next.bounds + view?.needsDisplay = true + } +} + +@_cdecl("ocx_native_tray_update_dot") +@MainActor +public func nativeTrayUpdateDot(_ item: UnsafeMutableRawPointer?, _ show: Int32) { + guard Thread.isMainThread, let item else { return } + let statusItem = Unmanaged<NSStatusItem>.fromOpaque(item).takeUnretainedValue() + UpdateDot.set(statusItem, visible: show != 0) +} +~~~ + +This uses the status button's imageRect, not the window theme. A title change calls the bridge again below; autoresizing/layout tracks button bounds. hitTest returning nil preserves status-button clicks and right-click menus. The overlay remains blue when AppKit highlights the template glyph; clearing removes the child view. A recreated status item replaces the weak button/view pair. The 7 pt dot and 1.25 pt halo are visual-QA values, not an API contract. + +MODIFY desktop/src-tauri/src/tray.rs:20-39: add update_pending: AtomicBool to TrayState and initialize it false. Add this exact helper after Default: + +~~~rust +#[cfg(any(not(target_os = "macos"), test))] +fn tray_icon_bytes(pending: bool) -> &'static [u8] { + if pending { include_bytes!("../icons/tray/icon-update.png") } + else { include_bytes!("../icons/tray/icon.png") } +} + +fn apply_update_indicator(app: &AppHandle, pending: bool) { + #[cfg(target_os = "macos")] + popup::set_update_dot(app, pending); + #[cfg(not(target_os = "macos"))] + if let Some(tray) = app.tray_by_id("main") { + let image = tauri::image::Image::from_bytes(tray_icon_bytes(pending)) + .expect("generated tray icon"); + let _ = tray.set_icon(Some(image)); + } +} + +fn update_pending(app: &AppHandle) -> bool { + app.try_state::<TrayState>() + .is_some_and(|state| state.update_pending.load(Ordering::Acquire)) +} +~~~ + +For show_update_available() at tray.rs:284-291, after its existing menu block insert only the setter-side redraw; `check_and_show` owns the atomic model write under `CheckGeneration`: + +~~~rust + apply_update_indicator(app, true); +~~~ + +For show_up_to_date() at :293-301, after its existing menu block insert: + +~~~rust + apply_update_indicator(app, false); +~~~ + +In TrayState at :20-39 add the exact field and Default initializer: + +~~~rust +pub update_pending: AtomicBool, +// inside Default Self +update_pending: AtomicBool::new(false), +~~~ + +Keep it true through set_installing() at :308-319 and set_install_failed() at :321-325; additionally publish updater phase at those two functions. In commit 3, move these phase writes into gate-owned pure transitions and leave the renamed `show_installing` as a setter-only function called by `start_ui_projection_worker`: + +~~~rust +// At end of set_installing: +app.state::<updater::DesktopUpdateState>().retain_phase("installing"); +// At end of set_install_failed: +app.state::<updater::DesktopUpdateState>().retain_phase("install-failed"); +~~~ + +Change both refresh_title call sites at tray.rs:244,261 to refresh_title(app, &tray, &proxy), and replace the function signature/body at :328-341 with: + +~~~rust +fn refresh_title(app: &AppHandle, tray: &tauri::tray::TrayIcon<Wry>, proxy: &ProxyClient) { + let app = app.clone(); + let proxy = proxy.clone(); + let tray = tray.clone(); + tauri::async_runtime::spawn(async move { + let Ok(settings) = proxy.companion_settings().await else { return; }; + let Ok(usage) = proxy.usage_today().await else { return; }; + let quotas = proxy.quotas().await.unwrap_or(Value::Null); + let title = render_title(&settings, &usage, "as); + let _ = tray.set_title(title.as_deref()); + #[cfg(target_os = "macos")] + apply_update_indicator(&app, update_pending(&app)); + }); +} +~~~ + +Before this redraw edit, `apply_update_indicator(&app, update_pending(&app));` ran unconditionally after `tray.set_title`. The after-body above wraps that call in `#[cfg(target_os = "macos")]`: it repositions only the Swift dot against the image whenever the title changes. Windows/Linux do not call set_icon on the minute title refresh. Also insert apply_update_indicator(app, update_pending(app)); after tray construction at tray.rs:231 so late tray registration paints a pending state. Neither title refresh nor tray construction reads `CheckGeneration`, and neither holds its mutex across an AppKit call. On Windows/Linux set_icon only swaps RGBA PNGs at actual pending-state transitions or initial tray construction; the base glyph remains the existing 44 px image and is not redesigned. + +MODIFY desktop/scripts/generate-icons.ts:62-145. Add next to TRAY_OUTPUT/TRAY_SIZE: + +~~~ts +const DOTTED_TRAY_OUTPUT = "tray/icon-update.png"; +const DOTTED_TRAY_SVG = '<g id="update-dot"><circle cx="409" cy="395" r="48" fill="#ffffff"/><circle cx="409" cy="395" r="34" fill="#2f81f7"/></g>'; + +function renderDottedTray(target: string): void { + const dottedSvg = join(target, ".tray-update.svg"); + const base = readFileSync(traySource, "utf8"); + if (!base.includes("</svg>")) throw new Error("tray icon source is not SVG"); + writeFileSync(dottedSvg, base.replace("</svg>", DOTTED_TRAY_SVG + "</svg>")); + try { render(TRAY_SIZE, join(target, DOTTED_TRAY_OUTPUT), dottedSvg); } + finally { rmSync(dottedSvg, { force: true }); } +} +~~~ + +After existing render(TRAY_SIZE, join(target, TRAY_OUTPUT), traySource); produced.push(TRAY_OUTPUT); at generate-icons.ts:126-128, add: + +~~~ts + renderDottedTray(target); + produced.push(DOTTED_TRAY_OUTPUT); +~~~ + +The source is the existing desktop/src-tauri/icons/tray/icon.svg plus this SVG halo/dot, rendered by the same SVG renderer. That is a single source for both variants: no copied glyph paths, no Windows light/dark redesign. The existing generator's produced list makes --check compare the new PNG byte-for-byte on the same renderer (generate-icons.ts:133-175). The generated PNG is the complete content of the NEW binary file; B runs bun run icons in desktop/ and commits that output. No literal PNG bytes belong in a text PRD. + +### Embedded GUI poll + +MODIFY gui/src/lib/desktop-shell.ts:1-32. Add after isDesktopShell() at :11-13. This deliberately requires both the desktop user agent (desktop/src-tauri/src/window.rs:4-14) and a valid UUID v4 in the URL, so a copied query in a normal browser cannot silently select the desktop surface. A desktop shell with no session still asks for the desktop surface and receives unknown; it never falls back to package state. + +~~~ts +const DESKTOP_SESSION = /^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/; + +export function desktopSession(search = typeof location === "undefined" ? "" : location.search): string | null { + const value = new URLSearchParams(search).get("desktop_session"); + return value && DESKTOP_SESSION.test(value) ? value : null; +} + +export function updateBadgeUrl(apiBase: string, ua?: string, search?: string): string { + const base = apiBase + "/api/update/badge"; + if (!isDesktopShell(ua)) return base; + const session = desktopSession(search); + return base + "?surface=desktop" + (session ? "&session=" + encodeURIComponent(session) : ""); +} +~~~ + +MODIFY gui/src/components/sidebar-github-row.tsx:18-40,67-76. Import updateBadgeUrl and isDesktopShell from ../lib/desktop-shell. Before badgePoll, compute badgeUrl = updateBadgeUrl(apiBase); replace the badge keyed resource's key/dependency/fetch/poll settings with: + +~~~tsx + const badgeUrl = updateBadgeUrl(apiBase); + const badgePoll = useKeyedClientResource( + "sidebar-update-badge:" + badgeUrl, + [badgeUrl], + (signal) => readJson<UpdateBadge>(badgeUrl, signal), + { pollMs: isDesktopShell() ? 60_000 : BADGE_POLL_MS }, + ); +~~~ + +Widen the local UpdateBadge.installer union to include "desktop" at sidebar-github-row.tsx:23-30. No new visible copy or i18n key is introduced here. Commit 3 routes the two desktop click actions to the app-origin update page; until then this commit only changes the signal. Keep normal-browser /api/update/badge unchanged. + +## Tests and exact additions + +NEW tests/update/update-desktop-badge.test.ts — full content: + +~~~ts +import { describe, expect, test } from "bun:test"; +import { + DESKTOP_SNAPSHOT_MAX_SESSIONS, + DESKTOP_SNAPSHOT_TTL_MS, + DesktopBadgeStore, + parseDesktopSnapshot, +} from "../../src/update/desktop-badge"; + +const A = "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa"; +const B = "bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb"; +const WALL_MS = 1_700_000_000_000; +const snapshot = (sessionId: string) => ({ + sessionId, currentVersion: "2.61.0", latestVersion: "2.62.0", + available: true, checkedAtMs: WALL_MS, phase: "available", +}); + +describe("desktop badge snapshot store", () => { + test("rejects extra fields, malformed values and forged availability", () => { + expect(parseDesktopSnapshot({ ...snapshot(A), token: "unwanted" }, WALL_MS)).toBeNull(); + expect(parseDesktopSnapshot({ ...snapshot(A), sessionId: "short" }, WALL_MS)).toBeNull(); + expect(parseDesktopSnapshot({ ...snapshot(A), currentVersion: "x".repeat(65) }, WALL_MS)).toBeNull(); + expect(parseDesktopSnapshot({ ...snapshot(A), latestVersion: null }, WALL_MS)).toBeNull(); + expect(parseDesktopSnapshot({ ...snapshot(A), phase: "installed" }, WALL_MS)).toBeNull(); + expect(parseDesktopSnapshot({ ...snapshot(A), checkedAtMs: WALL_MS + 60_001 }, WALL_MS)).toBeNull(); + expect(parseDesktopSnapshot({ ...snapshot(A), checkedAtMs: 946_684_799_999 }, WALL_MS)).toBeNull(); + expect(parseDesktopSnapshot({ ...snapshot(A), phase: "current" }, WALL_MS)).toBeNull(); + expect(parseDesktopSnapshot({ ...snapshot(A), phase: "installing", available: false, latestVersion: null }, WALL_MS)).toBeNull(); + expect(parseDesktopSnapshot(snapshot(A), WALL_MS)).not.toBeNull(); + }); + + test("an absent session is unknown desktop state, never package state", () => { + const store = new DesktopBadgeStore(() => WALL_MS, () => 1_000); + expect(store.read(A)).toMatchObject({ + installer: "desktop", unknown: true, updateAvailable: false, + latestVersion: null, currentVersion: "?", + }); + expect(store.read(null).unknown).toBe(true); + }); + + test("two desktop sessions do not see one another", () => { + let now = 1_000; + const store = new DesktopBadgeStore(() => WALL_MS, () => now); + expect(store.put(snapshot(A))).toBe(true); + expect(store.read(A).updateAvailable).toBe(true); + expect(store.read(B).unknown).toBe(true); + now += 1_000; + expect(store.put({ + ...snapshot(B), latestVersion: null, available: false, + phase: "current", checkedAtMs: WALL_MS, + })).toBe(true); + expect(store.read(A).updateAvailable).toBe(true); + expect(store.read(B)).toMatchObject({ + installer: "desktop", updateAvailable: false, unknown: false, + }); + }); + + test("a failed check is unknown without pending state but retains a known update", () => { + const store = new DesktopBadgeStore(() => WALL_MS, () => 1_000); + expect(store.put({ + ...snapshot(A), latestVersion: null, available: false, + checkedAtMs: null, phase: "error", + })).toBe(true); + expect(store.read(A)).toMatchObject({ unknown: true, updateAvailable: false }); + expect(store.put({ ...snapshot(A), phase: "error" })).toBe(true); + expect(store.read(A)).toMatchObject({ unknown: false, updateAvailable: true }); + }); + + test("a heartbeat extends receipt expiry and stale state disappears", () => { + let now = 1_000; + const store = new DesktopBadgeStore(() => WALL_MS, () => now); + expect(store.put(snapshot(A))).toBe(true); + now += 60_000; + expect(store.put(snapshot(A))).toBe(true); + now += DESKTOP_SNAPSHOT_TTL_MS - 1; + expect(store.read(A).updateAvailable).toBe(true); + now += 1; + expect(store.read(A).unknown).toBe(true); + }); + + test("a 25-hour-old check remains visible while heartbeats renew receipt", () => { + let received = 1_000; + const store = new DesktopBadgeStore(() => WALL_MS + 25 * 60 * 60_000, () => received); + expect(store.put(snapshot(A))).toBe(true); + received += 60_000; + expect(store.put({ ...snapshot(A), phase: "error" })).toBe(true); + expect(store.read(A)).toMatchObject({ updateAvailable: true, unknown: false }); + received += DESKTOP_SNAPSHOT_TTL_MS - 1; + expect(store.read(A).updateAvailable).toBe(true); + received += 1; + expect(store.read(A)).toMatchObject({ updateAvailable: false, unknown: true }); + }); + + test("new sessions evict the oldest after the fixed entry limit", () => { + const store = new DesktopBadgeStore(() => WALL_MS, () => 1_000); + expect(store.put(snapshot(A))).toBe(true); + for (let index = 0; index < DESKTOP_SNAPSHOT_MAX_SESSIONS; index++) { + const id = "00000000-0000-4000-8000-" + index.toString(16).padStart(12, "0"); + expect(store.put(snapshot(id))).toBe(true); + } + expect(store.read(A).unknown).toBe(true); + expect(store.read("00000000-0000-4000-8000-00000000001f").updateAvailable).toBe(true); + }); +}); +~~~ + +MODIFY tests/server/sidebar-routes.test.ts:23-38. Keep its existing `call()` helper and add the following helper immediately after its closing brace at line 38, before `withStarDeps()`. Direct dispatch already models principal selection (lines 23-38); no real admin token is put into fixtures. + +~~~ts +const DESKTOP_A = "cccccccc-cccc-4ccc-8ccc-cccccccccccc"; +const DESKTOP_B = "dddddddd-dddd-4ddd-8ddd-dddddddddddd"; +const desktopPayload = (sessionId = DESKTOP_A) => ({ + sessionId, currentVersion: "2.61.0", latestVersion: "2.62.0", + available: true, checkedAtMs: Date.now(), phase: "available", +}); + +async function desktopPost(body: string | Uint8Array, principal?: "admin-token" | "gui-session", + contentType = "application/json") { + const url = new URL("http://127.0.0.1:10100/api/update/desktop-snapshot"); + const req = new Request(url, { + method: "POST", + headers: { host: "127.0.0.1:10100", "content-type": contentType }, + body, + }); + const response = await handleManagementAPI(req, url, config, {}, principal); + expect(response).not.toBeNull(); + return { status: response!.status, body: await response!.json() as Record<string, unknown> }; +} +~~~ + +Add after the existing GET badge describe, which now includes commit 1's read-only cache test and closes at sidebar-routes.test.ts:139; insert before the `GET /api/github/star` describe at line 141: + +~~~ts +describe("desktop snapshot route", () => { + test("requires the raw admin-token principal, not a GUI session or missing principal", async () => { + const body = JSON.stringify(desktopPayload()); + expect((await desktopPost(body)).status).toBe(403); + expect((await desktopPost(body, "gui-session")).status).toBe(403); + expect((await desktopPost(body, "admin-token")).status).toBe(200); + }); + + test("rejects extra fields, malformed JSON and over-1KiB streams without echoing input", async () => { + expect((await desktopPost(JSON.stringify({ ...desktopPayload(), token: "sentinel" }), "admin-token")).status).toBe(400); + expect((await desktopPost("{", "admin-token")).status).toBe(400); + expect((await desktopPost("", "admin-token")).status).toBe(400); + expect((await desktopPost(new Uint8Array([0xff]), "admin-token")).status).toBe(400); + expect((await desktopPost(JSON.stringify(desktopPayload()), "admin-token", "text/plain")).status).toBe(400); + const oversized = await desktopPost("x".repeat(1025), "admin-token"); + expect(oversized.status).toBe(413); + expect(JSON.stringify(oversized.body)).not.toContain("x".repeat(32)); + }); + + test("desktop GET isolates sessions and never reads the package badge for an absent session", async () => { + expect((await desktopPost(JSON.stringify(desktopPayload()), "admin-token")).status).toBe(200); + const seen = await call("GET", "/api/update/badge?surface=desktop&session=" + DESKTOP_A); + expect(seen.body).toMatchObject({ installer: "desktop", updateAvailable: true, unknown: false }); + expect(seen.raw).not.toContain(DESKTOP_A); + const other = await call("GET", "/api/update/badge?surface=desktop&session=" + DESKTOP_B); + expect(other.body).toMatchObject({ installer: "desktop", updateAvailable: false, unknown: true }); + const missing = await call("GET", "/api/update/badge?surface=desktop"); + expect(missing.body).toMatchObject({ installer: "desktop", unknown: true }); + expect((await call("GET", "/api/update/badge")).body).not.toMatchObject({ installer: "desktop" }); + expect((await call("GET", "/api/update/badge?surface=typo")).status).toBe(400); + }); +}); +~~~ + +MODIFY tests/server/management-route-registry.test.ts:214-260: add this test inside the exemption describe: + +~~~ts + test("desktop snapshot is declared as a bounded internal shell mutation", () => { + const row = MANAGEMENT_ROUTES.find(r => + r.method === "POST" && r.path === "/api/update/desktop-snapshot"); + expect(row).toMatchObject({ + module: "server/management/sidebar-routes", mutates: true, + exempt: { reason: "desktop-internal" }, + }); + }); +~~~ + +MODIFY tests/cli/cli-capabilities.test.ts:342-371: add inside the capability/route parity describe: + +~~~ts + test("desktop snapshot has an explicit internal exemption, not an operator CLI verb", async () => { + const { MANAGEMENT_ROUTES } = await import("../../src/server/management/route-registry"); + const row = MANAGEMENT_ROUTES.find(r => r.method === "POST" + && r.path === "/api/update/desktop-snapshot"); + expect(row?.exempt?.reason).toBe("desktop-internal"); + expect(capabilityRouteKeys().has("POST /api/update/desktop-snapshot")).toBe(false); + }); +~~~ + +MODIFY tests/ci-workflows/build-desktop-icon-set.test.ts:175-251: add next to the menu bar size test. repoPath is already imported at :4-5; this test follows its source-oracle convention and requires no image renderer in CI. + +~~~ts + test("the generated dotted tray variant has the declared size and SVG halo", () => { + const generator = generatorSource(); + expect(generator).toContain('const DOTTED_TRAY_OUTPUT = "tray/icon-update.png"'); + expect(generator).toContain('renderDottedTray(target);'); + expect(generator).toContain('produced.push(DOTTED_TRAY_OUTPUT);'); + expect(generator).toContain('fill="#ffffff"'); + expect(generator).toContain('fill="#2f81f7"'); + const normal = readFileSync(join(ICONS_DIR, "tray", "icon.png")); + const dotted = readFileSync(join(ICONS_DIR, "tray", "icon-update.png")); + expect(pngDimensions(dotted)).toEqual({ width: 44, height: 44 }); + expect(dotted[25]).toBe(RGBA); + expect(dotted.equals(normal)).toBe(false); + }); +~~~ + +MODIFY gui/tests/desktop-shell.test.ts:1-37: import desktopSession and updateBadgeUrl from ../src/lib/desktop-shell; add: + +~~~ts + test("desktop session selects its badge, ordinary browser keeps package badge", () => { + const id = "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa"; + expect(desktopSession("?desktop_session=" + id)).toBe(id); + expect(desktopSession("?desktop_session=not-a-uuid")).toBeNull(); + expect(updateBadgeUrl("", tauriMac, "?desktop_session=" + id)) + .toBe("/api/update/badge?surface=desktop&session=" + id); + expect(updateBadgeUrl("", tauriMac, "")) + .toBe("/api/update/badge?surface=desktop"); + expect(updateBadgeUrl("", "Mozilla/5.0 Chrome/140.0", "?desktop_session=" + id)) + .toBe("/api/update/badge"); + }); +~~~ + +MODIFY desktop/src-tauri/src/updater.rs:116-144: inside the existing tests module import `CheckGeneration` and `DesktopUpdateState`, plus `std::sync::atomic::AtomicBool`, and add: + +~~~rust + #[test] + fn desktop_snapshot_serializes_the_bounded_wire_fields() { + let state = DesktopUpdateState::new("2.61.0".into()); + // A realistic epoch-millisecond value: the proxy parser rejects checkedAtMs before 2000-01-01. + state.publish("available", Some("2.62.0".into()), Some(1_790_000_000_000)); + let value = serde_json::to_value(state.tx.borrow().clone()).unwrap(); + assert!(uuid::Uuid::parse_str(state.session_id()).is_ok()); + assert_eq!(value["sessionId"], state.session_id()); + assert_eq!(value["currentVersion"], "2.61.0"); + assert_eq!(value["latestVersion"], "2.62.0"); + assert_eq!(value["available"], true); + assert_eq!(value["checkedAtMs"], 1_790_000_000_000u64); + assert!(value["checkedAtMs"].as_u64().unwrap() >= 946_684_800_000); // same lower bound as the proxy parser + assert_eq!(value["phase"], "available"); + assert_eq!(value.as_object().unwrap().len(), 6); + } + + #[test] + fn a_delayed_older_none_cannot_clear_a_newer_pending_update() { + let checks = CheckGeneration::default(); + let installing = AtomicBool::new(false); + let (older, _) = checks.begin_if_not_installing(&installing, || {}).unwrap(); + let (newer, _) = checks.begin_if_not_installing(&installing, || {}).unwrap(); + let mut pending: Option<&str> = None; + let mut phase = "checking"; + assert_eq!(checks.apply_if_current(newer, || { + pending = Some("2.62.0"); + phase = "available"; + }), Some(())); + // The first lookup completes after the second one; model its Ok(None) transition. + assert_eq!(checks.apply_if_current(older, || { + pending = None; + phase = "current"; + }), None); + assert_eq!(pending, Some("2.62.0")); + assert_eq!(phase, "available"); + let (third, _) = checks.begin_if_not_installing(&installing, || {}).unwrap(); + assert_eq!(checks.apply_if_current(newer, || { pending = None; }), None); + assert_eq!(pending, Some("2.62.0")); + assert_eq!(checks.apply_if_current(third, || { pending = None; }), Some(())); + assert_eq!(pending, None); + } + + #[test] + fn checking_publication_rechecks_install_claim_inside_the_gate() { + let checks = CheckGeneration::default(); + let installing = AtomicBool::new(false); + assert!(checks.claim_install(&installing)); + let mut published = false; + assert_eq!(checks.begin_if_not_installing(&installing, || { published = true; }), None); + assert!(!published); + } +~~~ + +MODIFY desktop/src-tauri/src/tray.rs:496-623: inside its existing tests module import tray_icon_bytes and add: + +~~~rust + #[test] + fn dotted_tray_variant_is_distinct_and_both_variants_are_png() { + let normal = tray_icon_bytes(false); + let dotted = tray_icon_bytes(true); + assert_eq!(&normal[..8], b"\x89PNG\r\n\x1a\n"); + assert_eq!(&dotted[..8], b"\x89PNG\r\n\x1a\n"); + assert_ne!(normal, dotted); + } +~~~ + +All three proposed Rust tests are pure and run on this Mac. The Windows/Linux set_icon path still needs hosted builds and human visual review at 16/20/24/32 px. The Swift overlay needs local highlighted, dark/light, title-width and 1×/2× QA; no unit assertion proves its final pixels. + +MODIFY scripts/test-layout/layout.json explicit object, alphabetical insertion between update-bun-ownership-lease and update-desktop-owner (layout.json:1655-1659): + +~~~json + "update-bun-ownership-lease.test.ts": "update", + "update-desktop-badge.test.ts": "update", + "update-desktop-owner.test.ts": "update", +~~~ + +MODIFY tests/fixtures/test-layout-expected.json similarly at :1481-1485: + +~~~json + "update-bun-ownership-lease.test.ts": "update", + "update-desktop-badge.test.ts": "update", + "update-desktop-owner.test.ts": "update", +~~~ + +The existing tests/test-layout-tooling.test.ts:248-250 asserts equality of both explicit maps. The new test imports source relatively as existing tests/update files do; any source-oracle path used by these additions goes through tests/helpers/repo-root.ts, as in the icon test. + +## Field/value chains — PLAN-FIELD-CHAIN-01 + +| New value | Creation | Serialization | Deserialization | Consumers | +| --- | --- | --- | --- | --- | +| DesktopSnapshot.sessionId | UUID v4 at desktop/src-tauri/src/updater.rs DesktopUpdateState::new; one id per app process | serde_json in publisher, POST JSON via desktop/src-tauri/src/proxy.rs | strict UUID validation in src/update/desktop-badge.ts parseDesktopSnapshot | Map key there; embedded URL from desktop/src-tauri/src/startup.rs; gui/src/lib/desktop-shell.ts query; sidebar GET. No persistence. Never included in GET body. | +| currentVersion | app.package_info().version in desktop/src-tauri/src/lib.rs | Same POST | 64-character version validation in src/update/desktop-badge.ts | Desktop GET currentVersion and channel (defaultUpdateTag); GUI's existing badge DTO. | +| latestVersion | signed Tauri Update.version at desktop/src-tauri/src/updater.rs:100-104, or null on no update | Same POST | nullable validated version in src/update/desktop-badge.ts | Desktop GET latestVersion; gui/src/components/sidebar-github-row.tsx:89,138-140 label/dot; tray menu uses existing updater::update_label. | +| available | Derived solely from latestVersion in DesktopUpdateState::publish | Same POST | strict boolean and equivalence validation in desktop-badge.ts | Desktop GET updateAvailable; GUI orb and tray indicator. It grants no install authority. | +| checkedAtMs | now_ms() after an accepted settled Tauri check, null before first check | Same POST | finite safe integer, at or after 2000-01-01 UTC, no future over 60 s; no relative-age expiry | Stored for validation and future page status; not echoed by GET. A 25-hour-old check remains visible when heartbeat receipt is fresh. | +| phase enum idle/checking/available/current/error/installing/install-failed | updater.rs constructor, check_and_show(), tray.rs installing/failure arms | String in serde JSON POST | fixed set in desktop-badge.ts | Badge unknown for idle or error without pending; available remains true through checking/error/install; future commit-3 page may read native state, not this HTTP enum. N/A in GET because no phase field is exposed. | +| native check generation, install epoch, UI revision | CheckGeneration::begin_if_not_installing before each check() await; claim_install increments epoch and invalidates queued UI; accepted model transitions increment latest_ui_revision | Process-local atomics and in-memory watch projection; no wire representation | apply_if_current and claim_install share one mutex; inspect reads page status under it; one worker rechecks revision without taking that mutex | Only the latest-started check can change PendingUpdate, update_pending, or DesktopUpdateState; an install claim cannot interleave with model writes. Blocking Tauri setters run later on the serialized worker, so a superseded projection cannot be the final UI. | +| receipt time and Map entry | performance.now at desktop-badge.ts put; wall Date.now is used only for checkedAt validation | N/A; process memory only | N/A | monotonic prune on GET/POST; expires at 180 s even if wall clock moves; oldest eviction at 32; no disk/cache/CLI consumer. | +| surface=desktop and session query | URL from gui/src/lib/desktop-shell.ts only in Tauri UA; native Rust main URL carries desktop_session | HTTP GET query | sidebar-routes.ts URLSearchParams; validDesktopSession in store | desktop projection only; ordinary GET remains package badge; missing session yields unknown desktop. | +| installer="desktop" | src/update/desktop-badge.ts read | jsonResponse in sidebar-routes.ts | gui/src/components/sidebar-github-row.tsx UpdateBadge union | Commit-3 desktop click routing is the next consumer. The value is never passed to src/update/job.ts. | +| desktop-internal exemption | src/server/management/route-registry.ts declaration | N/A, static registry data | tests/server/management-route-registry.test.ts and tests/cli/cli-capabilities.test.ts import | Explains why no CLI capability maps the snapshot POST. | +| macOS dot state | TrayState.update_pending at desktop/src-tauri/src/tray.rs | i32 over native_tray.rs @_cdecl ABI | app/Sources/NativeTray/Popover.swift: nativeTrayUpdateDot | NSView child of current NSStatusBarButton; N/A to HTTP. | +| Windows/Linux dotted icon | desktop/scripts/generate-icons.ts SVG source composition | PNG at desktop/src-tauri/icons/tray/icon-update.png | tauri::image::Image::from_bytes in tray.rs | set_icon on available tray, normal PNG when cleared; Linux no host means no icon consumer. | + +No new GUI i18n catalog key: the orb and its existing aria label remain at gui/src/components/sidebar-github-row.tsx:137-159. All ten gui/src/i18n/{en,de,fr,ja,ko,ru,tr,vi,zh,zh-TW}.ts files remain unchanged in this commit; commit 3 owns update-page copy. + +## Conditional paths — C-ACTIVATION-GROUNDING-01 + +| Guard / fallback / timeout / error | Test activation | Observable effect | +| --- | --- | --- | +| Raw admin-token principal only | sidebar-routes test posts identical valid JSON with undefined, gui-session, admin-token | 403/403/200; no store write until third call. Management-auth.ts:557-561 supplies principals at real ingress. | +| Wrong Content-Type, malformed/extra JSON, version, UUID, phase, availability relation, future/pre-2000 check time | POST test includes text/plain and malformed JSON; parseDesktopSnapshot unit covers the schema and timestamp bounds | Fixed 400, no input echo, no update to previous valid snapshot. | +| >1024 body with and without Content-Length | Route test's 1025-byte Request stream plus a declared length request in an ingress fixture | 413 before storage; stream cancel on overrun. Existing 2 MiB management cap is independent. | +| No body / stream read failure / invalid UTF-8 | POST with empty body and Uint8Array invalid UTF-8; an aborting stream in a direct-dispatch route test at B | Fixed 400, never log parser text. | +| Unknown surface | GET surface=typo | 400, no package fallback. | +| Missing/malformed/foreign session | GET desktop without session, invalid UUID, and B while A is stored | Desktop installer, unknown true, updateAvailable false; package cache not read. | +| Initial shell before bind; bind failure or later proxy replacement | Publisher starts before AppState has a proxy; failed ProxyClient identity; startup finish wake after valid bind | No token/body sent to foreign endpoint; first successful bound send after wake; 60 s retry after transient failure. ProxyClient's existing 4 s timeout bounds each attempt. | +| Two checks settle in reverse order | Rust CheckGeneration state-transition test begins old then new, applies new Some, then attempts old None, followed by a third generation | Old None is rejected and leaves newer pending/phase intact; only the latest-started result changes native state. The separate serial publisher then sends that state in order. | +| Check projection waits on an AppKit setter while status is read | Commit-3 Rust `status_read_completes_while_check_ui_setter_is_blocked`: fake setter waits for concurrent `inspect` result | Status read completes before setter release because the gate protects only pure model writes and projection enqueue; the serialized worker never holds it across Tauri setters. | +| Check while installing | Rust `checking_publication_rechecks_install_claim_inside_the_gate` claims first, then tries `begin_if_not_installing` | No `"checking"` publication; pending state/dot stays. | +| Failed update check with/without prior pending | Inject failed updater result in a focused Rust state test; unit store reads error with/without latest | Prior pending stays visible; no pending reports unknown. Existing fixed-context log remains. | +| Install failed after PendingUpdate was taken | Existing tray.rs:198-222 failure arm with updater::install error | Pending restored, menu re-enabled, dot stays, phase install-failed. Signed download/install sequence unchanged. | +| Snapshot TTL, heartbeat, bounded entries | New Bun store tests advance fake monotonic clock 60 s / 180 s and insert 33 sessions; the 25-hour-old checkedAtMs test re-POSTs an error phase with pending state | A fresh receipt keeps an old but still-pending check visible; after 180 s without receipt it becomes unknown; oldest entry is evicted. | +| Minute title refresh on Windows/Linux | Review the cfg(target_os = "macos") line and platform build; title-refresh smoke on Windows/Linux | No set_icon call from title refresh; set_icon remains for pending transitions and first tray construction. | +| macOS status item absent/recreated; title width; highlight | Native visual QA: check before tray exists, change title, recreate status item, open menu | No crash if absent; child moves relative to image and stays blue while glyph template tint changes; clear removes it. | +| Windows/Linux tray host absent or set_icon failure | Linux session without AppIndicator; platform smoke with failing icon setter | No claimed tray signal, GUI badge still works; setter error cannot affect updater check. | +| Desktop UA without valid URL nonce | GUI helper test | Requests surface=desktop with no session and sees unknown; ordinary browser keeps package URL. | + +## Structure and public documentation edits + +The source ownership table in structure/INDEX.md:154 maps src/update/ to *both* structure/runtime.md and structure/ops/service-and-sidecars.md. This corrects the shorter commit-2 doc list in 000_plan.md; it does not change a decision D1-D3. structure/manifest.json:3 caps each structure page at 600 lines. structure/runtime.md is exactly 600/600 now, so replace one existing line in place and do not append. No new structure document or manifest entry is necessary. The other source-area owners listed in structure/INDEX.md:118-154 were reviewed for consequences; their existing transport/adapter contracts do not change. + +MODIFY structure/runtime.md:159, exact one-line replacement. Preserve commit 1's package-refresh link and add the desktop state contract on the same line (600/600 stays 600/600). Current before: + +~~~md +| Support | `src/lib/`, `src/storage/`, `src/usage/`, `src/update/` ([package refresh](ops/service-and-sidecars.md#package-cache-refresh)), `src/generated/` | +~~~ + +After: + +~~~md +| Support | `src/lib/`, `src/storage/`, `src/usage/`, `src/update/` ([package refresh](ops/service-and-sidecars.md#package-cache-refresh); `desktop-badge.ts` holds bounded process-local display state, never install authority), `src/generated/` | +~~~ + +MODIFY structure/ops/service-and-sidecars.md:250-254, append the paragraph after the existing Package cache refresh text at line 254. That heading and its commit-1 scheduler/async-check paragraphs remain intact: + +~~~md +The desktop badge snapshot in src/update/desktop-badge.ts is process-local display state keyed by a Tauri session id. A 60-second shell heartbeat renews receipt time; entries expire after 180 seconds and the store retains at most 32 sessions. It is separate from the package version cache and from the updater job/ownership transaction. A proxy restart reports unknown until a bound desktop shell republishes; no update installation can be authorized by this snapshot. +~~~ + +MODIFY structure/desktop-shell.md:149-164, insert after the existing in-app update paragraph: + +~~~md +The Tauri updater also publishes a bounded desktop snapshot over its identity-bound ProxyClient. A random process-session id travels in the embedded dashboard URL, and the dashboard requests GET /api/update/badge?surface=desktop&session=<id>. A normal browser keeps the package badge. The shell posts each updater-state change and a 60-second heartbeat; if the proxy loses the snapshot or the shell stops, the desktop badge becomes unknown after 180 seconds. This display path never installs an update or replaces the signed Tauri result. The tray shows the same pending state: macOS draws a blue child NSView dot over the template status-item image; Windows/Linux swap a generated dotted PNG when a tray host exists. The Windows base glyph is unchanged. +~~~ + +MODIFY structure/companion.md:50-56, insert after the Tauri title paragraph: + +~~~md +The update dot is independent of companion usage and title filtering. A title refresh asks the macOS status-button overlay to redraw against the current image rectangle; update availability still comes only from the Tauri updater state in desktop/src-tauri/src/updater.rs. +~~~ + +MODIFY structure/gui-and-management-api.md:168 and :195. Keep the current Updates row's asynchronous check/run and 40-hour package-cache sentences verbatim. Replace only its final sentence `The badge links to the update surface rather than gating other actions.` with the three-sentence text below, keeping it in the same table row; replace the complete current Sidebar row (`GET/POST /api/github/star` and `GET /api/update/badge`; cosmetic failure) with the second block: + +~~~md +The badge links to the update surface rather than gating other actions. `GET /api/update/badge?surface=desktop&session=<id>` projects only that process-local Tauri session; missing or expired state is unknown and never falls back to the package cache. `POST /api/update/desktop-snapshot` is a 1 KiB bounded, admin-token-principal-only display-state mutation with no install permission. + +| Sidebar | `src/server/management/sidebar-routes.ts` — `GET/POST /api/github/star`, `GET /api/update/badge`, and `POST /api/update/desktop-snapshot`. The POST accepts only the raw admin-token principal; GUI sessions cannot publish desktop state. Badge state is cosmetic and a failed poll degrades silently. | +~~~ + +MODIFY docs-site/src/content/docs/reference/management-api.md:547. Exact current before: + +~~~md +| `GET /api/update/badge` | Read cached package badge state without a registry lookup; missing, wrong-channel or 40-hour-old cache returns `unknown: true`. | — | +~~~ + +Replace that row with these exact two rows. Put the desktop paragraph after the table and before the existing automatic-check paragraph at line 549; preserve that paragraph and the async check/run rows at lines 268-269: + +~~~md +| `GET /api/update/badge` | Read cached package badge without a registry lookup; missing, wrong-channel or 40-hour-old cache returns `unknown: true`. `surface=desktop&session=<id>` reads only that desktop app session. | 400 invalid surface; missing or expired desktop session returns `unknown: true` | +| `POST /api/update/desktop-snapshot` | Desktop shell publishes its Tauri-updater display state through the bound proxy client | 403 unless the raw `admin-token` principal; 400 invalid fields; 413 over 1 KiB | + +The desktop snapshot is temporary display state, not an install request. The proxy stores at most 32 sessions in memory and expires one 180 seconds after its last heartbeat. A normal browser without surface=desktop continues to read the package badge. +~~~ + +MODIFY docs-site/src/content/docs/guides/desktop-app.md: its existing "## Updates" section (currently below the tray usage section). Insert these exact sentences after the check cadence sentence: + +~~~md +When the Tauri updater finds a newer app version, a blue dot appears on the macOS menu-bar icon or the Windows/Linux tray icon where a tray host is available. The embedded dashboard shows the same desktop update signal. A normal browser connected to the same proxy still shows the proxy package update state. If the shell stops reporting for about three minutes, the embedded badge becomes unknown until it reconnects. The dot reports availability; installation remains an explicit action. +~~~ + +MODIFY the existing sibling pages at: + +- docs-site/src/content/docs/{fr,ja,ko,ru,tr,zh-cn,zh-tw}/reference/management-api.md +- docs-site/src/content/docs/{fr,ja,ko,ru,tr,zh-cn,zh-tw}/guides/desktop-app.md + +For each management API sibling, replace its existing commit-1 GET /api/update/badge row (which already says missing, wrong-channel or 40-hour-old cache returns `unknown: true`) with translations of the two English rows above. Add the 32-session/180-second paragraph after the table but before the existing translated automatic-check paragraph; preserve each sibling's async check/run rows. Keep literal route paths, surface=desktop, session=<id>, status numbers, admin-token, unknown and 1 KiB unchanged. For each desktop guide sibling, insert a translation of the five English sentences above in its existing Updates section. Say explicitly that Linux shows a dot only with a tray host, an ordinary browser sees package state, expiry yields unknown, and installation is explicit. Do not claim the commit-3 update page exists yet. These seven locales are the exact existing siblings found in docs-site/src/content/docs; there are no de/vi guide siblings. English remains the source of truth. + +No GUI i18n catalog edits: no new visible string is introduced in commit 2. The existing sidebar.updateAvailable and sidebar.checkUpdate keys already cover the label (gui/src/components/sidebar-github-row.tsx:137-141); the ten catalogs named earlier stay unchanged. Commit 3 adds localized update-page copy. + +## Ratchets and merge-head check + +The 2026-09-24 HEAD b429895f4a counts below are physical lines before B. `tests/fixtures/file-size-baseline.json` has no explicit cap for any row here. The scanned-file default is **strictly under 2,000**, so the last allowed count is 1,999 and headroom is `1,999 − current`; `structure/manifest.json` independently caps structure pages at 600. New files show 0 before creation. Recount if HEAD moves. `devlog/` is excluded from the file-size ratchet. Rust, Swift and PNG are outside its scanned extensions. The replacement at `structure/runtime.md:159` must be one line for one line. + +| Growing path | Current | Last allowed | Headroom | +| --- | ---: | ---: | ---: | +| `src/update/badge.ts` | 67 | 1999 | 1932 | +| `src/update/desktop-badge.ts` | 0 | 1999 | 1999 | +| `src/server/management/sidebar-routes.ts` | 106 | 1999 | 1893 | +| `src/server/management/route-registry.ts` | 395 | 1999 | 1604 | +| `desktop/scripts/generate-icons.ts` | 178 | 1999 | 1821 | +| `gui/src/lib/desktop-shell.ts` | 32 | 1999 | 1967 | +| `gui/src/components/sidebar-github-row.tsx` | 161 | 1999 | 1838 | +| `gui/tests/desktop-shell.test.ts` | 37 | 1999 | 1962 | +| `tests/update/update-desktop-badge.test.ts` | 0 | 1999 | 1999 | +| `tests/server/sidebar-routes.test.ts` | 327 | 1999 | 1672 | +| `tests/server/management-route-registry.test.ts` | 274 | 1999 | 1725 | +| `tests/cli/cli-capabilities.test.ts` | 372 | 1999 | 1627 | +| `tests/ci-workflows/build-desktop-icon-set.test.ts` | 251 | 1999 | 1748 | +| `scripts/test-layout/layout.json` | 1816 | 1999 | 183 | +| `tests/fixtures/test-layout-expected.json` | 1622 | 1999 | 377 | +| `structure/runtime.md` | 600 | 600 | 0 | +| `structure/ops/service-and-sidecars.md` | 254 | 600 | 346 | +| `structure/desktop-shell.md` | 388 | 600 | 212 | +| `structure/companion.md` | 77 | 600 | 523 | +| `structure/gui-and-management-api.md` | 395 | 600 | 205 | +| `docs-site/src/content/docs/reference/management-api.md` | 651 | 1999 | 1348 | +| `docs-site/src/content/docs/guides/desktop-app.md` | 122 | 1999 | 1877 | +| `docs-site/src/content/docs/fr/reference/management-api.md` | 406 | 1999 | 1593 | +| `docs-site/src/content/docs/fr/guides/desktop-app.md` | 79 | 1999 | 1920 | +| `docs-site/src/content/docs/ja/reference/management-api.md` | 348 | 1999 | 1651 | +| `docs-site/src/content/docs/ja/guides/desktop-app.md` | 79 | 1999 | 1920 | +| `docs-site/src/content/docs/ko/reference/management-api.md` | 374 | 1999 | 1625 | +| `docs-site/src/content/docs/ko/guides/desktop-app.md` | 79 | 1999 | 1920 | +| `docs-site/src/content/docs/ru/reference/management-api.md` | 396 | 1999 | 1603 | +| `docs-site/src/content/docs/ru/guides/desktop-app.md` | 124 | 1999 | 1875 | +| `docs-site/src/content/docs/tr/reference/management-api.md` | 432 | 1999 | 1567 | +| `docs-site/src/content/docs/tr/guides/desktop-app.md` | 79 | 1999 | 1920 | +| `docs-site/src/content/docs/zh-cn/reference/management-api.md` | 342 | 1999 | 1657 | +| `docs-site/src/content/docs/zh-cn/guides/desktop-app.md` | 79 | 1999 | 1920 | +| `docs-site/src/content/docs/zh-tw/reference/management-api.md` | 324 | 1999 | 1675 | +| `docs-site/src/content/docs/zh-tw/guides/desktop-app.md` | 79 | 1999 | 1920 | + +Outside the scanned extensions: `desktop/src-tauri/src/proxy.rs` 277, `updater.rs` 144, `lib.rs` 288, `startup.rs` 2161, `native_tray.rs` 273, `tray.rs` 620, `popup.rs` 390; `app/Sources/NativeTray/Popover.swift` 68; generated `desktop/src-tauri/icons/tray/icon-update.png` is binary. Their file-size-ratchet headroom is N/A. + +## Verifiers — PLAN-VERIFIER-REAL-01 + +The commands below record the earlier *pre-B* runs on 2026-09-24 after root and GUI dependencies were installed; they are historical baseline evidence, not reruns against commit 1. This plan is now tracked in the shared worktree. The existing source tests do not read it except for the explicit transpiler command shown below. The privacy scanner uses git ls-files (scripts/privacy-scan.ts:60-69), so a fresh B run will include this tracked plan. `structure:check` reads `structure/` and its index, not this plan. The commands prove only the stated earlier baseline, not the future implementation. + +| Command, exact working directory | Exit | What ran / reads this plan? | +| --- | --- | --- | +| bun test tests/server/sidebar-routes.test.ts tests/server/management-route-registry.test.ts tests/ci-workflows/build-desktop-icon-set.test.ts tests/cli/cli-capabilities.test.ts (repository root) | 0 | 52 pass, 0 fail across four files; existing source only, no plan read. | +| bun test tests/test-layout-tooling.test.ts (repository root) | 0 | 16 pass, 0 fail; reads existing layout JSON and fixture, not the plan. | +| bun test tests/desktop-shell.test.ts (gui/) | 0 | 3 pass, 0 fail; reads existing GUI helper, not the plan. | +| bun -e 'const text = require("node:fs").readFileSync("devlog/_plan/260924_update_indicator/020_phase2_desktop_state_icons.md", "utf8"); const transpiler = new Bun.Transpiler({ loader: "ts" }); for (const marker of ["### New src/update/desktop-badge.ts", "NEW tests/update/update-desktop-badge.test.ts"]) { const part = text.slice(text.indexOf(marker)); const match = part.match(/~~~ts\n([\s\S]*?)\n~~~/); if (!match) throw new Error("missing " + marker); transpiler.transformSync(match[1]); } console.log("2 complete new TypeScript blocks parse");' (repository root) | 0 | Both complete new TypeScript blocks parsed; this command reads this plan directly. It is syntax evidence, not typechecking or behavior. | +| bun run icons:check (desktop/) | 0 | 18 generated icons match existing source; reads icon generator/assets, not the plan. The earlier trial spelling bun --cwd desktop run icons:check printed Bun usage with exit 0 and was not treated as a verifier. | +| cargo test --manifest-path desktop/src-tauri/Cargo.toml (repository root) | 101 | The earlier run stopped before Rust tests because binaries/ocx-aarch64-apple-darwin was absent. This is environment setup, not a Rust test result. CI prepares an empty host-triple placeholder; B must do the same locally before retesting. | +| swift -e 'import AppKit; let b = NSStatusBarButton(); print(b.cell?.imageRect(forBounds: b.bounds) as Any)' (repository root) | 0 | Confirmed the existing AppKit imageRect API resolves; this is not an overlay visual test and does not read plan. | +| swift -e 'import AppKit; @MainActor enum Dot { static weak var button: NSStatusBarButton? }; @MainActor final class DotView: NSView { override func hitTest(_ point: NSPoint) -> NSView? { nil }; override func draw(_ dirtyRect: NSRect) { let b = NSStatusBarButton(); let r = b.cell?.imageRect(forBounds: b.bounds) ?? b.bounds; print(r.isEmpty) } }' (repository root) | 0 | Type-checked the proposed weak/button, hit-test and image-rect call shapes; it does not render the overlay or read plan. | +| bun run privacy:scan (repository root) | 0 | Earlier tracked-file scan passed; this plan was untracked then, so the result did not cover it. Re-run in B now that the plan is tracked. | +| bun run structure:check (repository root) | 0 | Structure SSOT checks passed on current docs; does not read plan or future source. | +| bun run typecheck (repository root) | 0 | Existing TypeScript compiled with no diagnostics; no proposed source exists yet. Required again after B. | + +For local desktop Cargo tests, mirror `.github/workflows/ci.yml:1399-1415`: derive `triple` from `rustc -vV`, create `desktop/src-tauri/binaries/ocx-${triple}` as an empty executable placeholder, and create the ignored GUI resource placeholder if the Tauri build needs it. `.gitignore:85-86` ignores `desktop/src-tauri/binaries/` and `desktop/src-tauri/resources/`; B must not commit these test prerequisites. The empty file is for compilation/tests, not a runnable sidecar. + +Runs after B because the new implementation and test files do not exist today: bun test tests/update/update-desktop-badge.test.ts tests/server/sidebar-routes.test.ts tests/server/management-route-registry.test.ts tests/cli/cli-capabilities.test.ts tests/ci-workflows/build-desktop-icon-set.test.ts; bun test tests/test-layout-tooling.test.ts; cd gui && bun test tests/desktop-shell.test.ts && bun run lint && bun run build; cd desktop && bun run icons:check; cargo test --manifest-path desktop/src-tauri/Cargo.toml after making the same ignored empty host-triple sidecar placeholder as CI; bun run typecheck; bun run structure:check; bun run privacy:scan after staging any new B files; cd docs-site && bun install --frozen-lockfile && bun run build. These are future B/C gates, not pass claims. Host CI must be inspected at the final PR head; Windows and Linux Tauri builds and native visual QA remain distinct evidence. + +## Risks and rollback + +The in-memory snapshot is cosmetic and forged only by a process already holding the raw management token. It cannot invoke the package update job or weaken Tauri signature verification. A proxy restart, shell death, port rebind, missed heartbeat, or >32 active shells makes the affected desktop badge unknown rather than borrowing another session's state. A 60 s poll/heartbeat with 180 s expiry leaves two missed intervals before expiry. The 1 KiB streamed ingress bound and strict six-field allowlist prevent an oversized or nested payload from being retained or logged. + +The macOS overlay must be checked against changing title width, AppKit highlight, multiple displays and 1×/2× scaling. Windows and Linux generated icons need visual checking on their hosts; this Mac can validate image generation and Rust logic but cannot claim their rendered tray pixels. Linux without an AppIndicator host still has the embedded badge and no tray icon. If this commit must be backed out before commit 3, revert commit 2 as one unit: the default package badge and original template/PNG tray behavior resume, and no persisted desktop schema needs migration. Do not roll back the independent commit-1 package cache change. + +Cross-document handoff for main: structure/INDEX.md maps src/update/ to both runtime.md and ops/service-and-sidecars.md, so commit 2 edits both despite the abbreviated D7 row in 000_plan.md. The binding main decision is net-zero for runtime.md at its 600/600 structure budget: use the one-line replacement above, and coordinate any commit-1 or commit-4 replacements on their final merged text. Commit 3 must consume `CheckGeneration::{begin_if_not_installing,claim_install,apply_if_current,epoch_is_current,inspect}`, extend the UI projection for install state, put its install-epoch comparison inside the application closure, and return an explicit checking page status while a newer check remains active. No D1/D2/D3 disposition changes. Commit 3 still owns the desktop update click before the PR is marked ready. + + +## wp2 P revalidation + +Revalidated against HEAD `b429895f4a` after commit 1; only this 020 plan was edited. Drifts and fixes: + +- `src/update/badge.ts`: the header and type moved from the former lines 19-29 to 6-18/12, and commit 1 added `UpdateBadgeDeps.now` plus the 40-hour unknown guard. The type-only replacement above uses the exact current installer declaration and leaves package cache semantics intact. The desktop projection keeps all seven `UpdateBadge` fields and uses its own heartbeat receipt TTL. +- `src/update/index.ts` and `src/server/management/context.ts`: `defaultUpdateTag` now starts at 176, and `ManagementContext.principal` is at 151 after new package-check deps imports. The opening anchors were corrected; neither API shape forces a desktop design change. +- `tests/server/sidebar-routes.test.ts`: commit 1 added the read-only package badge test, moving `call()` to 23-38 and the GET badge describe end to 139. The helper and new describe insertion points now target those exact boundaries and preserve the package test. +- `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`: new update-async and update-refresh entries shifted the alphabetical insertion anchors to 1655-1659 and 1481-1485. The three-line before/after windows above still match HEAD and insert `update-desktop-badge.test.ts` between `update-bun-ownership-lease` and `update-desktop-owner` in both maps. +- `structure/runtime.md`: commit 1 replaced the 600/600 Support row with a package-refresh link. The revised exact before/after block preserves that link and replaces one line with one line. `structure/ops/service-and-sidecars.md` gained a Package cache refresh section at 250-254; the desktop paragraph now appends there, preserving scheduler and async-check contracts. +- `structure/gui-and-management-api.md`: the Updates row now specifies asynchronous check/run and package-cache age. Its revised replacement changes only the final sentence; the Sidebar row replacement uses the current row as its before text. +- English and seven translated `docs-site` management API pages: commit 1 expanded the package badge row and added automatic-check paragraphs, and changed the check/run rows. The English exact-before row, the two replacement rows, and the sibling instructions now preserve those facts and insert the desktop paragraph before the existing automatic-check paragraph. +- Ratchets: the old pre-wp1 counts were stale for the changed package badge, route test, layout maps, ops structure page, and all management API pages. The per-file current count/last-allowed/headroom table above is recalculated from HEAD and the baseline. `structure/runtime.md` has zero spare lines. +- Desktop Cargo prerequisite: CI already creates an empty executable `ocx-${triple}` and GUI resource placeholder before `cargo test` (`.github/workflows/ci.yml:1399-1415`). B will prepare the same ignored local placeholders without committing them; `.gitignore:85-86` covers both directories. + +Fresh P checks: exact HEAD before-anchor assertions passed for the package badge, structure rows, layout neighbors, and all eight management API pages; all 36 headroom rows matched current file counts and absent baseline caps; both complete proposed TypeScript blocks parsed; `git diff --check` passed. These checks validate this plan, not the unimplemented B behavior. + +No D1/D2/D3 design decision changes. diff --git a/devlog/_plan/260924_update_indicator/030_phase3_desktop_update_page.md b/devlog/_plan/260924_update_indicator/030_phase3_desktop_update_page.md new file mode 100644 index 0000000000..2e555612b6 --- /dev/null +++ b/devlog/_plan/260924_update_indicator/030_phase3_desktop_update_page.md @@ -0,0 +1,993 @@ +# 030 — Desktop update page (wp3, commit 3) + +## Outcome and boundary + +Commit 3 makes both update entries in the *embedded* dashboard open a bundled desktop update page. The page can inspect, check, install, retry, and return to the dashboard through four native commands. Tray and page installation share one atomic claim and restore the pending signed update on failure. The browser dashboard retains its package update dialog. The dependency is commit 1's package badge freshness and commit 2's per-session desktop badge, updater-state publication, and tray icon state; rebase this diff on those commits and preserve their publication calls. This document proposes code; it does not implement it. + +IN: `desktop/ui/update.html`, the four commands, shared install claim, desktop GUI routing, tests, structure and public docs. OUT: `/api/update/run` changes, remote-origin updater IPC, updater key/signature policy, auto-install, tray icon art, package cache, npm PowerShell tray, and release/deployment. Commit 2 owns the `desktop_session` query key and the desktop badge request through `gui/src/lib/desktop-shell.ts::desktopSession`, `gui/src/lib/desktop-shell.ts::updateBadgeUrl`, and `gui/src/components/sidebar-github-row.tsx` ([020_phase2_desktop_state_icons.md](020_phase2_desktop_state_icons.md), “Embedded GUI poll”); wp3 leaves that code intact. That session is display identity only, never install authority (D3/D4 in `000_plan.md`). + +Current anchors at `9ffa2261ba`: `generate_handler!` and `WebviewUrl::App("index.html".into())` in `desktop/src-tauri/src/lib.rs:214-250`; `CheckGeneration`, `UiProjection`, `DesktopUpdateState`, `PendingUpdate`, and `check_and_show` in `desktop/src-tauri/src/updater.rs:12-238,329-378`; the still-ungated install arm, `TrayState`, and setter functions in `desktop/src-tauri/src/tray.rs:20-40,219-250,312-360`; `startup::open_dashboard`, `ready_dashboard`, and `navigate_dashboard` in `desktop/src-tauri/src/startup.rs:498-504,1454-1495`; app-origin policy in `desktop/src-tauri/src/window.rs:37-77`; `frontendDist`, `withGlobalTauri`, and CSP in `desktop/src-tauri/tauri.conf.json:6-15`; `window.__TAURI__.core.invoke` and nonce pattern in `desktop/ui/index.html:75-123`; GUI entry points in `gui/src/App.tsx:454-462`, `gui/src/components/sidebar-github-row.tsx:127-156`, and `gui/src/pages/use-dashboard-data.ts:861-901`. The maintenance anchor calls `openUpdateDialog` at `gui/src/pages/dashboard-overview-sections.tsx:219-229`; no edit there. `gui/src/lib/desktop-shell.ts:7-33` now owns desktop/session/OS detection and `updateBadgeUrl`. + +## File change map and executable edits + +No DELETE paths. Code blocks show the full new file or the exact replacement/addition at the named current-HEAD anchor. Preserve commit 2's snapshot publication and serialized UI projection worker when replacing check/install logic. + +### NEW `desktop/ui/update.html` — full file + +The page stays self-contained and English-only, matching `desktop/ui/index.html:1-74`. The inline script's Tauri nonce is required by the tested Linux CSP behavior at `desktop/ui/index.html:75-87` and `tests/clients/desktop-startup-surface.test.ts:313-323`. It does not call HTTP, parse a token, or use browser alert/confirm/prompt. + +```html +<!doctype html> +<html lang="en"> + <head> + <meta charset="UTF-8" /> + <meta name="viewport" content="width=device-width, initial-scale=1.0" /> + <title>OpenCodex update + + + +
    +

    OpenCodex update

    +

    Reading update status…

    + +
    + + + +
    +
    + + + +``` + +### MODIFY `desktop/src-tauri/src/updater.rs` + +Before adding the page command, extend the current `#[derive(Clone)] pub enum UiProjection` at `updater.rs:12-16` with `Installing(String)`; keep the derive. Add `UiProjection::Installing(version) => tray::show_installing(&app, &version),` to `start_ui_projection_worker`'s match at `updater.rs:137-140`. Replace the full `CheckGeneration::claim_install` at `updater.rs:64-80` with this block and add `InstallClaim` at module level. `pending_version` runs under the gate **before** anything is committed: `NoPending` leaves the flag, `install_epoch`, and `latest_ui_revision` untouched, so an in-flight check remains valid. `on_claim` only publishes the in-memory snapshot under the gate; neither closure calls a Tauri setter. Remove the commit-2 `#[allow(dead_code)]` markers from `claim_install`, `epoch_is_current`, and `inspect` when wp3 wires them (`updater.rs:65,82,98`). Leave the unrelated `popup_nonmac_test` marker at `updater.rs:242` intact. + +```rust +// Module level in updater.rs, next to CheckGeneration: +#[derive(Debug, PartialEq, Eq)] +pub enum InstallClaim { Claimed, Busy, NoPending } + +// Inside impl CheckGeneration, replacing commit 2's claim_install: +pub fn claim_install( + &self, installing: &std::sync::atomic::AtomicBool, + pending_version: impl FnOnce() -> Option, + on_claim: impl FnOnce(), +) -> InstallClaim { + let _guard = self.application.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + if installing.load(Ordering::Acquire) { + return InstallClaim::Busy; + } + // Verify pending under the gate before committing anything: a stale Install click with no + // pending update must not bump the epoch and orphan an in-flight check. + let Some(version) = pending_version() else { return InstallClaim::NoPending; }; + if installing.compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire).is_err() { + return InstallClaim::Busy; + } + self.install_epoch.fetch_add(1, Ordering::AcqRel); + self.latest_ui_revision.fetch_add(1, Ordering::AcqRel); // invalidate queued check UI + on_claim(); + self.queue_ui(UiProjection::Installing(version)); + InstallClaim::Claimed +} +``` + +The current tray install arm does **not** call commit 2's `claim_install(&installing) -> bool`: it still takes `PendingUpdate` directly at `tray.rs:228-233`. Replace that entire arm with `install_pending` below; the new three-way claim lives in that shared function and runs before either entry takes pending. The existing Rust test at `updater.rs:467-480` is the only live old-signature call site and must be changed as shown below. + +Reuse commit 2's `use serde::Serialize;` and `use std::sync::atomic::{AtomicU64, Ordering};` imports; do not add a second `Ordering` import ([020_phase2_desktop_state_icons.md](020_phase2_desktop_state_icons.md), “Rust transport and state”, `desktop/src-tauri/src/updater.rs`). Keep `PendingUpdate`'s single `Mutex>`; there is no second copy of signed `Update`. Add this after `PendingUpdate` and make all app-origin commands return the same projection. `PageUpdateStatus` is a new serialized type, not a new persisted schema. + +```rust +#[derive(Serialize)] +#[serde(rename_all = "camelCase")] +pub struct PageUpdateStatus { + current_version: String, + latest_version: Option, + available: bool, + installing: bool, + checking: bool, +} + +pub fn page_status(app: &AppHandle) -> PageUpdateStatus { + app.state::().inspect(|| { + let pending = app.state::(); + let pending = pending.0.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + let latest_version = pending.as_ref().map(|update| update.version.clone()); + let installing = tray::is_installing(app); + let checking = app.state::().tx.borrow().phase == "checking"; + PageUpdateStatus { + current_version: env!("CARGO_PKG_VERSION").to_owned(), + available: latest_version.is_some(), + latest_version, + installing, + checking, + } + }) +} + +pub async fn install_pending(app: &AppHandle) -> Result { + let state = app.state::(); + let gate = app.state::(); + match gate.claim_install( + &state.installing, + || app.state::().0.lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .as_ref().map(|update| update.version.clone()), + || app.state::().retain_phase("installing"), + ) { + InstallClaim::Claimed => {} + InstallClaim::Busy => return Err("an update is already installing".into()), + InstallClaim::NoPending => return Err("no update is ready to install".into()), + } + let pending = app.state::(); + let update = pending.0.lock().unwrap_or_else(std::sync::PoisonError::into_inner).take(); + let Some(update) = update else { + // Unreachable while the claim holds: only a claimed installer takes PendingUpdate and + // checks are rejected by the bumped epoch. Release under the gate defensively. + gate.inspect(|| { + state.installing.store(false, Ordering::Release); + app.state::().retain_phase("current"); + gate.queue_ui(UiProjection::Current); + }); + return Err("no update is ready to install".into()); + }; + let version = update.version.clone(); + let retry_update = update.clone(); + let result = install(app, update).await; + if let Err(error) = result { + gate.inspect(|| { + let pending = app.state::(); + *pending.0.lock().unwrap_or_else(std::sync::PoisonError::into_inner) = Some(retry_update); + state.installing.store(false, Ordering::Release); + state.update_pending.store(true, Ordering::Release); + app.state::().retain_phase("install-failed"); + gate.queue_ui(UiProjection::Available(version)); + }); + return Err(error); + } + state.installing.store(false, Ordering::Release); + Ok(page_status(app)) +} +``` + +The existing `install(app, update)` body at `updater.rs:288-313` remains unchanged: its signed `download` precedes `exit::prepare_restart`, then `update.install` and `exit::complete_restart`. The claim happens *before* taking `PendingUpdate` and increments the gate-owned epoch inside the same short mutex used by check-result application. Its `pending_version` closure reads the pending version; `on_claim` publishes the pure `"installing"` snapshot; `claim_install` queues the install UI projection while still in the gate. None makes a Tauri setter call. Keep commit 2's `DesktopUpdateState` and `start_snapshot_publisher` (`updater.rs:156-236`), whose 60-second wait uses `tokio::time::timeout`, not `select!`; `wake()` notifies via `send_modify` without replacing the snapshot (`updater.rs:197-204`). On a failed download or drain, the gate protects pending restoration, claim release, `"install-failed"` publication, and a new available projection together. The serialized UI worker performs menu/icon/overlay setters afterward. No second app update is fetched during installation. Do not return a raw updater error over IPC if it can contain a URL or local path; map it to fixed user-facing copy and keep detailed error in `logging::log_once` (current tray logger at `tray.rs:247`). + +Commit 2 owns `desktop/src-tauri/src/updater.rs::CheckGeneration` and `check_and_show` ([020_phase2_desktop_state_icons.md](020_phase2_desktop_state_icons.md), “Rust transport and state”). Change only `check_and_show`'s return/error contract; the gate already owns `install_epoch` and `latest_ui_revision`. The page command in `lib.rs::update_check` below calls this same function as the tray action and background loop; it must not call `check(app)` or create a second generation counter. `begin_if_not_installing` captures the epoch and publishes `"checking"` only after a locked install-flag recheck. Compare the epoch **inside** `apply_if_current`, before any pending, tray model, snapshot, or UI projection mutation. A superseded check returns `Ok(())` without publishing or reporting its obsolete error, then `page_status` returns `checking: true` while the newer generation is in flight; the page polls until a settled status arrives. + +```rust +pub async fn check_and_show(app: &AppHandle) -> Result<(), String> { + let gate = app.state::(); + let state = app.state::(); + let Some((generation, epoch)) = gate.begin_if_not_installing(&state.installing, || { + app.state::().retain_phase("checking"); + }) else { return Ok(()); }; + let answer = check(app).await; + let applied = gate.apply_if_current(generation, || { + if tray::is_installing(app) || !gate.epoch_is_current(epoch) { + return Ok(()); + } + match answer { + Ok(Some(update)) => { + let version = update.version.clone(); + if let Ok(mut pending) = app.state::().0.lock() { + *pending = Some(update); + } + app.state::() + .publish("available", Some(version.clone()), Some(now_ms())); + state.update_pending.store(true, Ordering::Release); + gate.queue_ui(UiProjection::Available(version)); + Ok(()) + } + Ok(None) => { + if let Ok(mut pending) = app.state::().0.lock() { + *pending = None; + } + app.state::().publish("current", None, Some(now_ms())); + state.update_pending.store(false, Ordering::Release); + gate.queue_ui(UiProjection::Current); + Ok(()) + } + Err(error) => { + app.state::().retain_phase("error"); + Err(error) + } + } + }); + if let Some(Err(error)) = &applied { logging::log_once("updater check failed", error); } + applied.unwrap_or(Ok(())) +} +``` + +`CheckGeneration::apply_if_current` makes pending, the pure tray flag, snapshot, and UI projection one ordered model transition with the install claim. It rejects a page result once the later tray check has started, regardless of completion order. The gate-owned `install_epoch` rejects a check that spans an install even if its generation is still current. `begin_if_not_installing` checks the flag again while holding that gate and publishes `"checking"` before releasing it. The single `start_ui_projection_worker` rechecks `latest_ui_revision` and performs Tauri setters without the gate; if a setter is already waiting on AppKit when a newer revision arrives, the worker applies the newer projection afterward. `page_status` uses `inspect` to read pending and the native phase against the same gate, so it cannot combine pending from before a completed check with phase from after it. In `start_background_checks` and the tray `check-updates` arm use `let _ = check_and_show(&app).await;` because the function itself logs an applied error and publishes `"error"`; the page command maps an applied error to fixed copy. A rejected stale error returns success without falsely replacing the current native state. + +Append Rust tests under the existing `#[cfg(test)] mod tests` at `updater.rs:380-505`. Retain commit 2's `CheckGeneration` and `AtomicBool` imports; add `InstallClaim`, `UiProjection`, `Ordering`, and `std::sync::{mpsc, Arc}`. `CheckGeneration` and its private `ui`, `queue_ui`, and `apply_ui_projection_if_current` methods are accessible to these child-module tests. + +Commit 2's inherited test `checking_publication_rechecks_install_claim_inside_the_gate` calls the commit-2 one-argument `claim_install(&installing)` and would stop compiling here. Replace its claim line in this commit: + +```rust +- assert!(checks.claim_install(&installing)); ++ assert_eq!(checks.claim_install(&installing, || Some("2.66.0".into()), || {}), InstallClaim::Claimed); +``` + +The tray arm currently contains no `claim_install` call. After this change, `rg -n 'claim_install\(' desktop/src-tauri/src/{updater,tray}.rs` should show the new declaration, the shared `install_pending` call, and three-argument test calls, with no old one-argument call. + +```rust +#[test] +fn install_claim_has_one_winner_and_can_retry_after_failure() { + let gate = CheckGeneration::default(); + let installing = AtomicBool::new(false); + assert_eq!(gate.claim_install(&installing, || Some("2.66.0".into()), || {}), InstallClaim::Claimed); + assert_eq!(gate.install_epoch.load(Ordering::Acquire), 1); + assert_eq!(gate.claim_install(&installing, || Some("2.66.0".into()), || {}), InstallClaim::Busy); + assert_eq!(gate.install_epoch.load(Ordering::Acquire), 1); + installing.store(false, Ordering::Release); + assert_eq!(gate.claim_install(&installing, || Some("2.66.0".into()), || {}), InstallClaim::Claimed); + assert_eq!(gate.install_epoch.load(Ordering::Acquire), 2); +} + +#[test] +fn install_click_without_pending_leaves_in_flight_check_valid() { + let gate = CheckGeneration::default(); + let installing = AtomicBool::new(false); + let (generation, epoch) = gate.begin_if_not_installing(&installing, || {}).unwrap(); + let revision = gate.latest_ui_revision.load(Ordering::Acquire); + let mut claimed_hook = false; + assert_eq!(gate.claim_install(&installing, || None, || { claimed_hook = true; }), InstallClaim::NoPending); + assert!(!claimed_hook); + assert!(!installing.load(Ordering::Acquire)); + assert_eq!(gate.install_epoch.load(Ordering::Acquire), 0); + assert_eq!(gate.latest_ui_revision.load(Ordering::Acquire), revision); + // The check that was running when the stale Install click arrived still settles. + assert!(gate.epoch_is_current(epoch)); + assert_eq!(gate.apply_if_current(generation, || "current"), Some("current")); +} + +#[test] +fn page_check_started_before_tray_check_cannot_override_it_in_either_completion_order() { + let gate = CheckGeneration::default(); + let installing = AtomicBool::new(false); + let mut pending = Some("previous"); + let mut phase = "available"; + + let (page, _) = gate.begin_if_not_installing(&installing, || { phase = "checking"; }).unwrap(); + let (tray, _) = gate.begin_if_not_installing(&installing, || { phase = "checking"; }).unwrap(); + assert_eq!(gate.apply_if_current(page, || { pending = None; phase = "current"; }), None); + assert_eq!((pending, phase), (Some("previous"), "checking")); + assert_eq!(gate.apply_if_current(tray, || { pending = Some("tray"); phase = "available"; }), Some(())); + assert_eq!((pending, phase), (Some("tray"), "available")); + + let (page, _) = gate.begin_if_not_installing(&installing, || { phase = "checking"; }).unwrap(); + let (tray, _) = gate.begin_if_not_installing(&installing, || { phase = "checking"; }).unwrap(); + assert_eq!(gate.apply_if_current(tray, || { pending = Some("new tray"); phase = "available"; }), Some(())); + assert_eq!(gate.apply_if_current(page, || { pending = None; phase = "current"; }), None); + assert_eq!((pending, phase), (Some("new tray"), "available")); +} + +#[test] +fn install_claim_cannot_land_between_check_guard_and_pending_tray_write() { + let gate = Arc::new(CheckGeneration::default()); + let installing = Arc::new(AtomicBool::new(false)); + let (generation, epoch) = gate.begin_if_not_installing(&installing, || {}).unwrap(); + let (attempt_tx, attempt_rx) = mpsc::channel(); + let (claimed_tx, claimed_rx) = mpsc::channel(); + let mut pending = None; + let mut tray_visible = false; + + let claim_thread = gate.apply_if_current(generation, || { + assert!(!installing.load(Ordering::Acquire)); // result guard + assert!(gate.epoch_is_current(epoch)); + let claim_gate = Arc::clone(&gate); + let claim_flag = Arc::clone(&installing); + let thread = std::thread::spawn(move || { + attempt_tx.send(()).unwrap(); + claimed_tx.send(claim_gate.claim_install(&claim_flag, || Some("signed update".into()), || {})).unwrap(); + }); + attempt_rx.recv_timeout(std::time::Duration::from_secs(1)).unwrap(); + // Force the claim attempt while this application still owns the gate. + assert_eq!(claimed_rx.recv_timeout(std::time::Duration::from_millis(25)), + Err(mpsc::RecvTimeoutError::Timeout)); + pending = Some("signed update"); + tray_visible = true; + assert!(!installing.load(Ordering::Acquire)); + thread + }).unwrap(); + assert_eq!((pending, tray_visible), (Some("signed update"), true)); + assert_eq!(claimed_rx.recv_timeout(std::time::Duration::from_secs(1)).unwrap(), InstallClaim::Claimed); + claim_thread.join().unwrap(); + assert!(installing.load(Ordering::Acquire)); + assert!(!gate.epoch_is_current(epoch)); +} + +#[test] +fn status_read_completes_while_check_ui_setter_is_blocked() { + let gate = Arc::new(CheckGeneration::default()); + let installing = AtomicBool::new(false); + let (generation, _) = gate.begin_if_not_installing(&installing, || {}).unwrap(); + let mut pending = None; + assert_eq!(gate.apply_if_current(generation, || { + pending = Some("signed update"); + gate.queue_ui(UiProjection::Available("2.66.0".into())); + }), Some(())); + let projected = gate.ui.borrow().clone().unwrap(); + let (setter_entered_tx, setter_entered_rx) = mpsc::channel(); + let (status_returned_tx, status_returned_rx) = mpsc::channel(); + let setter_gate = Arc::clone(&gate); + let setter = std::thread::spawn(move || setter_gate.apply_ui_projection_if_current( + projected, |_| { + setter_entered_tx.send(()).unwrap(); + // Fake a Tauri setter waiting on AppKit; it finishes only after status returns. + status_returned_rx.recv_timeout(std::time::Duration::from_secs(1)).unwrap(); + }, + )); + setter_entered_rx.recv_timeout(std::time::Duration::from_secs(1)).unwrap(); + let read_gate = Arc::clone(&gate); + let (read_tx, read_rx) = mpsc::channel(); + let reader = std::thread::spawn(move || { + read_tx.send(read_gate.inspect(|| "available")).unwrap(); + }); + assert_eq!(read_rx.recv_timeout(std::time::Duration::from_secs(1)).unwrap(), "available"); + status_returned_tx.send(()).unwrap(); + reader.join().unwrap(); + assert!(setter.join().unwrap()); + assert_eq!(pending, Some("signed update")); +} + +#[test] +fn superseded_ui_projection_never_enters_its_setter() { + let gate = CheckGeneration::default(); + gate.inspect(|| gate.queue_ui(UiProjection::Current)); + let old = gate.ui.borrow().clone().unwrap(); + gate.inspect(|| gate.queue_ui(UiProjection::Available("2.66.0".into()))); + let newest = gate.ui.borrow().clone().unwrap(); + assert!(!gate.apply_ui_projection_if_current(old, |_| panic!("stale setter ran"))); + let mut applied = false; + assert!(gate.apply_ui_projection_if_current(newest, |_| applied = true)); + assert!(applied); +} +``` + +The first test checks the successful/busy/retry claim sequence. The second checks that `NoPending` leaves an in-flight check valid. The third forces a page `None` result before and after a later-started tray `Some` result; both orders leave the tray's pending version and phase intact. The fourth injects a competing claim after the result guard but before the modeled pending/tray model writes, proves the claim cannot complete while the application closure owns the mutex, then proves it completes afterward and increments the epoch. The fifth runs the production projection-application seam with a fake setter that blocks until the concurrent `inspect` read completes; the one-second channel deadline makes a regression fail instead of hanging. The sixth rejects a queued projection superseded before its setter starts. The page contract test below checks that both page and tray call `check_and_show` and `install_pending`. A packaged signed-update test is left to the desktop CI lane; a unit test cannot safely replace the running application. + +### MODIFY `desktop/src-tauri/src/tray.rs` + +At current HEAD, `tray.rs:225-250` takes `PendingUpdate` before any gate claim and calls the menu/snapshot setters directly. This is the reviewer High carried from wp2. The exact current arm is the **Before** block: + +```rust +"install-update" => { + let app = app.clone(); + tauri::async_runtime::spawn(async move { + let update = app + .state::() + .0 + .lock() + .ok() + .and_then(|mut pending| pending.take()); + let Some(update) = update else { return; }; + let version = update.version.clone(); + let retry_update = update.clone(); + set_installing(&app, &version); + if let Err(error) = updater::install(&app, update).await { + if let Ok(mut pending) = app.state::().0.lock() { + *pending = Some(retry_update); + } + set_install_failed(&app, &version); + crate::logging::log_once("updater install failed", &error); + } + }); +} +``` + +The **After** block replaces it. Both page and tray then enter `updater::install_pending`, which calls `gate.claim_install(&state.installing, pending_version, on_claim)` and handles `Claimed`, `Busy`, and `NoPending` before taking the signed update: + +```rust +"install-update" => { + let app = app.clone(); + tauri::async_runtime::spawn(async move { + if let Err(error) = updater::install_pending(&app).await { + crate::logging::log_once("updater install failed", &error); + } + }); +} +``` + +Keep `TrayState.installing` as the single flag; `CheckGeneration` owns `install_epoch`, so do not add an epoch field to `TrayState`. Replace current `set_installing` at `tray.rs:338-351` with this setter-only function, invoked solely by `start_ui_projection_worker`: + +```rust +pub fn show_installing(app: &AppHandle, version: &str) { + if let Some(menu) = menu_handles(app) { + let _ = menu.install_update.set_text(format!("Installing update v{version}…")); + let _ = menu.install_update.set_enabled(false); + let _ = menu.check_updates.set_enabled(false); + } +} +``` + +The deleted before-lines are `state.installing.store(true, Ordering::Release)` and `DesktopUpdateState::retain_phase("installing")`; the gate's claim now owns both. Delete private `set_install_failed` at `tray.rs:353-360`; `install_pending` publishes `"install-failed"` under the gate after restoration. Keep `is_installing`, `show_update_available`, `show_up_to_date`, `TrayState.update_pending`, and commit-2 icon updates; all updater-driven calls to those menu/icon/overlay setters go through the worker. In the check arm at `tray.rs:219-224`, use `let _ = updater::check_and_show(&app).await;` for its new `Result` contract. Both menu handlers already spawn async tasks before touching updater state; keep gate reads inside those tasks. Title refresh reads only `TrayState.update_pending` atomically. `TrayState` still exists without a rendered tray: `lib.rs:232` manages it before `startup::begin` at `lib.rs:272`, so Linux without an AppIndicator host has the same claim path. + +### MODIFY `desktop/src-tauri/src/lib.rs`, `desktop/src-tauri/src/startup.rs`, `desktop/src-tauri/src/window.rs` + +Add these commands before `run()` in `lib.rs`, and append their names to the existing `generate_handler!` at current `lib.rs:214-222`. The origin guard uses the actual requesting `WebviewWindow`, not a URL string supplied by JavaScript. `window::require_update_page` is defined below. + +```rust +#[tauri::command] +async fn update_status(window: tauri::WebviewWindow, app: tauri::AppHandle) -> Result { + window::require_update_page(&window)?; + Ok(updater::page_status(&app)) +} + +#[tauri::command] +async fn update_check(window: tauri::WebviewWindow, app: tauri::AppHandle) -> Result { + window::require_update_page(&window)?; + updater::check_and_show(&app).await + .map_err(|_| "the update check failed; try again".to_owned())?; + Ok(updater::page_status(&app)) +} + +#[tauri::command] +async fn update_install(window: tauri::WebviewWindow, app: tauri::AppHandle) -> Result { + window::require_update_page(&window)?; + updater::install_pending(&app).await.map_err(|error| { + logging::log_once("updater install failed", &error); + "the update could not be installed; try again".to_owned() + }) +} + +#[tauri::command] +fn return_to_dashboard(window: tauri::WebviewWindow, app: tauri::AppHandle) -> Result<(), String> { + window::require_update_page(&window)?; + startup::return_to_dashboard(&app) +} +``` + +`update_status` is deliberately an async Tauri command: the pinned synchronous command wrapper can run inline on the AppKit thread, and `page_status` waits for `CheckGeneration::inspect`. `update_check` and `update_install` are already async and may call `page_status` or `claim_install` after an await. `return_to_dashboard` has no gate read and stays synchronous. Do not add a direct `inspect`, `claim_install`, or pending-lock read to a synchronous Tauri command or tray menu callback; spawn an async task first. This command contract and the blocked-setter regression below must be checked before the wp3 audit. + +In `window.rs` add after the existing private `is_app_origin` at lines 71-77: + +```rust +pub fn require_update_page(window: &WebviewWindow) -> Result<(), String> { + if window.label() != "main" { return Err("update page unavailable".into()); } + let url = window.url().map_err(|_| "update page unavailable")?; + if !is_update_page_url(&url) { + return Err("update page unavailable".into()); + } + Ok(()) +} + +fn is_update_page_url(url: &Url) -> bool { + is_app_origin(url) && url.path() == "/update.html" +} +``` + +Tighten `is_app_origin`'s `"tauri"` arm from `true` to `url.host_str() == Some("localhost") && url.port().is_none()`; keep the Windows `http://tauri.localhost` arm. This is the exact packaged URL: `tauri://localhost/update.html` on macOS/Linux and `http://tauri.localhost/update.html` on Windows (`window.rs:61-75`). The bound loopback dashboard is `http://127.0.0.1:` (`window.rs:45-56`) and fails the command guard. Extend the existing `window.rs:131-226` test module's `use super` import with `is_update_page_url` and add: + +```rust +#[test] +fn only_the_bundled_update_page_has_update_commands() { + for value in ["tauri://localhost/update.html", "http://tauri.localhost/update.html"] { + assert!(is_update_page_url(&url(value)), "{value}"); + } + for value in ["http://127.0.0.1:10100/update.html", "tauri://evil/update.html", + "tauri://localhost/index.html", "http://tauri.localhost/update.html.evil"] { + assert!(!is_update_page_url(&url(value)), "{value}"); + } +} +``` + +`startup::open_dashboard` cannot serve as the return command: its `navigate_once` gate at `startup.rs:1472-1481` has already been consumed. Add below `open_dashboard` at line 1461: + +```rust +pub fn return_to_dashboard(app: &AppHandle) -> Result<(), String> { + let startup = app.try_state::().ok_or("dashboard is not ready")?; + let dashboard = startup.ready_dashboard(); + let window = app.get_webview_window("main").ok_or("dashboard window is unavailable")?; + return_ready_dashboard(dashboard.as_deref(), |url| navigate_dashboard(&window, url))?; + crate::window::show(&window); + Ok(()) +} + +fn return_ready_dashboard(dashboard: Option<&str>, navigate: impl FnOnce(&str) -> bool) -> Result<(), String> { + let dashboard = dashboard.ok_or("dashboard is not ready")?; + if !navigate(dashboard) { return Err("dashboard could not be opened".into()); } + Ok(()) +} +``` + +The existing `ready_dashboard` accessor is at `startup.rs:498-503`, and `navigate_dashboard` is at `startup.rs:1483-1489`. Add this inline test in `startup.rs`'s existing test module, importing `return_ready_dashboard`: + +```rust +#[test] +fn update_page_return_requires_a_ready_dashboard_and_retries_refused_navigation() { + assert_eq!(return_ready_dashboard(None, |_| true).unwrap_err(), "dashboard is not ready"); + assert_eq!(return_ready_dashboard(Some("http://127.0.0.1:10100/#/usage"), |_| false).unwrap_err(), "dashboard could not be opened"); + let mut visited = None; + assert!(return_ready_dashboard(Some("http://127.0.0.1:10100/#/usage"), |url| { + visited = Some(url.to_owned()); + true + }).is_ok()); + assert_eq!(visited.as_deref(), Some("http://127.0.0.1:10100/#/usage")); +} +``` + +Do not bypass `ready_dashboard` with a hardcoded port or local storage. Do not widen `capabilities/dashboard-zoom.json:1-8`: it grants only zoom to the loopback origin. `capabilities/default.json:1-12` is local-app-only and needs no new permission entry for these app commands; custom commands are registered in `generate_handler!`, and the Rust page-origin guard is the authority. If Tauri's generated capability schema demands a custom-command permission at implementation, add a separate capability scoped to the app origin only; never place updater commands in the `remote.urls` capability. + +### MODIFY `gui/src/lib/desktop-shell.ts`, `gui/src/App.tsx`, `gui/src/components/sidebar-github-row.tsx`, `gui/src/pages/use-dashboard-data.ts` + +Append to `desktop-shell.ts` (using the existing `isDesktopShell` at lines 11-13 and `hostOs` at lines 29-34): + +```ts +export function desktopUpdatePageUrl(ua = currentUserAgent()): string | null { + if (!isDesktopShell(ua)) return null; + const os = hostOs(ua); + if (os === "windows") return "http://tauri.localhost/update.html"; + if (os === "macos" || os === "linux") return "tauri://localhost/update.html"; + return null; +} + +export function openDesktopUpdatePage(ua = currentUserAgent()): boolean { + const url = desktopUpdatePageUrl(ua); + if (!url) return false; + window.location.assign(url); + return true; +} +``` + +In `App.tsx:29`, import `openDesktopUpdatePage`; replace the `onOpenUpdate` body at `App.tsx:456-462` with: + +```tsx +onOpenUpdate={() => { + setNavOpen(false); + if (openDesktopUpdatePage()) return; + navigateToPage("dashboard", "update"); +}} +``` + +In `use-dashboard-data.ts:1-6`, import `openDesktopUpdatePage`; prepend `if (openDesktopUpdatePage()) return;` to `openUpdateDialog` at line 861. This catches the dashboard maintenance anchor (`dashboard-overview-sections.tsx:219-229`) and cold `#dashboard/update` deep links consumed at `use-dashboard-data.ts:887-901`; it executes before `fetchUpdateCheck` or `/api/update/run`. The normal browser path then executes the unchanged package dialog. `runUpdate` at lines 903-919 remains unchanged and unreachable from the desktop entry path. + +Commit 2 already replaces `sidebar-github-row.tsx`'s badge request with `updateBadgeUrl`, changes the desktop poll to 60 seconds while retaining the 10-minute `BADGE_POLL_MS` browser default, and extends `UpdateBadge.installer` to `"desktop"` ([020_phase2_desktop_state_icons.md](020_phase2_desktop_state_icons.md), “Embedded GUI poll”, `badgePoll`). Keep that exact code. In wp3 import `isDesktopShell` from `../lib/desktop-shell` and replace only the label expression at current lines 127-131 with: + +```tsx +const updateLabel = updateAvailable && latestVersion + ? t("sidebar.updateAvailable", { version: latestVersion }) + : isDesktopShell() ? t("sidebar.desktopUpdate") : t("sidebar.checkUpdate"); +``` + +This is the only new GUI copy. Update the top comment at lines 10-14 and prop comment at line 51 to describe the two destinations. + +### MODIFY all ten GUI catalogs + +Add one key adjacent to `sidebar.checkUpdate` (source at `gui/src/i18n/en.ts:92-93`) in every catalog. The page's English copy is outside React/i18n, matching the English-only bootstrap page. + +| Path | Exact key/value to add | +| --- | --- | +| `gui/src/i18n/en.ts` | `"sidebar.desktopUpdate": "Open desktop updates",` | +| `gui/src/i18n/de.ts` | `"sidebar.desktopUpdate": "Desktop-Updates öffnen",` | +| `gui/src/i18n/fr.ts` | `"sidebar.desktopUpdate": "Ouvrir les mises à jour de l’application",` | +| `gui/src/i18n/ja.ts` | `"sidebar.desktopUpdate": "デスクトップアプリの更新を開く",` | +| `gui/src/i18n/ko.ts` | `"sidebar.desktopUpdate": "데스크톱 앱 업데이트 열기",` | +| `gui/src/i18n/ru.ts` | `"sidebar.desktopUpdate": "Открыть обновления приложения",` | +| `gui/src/i18n/tr.ts` | `"sidebar.desktopUpdate": "Masaüstü güncellemelerini aç",` | +| `gui/src/i18n/vi.ts` | `"sidebar.desktopUpdate": "Mở cập nhật ứng dụng máy tính",` | +| `gui/src/i18n/zh.ts` | `"sidebar.desktopUpdate": "打开桌面应用更新",` | +| `gui/src/i18n/zh-TW.ts` | `"sidebar.desktopUpdate": "開啟桌面應用程式更新",` | + +### NEW `tests/clients/desktop-update-surface.test.ts` — full file + +```ts +import { describe, expect, test } from "bun:test"; +import { readFileSync } from "node:fs"; +import { runInNewContext } from "node:vm"; +import { repoPath } from "../helpers/repo-root"; + +const read = (path: string) => readFileSync(repoPath(path), "utf8"); +const page = read("desktop/ui/update.html"); +const lib = read("desktop/src-tauri/src/lib.rs"); +const updater = read("desktop/src-tauri/src/updater.rs"); +const tray = read("desktop/src-tauri/src/tray.rs"); +const windowPolicy = read("desktop/src-tauri/src/window.rs"); + +function evaluatePage(invoke?: (name: string) => Promise) { + const script = page.match(/ + + diff --git a/docs-site/src/content/docs/contributing.md b/docs-site/src/content/docs/contributing.md index e6f9844c1f..a7c1dd547d 100644 --- a/docs-site/src/content/docs/contributing.md +++ b/docs-site/src/content/docs/contributing.md @@ -12,17 +12,24 @@ Bun runtime for users, but this checkout's scripts run through your local Bun in git clone https://github.com/lidge-jun/opencodex.git cd opencodex bun install +bun run setup:hooks # retire the managed pre-push and post-merge hooks bun run dev:proxy # proxy API in dev mode bun run dev:gui # dashboard dev server (another terminal) bun run typecheck # bun x tsc --noEmit -bun run test:changed # routine import-graph test selection -bun test tests/routing/router.test.ts # routine focused test -bun run test # complete suite (PR-ready / explicit ask) +bun run test # full suite (default) ``` +`bun run setup:hooks` removes an unmodified retired managed `pre-push` or `post-merge` +hook, preserving custom hooks. A pre-push hook is no longer required. +`bun run prepush` remains an optional manual check. + `bun run dev` remains an alias for `bun run dev:proxy`. The dashboard dev server is `bun run dev:gui`; the packaged dashboard at `GET /` is produced by `bun run build:gui` (`gui/dist`). +The retired `post-merge` hook used to rebuild `gui/dist` after every merge. With the hook gone, +a merge that changes `gui/` leaves the packaged dashboard stale until you run `bun run build:gui` +yourself — the dev server is unaffected because it rebuilds on demand. + ## Build and test commands The root package is Bun-native TypeScript; there is no separate server compile step. Use the checked-in @@ -31,13 +38,21 @@ scripts so local commands match CI: ```bash bun run typecheck # strict TypeScript check bun run test:changed # import-graph tests against the resolved dev merge base -bun run test # complete tests/ suite (PR-ready / explicit ask) +bun run test # full suite (default) bun test tests/routing/router.test.ts # focused test file bun run build:gui # Vite GUI build + package preparation bun run privacy:scan # credential/privacy scan used by CI bun run prepare:package # refresh package launchers/assets ``` +Run `bun run test` by default. If a full run is disproportionately expensive for the task size, +available machine resources, or concurrent worktrees, you must at least run focused regression tests +that exercise the changed behavior, such as `bun test tests//.test.ts`. Explain the +reason for narrowing the run and report the exact commands, results, and untested scope. +`bun run test:changed` can supplement this coverage, but it cannot detect every indirect dependency. +Neither relying only on CI nor skipping local testing is a blanket exemption. Before merge, all +required CI checks must pass for the exact current PR head. + `test:changed` selects the first comparison ref that exists, in order: `upstream/dev`, `origin/dev`, then local `dev`. It reports that ref and the exact `git merge-base HEAD ` commit, then passes the merge-base SHA to Bun. @@ -54,8 +69,7 @@ and `tests/test-layout.test.ts` enforces it, so a new test goes into its domain an entry in the map (the tooling test tells you which one is missing). `tests/helpers/` holds shared fixtures and `tests/helpers/repo-root.ts` is how a test reaches repository files; `tests/e2e-style/` holds broader native-parity scenarios. Keep a focused regression near the -existing tests for the subsystem you change (`bun test tests/` runs one subsystem); run -the full suite for shared routing, adapters, config, or server behavior. +existing tests for the subsystem you change (`bun test tests/` runs one subsystem). The docs site you're reading lives in `docs-site/` (Astro + Starlight): @@ -144,7 +158,7 @@ description. - Target **`dev`**. Do not open feature or fix pull requests against **`main`**. - Branch from the current **`dev`** tip, not from **`main`**. The required **`enforce-target`** check rejects heads whose merge base sits on the **`main`** tip while the branch is far behind the pull request base (the failure mode seen in #644). - Write a real description: a **Summary** of what changed and why, plus a **Test plan** (or equivalent substance). Empty bodies, placeholder-only text, and descriptions that use escaped `\n` instead of real line breaks fail the check. -- If the title or description mentions `gui`, include a screenshot of the UI change in the description; the `enforce-target` check re-runs on description edits until the screenshot is present. +- If the pull request changes files under `gui/`, include a screenshot of the UI change in the description; the `enforce-target` check re-runs on description edits until the screenshot is present. Drag the image into the description editor rather than committing it: an image on your branch rides the squash merge into `dev`. Maintainers uploading from the command line use the `pr-assets` branch and link by commit SHA. - Workflow changes in this repository use **`pull_request_target`**. Updated enforcement logic applies only after the workflow is promoted to the repository default branch — the same operational caveat documented in #631. ## Project maintainers @@ -167,7 +181,7 @@ does not change `main`/`preview` review rules or allow direct pushes, force-push - **Handle async errors at boundaries** — sidecars never throw into the request path; they degrade to a graceful marker. - **Structure SOT** — current maintainer invariants live in `structure/`. Keep public user workflows - in `docs-site/` and historical investigation notes in `docs/`. + in `docs-site/` and planning and investigation notes in `devlog/`. - **Preserve exports** — other modules may depend on them. ## Adding a provider to the catalog @@ -250,6 +264,6 @@ startup path must not import the manifest catalog or activate Compatibility Lab. ## Verify before you claim done -Run the narrowest command that proves your change — `bun run typecheck` for types, a focused -`bun test tests//.test.ts` or runtime probe for behavior, then the broader gates appropriate to -the affected surface. opencodex favors small, verifiable commits over large batches. +Follow the testing policy above and run `bun run typecheck` for type changes, plus the checks +required for the affected surface. Report the commands, results, and remaining untested scope; +only claim the validation you actually completed. diff --git a/docs-site/src/content/docs/contributing/pr-quality.md b/docs-site/src/content/docs/contributing/pr-quality.md index 27bf69f555..bac390abde 100644 --- a/docs-site/src/content/docs/contributing/pr-quality.md +++ b/docs-site/src/content/docs/contributing/pr-quality.md @@ -51,8 +51,8 @@ tells you exactly what to change: self-waive the screenshot requirement. Contributor PRs (authors without repository push permission) open in draft and stay there until a four-box review-readiness checklist in the - description is complete: local CI green, the branch on the latest `dev` - commit, all correct Codex and CodeRabbit findings fixed, and the + description is complete: required local validation passed with its scope + documented, the branch on the latest `dev` commit, all correct Codex and CodeRabbit findings fixed, and the ready-for-review confirmation. Once every box is ticked the check marks the PR ready for review and notifies the maintainers listed in `MAINTAINERS.md` (excluding the author). The gate's status and "what to do" live in a single @@ -67,8 +67,9 @@ tells you exactly what to change: can check itself: the branch must be on the latest `dev` commit or at most 10 commits behind it, and every Codex and CodeRabbit review thread authored by a review bot on the current head must be resolved (unresolved threads - from other authors do not block). The local-CI box is an author attestation - only — fork contributors cannot start repository CI; a maintainer has to — + from other authors do not block). The local-validation box follows the [test-scope policy](/contributing/#build-and-test-commands): + run the full suite by default; when it is too costly, run focused regressions + and document the exception. It is an author attestation only — fork contributors cannot start repository CI; a maintainer has to — so the gate never disproves it; a new push still resets every box. CodeRabbit findings that fall outside the diff range and are reported only in a review body on the current head add to the unresolved count while a bot review @@ -131,3 +132,14 @@ A PR that stalls with unresolved review feedback may be closed, with the reason stated plainly. Closure is not a verdict on the contributor: reopen it once the stated reason is resolved, or replace it with a clean one. Ask if the reason is not clear. + +## Updating an older readiness checklist + +If the gate reports that your checklist still uses the retired local-CI wording, it preserves +your description and asks you to update the first item. Change that item to the wording in +the bot notice, clear all four boxes and save. Wait for the bot to acknowledge the cleared +checklist, validate the displayed head, then tick all four boxes and save again. Changing +only the wording while leaving four ticks does not count as a new attestation. A push or +retarget invalidates the checkpoint. The gate stays red and the PR stays draft until this +sequence and the ordinary quality checks complete. If your save shares the checkpoint's +timestamp, make another body edit and save later; editing only the title does not count. diff --git a/docs-site/src/content/docs/fr/contributing.md b/docs-site/src/content/docs/fr/contributing.md index 345d31359f..b27a50689f 100644 --- a/docs-site/src/content/docs/fr/contributing.md +++ b/docs-site/src/content/docs/fr/contributing.md @@ -12,14 +12,17 @@ runtime Bun aux utilisateurs, mais les scripts de ce dépôt utilisent votre ins git clone https://github.com/lidge-jun/opencodex.git cd opencodex bun install +bun run setup:hooks # retirer les anciens hooks gérés pre-push et post-merge bun run dev:proxy # proxy API in dev mode bun run dev:gui # dashboard dev server (another terminal) bun run typecheck # bun x tsc --noEmit -bun run test:changed # routine import-graph test selection -bun test tests/routing/router.test.ts # routine focused test -bun run test # complete suite (PR-ready / explicit ask) +bun run test # suite complète (par défaut) ``` +`bun run setup:hooks` supprime les anciens hooks gérés `pre-push` et `post-merge` non modifiés, +tout en préservant les hooks personnalisés. Le hook `pre-push` +n’est plus obligatoire. `bun run prepush` reste une vérification manuelle facultative. + `bun run dev` reste un alias pour `bun run dev:proxy`. Le serveur de développement du tableau de bord est `bun run dev:gui` ; le tableau de bord packagé en `GET /` est produit par `bun run build:gui` (`gui/dist`). @@ -31,17 +34,25 @@ distincte. Utilisez les scripts enregistrés afin que les commandes locales corr ```bash bun run typecheck # strict TypeScript check bun run test:changed # import-graph tests against the resolved dev merge base -bun run test # complete tests/ suite (PR-ready / explicit ask) +bun run test # suite complète (par défaut) bun test tests/routing/router.test.ts # focused test file bun run build:gui # Vite GUI build + package preparation bun run privacy:scan # credential/privacy scan used by CI bun run prepare:package # refresh package launchers/assets ``` +Exécutez `bun run test` par défaut. Si une exécution complète est disproportionnée par rapport +à la taille de la tâche, aux ressources de la machine ou aux worktrees utilisés en parallèle, vous +devez au minimum exécuter des tests de régression ciblés qui exercent le comportement modifié, par +exemple `bun test tests//.test.ts`. Expliquez ce choix et indiquez les commandes +exactes, leurs résultats et le périmètre non testé. `bun run test:changed` peut compléter cette +couverture, mais ne détecte pas toutes les dépendances indirectes. Se reposer uniquement sur la CI +ou omettre les tests locaux ne constitue pas une exemption générale. Avant la fusion, tous les +contrôles CI obligatoires doivent réussir sur le commit exact de la tête actuelle de la PR. + Les tests Bun vivent dans des répertoires par domaine calqués sur `src/` (`tests//`), la carte étant `scripts/test-layout/layout.json`. `tests/helpers/` contient les fixtures partagées et `tests/e2e-style/` des scénarios plus larges de parité native. Placez une régression ciblée près -des tests existants du sous-système modifié. Exécutez la suite complète pour le routage partagé, les adaptateurs, -la configuration ou le comportement du serveur. +des tests existants du sous-système modifié. Le site de documentation que vous lisez se trouve dans `docs-site/` (Astro + Starlight) : @@ -136,8 +147,11 @@ une contribution normale ; indiquez les commits sources dans la description. - Rédigez une vraie description : un **Résumé** de la modification et de sa raison, ainsi qu’un **Plan de test** ou un contenu équivalent. Les corps vides, les textes composés seulement d’espaces réservés et les descriptions contenant des `\n` échappés au lieu de véritables sauts de ligne échouent au contrôle. -- Si le titre ou la description mentionne `gui`, incluez dans la description une capture d’écran de la modification - de l’interface. `enforce-target` est réexécuté après chaque modification de la description jusqu’à sa présence. +- Si la pull request modifie des fichiers sous `gui/`, ajoutez une capture d’écran de la modification de + l’interface dans sa description. `enforce-target` est réexécuté après chaque modification de la description + jusqu’à sa présence. Glissez l’image dans la description au lieu de la committer sur la branche : elle serait + incluse dans le squash merge vers `dev`. Pour un envoi en ligne de commande, les responsables utilisent la + branche `pr-assets` avec un lien vers le SHA du commit. - Les workflows de ce dépôt utilisent **`pull_request_target`**. Une nouvelle logique d’application ne prend effet qu’après la promotion du workflow vers la branche par défaut, conformément à l’avertissement opérationnel de #631. @@ -155,7 +169,7 @@ du dépôt et des chemins sensibles du point de vue de la sécurité est déclar - **Gérer les erreurs asynchrones aux frontières** — les services auxiliaires ne propagent jamais d’exception dans le chemin de requête ; ils se dégradent en un marqueur explicite. - **Structure, source de vérité** — les invariants actuels des responsables résident dans `structure/`. Conservez - les parcours utilisateurs publics dans `docs-site/` et les notes d’enquête historiques dans `docs/`. + les parcours utilisateurs publics dans `docs-site/` et les notes de planification et d’enquête dans `devlog/`. - **Préserver les exportations** — d'autres modules peuvent en dépendre. ## Ajout d'un fournisseur au catalogue @@ -224,6 +238,6 @@ la fabrique depuis `src/index.ts` lorsqu’elle appartient à l’API publique d ## Vérifiez avant de déclarer que c'est fait -Exécutez la commande la plus étroite qui prouve votre changement — `bun run typecheck` pour les types, un -`bun test tests/.test.ts` ou une sonde d'exécution pour le comportement, puis les portes plus larges appropriées à -la surface affectée. opencodex privilégie les petits commits vérifiables plutôt que les gros lots. +Suivez la politique de test ci-dessus et exécutez `bun run typecheck` pour les changements de types, +ainsi que les vérifications requises pour la zone concernée. Indiquez les commandes, les résultats +et le périmètre non testé ; ne revendiquez que les validations réellement effectuées. diff --git a/docs-site/src/content/docs/fr/contributing/pr-quality.md b/docs-site/src/content/docs/fr/contributing/pr-quality.md index 448432cbc1..0711cb3b0e 100644 --- a/docs-site/src/content/docs/fr/contributing/pr-quality.md +++ b/docs-site/src/content/docs/fr/contributing/pr-quality.md @@ -44,8 +44,8 @@ Trois contrôles déterministes précèdent la revue humaine. Chaque message d déclenchent plus le contrôle privilégié. Un contributeur ne peut pas lever lui-même cette exigence. Les PR de contributeurs sans droit de push sur le dépôt s’ouvrent en brouillon et le restent jusqu’à ce que - les quatre cases de préparation à la revue soient cochées dans la description : CI locale verte, branche sur - le dernier commit de `dev`, tous les constats valides de Codex et CodeRabbit corrigés, et confirmation de + les quatre cases de préparation à la revue soient cochées dans la description : validation locale requise réussie + (commandes, résultats et toute exception à la suite complète documentés), branche sur le dernier commit de `dev`, tous les constats valides de Codex et CodeRabbit corrigés, et confirmation de disponibilité pour la revue. Lorsque les quatre cases sont cochées, le contrôle marque la PR comme prête et avertit les responsables répertoriés dans `MAINTAINERS.md`, à l’exclusion de l’auteur. L’état du contrôle et les actions attendues figurent dans un unique commentaire consolidé, réécrit à chaque exécution. @@ -58,7 +58,7 @@ Trois contrôles déterministes précèdent la revue humaine. Chaque message d Avant d’accepter la liste, le contrôle vérifie les affirmations qu’il peut lui-même confirmer : la branche doit être sur le dernier commit de `dev`, ou au plus 10 commits derrière, et tous les fils de revue Codex et CodeRabbit créés par un robot sur la tête actuelle doivent être résolus. Les fils non résolus d’autres auteurs - ne bloquent pas. La case de CI locale est uniquement une attestation de l’auteur : les contributeurs depuis un + ne bloquent pas. La case de validation locale requise est uniquement une attestation de l’auteur : les contributeurs depuis un fork ne peuvent pas démarrer la CI du dépôt, seul un responsable le peut. Le contrôle ne contredit donc jamais cette case, mais tout nouveau push réinitialise toutes les cases. @@ -76,7 +76,8 @@ Trois contrôles déterministes précèdent la revue humaine. Chaque message d - **Hygiène.** Les changements de comportement exigent un test. Les nouvelles suppressions de règles de lint ou de types, les tests ciblés ou ignorés, les blocs catch vides, la modification de sorties générées et celle d’un lockfile sans son manifeste nécessitent chacun un label d’approbation explicite. Une modification limitée - à un commentaire dans un fichier source ne change pas le comportement et n’exige aucun test. + à un commentaire dans un fichier source ne change pas le comportement et n’exige aucun nouveau test + de régression. La [politique de test locale](/fr/contributing/) reste applicable. - **CI multiplateforme.** Pour les changements concernés, la suite est fragmentée sous Linux et exécutée intégralement sous macOS pour chaque pull request. La voie Windows principale ne s’exécute actuellement que @@ -117,3 +118,15 @@ seules surfaces soumises à cette règle. Toutes les autres restent ouvertes. Une PR bloquée par des remarques de revue non résolues peut être fermée, avec une raison clairement indiquée. La fermeture n’est pas un jugement sur le contributeur : rouvrez la PR lorsque la raison donnée est résolue, ou remplacez-la par une nouvelle PR propre. Demandez des précisions si la raison n’est pas claire. + +## Mettre à jour une ancienne liste de préparation + +Si le contrôle signale l’ancienne formulation sur la CI locale, il conserve votre description. +Remplacez le premier élément par le texte indiqué dans le commentaire du bot, décochez les +quatre cases et enregistrez. Attendez que le bot confirme cette étape, validez le commit +indiqué, puis cochez les quatre cases et enregistrez de nouveau. Modifier uniquement le +texte en conservant les quatre coches ne renouvelle pas l’attestation. Un push ou un +changement de branche cible invalide cette étape. Le contrôle reste en échec et la PR en +brouillon jusqu’à la fin de cette procédure et des contrôles habituels. Si l’enregistrement +partage l’horodatage de l’étape, modifiez de nouveau le corps et enregistrez plus tard ; +modifier seulement le titre ne suffit pas. diff --git a/docs-site/src/content/docs/fr/getting-started/quickstart.md b/docs-site/src/content/docs/fr/getting-started/quickstart.md index 73512a4a6c..3cf3a91b39 100644 --- a/docs-site/src/content/docs/fr/getting-started/quickstart.md +++ b/docs-site/src/content/docs/fr/getting-started/quickstart.md @@ -13,7 +13,7 @@ ocx init `ocx init` vous accompagne dans les étapes suivantes : -1. **Choix d’un fournisseur** — sélectionnez l’un des 96 préréglages intégrés au registre, ou `custom` pour saisir une +1. **Choix d’un fournisseur** — sélectionnez l’un des 100 préréglages intégrés au registre, ou `custom` pour saisir une URL de base et un adaptateur. 2. **Clé API** — collez une clé ou référencez une variable d’environnement telle que `${ANTHROPIC_API_KEY}`. 3. **Modèle par défaut** — pour les fournisseurs clés, locaux et personnalisés, acceptez le préréglage ou saisissez un identifiant de modèle. diff --git a/docs-site/src/content/docs/fr/guides/claude-code.md b/docs-site/src/content/docs/fr/guides/claude-code.md index 268c562326..e2f9b6ece5 100644 --- a/docs-site/src/content/docs/fr/guides/claude-code.md +++ b/docs-site/src/content/docs/fr/guides/claude-code.md @@ -28,8 +28,7 @@ Comportement lorsque cette option est activée : - Un **429** en amont place le compte en temporisation selon `Retry-After` lorsqu'il est présent, ou selon un délai de repli, efface ses affinités et peut faire basculer la requête vers un autre compte admissible, dans les limites prévues. - L'affinité est **locale au processus** et disparaît au redémarrage du proxy. -- Les erreurs d'identification **401/403** mettent le compte en quarantaine (`needsReauth`) afin de l'exclure de la - sélection jusqu'à sa réauthentification. +- Les erreurs de renouvellement du jeton conservent la règle `needsReauth`. Un 403 confirmé lié à un abonnement ou à la facturation du compte peut déclencher un basculement avant la sortie, avec une temporisation selon `Retry-After` ou de dix minutes. Un refus d’autorisation ordinaire reste terminal. Voir la [reprise des comptes](/guides/claude-code/). - Si chaque compte éligible est en temporisation, le proxy renvoie **429** (et non 401) avec `Retry-After` lorsqu'il est connu. - La récupération, y compris le basculement 429, utilise `quotaWindow` pour classer les comptes de @@ -54,10 +53,10 @@ ocx claude | `CLAUDE_CODE_AUTO_COMPACT_WINDOW` | Seuil de compactage du contexte automatique (par défaut `829800`) ; injecté uniquement lorsque le contexte automatique est activé | | `ANTHROPIC_MODEL` | `claudeCode.model` (facultatif) | | `ANTHROPIC_DEFAULT_HAIKU_MODEL` | `claudeCode.tierModels.haiku ?? claudeCode.smallFastModel` (facultatif ; ancien `ANTHROPIC_SMALL_FAST_MODEL` également) | -| `ANTHROPIC_DEFAULT_{OPUS,SONNET,FABLE}_MODEL` | `claudeCode.tierModels.*` (facultatif) | +| `ANTHROPIC_DEFAULT_{OPUS,SONNET,FABLE}_MODEL` | `claudeCode.tierModels.*` ; lors d'un lancement par abonnement, si non défini, `claude-opus-5-5[1m]` / `claude-sonnet-5[1m]` / `claude-fable-5-1[1m]` natif | | `CLAUDE_CODE_ALWAYS_ENABLE_EFFORT` | `1` lorsque `alwaysEnableEffort` est activé (conditionnel) | | `ENABLE_TOOL_SEARCH` | `claudeCode.toolSearch` lorsqu'il est défini (conditionnel ; désactivé par défaut) | -| `CLAUDE_CODE_MAX_CONTEXT_TOKENS` / `DISABLE_COMPACT` | Remplacement du contexte hérité lorsque `maxContextTokens` est défini (conditionnel) | +| `CLAUDE_CODE_MAX_CONTEXT_TOKENS` | Remplacement du contexte hérité lorsque `maxContextTokens` est défini (conditionnel) | Les variables que vous exportez vous-même gagnent toujours. Les arguments supplémentaires passent par : `ocx claude -p "hello"`. Une exception porte sur *l'origine* d'une variable, et non sur sa priorité. L'environnement d'exécution Bun fourni @@ -121,35 +120,102 @@ Sur macOS, l'intégration automatique (`claudeCode.systemEnv`) suit la même ré `claude` lancée sans passer par `ocx` se comporte donc de la même manière. Le fichier d'environnement est un instantané actualisé au démarrage du proxy ou lors de l'enregistrement des paramètres, tandis que `ocx claude` effectue toujours une résolution immédiate. -## Modes Claude Desktop : first-party (par défaut) et passerelle - -Claude Desktop utilise OpenCodex dans l'un de deux modes mutuellement exclusifs. Choisissez-le dans -**Claude → Bureau → Mode de connexion** du tableau de bord ou avec -`ocx claude desktop apply --first-party|--gateway`. - -- **First-party (par défaut)** : Desktop lui-même n'est pas reconfiguré. La connexion claude.ai, - l'onglet Chat, les connecteurs et le contrôle à distance continuent de fonctionner. OpenCodex - n'écrit que deux valeurs dans le bloc `env` de `~/.claude/settings.json` : - `HTTPS_PROXY=http://127.0.0.1:` et - `NODE_EXTRA_CA_CERTS=~/.opencodex/claude-intercept/ca.pem`. Seuls Claude Code lancé par Desktop - pour l'onglet Code (sous-agents compris) et la CLI `claude` du terminal les lisent et passent par le - proxy d'interception local ; seuls `POST /v1/messages` et `count_tokens` sont traités par OpenCodex, - les autres chemins de `api.anthropic.com` sont relayés tels quels vers Anthropic. L'AC n'est jamais - installée dans le magasin de confiance du système. -- **Passerelle (tiers)** : l'ancien mode ; le profil ci-dessous fait basculer toute l'application sur - OpenCodex comme passerelle. Sélectionnez-le explicitement (`--gateway`, ou les anciens - `--static`/`--hybrid`/`--discovery-only`). - -Le mode est enregistré dans `claudeCode.desktopMode`. Les installations ayant déjà appliqué un profil -passerelle le conservent après mise à jour ; seules les nouvelles installations démarrent en -first-party. Changer de mode supprime la configuration de l'autre mode (uniquement les valeurs -écrites par OpenCodex) ; un `HTTPS_PROXY`/`NODE_EXTRA_CA_CERTS` étranger (proxy d'entreprise) n'est -jamais écrasé et l'application est refusée. Quittez complètement Desktop puis rouvrez-le après un -changement. Les détails et la compatibilité de la CLI Claude Code sont décrits dans la documentation -anglaise. +## Modes Claude Desktop : passerelle (par défaut) et first-party + +Choisissez le mode dans **Claude → Bureau → Mode de connexion** ou avec +`ocx claude desktop apply --first-party|--gateway`. Les deux modes sont exclusifs. + +### Passerelle (par défaut) + +Une nouvelle installation applique par défaut le profil passerelle : toute l'application utilise +OpenCodex, y compris l'onglet Chat. Les fonctions réservées à claude.ai ne sont alors pas disponibles. +Les anciens indicateurs `--static`, `--hybrid` et `--discovery-only` sélectionnent aussi ce mode. + +### First-party (sur demande) + +:::caution[Risque pour le compte] +Le mode first-party fait passer le trafic de votre abonnement Claude par un proxy local d'interception. +Anthropic peut y voir une violation de ses conditions et suspendre votre compte. La passerelle est +le choix par défaut ; n'activez first-party que si vous acceptez ce risque. +::: + +Le mode first-party de Desktop route son onglet Code et ses sous-agents via OpenCodex. La CLI Claude Code autonome possède un interrupteur distinct. Les deux lisent les mêmes réglages de proxy et d’autorité dans `settings.json` : si un seul mode est actif, l’autre client traverse encore le proxy local, où TLS se termine, mais ses requêtes Messages sont relayées sans modification vers Anthropic. + +Le mode est enregistré dans `claudeCode.desktopMode`. Une installation ayant déjà appliqué le mode +first-party, même avant cette version, le conserve ; un profil passerelle existant reste aussi en +passerelle. Sans mode explicite, le profil passerelle sélectionné et détenu par OpenCodex, puis son +empreinte enregistrée, priment sur les réglages first-party détenus dans `settings.json` ; sans ces +indices, le mode est passerelle. Un environnement écrit uniquement pour le first-party de la CLI ne constitue pas une preuve que Desktop est en mode first-party. La synchronisation du catalogue et la mise à jour de la liste des +modèles n'écrivent jamais un profil passerelle sur une installation first-party. Si +`claudeCode.intercept.enabled: false`, l'application d'un mode first-party existant est refusée +(`intercept_disabled`) ; une nouvelle installation applique la passerelle. Un proxy d'entreprise +étranger n'est pas écrasé. Quittez complètement Desktop puis rouvrez-le après un changement. + +### First-party de la CLI Claude Code + +Activez l’interrupteur dans Claude → Code ou lancez `ocx claude config set --first-party on` ; utilisez `off` pour désactiver. L’activation est refusée si le proxy local est indisponible, si l’autorité ne peut être préparée, si les réglages sont illisibles ou si des clés appartiennent à un autre programme. La désactivation reste enregistrable. Avec le seul first-party Desktop actif, définissez `NO_PROXY='*'` dans le shell pour un terminal entièrement natif. Le risque pour le compte décrit ci-dessus s’applique aussi à la CLI. +Désactiver le routage Claude conserve les variables de proxy gérées. Tant que le listener fonctionne, toutes les requêtes Messages sont relayées sans modification ; après son arrêt, `claude` ne peut plus se connecter avant le lancement d’OpenCodex ou la désactivation du first-party Desktop/CLI. Le lancement natif via `ocx claude` ne définit `NO_PROXY=*` que pour un environnement géré sans proxy HTTPS hérité d’un autre programme. Sinon il conserve ce proxy et avertit que l’interception définie dans les réglages reste active ; désactivez le first-party ou retirez ce réglage. +L’interface distingue les réglages illisibles (unknown), une URL opencodex avec jeton mais une AC étrangère (foreign : corrigez HTTPS_PROXY / NODE_EXTRA_CA_CERTS à la main) et le routage Claude désactivé avec un listener encore actif qui relaie sans modification (disabled : désactivez first-party avant le redémarrage). Sans listener, l’état est stopped ; avec une AC gérée mais un port ou jeton incorrect, il est broken. Avec le first-party actif et une interception indisponible, stopped et broken affichent routingOff : le routage Claude ou l’interception est désactivé, ou cette machine est cliente d’un autre hub opencodex ; réactivez l’interception ici ou désactivez first-party pour supprimer les réglages. Si l’interception est disponible, stopped demande de lancer opencodex et broken conseille `ocx ensure` ou un redémarrage. La CLI activée sans proxy est non appliquée ; un seul client activé avec un proxy opérationnel partage le relais ; aucun client activé avec un proxy restant produit un avertissement de réglage résiduel. +unknown signifie qu’opencodex ne peut pas déterminer si les réglages pointent encore vers son proxy. Un proxy sans jeton sur 127.0.0.1 avec une AC étrangère est local : sa propriété est incertaine ; supprimez HTTPS_PROXY de ~/.claude/settings.json si vous ne l’utilisez plus. disabled exige des réglages appliqués correspondant au listener ; un port ou jeton différent donne broken même si le routage est désactivé. + +### Mode picker : modèles opencodex dans le sélecteur Code first-party + +Le mode picker fait partie du mode first-party. Sur macOS, il est activé par défaut lorsque first-party +est sélectionné, sauf si `claudeCode.intercept.picker: false` est défini. Il modifie le sélecteur de +modèles de l'onglet Code de Desktop first-party pour y afficher les modèles opencodex disponibles par +leur nom. Lors de la première activation, macOS peut demander l'autorisation d'une autorité de certification +locale dans le trousseau de connexion. Cette autorité est limitée à `claude.ai` et à ses sous-domaines. +Sa clé de signature n'existe que dans le processus OpenCodex en cours : chaque redémarrage d'OpenCodex +publie une nouvelle autorité et macOS demande donc de nouveau votre confiance — approuvez la demande, +ou lancez ensuite `ocx claude desktop picker trust`, après chaque redémarrage. + +Lorsque le mode picker est actif, Claude Desktop accède au réseau par OpenCodex. Si OpenCodex s'arrête, +Desktop reste hors ligne jusqu'à son redémarrage complet ou jusqu'à la désactivation du mode picker. +Consultez l'état avec `ocx claude desktop picker status`, relancez l'étape de confiance avec +`ocx claude desktop picker trust`, ou désactivez-le avec `ocx claude desktop picker off`. Le tableau +de bord propose le même interrupteur dans **Claude → Bureau**. Après la sélection du profil picker, +quittez complètement puis rouvrez Claude Desktop. + +Le mode picker fait partie de first-party : le [risque pour le compte du mode first-party](#first-party-sur-demande) +s'applique donc aussi à ce mode. + +### Utiliser les modèles opencodex depuis l'onglet Code de Desktop (associations first-party) + +En mode first-party, le sélecteur de modèles de l'onglet Code appartient à claude.ai : ses lignes +(Opus 5.5, Sonnet 5, Haiku 4.5 et les anciens modèles sous **More models**) viennent de votre +compte, et aucun réglage local ne peut y ajouter une ligne opencodex. Ce qui arrive à OpenCodex, +c'est l'identifiant de modèle Anthropic du sélecteur à chaque requête ; on associe donc une ligne +du sélecteur à une route opencodex : + +```bash +ocx claude desktop bind claude-sonnet-4-6 xai/grok-4.7 +ocx claude desktop bind claude-opus-4-6 native/gpt-6.1-sol +ocx claude desktop unbind claude-opus-4-6 +``` + +ou utilisez **Claude → Bureau → Associations de modèles de l'onglet Code** dans le tableau de bord. +Choisir **Sonnet 4.6** dans l'onglet Code est alors servi par `xai/grok-4.7`. Le sélecteur garde le +nom Anthropic, et le modèle continue d'être présenté comme ce modèle Claude par le prompt système de +Claude Code ; préférez donc des lignes que vous n'utilisez pas par ailleurs (les entrées +**More models** sont de bonnes candidates). Les associations s'appliquent dès la requête suivante ; +Desktop n'a pas besoin d'être relancé. + +- Les routes suivent le vocabulaire des routes Desktop : `provider/model`, ou `native/` pour + le pool OpenAI natif. La route doit figurer comme disponible dans le tableau de bord. +- Un identifiant de sélecteur daté (`claude-haiku-4-5-20251001`) correspond à une association non + datée (`claude-haiku-4-5`), et les sélections `[1m]` et du mode rapide suivent la même association. +- Les associations sont enregistrées dans `claudeCode.intercept.modelMap` et ne s'appliquent qu'au + trafic Claude Code qui passe par le proxy d'interception local : l'onglet Code de Desktop et la CLI + `claude` autonome en mode first-party. Les sessions `ocx claude` et le point d'entrée public + `/v1/messages` les ignorent ; le `claudeCode.modelMap` global continue de s'appliquer partout, et + une association l'emporte sur lui pour le même identifiant. +- `ocx claude desktop status --json` rapporte les associations en vigueur sous + `firstParty.modelBindings`. ## Profil Claude Desktop (mode passerelle) +Le profil ci-dessous n'est écrit que lorsque le mode passerelle est sélectionné. + Claude Desktop utilise un profil distinct de Claude Code. Ouvrez **Claude → Bureau** dans le tableau de bord afin de placer chaque route disponible dans l'une des quatre familles : Opus, Fable, Sonnet ou Haiku. Dans un nouveau profil, toutes les routes appartiennent initialement à Opus. La première route Opus devient la route globale @@ -174,7 +240,8 @@ ocx claude desktop export ocx claude desktop import [--apply] ``` -`ocx claude desktop` et `apply` écrivent tous deux le profil actuel dans Claude Desktop. `show` affiche un +`ocx claude desktop` et `apply` appliquent le mode sélectionné : first-party écrit l'environnement +du proxy Claude Code, tandis que passerelle écrit le profil Desktop. `show` affiche un résumé lisible ; ajoutez `--json` pour les scripts. `export -` écrit le document JSON versionné sur la sortie standard. L'importation valide le fichier entier avant tout enregistrement : un fichier invalide laisse donc le profil actuel inchangé. Ajoutez `--apply` pour écrire immédiatement un profil importé valide dans Claude Desktop. Utilisez `none` uniquement @@ -187,9 +254,9 @@ Support/Claude/configLibrary` sur macOS, `%APPDATA%\Claude\configLibrary` sur Wi `CLAUDE_USER_DATA_DIR` pour utiliser une autre racine de données Claude Desktop. L'ancien répertoire `Claude-3p` n'est ni lu ni supprimé automatiquement. -Les routes non Anthropic reçoivent des alias stables comme `claude-opus-4-8-YYYYMMDD`, dont l'année va de 2026 à 2035. La partie qui ressemble à une date -est un emplacement synthétique de route, et non la date de publication du modèle. Les emplacements de 2026 sont attribués en premier, de sorte que les alias -existants conservent leur identifiant ; les années suivantes ne sont utilisées qu'une fois 2026 saturée. Les véritables routes Anthropic Claude conservent +Les routes non Anthropic reçoivent des alias stables comme `claude-opus-4-8-p01q`, avec un code de quatre caractères préfixé par `p`. OpenCodex conserve +un emplacement synthétique daté en interne pour stabiliser les affectations du profil, mais n'expose pas cette date comme identifiant Desktop : les versions +actuelles de Desktop retirent les dates finales lors de la comparaison des modèles d'une session active, ce qui peut empêcher un changement. Les véritables routes Anthropic Claude conservent leur identité. Les nouvelles routes appartiennent par défaut à la famille Opus, mais déplacer une route ne change ni le fournisseur ni le modèle qu'elle appelle. Les anciens indicateurs `--static`, `--hybrid` et `--discovery-only` restent disponibles pour les scripts existants. @@ -233,6 +300,8 @@ la résolution par alias et carte des modèles renvoie le même modèle sans mod l'en-tête d'admission dédié du proxy est valide. Par conséquent, l'avertissement « Les connecteurs claude.ai sont désactivés » n'apparaît plus avec `ocx claude`. +La seule modification du corps concerne les identifiants d'appel d'outil. Un `tool_use.id` ou `tool_result.tool_use_id` qu'Anthropic refuserait (caractères hors de `a-zA-Z0-9_-`, ou plus de 64), par exemple créé plus tôt dans la session par un modèle routé, est réécrit en identifiant conforme sans rompre l'appariement appel/résultat. Les identifiants conformes sont envoyés tels quels, et un identifiant vide reçoit une erreur 400 locale. + Désactivez ce comportement avec `claudeCode.nativePassthrough: false` ; définissez une autre destination avec `claudeCode.anthropicBaseUrl`. @@ -300,23 +369,27 @@ pas les copies externes ; révoquez-la séparément sur le hub si nécessaire. ## Le sélecteur /model (« Depuis la passerelle ») Claude Code 2.1.129+ découvre les modèles de passerelle via `GET /v1/models?limit=1000` et les répertorie dans -le sélecteur natif `/model` intitulé « Depuis la passerelle ». Comme ce sélecteur n'accepte que les identifiants commençant -par `claude` ou `anthropic`, opencodex expose les modèles routés sous forme d'alias stables et réversibles : +le sélecteur natif `/model`. Une ligne sans `description` affiche « From gateway » ; opencodex en envoie une pour +chaque ligne du CLI Claude Code (`Routed by OpenCodex to /` ; lignes natives : `Routed by OpenCodex to native ` ; les lignes Fast ajoutent ` · Fast` et les lignes 1M gardent la description de base), que Claude Code 2.1.257+ affiche +à la place. Claude Code 2.1.278 accepte un identifiant qui contient `claude` ou `anthropic`. Un identifiant inconnu qui commence par `claude-` est compté à 200k sauf si le compactage est désactivé, donc opencodex expose les modèles routés sous forme d'alias stables et réversibles qui contiennent `claude` sans commencer par `claude-` : | Surface | Format | Exemple | | --- | --- | --- | -| Claude Code CLI | `claude-ocx---` (simple) ou `claude-ocx2-…` (échappé) | `claude-ocx-native--gpt-5.6-sol` | -| Claude Desktop 3P | `claude-opus-4-8-` (hachage base36 de 3 caractères) | `claude-opus-4-8-ncb` | +| Claude Code CLI | `ocx-claude---` (simple) ou `ocx-claude2-…` (échappé) | `ocx-claude-native--gpt-5.6-sol` | +| Claude Desktop 3P | `claude-opus-4-8-p` (emplacement base36 de 3 caractères) | `claude-opus-4-8-p01q` | Le proxy choisit la famille pour chaque requête : `?ids=cli` ou `?ids=desktop` est prioritaire ; à défaut, l'agent utilisateur -`claude-code/*` reçoit la forme lisible de la CLI et les autres clients reçoivent la forme hachée de Claude Desktop. +`claude-code/*` reçoit la forme lisible de la CLI et les autres clients reçoivent le code Claude Desktop. Les deux familles restent toujours décodables : un modèle enregistré sous l'une ou l'autre forme dans `settings.json` continue de fonctionner. Chaque entrée porte un nom d'affichage explicite, comme `gemini-3-pro (gemini)`, ainsi que toutes les capacités du modèle (échelle d'effort de raisonnement et types de réflexion) dans la structure officielle ModelInfo. Le mode passerelle tierce de Claude Desktop peut ainsi proposer son sélecteur d'effort. Les véritables modèles Anthropic conservent leurs identifiants canoniques. La date synthétique 2026 désigne un emplacement interne, et non une date de publication. Les anciens alias hachés et les identifiants `claude-ocx---` des configurations antérieures sont -toujours résolus. +toujours résolus, tout comme les identifiants échappés `claude-ocx2---`. Un identifiant hérité +enregistré est toujours acheminé, mais Claude Code continue de le compter à 200k. Choisissez une fois `ocx-claude-` +à la place d'un `claude-ocx-` enregistré, et `ocx-claude2-` à la place d'un `claude-ocx2-` échappé, pour que la vraie +fenêtre de contexte et le compactage s'appliquent tous les deux. Si le sélecteur situé au bas de Claude Desktop ne modifie pas le modèle d'une conversation 3P déjà en cours, vous pouvez essayer `/model `, mais ce contournement peut également échouer sur les versions de Desktop @@ -338,9 +411,9 @@ résolu vers le modèle routé. Avec les anciennes versions de Claude Code, le s `ANTHROPIC_MODEL` ou tapez n'importe quel identifiant routé avec `/model` (Claude Code fait passer les chaînes). **Règles de grammaire des alias :** le fournisseur ne doit contenir ni `/` ni `--`, et ne doit pas être égal à `native`. -Les identifiants de modèle simples, sans `/` ni `~`, conservent le préfixe v1 `claude-ocx-…`. Ceux qui contiennent `/` ou -`~` utilisent le préfixe v2 `claude-ocx2-…` avec des échappements (`/` → `~s`, `~` → `~t`), par exemple : -`openrouter/anthropic/claude-opus-4-8` → `claude-ocx2-openrouter--anthropic~sclaude-opus-4-8`. +Les identifiants de modèle simples, sans `/` ni `~`, conservent le préfixe v1 `ocx-claude-…`. Ceux qui contiennent `/` ou +`~` utilisent le préfixe v2 `ocx-claude2-…` avec des échappements (`/` → `~s`, `~` → `~t`), par exemple : +`openrouter/anthropic/claude-opus-4-8` → `ocx-claude2-openrouter--anthropic~sclaude-opus-4-8`. Les alias v1 décodent littéralement (donc un identifiant de modèle historique qui contenait les séquences de deux caractères `~s` / `~t` est conservé) ; les alias v2 développent les échappements. Les routes impossibles à représenter sous une forme lisible utilisent l'alias haché. Les identifiants de modèle peuvent contenir `--` (la résolution se sépare uniquement au premier @@ -400,6 +473,8 @@ Les valeurs de configuration invalides définies manuellement reviennent à 829, `ANTHROPIC_SMALL_FAST_MODEL`. Le modèle Haiku effectif vaut `tierModels.haiku ?? smallFastModel` et alimente les deux variables Haiku. +Quand `ocx claude` se lance en mode abonnement, la connexion propre de Claude Code envoie un identifiant Claude nu comme `claude-sonnet-5` directement à Anthropic ; ces identifiants prennent donc leur fenêtre de contexte dans le registre des fournisseurs, quoi qu'un autre fournisseur indique pour le même identifiant. Un emplacement Opus, Sonnet ou Fable non défini reçoit alors l'identifiant natif vers lequel Claude Code résout cet alias, avec le marqueur `[1m]`, car derrière une passerelle Claude Code compte un identifiant sans marqueur à 200k. Une ligne `anthropic` plafonnée sous 1M ou une entrée `claudeCode.modelMap` laisse son identifiant sans marqueur, et Haiku n'est jamais rempli ni marqué. Avec une authentification par proxy ou `nativePassthrough` désactivé, c'est le routeur qui décide, et seule la fenêtre d'une ligne routée compte. L'environnement système et le fichier du shell laissent les emplacements non définis vides, car leurs valeurs atteignent aussi les lancements qui passent par un hub. + Lorsque `tierModels.haiku` et `smallFastModel` sont absents, OpenCodex laisse les deux variables auxiliaires non définies ; Claude Code choisit ensuite son modèle d'assistance natif (actuellement Sonnet), qui peut entraîner des frais de fournisseur natif. ## Agents de la liste (injectAgents) @@ -434,7 +509,8 @@ Le transfert Anthropic natif reste intact. remplacé par un contenu minimal lorsque l'entrée JSON en minuscules contient un nom bloqué. 2. **Vecteur de bloc de texte :** un bloc de texte utilisateur d'au moins 10 000 caractères commençant par `Base directory for this skill: ` — est reconnu lorsque le nom de base du répertoire correspond à un nom bloqué - (insensible à la casse). + (insensible à la casse). La ligne du répertoire n'est inspectée que jusqu'à 4 096 unités de code UTF-16 ; + une ligne plus longue est envoyée telle quelle, y compris sans saut de ligne final. Configurez cette fonction avec `claudeCode.blockedSkills` (`["claude-api"]` par défaut ; `[]` désactive entièrement l'élision). Le contenu de remplacement préserve l'association entre l'appel d'outil et son résultat. @@ -543,6 +619,8 @@ Le proxy traduit chaque requête Anthropic Messages API au format Codex Response | `max_tokens` | `max_output_tokens` | | `stop_sequences` | `stop` | +Le mode automatique de Claude Code envoie toujours `stop_sequences`. Pour les modèles de la liste `noStopModels` du fournisseur routé, OpenCodex omet `stop` sur les fils Chat Completions et Responses, afin que les modèles de raisonnement xAI comme grok-4.7 et grok-4.6 ne renvoient pas `400 invalid-argument` et ne soient pas marqués temporairement indisponibles. Voir [`noStopModels`](/fr/reference/configuration/providers/). + Sur l’adaptateur Anthropic prévu, les blocs signés non masqués (y compris thinking vide) et les blocs redacted opaques sont préservés. `hideThinkingSummary` reste inchangé : le texte signé masqué localement n’est pas exposé aux clients Claude ; sa relecture sans perte via cette frontière reste non établie. Les anciennes enveloppes combinées ne permettent pas de rétablir l’ordre après émission du texte en streaming. `claudeCode.compatibility: "enforce"` refuse toujours la relecture thinking. Cela ne prouve ni l’acceptation réelle par Anthropic ni une amélioration du cache ; [#3719](https://github.com/lidge-jun/opencodex/issues/3719) reste ouvert. **Cas d'erreur (400) :** JSON mal formé ; `model` absent ou vide ; `messages` absent ou vide ; rôle non pris en charge ; @@ -655,4 +733,8 @@ Utilisez `"haiku"` comme valeur de remplacement pour le modèle. Dans `config.json`, `claudeCode.stabilizePromptCache: true` déplace les notices Claude reconnues en fin des instructions système vers un dernier message utilisateur sur les routes traduites. La valeur par défaut est `false`. Activez cette option seulement si ce changement de rôle convient à vos clients. Les exemples dans des blocs de code et le texte non reconnu sont conservés ; le transfert Anthropic natif reste inchangé. Sans métadonnées, la clé de cache suit les instructions stabilisées. Cette option ne crée pas une identité de conversation et ne garantit aucun succès du cache amont. -Sur toutes les routes Chat traduites, les rappels de l’historique conservent leur position dans la conversation, après les résultats d’outils encore attendus, et sont transmis avec le rôle `developer`. L’ajout d’un rappel ne réécrit donc pas le prompt système initial, et une instruction placée au milieu de la conversation n’arrive plus avant les tours qu’elle était censée suivre. Si le service en amont refuse le rôle `developer`, activez `foldDeveloperRoleToSystem` sur ce fournisseur : le rappel est alors envoyé en `system`, à la même position. Ce comportement s’applique avec ou sans `stabilizePromptCache` ; le transfert Anthropic natif reste inchangé. La réutilisation du cache exige toujours une identité de session stable et un cache disponible en amont. Les changements des instructions ou outils antérieurs et la compaction de la conversation peuvent aussi affecter les succès du cache ; préserver l’ordre des rappels ne suffit pas à garantir sa réutilisation. +Sur toutes les routes Chat traduites, les rappels de l’historique conservent leur position dans la conversation, après les résultats d’outils encore attendus. L’ajout d’un rappel ne réécrit donc pas le prompt système initial, et une instruction placée au milieu de la conversation n’arrive plus avant les tours qu’elle était censée suivre. Le rôle porté par cet emplacement se décide séparément : un rappel part en `system`, sauf si le fournisseur enregistre `foldDeveloperRoleToSystem: false`, ce qui indique que le service en amont accepte le rôle `developer` et le transmet à la même position. Un service qui ne l’accepte pas répond `400 role 'developer' is not allowed` et le tour ne démarre pas, d’où le repli d’une destination non enregistrée. Ce comportement s’applique avec ou sans `stabilizePromptCache` ; le transfert Anthropic natif reste inchangé. La réutilisation du cache exige toujours une identité de session stable et un cache disponible en amont. Les changements des instructions ou outils antérieurs et la compaction de la conversation peuvent aussi affecter les succès du cache ; préserver l’ordre des rappels ne suffit pas à garantir sa réutilisation. + +### `anthropicAccountPool.routes` + +Les règles `anthropicAccountPool.routes` limitent la sélection et les reprises 429 aux comptes enregistrés du premier modèle correspondant lorsque le pool est activé. Sans compte éligible, la requête échoue localement; `fallback: true` autorise alors le pool ordinaire. Les règles ne prouvent pas l’accès du compte au modèle. diff --git a/docs-site/src/content/docs/fr/guides/codex-app-models.md b/docs-site/src/content/docs/fr/guides/codex-app-models.md index 28fa0ee8a3..0e69068a71 100644 --- a/docs-site/src/content/docs/fr/guides/codex-app-models.md +++ b/docs-site/src/content/docs/fr/guides/codex-app-models.md @@ -143,8 +143,8 @@ approximation fondée sur un ancien modèle d'entrée. | Connexion Codex (ligne Daybreak transférée explicitement) | `openai/gpt-daybreak-blue-latest` uniquement lorsque l'entrée `customModels` exacte est configurée sur le fournisseur canonique `openai`. Elle conserve l'identifiant Daybreak transmis et utilise l'instantané de capacités Sol épinglé (contexte de 372 000 jetons ; compactage automatique à 334 800 jetons). | | OpenAI (clé API) | Exactement dix lignes avec espace de noms : `gpt-5.5`, `gpt-5.6`, Sol/Terra/Luna, les trois identifiants virtuels `*-pro` et les deux alias Daybreak (contexte de 1 050 000 jetons ; entrée maximale de 922 000 jetons pour les dix) | | OpenRouter | `openrouter/openai/gpt-5.6-sol`, `openrouter/openai/gpt-5.6-terra`, `openrouter/openai/gpt-5.6-luna` (1 050 000) | -| Cursor | Le repli statique comprend `cursor/gpt-5.6-sol`, `cursor/gpt-5.6-terra` et `cursor/gpt-5.6-luna` (1 000 000), ainsi que des lignes ordinaires/rapides pour Grok 4.5 et 4.6 (500 000) ; 4.6 ajoute `xhigh`, et la découverte dynamique propre au compte détermine quelles lignes restent visibles. | -| xAI | La découverte dynamique fait autorité. Le catalogue de secours comprend `xai/grok-4.6` et utilise `xai/grok-4.5` par défaut ; les deux ont une fenêtre de 500 000 jetons. Grok 4.6 propose `low` / `medium` / `high` / `xhigh` (valeur amont par défaut : `high`), tandis que Grok 4.5 s'arrête à `high`. | +| Cursor | Le repli statique comprend `cursor/gpt-5.6-sol`, `cursor/gpt-5.6-terra` et `cursor/gpt-5.6-luna` (1 000 000), ainsi que des lignes ordinaires/rapides pour Grok 4.5, 4.6 et 4.7 (500 000) ; 4.6 et 4.7 ajoutent `xhigh`, et la découverte dynamique propre au compte détermine quelles lignes restent visibles. | +| xAI | La découverte dynamique fait autorité. Le catalogue de secours comprend `xai/grok-4.6` et `xai/grok-4.7` et utilise `xai/grok-4.5` par défaut ; les trois ont une fenêtre de 500 000 jetons. Grok 4.6 et 4.7 proposent `low` / `medium` / `high` / `xhigh` (valeur amont par défaut : `high`), tandis que Grok 4.5 s’arrête à `high`. | Les entrées GPT-5.6 épinglées préservent exactement l'échelle amont. Sol et Terra proposent les niveaux de `low` à `ultra` ; Luna s'arrête à `max`. Sol utilise `low` par défaut, contre `medium` pour Terra et Luna. diff --git a/docs-site/src/content/docs/fr/guides/codex-integration.md b/docs-site/src/content/docs/fr/guides/codex-integration.md index b4467db270..aa0c8b988e 100644 --- a/docs-site/src/content/docs/fr/guides/codex-integration.md +++ b/docs-site/src/content/docs/fr/guides/codex-integration.md @@ -153,7 +153,8 @@ $CODEX_HOME/opencodex-catalog.json $CODEX_HOME/models_cache.json ``` -Sous WSL, si `CODEX_HOME` n'est pas défini et que `~/.codex/config.toml` n'existe pas côté Linux, opencodex +Sous WSL, si `CODEX_HOME` n'est pas défini et que le répertoire `~/.codex` côté Linux est absent ou ne contient aucun état Codex +(`config.toml`, `auth.json`, `sessions`, `history.jsonl`), opencodex recherche également un unique répertoire personnel de Codex Desktop pour Windows à l'emplacement `/mnt/c/Users/*/.codex/config.toml`. S'il trouve exactement un candidat, il utilise ce répertoire afin que le mode app-server sous WSL et Codex Desktop sous Windows partagent les mêmes fichiers de configuration et @@ -382,6 +383,8 @@ Si la lecture authentifiée des quotas avec le nouveau jeton OAuth confirme un q `ocx account refresh openai` et `ocx account list openai --quota --refresh` consultent uniquement les quotas. La validation du modèle consomme du quota et nécessite une session humaine du tableau de bord : après récupération, ouvrez `ocx gui` et cliquez sur **Refresh quotas**. Sur un hôte sans interface graphique, accédez à son tableau de bord depuis votre navigateur ; le jeton administrateur seul n’autorise pas la validation. Un compte en pause peut être validé sans être repris ni sélectionné. Les erreurs d’autorisation restent visibles jusqu’à une validation ou une réauthentification réussie. +Dans **Codex Set → Multi-auth**, activez le commutateur **Crédits Codex** dans l’en-tête **Codex Auth** pour afficher la dernière observation de chaque compte principal et du pool juste sous Week. Désactivé par défaut, il est enregistré dans `showCodexCredits`. Le solde utilise le format numérique local ; les mentions illimité ou plafond de dépassement atteint apparaissent si elles sont signalées. Sans plafond total fourni, la barre indique la disponibilité et non un pourcentage. Le commutateur ne contrôle que l’affichage ; une nouvelle connexion attend sa propre observation. + La revalidation en arrière-plan est distincte et désactivée par défaut. Elle nécessite Token Guardian, la politique `proactive` du fournisseur `openai` et `tokenGuardian.codexWarmupEnabled`, et ignore les comptes dont la validation d’inscription est en attente. ### Pourquoi un compte a cessé de servir les requêtes @@ -413,13 +416,13 @@ ocx restore # restore without stopping (alias: ocx eject) ocx restore back # point plain Codex at the running proxy again ``` -Lorsque opencodex s'exécute comme [service d'arrière-plan géré](/fr/reference/cli/lifecycle/#ocx-service-installrepairstartstopstatusuninstallremove), il définit +Lorsque opencodex s'exécute comme [service d'arrière-plan géré](/fr/reference/cli/lifecycle/#ocx-service-installrepairrestartstartstopstatusuninstallremove), il définit `OCX_SERVICE=1` afin qu'un redémarrage déclenché par le service ne modifie **pas** sans cesse la configuration Codex. Seule l'exécution explicite de `ocx stop` ou `ocx service stop` restaure Codex natif. ## Refus de sécurité pour l’historique paginé -Une transition de fournisseur peut renvoyer `history_paginated_requires_native_writer` si le stockage concerné prend en charge la pagination, même pour ses lignes legacy. Cette raison ne refuse plus la configuration Codex, le profil de référence ni le catalogue de modèles. `ocx sync` et `ocx start` écrivent toujours ces fichiers et définissent `model_catalog_json`, afin que le sélecteur de modèles Codex continue d’afficher tous les modèles routés par OpenCodex. Seule cette raison interrompt le réétiquetage de l’historique des conversations, car Codex attribue les numéros d’historique paginé dans son propre processus d’écriture et aucune nouvelle tentative n’y change rien. Toute autre raison de contrôle préalable de l’historique — une base d’état illisible, un historique dont l’identité a changé, ou un contrôle préalable qui n’a pas pu s’exécuter — refuse encore toute la transition et l’annule, car ces cas peuvent réussir plus tard. Dans cet état, OpenCodex ne modifie jamais les fichiers d’historique paginé ni les lignes de conversation. Les conversations existantes conservent le fournisseur déjà associé et ne sont pas migrées ; les nouvelles conversations passent par le proxy. Lorsque le réétiquetage est interrompu, une table `[model_providers.opencodex]` déjà présente dans le répertoire d’accueil est conservée plutôt que retirée, y compris sous la forme root-override (loopback), afin que les conversations dont les lignes sont étiquetées `opencodex` gardent un identifiant de fournisseur qui existe encore. Le CLI affiche `Codex resume history: left to Codex's native writer (history_paginated_requires_native_writer)`. `ocx restore` et la suppression de la configuration Codex refusent toujours sur `history_paginated_requires_native_writer`. Retirer la définition `[model_providers.opencodex]` alors que des lignes de conversation la référencent encore rendrait ces conversations irrésolubles, et le chemin de restauration n’a aucun moyen de conserver une table de fournisseur de compatibilité. Un répertoire d’accueil déjà paginé ne peut pas actuellement être désinstallé par le produit ; c’est un travail ouvert connu, et non le comportement voulu. +Une transition de fournisseur peut renvoyer `history_paginated_requires_native_writer` si le stockage concerné prend en charge la pagination, même pour ses lignes legacy. Cette raison ne refuse plus la configuration Codex, le profil de référence ni le catalogue de modèles. `ocx sync` et `ocx start` écrivent toujours ces fichiers et définissent `model_catalog_json`, afin que le sélecteur de modèles Codex continue d’afficher tous les modèles routés par OpenCodex. Seule cette raison interrompt le réétiquetage de l’historique des conversations, car Codex attribue les numéros d’historique paginé dans son propre processus d’écriture et aucune nouvelle tentative n’y change rien. Toute autre raison de contrôle préalable de l’historique — une base d’état illisible, un historique dont l’identité a changé, ou un contrôle préalable qui n’a pas pu s’exécuter — refuse encore toute la transition et l’annule, car ces cas peuvent réussir plus tard. Dans cet état, OpenCodex ne modifie jamais les fichiers d’historique paginé ni les lignes de conversation. Les conversations existantes conservent le fournisseur déjà associé et ne sont pas migrées ; les nouvelles conversations passent par le proxy. Lorsque le réétiquetage est interrompu, une table `[model_providers.opencodex]` déjà présente dans le répertoire d’accueil est conservée plutôt que retirée, y compris sous la forme root-override (loopback), afin que les conversations dont les lignes sont étiquetées `opencodex` gardent un identifiant de fournisseur qui existe encore. Le CLI affiche `Codex resume history: left to Codex's native writer (history_paginated_requires_native_writer)`. `ocx restore`, `ocx stop` et `ocx uninstall` ne refusent plus sur `history_paginated_requires_native_writer`. Ils retirent toutes les clés de routage racine d'OpenCodex et conservent la définition `[model_providers.opencodex]` sur le disque : les conversations dont les lignes nomment encore ce fournisseur restent résolubles, tandis que `codex` seul cesse de pointer vers le proxy. Le résultat est signalé comme une restauration partielle qui nomme les lignes conservées, et `ocx restore --remove-codex-provider-table` les supprime aussi, après quoi ces conversations ne s'ouvrent plus. Par ailleurs, activer l'intégration sous sa forme table de fournisseur sur un répertoire d'accueil dont les conversations marquées `openai` ont déjà été paginées par Codex était auparavant refusé d'emblée avec `history_paginated_openai_requires_native_writer` : rien n'était écrit et l'intégration restait désactivée. OpenCodex termine désormais cette transition en conservant la redéfinition racine gérée `openai_base_url` à côté de la table `[model_providers.opencodex]`. Codex fusionne cette redéfinition avec son fournisseur `openai` intégré, donc ces conversations continuent d'atteindre le proxy sans être réétiquetées, et aucun octet d'historique ni ligne de conversation n'est modifié. Seule une forme de routage exigeant l'en-tête d'admission `x-opencodex-api-key` refuse encore, car le fournisseur intégré de Codex ne peut pas porter cet en-tête ; son message nomme les deux réglages qui résolvent la situation — router Codex par l'écouteur loopback pour conserver la redéfinition, ou mettre `syncResumeHistory` à `false` en acceptant que ces conversations reprennent sur le point de terminaison OpenAI propre à Codex. Lors du retour au mode de remplacement de l’URL racine, OpenCodex conserve la définition `[model_providers.opencodex]` existante avant de valider la configuration, même si la vérification préalable de l’historique réussit. Les anciennes conversations `opencodex` peuvent ainsi toujours retrouver leur fournisseur si Codex migre l’historique après cette validation ou pendant le démarrage du traitement en arrière-plan. Les nouvelles conversations utilisent le fournisseur racine sélectionné ; la restauration explicite conserve ses contrôles de suppression distincts. diff --git a/docs-site/src/content/docs/fr/guides/codex-log-guard-reclaim.md b/docs-site/src/content/docs/fr/guides/codex-log-guard-reclaim.md new file mode 100644 index 0000000000..f4e8cda940 --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/codex-log-guard-reclaim.md @@ -0,0 +1,98 @@ +--- +title: Récupération d’espace avec Codex Log Guard +description: Récupérez manuellement les pages libres du stockage SQLite des journaux de diagnostic Codex par un compactage incrémental borné. +--- + +Reclaim est l’étape manuelle de récupération d’espace de Codex Log Guard. Elle compacte la base de données canonique `logs_2.sqlite` de Codex uniquement lorsque la base et l’environnement d’exécution passent les mêmes contrôles de sécurité que ceux de la protection Log Guard. + +Reclaim n’est **jamais planifié automatiquement** et ne s’exécute pas simplement parce que la page Stockage est ouverte. Le tableau de bord exige une action Compact explicite et une seconde confirmation avant d’envoyer la requête de modification. + +## Fonctionnement de Reclaim + +OpenCodex effectue une séquence de maintenance hors ligne et bornée : + +1. résoudre le chemin canonique de `logs_2.sqlite` à partir du `sqlite_home` effectif de Codex ; +2. vérifier l’identité du fichier et le schéma connu des journaux Codex ; +3. vérifier que l’énumération des processus a réussi et qu’aucun processus d’écriture Codex pris en charge n’est actif ; +4. acquérir le verrou Log Guard dédié, partagé entre processus ; +5. répéter la vérification des processus Codex pendant que ce verrou est détenu ; +6. ouvrir la base existante en lecture et écriture sans autoriser sa création et vérifier qu’un accès immédiat en écriture SQLite peut être obtenu ; +7. exiger que `PRAGMA auto_vacuum` soit déjà réglé sur `INCREMENTAL` ; +8. exécuter `PRAGMA quick_check` avant la maintenance ; +9. effectuer un checkpoint WAL complet et refuser un checkpoint occupé ou incomplet ; +10. exécuter des lots bornés de `PRAGMA incremental_vacuum(N)`, avec un checkpoint après chaque lot ; +11. exécuter de nouveau `PRAGMA quick_check` après la maintenance ; et +12. communiquer les mesures avant/après de la base, du WAL, du nombre de pages, de la liste des pages libres et des octets récupérables. + +La cible par défaut d’un lot est d’environ **8 Mio de pages SQLite**. Une exécution récupère au plus environ **256 Mio de pages**, avec une limite supplémentaire sur le nombre d’itérations. S’il reste des pages libres, le résultat est signalé comme partiel et vous pouvez relancer Compact plus tard. + +Les limites en octets sont converties en nombres de pages selon la taille réelle des pages SQLite de la base. Elles bornent les pages SQLite logiques traitées ; elles ne mesurent pas le volume d’écritures sur SSD/NAND. + +## Garanties de sécurité + +Reclaim ne fait délibérément **pas** les actions suivantes : + +- exécuter un `VACUUM` complet ; +- modifier le mode `auto_vacuum` d’une base Codex existante ; +- supprimer, tronquer, renommer ou manipuler directement les fichiers Codex `-wal` / `-shm` ; +- supprimer des lignes de diagnostic ; +- modifier les déclencheurs de protection Log Guard ou d’autres déclencheurs utilisateur ; +- s’exécuter lorsque Codex est détecté comme actif ; +- continuer si l’énumération des processus est incertaine ; +- continuer avec un futur schéma de journaux inconnu ; ou +- continuer après l’échec d’un contrôle d’intégrité SQLite. + +Un verrou Log Guard occupé, un accès SQLite en écriture occupé ou un checkpoint initial occupé entraîne un refus explicite, sans nouvelle tentative en arrière-plan. Si la contention du checkpoint apparaît seulement après la validation d’un lot de compactage incrémental, OpenCodex signale le travail déjà effectué comme un résultat partiel réussi avec `stopReason: "busy"`, au lieu de prétendre qu’aucun changement n’a eu lieu. + +## CLI + +Examinez d’abord l’espace récupérable : + +```bash +ocx storage codex-logs status +``` + +Exécutez une passe de maintenance bornée : + +```bash +ocx storage codex-logs compact +``` + +Pour obtenir des mesures avant/après lisibles par machine : + +```bash +ocx storage codex-logs compact --json +``` + +Si le résultat indique qu’il reste de l’espace récupérable, arrêtez-vous là, sauf si vous souhaitez explicitement une autre passe bornée. OpenCodex ne boucle pas indéfiniment et ne planifie pas de passe supplémentaire à votre place. + +## API de gestion + +Le compactage n’est exposé que par un point de terminaison de modification : + +```text +POST /api/storage/codex-logs/compact +``` + +Il n’existe aucun alias GET pour le compactage. Une réponse réussie contient un objet `report` avec les mesures avant/après, le nombre de pages récupérées, la variation de la taille physique du fichier principal de la base, le nombre d’itérations, l’état d’achèvement, le motif d’arrêt et l’état d’intégrité. + +Les états de refus habituels comprennent : + +- `codex_running` : Codex est en cours d’exécution ; +- `process_enumeration_failed` : l’énumération des processus a échoué ; +- `busy` : une ressource est occupée ; +- `unsupported_schema` : le schéma n’est pas pris en charge ; +- `auto_vacuum_not_incremental` : le mode de compactage requis est absent ; +- `unsafe_path` : le chemin n’est pas sûr ; +- `integrity_check_failed` : le contrôle d’intégrité a échoué ; +- `database_error` : une erreur de base de données s’est produite. + +Les échecs d’intégrité précisent s’ils se sont produits avant ou après la passe de maintenance. Un refus `busy` signifie qu’une contention a été détectée avant la validation du moindre lot ; `stopReason: "busy"` dans un rapport réussi signifie qu’au moins un lot a été validé avant qu’une contention ultérieure du checkpoint n’arrête la passe. + +## Interpréter le résultat + +`pagesReclaimed` et `logicalBytesReclaimed` décrivent les pages de la liste libre SQLite retirées pendant la passe. `physicalDatabaseBytesReclaimed` indique la réduction observée de la taille du fichier principal après les checkpoints de maintenance. + +Ces nombres peuvent différer. Le comportement de SQLite, du WAL et du système de fichiers signifie que récupérer des pages logiques ne garantit pas une réduction physique immédiate identique. Aucune de ces mesures ne doit être interprétée comme un nombre d’écritures NAND, une usure du SSD ou un volume TBW consommé ou économisé. + +`complete: true` signifie que la liste libre observée est vide. Un résultat partiel utilise `stopReason: "page_budget"` lorsque le budget de pages par exécution ou la limite finie d’itérations met fin à la passe, `stopReason: "no_progress"` lorsque SQLite cesse de réduire la liste libre, et `stopReason: "busy"` lorsqu’une contention du checkpoint apparaît après une récupération validée. Ces trois issues sont bornées et ne déclenchent aucune nouvelle tentative automatique. diff --git a/docs-site/src/content/docs/fr/guides/codex-log-guard.md b/docs-site/src/content/docs/fr/guides/codex-log-guard.md new file mode 100644 index 0000000000..25d26cec5b --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/codex-log-guard.md @@ -0,0 +1,141 @@ +--- +title: Codex Log Guard +description: Examinez et réduisez explicitement la persistance des journaux de diagnostic Codex sans exposer leur contenu. +--- + +OpenCodex peut examiner la base persistante des journaux de diagnostic Codex et, si vous l’activez, réduire les lignes de diagnostic que Codex conserve. L’inspection reste en lecture seule ; la protection est une modification explicite, refusée si le schéma connu des journaux Codex est absent ou si Codex est en cours d’exécution. + +## Ce que rapporte Inspect + +OpenCodex détermine le `sqlite_home` effectif de Codex selon l’ordre de priorité existant de Codex et examine la base canonique `logs_2.sqlite` qui s’y trouve. Un fichier `logs_N.sqlite` de numéro supérieur ou hérité ne remplace jamais la cible susceptible d’être modifiée. + +La vue Stockage indique : + +- la taille des fichiers de la base principale, du WAL et du SHM ; +- le nombre total de lignes de journaux et la part enregistrée au niveau `TRACE` ; +- les principales catégories de cibles de journaux par nombre de lignes, désignées par leur rang plutôt que par leur nom ; +- l’espace libre SQLite qui pourrait être récupéré plus tard ; et +- la compatibilité du schéma observé avec le schéma actuellement connu des journaux Codex. + +Si `sqlite_home` se trouve hors de `CODEX_HOME`, la base de diagnostic est affichée séparément. Ses octets ne sont pas ajoutés silencieusement au total de stockage `CODEX_HOME` existant. + +OpenCodex ne sélectionne ni n’expose `feedback_log_body` pour produire ces diagnostics. Les niveaux de journal sont réduits à l’ensemble fixe des niveaux connus, plus `OTHER`, et les noms des cibles ne sont pas sérialisés. + +## Modes de Protect + +La protection est **désactivée par défaut**. Son activation installe un déclencheur `BEFORE INSERT` appartenant à OpenCodex dans la base canonique `logs_2.sqlite` de Codex. OpenCodex ne remplace jamais un déclencheur inconnu portant l’un de ses noms réservés et ne supprime que les déclencheurs dont le SQL correspond à sa propre version. + +Deux modes sont disponibles : + +- **Compatibilité** (`compat`) est le mode recommandé. Il applique l’ensemble actuel de règles Log Guard v1 aux cibles à fort volume que Codex filtre déjà ou dont il abaisse le niveau dans son stockage persistant SQLite. Les autres lignes `TRACE` sont conservées. +- **Silencieux** (`quiet`) supprime toutes les nouvelles lignes `TRACE` tout en conservant les lignes `DEBUG`, `INFO`, `WARN` et `ERROR`. + +La protection réduit les lignes qui atteignent le stockage persistant SQLite. Elle ne supprime **pas** le travail de traçage effectué plus tôt par Codex : les événements peuvent encore être mis en forme, placés en file, regroupés en transactions et examinés par la logique d’élagage propre à Codex avant que le déclencheur n’ignore une ligne. Considérez Protect comme une protection contre les écritures persistantes répétées, et non comme un interrupteur qui désactive la production des diagnostics dans Codex. + +Log Guard filtre uniquement les lignes de journaux SQLite locaux persistants. Il ne modifie ni le traitement des diagnostics Codex, ni le [transport des adaptateurs](/fr/reference/adapters/), ni les charges utiles des fournisseurs, la sémantique du streaming, l’authentification, le routage, les quotas ou l’état des comptes. + +### Contrôles de sécurité + +Avant que Protect, Disable ou Repair ne modifie cette base externe, OpenCodex : + +1. résout exactement le chemin canonique de `logs_2.sqlite` ; +2. vérifie qu’il s’agit d’un fichier ordinaire, sans lien symbolique, et que le schéma connu correspond exactement ; +3. vérifie que l’énumération des processus a réussi et qu’aucun processus d’écriture Codex pris en charge n’est actif ; +4. acquiert un verrou Log Guard dédié, partagé entre processus ; +5. répète la vérification des processus Codex après l’acquisition du verrou ; +6. ouvre la base en lecture et écriture **sans** autoriser sa création et acquiert `BEGIN IMMEDIATE` dans SQLite sans attente en cas d’occupation ; +7. ne modifie que les déclencheurs Log Guard appartenant à OpenCodex et relit le résultat avant validation ; et +8. enregistre le mode demandé dans la configuration OpenCodex pendant que le verrou Log Guard est encore détenu. + +Si l’énumération des processus est incertaine, si la base est occupée, si le schéma est inconnu ou si un nom de déclencheur réservé appartient à un autre SQL, la modification est refusée par précaution. OpenCodex n’arrête pas Codex automatiquement. + +## Dérive et réparation + +Le mode de protection demandé est enregistré dans la configuration OpenCodex, séparément de la base des journaux Codex. C’est important, car une migration Codex peut reconstruire la table `logs`, et SQLite supprime les déclencheurs attachés à une table remplacée. + +Lorsque le mode enregistré est `compat` ou `quiet`, mais que le déclencheur correspondant n’est plus observé, Log Guard signale une **dérive**. `ocx doctor` la signale sans jamais la réparer automatiquement. + +La réparation est explicite : + +```bash +ocx storage codex-logs repair +``` + +OpenCodex ne recrée délibérément pas la protection à chaque démarrage. Une version ultérieure pourra réexaminer la réparation automatique lorsque suffisamment d’observations sur le terrain auront établi sa sûreté lors des migrations Codex. + +## CLI + +Consultez l’état : + +```bash +ocx storage codex-logs status +ocx storage codex-logs status --json +ocx doctor +``` + +Activez la politique de compatibilité recommandée : + +```bash +ocx storage codex-logs protect +``` + +Choisissez explicitement le mode silencieux : + +```bash +ocx storage codex-logs protect --mode quiet +``` + +Désactivez la protection OpenCodex ou réparez une dérive : + +```bash +ocx storage codex-logs unprotect +ocx storage codex-logs repair +``` + +Ajoutez `--json` aux commandes Log Guard pour obtenir une sortie lisible par machine. Consultez la [référence CLI](/fr/reference/cli/) pour la syntaxe canonique et le comportement JSON. + +La commande existante reste inchangée : + +```bash +ocx storage --json +``` + +Sa réponse contient le même état des journaux Codex que celui utilisé par la page Stockage. + +## API de gestion + +L’état est disponible à l’adresse : + +```text +GET /api/storage/codex-logs +``` + +Les modifications explicites utilisent : + +```text +POST /api/storage/codex-logs/protect +POST /api/storage/codex-logs/unprotect +POST /api/storage/codex-logs/repair +``` + +Le corps de Protect est soit `{"mode":"compat"}`, soit `{"mode":"quiet"}`. `GET /api/storage` inclut également le rapport sous `codexLogs`, afin que le tableau de bord puisse actualiser la répartition normale du stockage et les diagnostics des journaux Codex à partir d’un seul instantané. + +## Sémantique des instantanés en lecture seule + +L’inspection de l’état ouvre la base en lecture seule avec SQLite `immutable=1`. Une lecture de diagnostic ne peut ainsi ni créer ni mettre à jour les fichiers annexes `-wal` ou `-shm`. + +Ce choix a une conséquence importante : les agrégats SQL et les métadonnées de déclencheurs observées décrivent le dernier instantané de la base ayant fait l’objet d’un checkpoint. Si Codex écrit activement, le WAL courant peut contenir des lignes ou des pages de schéma plus récentes que l’instantané immuable. Une réponse de modification réussie s’appuie sur l’état du déclencheur vérifié par OpenCodex dans sa transaction d’écriture ; une requête ultérieure d’état en lecture seule peut rester temporairement en retard jusqu’à ce que SQLite effectue un checkpoint de ces pages de schéma. + +OpenCodex indique séparément la taille du fichier WAL et ne présente **pas** le résultat comme un débit d’écriture SSD, des écritures NAND ou une consommation d’usure/TBW du disque. + +## États de compatibilité + +Avec un schéma connu, l’inspection et la protection sont signalées comme prises en charge. Un schéma absent, illisible ou inconnu d’une version future reste inspectable sous forme de métadonnées, mais les opérations susceptibles de modifier la base sont indiquées comme non prises en charge. + +Un schéma inconnu n’est pas supposé compatible. Une version plus récente de Codex reste ainsi observable sans que Log Guard traite une structure de base non examinée comme sûre à modifier. + +## Reclaim reste distinct + +Protect n’exécute ni `VACUUM` ni compactage SQLite. [**Reclaim**](/fr/guides/codex-log-guard-reclaim/) fournit un flux explicite, hors ligne et borné de compactage incrémental, avec checkpoints et contrôles d’intégrité. + +Protect n’exécute jamais `VACUUM`, ne tronque ni ne supprime directement le WAL de Codex et ne planifie jamais de récupération d’espace. diff --git a/docs-site/src/content/docs/fr/guides/codex-native-context.md b/docs-site/src/content/docs/fr/guides/codex-native-context.md new file mode 100644 index 0000000000..697318dfb3 --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/codex-native-context.md @@ -0,0 +1,79 @@ +--- +title: Compatibilité du contexte natif +description: Éligibilité, configuration d’essai authentifiée et limites du relais d’historique et de notes Codex. +--- + +OpenCodex relaie déjà l’historique et les notes natifs de Codex. Il ne s’agit pas d’un service de mémoire général pour les fournisseurs routés, et l’exposition de ses points de terminaison HTTP ne prouve pas qu’une version, un compte ou un modèle Codex particulier peut les utiliser. Consultez [l’intégration Codex](/fr/guides/codex-integration/) pour les limites de propriété, d’annulation et d’identifiants du relais. + +## Deux conditions indépendantes + +Codex doit activer l’extension, et OpenCodex doit identifier l’appelant. Modifier l’URL du backend ne satisfait à elle seule aucune de ces conditions. + +Le contrat amont Codex examiné exige un modèle dont l’entrée du catalogue natif annonce `supports_experimental_context`, une connexion ChatGPT éligible et un fournisseur nommé exactement `OpenAI` dont l’URL de base se termine par `/backend-api/codex`. Son activation automatique rejette les fournisseurs qui utilisent `env_key`, `experimental_bearer_token`, une `auth` exécutée par commande ou l’authentification AWS. Le prédicat d’éligibilité examiné accepte ChatGPT Plus, Pro et ProLite ; cela ne signifie pas que tous les comptes de ces offres disposent de points de terminaison d’historique fonctionnels. + +OpenCodex exige en plus une **clé API du plan de données** active, aussi bien pour la requête de modèle réussie que pour les requêtes de contexte suivantes. L’injection loopback intégrée par défaut n’envoie pas cette clé : elle peut donc servir des modèles alors que les appels de contexte échouent avec `context_principal_required` (403). La seule forme authentifiée de la table des fournisseurs distants ne résout pas non plus le contexte natif : son `env_key` et son nom de fournisseur ne satisfont pas au contrat d’activation Codex ci-dessus. Ne supprimez jamais les contrôles de principal ou de propriété du compte pour masquer l’un ou l’autre problème. + +## Syntaxe d’activation explicite + +OpenCodex accepte les deux formes persistantes de fonctionnalité racine reconnues par `FeatureToml` de Codex : + +```toml +[features] +context_management = true +``` + +La forme équivalente en table fonctionne aussi et correspond à la syntaxe compatible avec les anciennes versions d’OpenCodex qui ne reconnaissaient pas encore la forme booléenne : + +```toml +[features.context_management] +experimental_mode = true +``` + +Utilisez une seule forme. Une valeur fausse, absente ou malformée laisse la fonction désactivée. Le proxy lit sa propre configuration du répertoire Codex ; une surcharge propre au CLI ou une activation présente uniquement dans un profil Codex n’active pas sa condition d’exécution. Ce changement ne déduit pas l’activation des métadonnées du modèle. + +## Profil d’essai natif authentifié + +Il s’agit d’une **configuration d’essai vérifiée dans le code source, pas d’une certification de bout en bout avec un compte réel**. Sauvegardez la configuration Codex et conservez un point de reprise durable avant l’essai. Utilisez un nouveau fil jetable ; ne changez pas l’identité du fournisseur d’un fil existant qui fonctionne. + +Fournissez une clé active existante du plan de données OpenCodex dans `OCX_CONTEXT_API_KEY`, dans l’environnement du processus Codex. N’utilisez pas un jeton de gestion ou d’administration et ne stockez pas la clé dans TOML. L’environnement d’un service n’est pas automatiquement hérité par une application de bureau lancée séparément. Conservez la connexion ChatGPT native habituelle de Codex ; l’en-tête supplémentaire ne remplace pas OAuth. + +Avec l’activation racine ci-dessus, le fournisseur de transfert ChatGPT canonique configuré dans OpenCodex et un catalogue de modèles natifs à jour, ajoutez ce fournisseur et ce profil **supplémentaires** à la même configuration Codex. Adaptez le port à celui du proxy local. Ne modifiez pas le `model_provider` racine ni les tables de fournisseurs existantes. + +```toml +[model_providers.ocx-native-context] +name = "OpenAI" +base_url = "http://127.0.0.1:10100/backend-api/codex" +wire_api = "responses" +requires_openai_auth = true +supports_websockets = false +env_http_headers = { "x-opencodex-api-key" = "OCX_CONTEXT_API_KEY" } + +[profiles.ocx-native-context] +model_provider = "ocx-native-context" +model = "gpt-6-astra" +``` + +Démarrez un nouveau fil CLI avec `codex --profile ocx-native-context`. L’exemple utilise HTTP/SSE pour limiter l’essai initial au chemin de propriété entre modèle et relais ; il ne modifie pas le transport des autres profils et ne certifie pas la parité WebSocket ou du pilotage en cours de tour. N’utilisez ce modèle que si le catalogue natif du compte annonce effectivement sa capacité de contexte ; ne forcez jamais cet indicateur sur une entrée Devin, Gemini ou d’un autre fournisseur routé. + +L’identifiant personnalisé du fournisseur est volontaire. Codex en amont ne remplace généralement pas les fournisseurs intégrés à partir de `model_providers.openai` ; y ajouter un en-tête peut rester sans effet et sans avertissement. L’identifiant personnalisé préserve le fournisseur normal, tandis que le nom exact `OpenAI` satisfait au prédicat du backend natif. N’ajoutez pas `env_key` à ce profil : `env_http_headers` transporte séparément l’autorisation locale, tandis que `Authorization` continue de transporter la connexion ChatGPT native. OpenCodex consomme la clé locale ; il ne la transmet pas à ChatGPT. + +L’activation racine concerne aussi les autres profils natifs éligibles. **Pendant cet essai, ne poursuivez pas les fils loopback intégrés ordinaires dépourvus de la clé supplémentaire.** Désactivez la fonctionnalité racine et exécutez `ocx sync` avant de reprendre ces fils. Il ne s’agit ni d’un changement d’intégration automatique ou par défaut, ni d’une preuve de prise en charge de la sélection de profil dans l’application de bureau. + +## Vérifier avant de réinitialiser le contexte + +Obtenez d’abord une réponse de modèle natif réussie dans le nouveau fil. Vérifiez ensuite l’écriture d’une note, relisez cette même note et interrogez l’historique du fil. Ce n’est qu’après la réussite de ces opérations qu’un essai jetable devrait utiliser `new_context` et vérifier que l’état enregistré peut être restauré. Conservez le point de reprise externe même si l’essai réussit. + +- **403 `context_principal_required` :** aucune clé locale valide du plan de données n’a atteint le proxy. +- **409 `context_account_unavailable` :** la propriété est absente ou incohérente ; ne substituez pas le compte actif courant et ne relancez pas aveuglément une écriture. +- **404 :** distinguez la réponse du proxy pour une fonction désactivée ou un point de terminaison inconnu d’une réponse 404 en amont. Cette dernière ne prouve ni un défaut de routage OpenCodex ni une panne touchant tous les comptes. + +Un appel de modèle réussi ou `ocx ready` ne prouve pas que les notes, l’historique ou la restauration de l’état fonctionnent. Le routage du modèle, les changements de compte, les redémarrages du proxy et la disponibilité des points de terminaison en amont restent des sujets distincts. Aucun indicateur local ne peut accorder une éligibilité backend manquante, et une opération de contexte échouée ne doit pas être présentée comme une réinitialisation réussie. Supprimez les tables d’essai et retirez la clé d’essai de l’environnement une fois l’essai terminé ; laissez la fonctionnalité désactivée sauf si vous utilisez un chemin authentifié vérifié. + +## Contrats amont examinés + +Ces liens fixent le contrat source utilisé pour la configuration ci-dessus, sans promettre son fonctionnement dans un déploiement : + +- [Formes booléenne et en table de FeatureToml](https://github.com/openai/codex/blob/78245b47af2a7aafcabe025828ceecca69db4df1/codex-rs/features/src/lib.rs) +- [Éligibilité du contexte natif](https://github.com/openai/codex/blob/78245b47af2a7aafcabe025828ceecca69db4df1/codex-rs/core/src/session/token_budget.rs) +- [Identité du fournisseur et règles de fusion des fournisseurs intégrés](https://github.com/openai/codex/blob/78245b47af2a7aafcabe025828ceecca69db4df1/codex-rs/model-provider-info/src/lib.rs) +- [L’historique et les notes utilisent les en-têtes de requête et l’authentification du fournisseur](https://github.com/openai/codex/blob/78245b47af2a7aafcabe025828ceecca69db4df1/codex-rs/ext/history-notes/src/backend.rs) diff --git a/docs-site/src/content/docs/fr/guides/combos.md b/docs-site/src/content/docs/fr/guides/combos.md index 4e9ccda71a..4cb051fc72 100644 --- a/docs-site/src/content/docs/fr/guides/combos.md +++ b/docs-site/src/content/docs/fr/guides/combos.md @@ -202,11 +202,12 @@ Les échecs d’un combo se répartissent entre ceux qui entraînent un **bascul | Erreur classée comme erreur d’authentification, d’abonnement, de quota, de limitation de débit, de surcharge ou de serveur en amont | Place la cible en période de refroidissement et bascule, même si le statut seul ne suffit pas. | | Annulation client (499), `origin_rejected`, refus de cyber-politique, débordement de contexte ou autre demande invalide | Arrêtez et renvoyez l'erreur ; une autre cible ne rendrait pas la demande valide. | | Rejet structuré de `user`, valeur non prise en charge pour `reasoning.effort`/`reasoning_effort`, ou rejet d'entrée d'image propre à un modèle (`param: input`) | Bascule vers la cible admissible suivante avant le début de la sortie, sans délai de refroidissement ; voir Compatibilité des paramètres facultatifs ci-dessous. | +| Premier appel d'outil d'un tour Responses exécuté par un adaptateur interne (`runTurn`) que la requête courante n'a pas déclaré, avant toute sortie et tout effet de bord non rejouable | Met la cible en refroidissement et bascule avec le même catalogue d'outils. Après une sortie visible ou un effet de bord non rejouable, le refus est définitif. Les requêtes Chat Completions et Anthropic Messages ne changent pas. | | Toute autre erreur non classifiée | Arrêtez et renvoyez l'erreur. | -Une cible sautée entre en temps de recharge pendant 60 secondes par défaut. Si la réponse en amont inclut un -valeur `Retry-After` valide, opencodex l’utilise à la place. Les secondes numériques et les valeurs de date HTTP sont -accepté, et chaque temps de recharge est limité à 10 minutes. +Une cible sautée utilise par défaut un temps de recharge en amont : 5 secondes pour les codes de limitation de débit `1302`/`1305`, 10 minutes pour une fenêtre d’utilisation épuisée (quel que soit le statut HTTP, y compris 502) ou pour un échec d’identifiants ou de facturation, et 60 secondes dans les autres cas. Si la réponse en amont inclut une +valeur `Retry-After` valide, opencodex l’utilise à la place ; les en-têtes de réinitialisation Codex viennent ensuite, puis le `cooldownMs` configuré. Les secondes numériques et les valeurs de date HTTP sont +acceptées, et un délai explicite `Retry-After` est plafonné à 24 heures ; les autres temps de recharge restent plafonnés à 10 minutes. La requête actuelle ne réessaye jamais la même cible tentée. Les demandes ultérieures l'ignorent jusqu'à ce qu'il soit le temps de recharge expire. S’il ne reste aucune cible éligible, le proxy renvoie HTTP 503 avec @@ -215,6 +216,7 @@ le temps de recharge expire. S’il ne reste aucune cible éligible, le proxy re :::note Le basculement est intentionnellement limité. Il facilite la disponibilité, l'authentification et l'authentification spécifiques à la cible. échecs de quota et de surcharge ; il ne cache pas les erreurs des appelants ni les refus de politique. +Sur une requête Responses hors combo, un 403 de politique xAI de la liste autorisée est réécrit en HTTP 200 `incomplete/content_filter` avant que Codex ne le relance comme un échec de transport ; voir [xAI policy refusals](/fr/reference/proxy-formats/#xai-policy-refusals). Les sauts de combo classent toujours le HTTP 403 d'origine comme un saut. ::: ## Effort de raisonnement par défaut @@ -331,7 +333,7 @@ Les combos sont stockés dans l'objet `combos` de niveau supérieur, saisi par l | --- | --- | --- | --- | | `targets` | Oui | — | Tableau ordonné non vide de `{ provider, model, weight? }` cibles configurées. Les paires provider/model en double sont rejetées. | | `targets[].weight` | Non | `1` | Entier de 1 à 10 000. Utilisé par `round-robin` et `random` ; ignoré par `failover`, `least-used` et `reset-window`. | -| `strategy` | Non | `"failover"` | Valeurs autorisées : `"failover"`, `"round-robin"`, `"random"`, `"least-used"` et `"reset-window"`. | +| `strategy` | Non | `"failover"` | Valeurs autorisées : `"failover"`, `"round-robin"`, `"random"`, `"least-used"`, `"reset-window"` et `"jev"`. JEV décide uniquement de la première cible éligible et de l’effort ; le fallback Combo ordinaire gère les tentatives suivantes. | | `stickyLimit` | Non | `1` | Nombre entier de 1 à 100 requêtes réussies par sélection à tour de rôle. S’applique uniquement à `round-robin`. | | `defaultEffort` | Non | `null` | `low`, `medium`, `high`, `xhigh`, `max` ou `ultra` ; appliqué uniquement lorsque l'appelant omet ses efforts et que la cible annonce son soutien. | | `reasoningEffortMode` | Non | `"strict"` | `strict` ou `adaptive` ; choisit l’intersection des capacités et la normalisation par cible. | @@ -352,8 +354,7 @@ exécution d'une instance opencodex qui reçoit des requêtes de modèle. Chaque cible est actuellement inéligible : par exemple, son fournisseur est désactivé, il est en phase de refroidissement, elle a déjà été tentée pour cette requête, ou une tâche v2 chiffrée l'exclut. Vérifier la cible -état du fournisseur et erreurs récentes en amont. Pour les temps de recharge, attendez la valeur par défaut de 60 secondes ou la -délai indiqué par `Retry-After` en amont (jamais plus de 10 minutes), puis réessayez. +état du fournisseur et erreurs récentes en amont. Pour les temps de recharge, suivez d’abord la valeur `Retry-After` observée, puis les en-têtes de réinitialisation Codex, puis le `cooldownMs` configuré ; à défaut, le fallback en amont s’applique (5 secondes pour les codes `1302`/`1305`, 10 minutes pour une fenêtre d’utilisation épuisée, quel que soit le statut HTTP, ou pour un échec d’identifiants ou de facturation, 60 secondes sinon). Un `Retry-After` explicite est plafonné à 24 heures, les autres temps de recharge à 10 minutes, puis réessayez. ### Pourquoi mon alias a-t-il été rejeté ? diff --git a/docs-site/src/content/docs/fr/guides/cursor-private-inference.md b/docs-site/src/content/docs/fr/guides/cursor-private-inference.md new file mode 100644 index 0000000000..53bb9c034f --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/cursor-private-inference.md @@ -0,0 +1,124 @@ +--- +title: Cursor Private Inference +description: Utilisez les modèles routés par opencodex dans la version de Cursor à agent local, sur macOS, Windows ou Linux, sans tunnel public. +--- + +Cursor classique ne peut pas communiquer avec un proxy sur votre propre machine. Lorsque vous définissez « Override OpenAI Base URL », le backend de Cursor construit la requête et appelle cette URL depuis les serveurs de Cursor, qui rejettent les adresses loopback, de réseau local et privées. C’est pourquoi les recettes communautaires pour Cursor avec des modèles locaux se terminent par ngrok, Cloudflare Tunnel ou un VPS. + +Cursor distribue également une seconde version de bureau, **Cursor Private Inference**, dont la boucle d’agent s’exécute localement et appelle une passerelle compatible OpenAI que vous configurez. Dirigée vers opencodex, elle utilise vos modèles routés sans tunnel, sans modifier l’application et sans TLS. Cette page décrit cette version. + +## Avant de commencer + +Lisez d’abord cette section : c’est celle qui est souvent négligée. + +- **opencodex ne distribue pas cette version.** Cursor ne la documente pas non plus. Elle n’est pas liée depuis cursor.com, peut changer sans préavis et peut cesser d’être disponible. Si vous ne l’avez pas déjà, ce guide ne s’applique pas ; utilisez plutôt le pont communautaire [`ocx-cursor`](https://www.npmjs.com/package/ocx-cursor) avec un point de terminaison HTTPS public. +- **La connexion à Cursor reste obligatoire.** L’écran de connexion apparaît avant la boîte de dialogue de la passerelle. +- **Les modèles propres à Cursor ne sont pas disponibles.** En mode local, le sélecteur ne répertorie que les modèles renvoyés par votre passerelle. La complétion Tab, le catalogue Cursor (Composer, Auto) et Cloud Agents sont désactivés. Vous pouvez toujours accéder aux modèles du fournisseur Cursor par les routes `cursor/*` d’opencodex si vous avez configuré ce fournisseur. +- **Chaque tour contient la consigne système locale de Cursor**, d’environ 23 000 jetons à partir du deuxième tour. Tenez-en compte lors du choix du modèle. +- **Cette version partage son identité avec Cursor classique.** Même identifiant de paquet, même `~/.cursor`, même `Application Support/Cursor` (macOS), `%APPDATA%\Cursor` (Windows) ou `~/.config/Cursor` (Linux). Lancez-la avec `--user-data-dir ` pour séparer les deux versions et laissez « Import data from existing Cursor installation » décoché au premier lancement, sauf si vous voulez copier vos réglages. + +## Identifier la version installée + +Les deux versions s’appellent « Cursor » dans le Dock et partagent un identifiant de paquet ; vérifiez donc `product.json` : + +| Plateforme | product.json | +|---|---| +| macOS | `/Applications/Cursor Private Inference.app/Contents/Resources/app/product.json` | +| Windows | `%LOCALAPPDATA%\\Programs\\cursor-private-inference\\resources\\app\\product.json` | +| Linux | `/resources/app/product.json` (une AppImage doit d’abord être extraite) | + +`nameLong` vaut `"Cursor Private Inference"` pour la version à agent local et `"Cursor"` pour la version classique ; `version` indique la version (3.18.25 au moment de la rédaction). La carte Cursor sous Integrations dans le tableau de bord effectue le même contrôle et affiche ce qu’elle a trouvé. Le mode local est activé dans le paquet de l’espace de travail, pas dans `product.json` ; aucun indicateur ne permet donc de le basculer. Si `nameLong` indique Cursor classique, cette installation ne peut pas joindre une passerelle loopback. + +La boucle d’agent qui communique avec la passerelle réside dans un fichier sous le même répertoire d’installation, `extensions/cursor-agent-exec/dist/main.js`. opencodex le lit en mode lecture seule, avec une limite de taille, pour connaître la table des efforts de raisonnement de Cursor ; voir « Modèles et effort de raisonnement ». + +## Configurer la passerelle + +opencodex doit être en cours d’exécution (`ocx service status`). Les deux méthodes suivantes aboutissent au même réglage. + +**Dans l’application.** Settings → Models → Gateway → Configure gateway : + +| Champ | Valeur | +|---|---| +| Base URL | `http://127.0.0.1:10100/v1` (incluez `/v1` ; le loopback en `http://` est accepté) | +| API Key | la valeur de `OPENCODEX_API_AUTH_TOKEN` si votre service utilise l’authentification API ; sinon, n’importe quelle valeur fictive comme `opencodex-loopback` | + +Cliquez sur **Refresh model list**. Le sélecteur se remplit avec la liste `/v1/models` d’opencodex ; activez les lignes souhaitées. + +**Avec des variables d’environnement.** L’application les lit au démarrage : + +```text +CURSOR_LOCAL_AGENT_BASE_URL=http://127.0.0.1:10100/v1 +CURSOR_LOCAL_AGENT_API_KEY=opencodex-loopback +CURSOR_LOCAL_AGENT_HEADERS= # optional, newline-separated "Header-Name: value" lines +``` + +`CURSOR_LOCAL_AGENT_HEADERS` rejette `User-Agent` et les espaces réservés `{...}` non résolus ; `{gitOrgRepo}` et `{gitBranch}` sont développés. + +Ordre de priorité, du plus élevé au plus faible : identifiants propres au modèle → passerelle enregistrée dans Settings → `CURSOR_LOCAL_AGENT_*` → `ANTHROPIC_BASE_URL` / `ANTHROPIC_AUTH_TOKEN` (solution de compatibilité). L’environnement ne remplace pas une passerelle enregistrée ; effacez d’abord celle-ci dans Settings si vous souhaitez utiliser les variables d’environnement. + +Cursor Private Inference est une application graphique : un profil de shell interactif ne suffit donc pas à lui seul. La variable doit se trouver dans l’environnement du processus qui lance l’application. + +| Système | Où la définir | +|---|---| +| macOS | `launchctl setenv CURSOR_LOCAL_AGENT_BASE_URL http://127.0.0.1:10100/v1` pour la session de connexion courante, ou un LaunchAgent avec `EnvironmentVariables` pour la rendre persistante. Lancer l’application depuis un terminal fonctionne aussi. | +| Windows | `setx CURSOR_LOCAL_AGENT_BASE_URL http://127.0.0.1:10100/v1` (portée utilisateur ; concerne les nouveaux processus) ou System Properties → Environment Variables. Redémarrez ensuite l’application. | +| Linux | `~/.profile` ou `~/.pam_environment` pour une session de gestionnaire d’affichage, ou `systemctl --user set-environment CURSOR_LOCAL_AGENT_BASE_URL=http://127.0.0.1:10100/v1` lorsque le bureau utilise une session systemd utilisateur. Une AppImage lancée depuis un terminal hérite de l’environnement de ce shell. | + +Cette version existe pour macOS (arm64, x64, universel), Windows (x64, arm64) et Linux (x64, arm64). Sa configuration est identique sur ces plateformes. + +## Depuis le tableau de bord + +Le tableau de bord opencodex comporte un onglet **Cursor** sous Integrations (`/#integrations/cursor`). Il n’écrit rien dans Cursor : ni base de réglages, ni entrée du trousseau, ni paquet de l’application. Il n’y a donc aucun commutateur à actionner. L’onglet vous fournit les valeurs et indique si elles ont fonctionné. + +- **Versions installées.** Il indique si Cursor Private Inference est présent, avec son chemin et sa version, ainsi que Cursor classique, avec son chemin. Si seul Cursor classique est trouvé, l’onglet le signale, renvoie ici et propose un bouton **Rechercher le programme d’installation de Private Inference**. Ce bouton demande au canal de mise à jour de Cursor quel programme d’installation local-mode il annonce pour cette plateforme et ce processeur (le canal `cursor-local` sur `api2.cursor.sh`), puis affiche cette version avec un lien vers l’installateur `downloads.cursor.com/local-mode/`. Rien n’est demandé tant que vous n’appuyez pas sur le bouton : l’ouverture de l’onglet et son actualisation périodique ne contactent jamais le canal de Cursor. opencodex se contente d’afficher le lien : il ne télécharge, ne lance et n’installe jamais rien. La réponse est mise en cache 30 minutes (5 après un échec). Si le canal est injoignable, répond de façon inexploitable ou si Cursor ne publie aucune version pour cet ordinateur (seuls x64 et arm64 sous Windows, macOS et Linux en ont une), l’onglet indique que l’installateur n’a pas pu être déterminé ; les valeurs de passerelle ci-dessous restent valables dans tous les cas. Cursor classique fait toujours passer les points de terminaison personnalisés par les serveurs Cursor et ne peut donc pas joindre un proxy loopback. +- **Valeurs de la passerelle.** L’URL de base utilise le port d’écoute propre au proxy, lu dans son enregistrement d’exécution. Ainsi, même si le tableau de bord passe par un proxy inverse, l’URL affichée indique le port joignable par Cursor sur cette machine. Un bouton Copy permet de la copier. La ligne API Key dépend de l’adresse d’écoute : sans besoin d’identifiant, elle affiche `opencodex-loopback` avec Copy ; si l’authentification API est active ou qu’une clé API opencodex est configurée, elle vous demande d’utiliser l’une de vos clés et renvoie à l’onglet API Keys. Toute clé configurée convient, pas seulement `OPENCODEX_API_AUTH_TOKEN`. +- **Connexion.** L’onglet affiche la dernière requête `/v1/models` dont le User-Agent est exactement `Cursor/`, l’en-tête envoyé par l’environnement d’agent local de Cursor, avec l’heure et la version. Il affiche « never seen » jusqu’à ce que Cursor appelle le proxy ; appuyer sur **Refresh model list** dans Cursor fait changer cet état. La carte s’actualise toutes les 15 secondes tant que l’onglet est ouvert. +- **Ce que Cursor affichera.** Un tableau Model / Reasoning / Context pour les modèles annoncés par opencodex, avec les mêmes exclusions de modèles désactivés et de listes d’autorisation des fournisseurs que la liste brute, selon les règles de la section suivante. C’est une prévision : Cursor choisit les niveaux de Reasoning dans sa propre table. + +## Modèles et effort de raisonnement + +Le sélecteur utilise la liste brute `/v1/models` d’opencodex. Deux conditions déterminent si une ligne de modèle reçoit un contrôle **Reasoning** : + +1. opencodex doit annoncer les capacités de la ligne (`api_types` et un objet `capabilities`). C’est le cas à partir de la v2.41. Les proxys plus anciens montrent les modèles, mais sans contrôle d’effort. +2. L’identifiant du modèle, après suppression de tout ce qui précède le dernier `/` et de tout suffixe `@…`, doit correspondre à la table d’effort propre à Cursor. Cette table est intégrée à l’application (`extensions/cursor-agent-exec/dist/main.js`) ; opencodex la lit dans l’installation détectée pour que la prévision du tableau de bord suive les mises à jour de Cursor. La carte indique la version lue ou « static mirror » si aucune n’est trouvée. Cursor détermine les niveaux, pas opencodex, et aucun champ `/v1/models` ne peut ajouter un modèle à cette table. La matrice ci-dessous est l’instantané 3.18.25 contenu dans la copie statique : + +| Identifiant du modèle (après le dernier `/`) | Niveaux affichés par Cursor | Champ transmis | +|---|---|---| +| `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna` | de Low à Extra High : Low, Medium, High, Extra High | `reasoning.effort` | +| `gpt-5`, `gpt-5.x` | de Low à Extra High : Low, Medium, High, Extra High | `reasoning.effort` | +| `claude-opus-5`, `claude-sonnet-5`, `claude-opus-4.7`, `claude-opus-4.8` | de Low à Max : Low, Medium, High, Extra High, Max | `output_config.effort` | +| `claude-opus-4.6`, `claude-opus-4.5`, `claude-sonnet-4.6` | de Low à Max : Low, Medium, High, Max | `output_config.effort` | +| `grok-4.3`, `grok-4.5`, `grok-4.6`, `grok-build-latest` | de Minimal à Extra High : Minimal, Low, Medium, High, Extra High | `reasoning_effort` | +| `gemini-*` (nécessite `supports_reasoning`) | Minimal, Low, Medium, High | `reasoning_effort` | +| tout autre modèle, y compris `claude-fable-5-1`, `kimi-k3` | aucun contrôle | — | + +Ainsi, `anthropic/claude-opus-5` fonctionne, mais les niveaux `max`/`ultra` d’opencodex pour GPT-5.6 ne sont pas accessibles depuis ce sélecteur. + +### Modèles sans contrôle + +`anthropic/claude-fable-5-1`, `cursor/kimi-k3` et tous les autres modèles absents de la table n’ont pas de contrôle Reasoning. Quand la passerelle annonce `supports_reasoning`, Cursor enregistre pour chaque identifiant concerné une ligne : « Local provider advertises reasoning support for a model with no hardcoded Bottlerocket effort family ». Deux options permettent néanmoins de choisir un effort : + +- **Lignes d’effort** (`cursorEffortRows: true` dans la configuration opencodex, désactivé par défaut) : la passerelle publie une entrée par effort dans le sélecteur pour les modèles absents de la table, comme `anthropic/claude-fable-5-1--high` ou `cursor/kimi-k3--max`, et route chacune vers le modèle de base avec l’effort correspondant. Les modèles pour lesquels Cursor affiche déjà un contrôle ne reçoivent pas de lignes supplémentaires, et un identifiant de modèle exact et connu prévaut toujours sur le suffixe `--`. Appuyez sur Refresh model list après l’activation. La carte du tableau de bord compte les lignes publiées par modèle. Choisir une ligne constitue un choix explicite : son effort prévaut donc aussi sur une directive `ocx-effort` dans la requête. +- **Une valeur par défaut fixe** (`modelDefaultReasoningEfforts` sur le fournisseur) : elle s’applique lorsque Cursor n’envoie aucun effort. + +### « Max » a deux sens différents + +Cursor classique affiche un commutateur **Max** à côté de certains modèles. C’est Max Mode, une fenêtre de contexte plus grande, et non un niveau de raisonnement. Dans la version à agent local, cette possibilité apparaît comme une entrée **Context** dans le menu du modèle. opencodex l’active pour la famille GPT-5.6 native : **272K** par défaut ou **922K** pour l’option 1M, signalée comme plus coûteuse. La valeur choisie plafonne le contexte de ce tour. Les modèles routés affichent une seule fenêtre et aucune entrée Context ; un plafond de contexte fournisseur inférieur à 922K retire aussi cette entrée des lignes natives. + +L’effort de raisonnement **Max** (les niveaux `max`/`ultra` d’opencodex) est l’autre sens, et celui-ci n’est pas accessible : Cursor prend ses niveaux d’effort dans sa propre table plutôt que dans la passerelle, et l’entrée GPT-5.6 s’arrête à Extra High. + +Comme opencodex annonce `responses` dans `api_types`, cette version envoie les tours de l’agent à `/v1/responses` avec `reasoning.effort`, et non à `/v1/chat/completions`. + +Ce choix de protocole a un effet secondaire pour les lignes Claude : Cursor n’envoie l’effort Claude que sous la forme `output_config.effort` sur le protocole Anthropic Messages. Avec une URL de base `/v1`, une ligne Claude qui affiche un contrôle utilise donc quand même la valeur par défaut du fournisseur. Une URL de base se terminant par `/messages` inverse la situation : l’effort Claude est envoyé et celui des modèles de la famille OpenAI est perdu. Une seule entrée de passerelle ne peut pas servir les deux familles ; les lignes d’effort ci-dessus contournent cette limite, car opencodex applique lui-même l’effort. + +## Vérification + +`ocx observe logs` affiche les tours avec `inboundProtocol: responses` et `admissionKind: loopback`. + +| Symptôme | Vérification | +|---|---| +| Réponse 401 de la passerelle | l’API Key envoyée n’est pas acceptée par la configuration d’authentification API active ; pour une adresse loopback sans authentification API, toute valeur convient | +| sélecteur vide | opencodex n’est pas en cours d’exécution ou il manque `/v1` dans la Base URL ; appuyez sur Refresh model list après correction | +| modèles présents, mais sans contrôle Reasoning | opencodex est antérieur à la v2.41, ou l’identifiant est absent de la table Cursor (le tableau de bord affiche —) ; activez `cursorEffortRows` ou définissez une valeur par défaut sur le fournisseur | +| un changement de schéma n’est pas pris en compte | Cursor met en cache `/models` pour chaque chaîne Base URL sans expiration ; Refresh model list relit la liste. Sinon, redémarrez l’application ou enregistrez temporairement une autre forme de l’URL (`localhost` au lieu de `127.0.0.1`) | +| premier tour de 23 000 jetons | comportement attendu : il s’agit de la consigne système locale de Cursor | diff --git a/docs-site/src/content/docs/fr/guides/desktop-app.md b/docs-site/src/content/docs/fr/guides/desktop-app.md new file mode 100644 index 0000000000..b2a87f91df --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/desktop-app.md @@ -0,0 +1,85 @@ +--- +title: Application de bureau +description: Installez et utilisez l’application de bureau OpenCodex sur macOS, Windows et Linux. +--- + +L’application de bureau OpenCodex associe une icône native de zone de notification au tableau de bord web. Son CLI intégré cherche un proxy local existant ; l’application ne démarre son environnement d’exécution intégré que si l’absence de proxy est établie. + +Le tableau de bord est servi depuis le point de terminaison du proxy local trouvé (port `10100` par défaut). L’application de bureau est une enveloppe locale autour de ce tableau de bord et de son environnement d’exécution intégré. + +## Installation + +### macOS + +Téléchargez `OpenCodex--macos.dmg` depuis la [dernière version](https://github.com/lidge-jun/opencodex/releases). Ouvrez le DMG et faites glisser `OpenCodex.app` vers Applications. L’application nécessite macOS 13 ou une version ultérieure. + +Les versions publiées de `OpenCodex.app` sont signées avec un Developer ID et notariées par Apple. Au premier lancement, macOS ne demande normalement que la confirmation habituelle pour une application téléchargée. S’il la bloque tout de même, utilisez **System Settings → Privacy & Security → Open Anyway**. + +### Windows + +Téléchargez `OpenCodex--windows-x64.msi` et lancez l’installation. Windows SmartScreen peut afficher un avertissement, car l’installateur n’est pas encore signé ; choisissez **More info → Run anyway** après avoir vérifié que le téléchargement provient de la page des versions. + +### Linux + +Téléchargez `OpenCodex--linux-x86_64.AppImage` ou `OpenCodex--linux-amd64.deb` depuis la page des versions. + +Pour l’AppImage : + +```bash +chmod +x OpenCodex--linux-x86_64.AppImage +./OpenCodex--linux-x86_64.AppImage +``` + +Pour les distributions fondées sur Debian : + +```bash +sudo apt install ./OpenCodex--linux-amd64.deb +``` + +L’icône de zone de notification nécessite un environnement de bureau compatible avec AppIndicator. + +## Premier lancement + +L’application demande à son CLI intégré d’exécuter `ocx resolve --json` et se connecte à un proxy local accessible s’il en existe déjà un. Elle ne démarre son environnement d’exécution intégré que lorsque le CLI établit l’absence de proxy ; un résultat incertain est affiché comme un échec de démarrage. Le tableau de bord s’ouvre alors dans la vue web de l’application, au point de terminaison loopback trouvé. Un lancement à l’ouverture de session qui démarre masqué dans la zone de notification conserve plutôt la page de démarrage légère et ne charge le tableau de bord qu’à sa première ouverture depuis la zone de notification ou à un nouveau lancement de l’application. + +Utilisez l’action **Open dashboard** ou **Open in browser** de la zone de notification pour passer du tableau de bord intégré à votre navigateur habituel. Le menu permet aussi de rechercher les mises à jour. + +Sur macOS, fermer le tableau de bord laisse l’application active dans la barre des menus. Ouvrez à nouveau OpenCodex depuis le Dock ou le Finder pour réafficher le tableau de bord sans redémarrer le proxy. + +## Utilisation dans la zone de notification + +Sur macOS et Windows, cliquez sur l’icône pour ouvrir un panneau compact d’utilisation. L’action **Show usage** l’ouvre également, notamment sous Linux lorsque la zone de notification ne transmet pas les clics. Sous Linux, le tableau de bord s’ouvre au démarrage, même si l’environnement de bureau n’affiche pas d’icône. + +Le panneau d’utilisation présente les totaux du jour et des 30 derniers jours, le graphique configuré, une liste compacte de modèles et les limites des fournisseurs et des comptes. Les comptes à rebours de réinitialisation des quotas figurent à côté des barres ; survolez-les pour voir l’heure exacte. Les réglages **Menu bar & widget** existants déterminent les sections et le graphique visibles. Les fournisseurs masqués sont exclus du titre, des totaux, des quotas et du graphique. Le graphique inclut l’activité de l’intervalle de temps en cours. Un indicateur de données partielles signifie qu’une partie des données ne peut pas être attribuée de façon fiable. Les mesures manquantes ne sont pas présentées comme une utilisation nulle. Sous Windows et Linux, faites défiler le panneau pour atteindre Refresh et Dashboard après une longue liste de comptes. + +Sur macOS, ce panneau utilise des contrôles SwiftUI natifs et un panneau AppKit défilant. Apple Liquid Glass est utilisé à partir de macOS 26 ; les systèmes plus anciens emploient le matériau natif des fenêtres contextuelles. L’en-tête et les boutons Refresh et Dashboard restent visibles pendant le défilement des longues listes de comptes. Vous pouvez aussi ouvrir le panneau par **View → Show Usage** (Command-Shift-U). Appuyez sur Échap ou cliquez hors du panneau pour le fermer. + +Le menu de la zone de notification affiche le nombre de requêtes et de jetons du jour, ainsi que le coût estimé lorsqu’il est activé. Il utilise la même utilisation en jour local que le widget. Choisissez **Refresh now** pour actualiser immédiatement ; l’application actualise aussi les données toutes les 60 secondes. Les préférences d’affichage restent dans la section **Menu bar & widget** du tableau de bord. Désactiver **Today** masque le résumé, et désactiver **Cost** en retire le coût. + +Une utilisation indisponible ou explicitement non mesurée est affichée sous la forme `—`, et non comme un zéro mesuré. Choisir un titre composé uniquement de l’icône efface l’ancien compteur. Les abréviations conservent les zéros des nombres entiers : dix millions de jetons s’affichent `10M`, et non `1M`. + +## Mises à jour + +Choisissez **Check for Updates…** dans le menu pour lancer immédiatement une recherche. Les versions publiées vérifient aussi automatiquement au démarrage, puis toutes les six heures. + +Quand l’updater Tauri trouve une version plus récente, un point bleu apparaît sur l’icône de la barre de menus macOS ou sur l’icône de la zone de notification Windows/Linux si un hôte de tray est disponible. Le tableau de bord intégré affiche le même signal. Un navigateur ordinaire connecté au même proxy affiche toujours l’état de mise à jour du paquet proxy. Si le shell cesse de signaler son état pendant environ trois minutes, le badge intégré devient unknown jusqu’à la reconnexion. Le point indique la disponibilité ; l’installation reste une action explicite. + +Dans l’application de bureau, le bouton de mise à jour du tableau de bord ouvre la page de mise à jour de l’application. Vous pouvez y vérifier à nouveau, installer une mise à jour signée en attente ou revenir au tableau de bord. La même installation est disponible dans le menu de la zone de notification. En cas d’échec, la mise à jour reste disponible pour une nouvelle tentative. Cette page fonctionne aussi sous Linux lorsque le bureau n’a pas d’icône de tray. Un tableau de bord ouvert dans un navigateur gère à la place l’installation du paquet sur ce proxy. + +Les mises à jour sont vérifiées avec la clé publique signée de l’outil de mise à jour du projet avant installation. Sur macOS, les mises à jour intégrées téléchargent `OpenCodex--macos.app.tar.gz` ; le DMG sert à la première installation. Le manifeste de publication n’est généré que si le secret de la clé de mise à jour est configuré ; les quatre plateformes doivent alors être signées. + +## Widget + +L’application macOS comprend l’extension OpenCodex WidgetKit. Consultez le [guide de l’application de barre de menus macOS](/fr/guides/macos-menu-bar/) pour installer le widget et comprendre les instantanés locaux. + +## Désinstallation + +Sur macOS, faites glisser `OpenCodex.app` d’Applications vers la Corbeille. Sous Windows, supprimez OpenCodex depuis **Installed apps**. Sous les systèmes Linux fondés sur Debian, exécutez : + +```bash +sudo apt remove opencodex +``` + +Pour une AppImage, supprimez le fichier téléchargé. + +Si les réglages enregistrés de la barre de menus sont illisibles, les modifications partielles sont refusées afin de préserver le fichier. Restaurez-le ou réinitialisez explicitement les réglages du composant associé avant de les modifier à nouveau. diff --git a/docs-site/src/content/docs/fr/guides/integrations.md b/docs-site/src/content/docs/fr/guides/integrations.md index 38fb208516..7588d5526f 100644 --- a/docs-site/src/content/docs/fr/guides/integrations.md +++ b/docs-site/src/content/docs/fr/guides/integrations.md @@ -1,10 +1,10 @@ --- title: Intégrations -description: Connectez opencodex à OpenCode, Pi, OMP, Hermes, OpenClaw, Kimi Code, gjc, DeepSeek Harness, MiniMax Code, ZCode, Prime Agent, Aside, Raycast et omo depuis le tableau de bord — un commutateur par client, avec une sauvegarde avant chaque écriture. +description: Connectez opencodex à OpenCode, Pi, OMP, Hermes, OpenClaw, Kimi Code, gjc, DeepSeek Harness, MiniMax Code, ZCode, Prime Agent, Aside, Raycast, omo, Cline CLI, Kilo et Factory Droid depuis le tableau de bord — un commutateur par client, avec une sauvegarde avant chaque écriture. --- L'onglet **Intégrations** écrit le bloc fournisseur d'opencodex dans le fichier de configuration du client, -puis peut le retirer. Quinze clients fonctionnent ainsi, chacun avec son propre commutateur : +puis peut le retirer. Dix-sept clients fonctionnent ainsi, chacun avec son propre commutateur : | Client | Fichier de configuration | Format | Prise d'effet de la modification | Identifiant | |---|---|---|---|---| @@ -23,6 +23,10 @@ puis peut le retirer. Quinze clients fonctionnent ainsi, chacun avec son propre | Raycast | `~/.config/raycast/ai/providers.yaml` | YAML | immédiatement à l'enregistrement — Raycast surveille le fichier | aucun — bouclage uniquement | | omo | `~/.omo/agent/models.json` | JSON | nouvelles sessions | espace réservé de bouclage | | Cline CLI | `~/.cline/data/settings/providers.json` + `models.json` | JSON | après arrêt et redémarrage | bouclage uniquement | +| Kilo | premier fichier existant parmi `kilo.jsonc`, `kilo.json`, `opencode.jsonc`, `opencode.json` ou `config.json` sous `~/.config/kilo` (`XDG_CONFIG_HOME` déplace ce répertoire ; `kilo.jsonc` est créé si aucun n'existe) | JSONC | nouvelles sessions | `OPENCODEX_KILO_API_KEY` | +| Factory Droid | `~/.factory/settings.json` (`%USERPROFILE%\.factory\settings.json` sous Windows) | JSON | dès la détection du fichier | boucle locale sans clé | + +Les modèles GJC dotés d'une échelle d'effort de raisonnement prise en charge exportent `reasoning: true`, `thinking.levels` et `compat.supportsReasoningEffort`, afin que GJC propose le choix de l'effort. Les modèles Codex natifs reçoivent leur échelle standard même si le catalogue l'omet. Ces champs sont absents sans échelle connue ; `none` n'envoie pas d'effort et `ultra` devient `max` sur le réseau. Actualisez l'intégration pour mettre à jour ces options. La prise en charge gérée de DSH exige au minimum **DSH 0.1.0-rc.6**. OpenCodex ne possède que le fragment `llm-pi-ai.providers.opencodex` : **Appliquer** et **Actualiser** remplacent ce fragment, **Désactiver** ne @@ -134,6 +138,8 @@ Kimi Code, gjc, MiniMax Code et Raycast — documents YAML, JSON5 et TOML rééc d'opencodex ont été modifiées, le commutateur se verrouille et la désactivation est refusée plutôt que de deviner quelles modifications vous appartiennent. +Exception pour Hermes : l'ajout de `session_affinity_header: session-id` seul dans un bloc déjà géré peut être adopté via **Apply** ; toute autre modification d'un champ géré reste un conflit. Jusqu'à cette application, l'actualisation automatique de la liste des modèles est également suspendue. Le réglage concerne tous les modèles du provider et nécessite une version de Hermes qui le prend en charge ; il ne garantit aucun taux de succès du cache. Voir le [guide de mise à niveau en anglais](/guides/integrations/#hermes-session-affinity). + ## Prévisualiser et confirmer les modifications Appliquer, Remplacer, Désactiver et Restaurer commencent désormais par un aperçu. La boîte de dialogue @@ -254,6 +260,36 @@ Les détails des clients ont été vérifiés par rapport au format de configura consultez les notes de recherche dans `devlog/_fin/260802_client_toggle_api/002_client_toggle_matrix.md` pour savoir ce qui a été contrôlé et quand. +## ZCode 3.14 et versions ultérieures + +ZCode 3.14 a déplacé ses fournisseurs personnalisés vers `~/.zcode/v2/provider_config.json` et ne +lit plus `~/.zcode/v2/config.json` qu'au travers d'un import unique, exécuté seulement quand le +nouveau fichier est absent. ZCode crée ce nouveau fichier au premier lancement : sur toute +installation déjà démarrée une fois, l'import a donc déjà eu lieu et une écriture dans +`config.json` n'atteint plus rien. + +opencodex écrit désormais `provider_config.json` directement quand il le peut. Activer +l'intégration ajoute la règle de fournisseur `opencodex` dans ce fichier, une actualisation du +catalogue la met à jour, et la désactivation retire exactement ce qu'opencodex y a mis. Toutes les +autres règles du fichier restent intactes, y compris celle qu'un autre fournisseur conserve pour un +identifiant de modèle qui figure aussi chez nous. Une règle portant l'identifiant `opencodex` +qu'opencodex n'a pas écrite est un conflit et non quelque chose à reprendre : réglez-la dans ZCode, +ou utilisez l'écrasement explicite. + +Deux situations refusent encore au lieu d'écrire. Un bloc écrit par opencodex avant le déplacement +du stockage maintient l'intégration sur `config.json` : désactivez-la d'abord à cet endroit, puis +réactivez-la pour écrire le nouveau stockage. Et un `provider_config.json` dont le +`schemaVersion` n'est pas un de ceux qu'opencodex a observés est signalé plutôt que fusionné : +ce fichier contient tous les fournisseurs de ZCode, et y affirmer une forme échangerait une +absence d'effet silencieuse contre une perte silencieuse. L'état nomme le fichier que ZCode lit dès +que l'intégration ne l'écrit pas. + +Dans ce second cas, ajoutez le fournisseur dans les réglages de ZCode : URL de base +`http://127.0.0.1:10100/v1` (ajustez le port à votre écoute), une clé non vide quelconque, et les +identifiants de modèle donnés par `ocx export --client zcode`. Supprimer +`provider_config.json` pour relancer l'import de ZCode n'est pas pris en charge : cela détruit +tous les fournisseurs que ZCode y conserve. + ## Cline CLI Cline CLI utilise providers.json et models.json. Quittez Cline avant toute modification ou synchronisation, puis redémarrez-le. Annuler restaure les deux originaux. Le fournisseur par défaut reste inchangé. Cette intégration ne migre pas le stockage des anciennes extensions VS Code. @@ -265,3 +301,17 @@ ocx integration client restore --op ``` [CLI / rollback / CLINE_PROVIDER_SETTINGS_PATH](/guides/integrations/#cline-cli). + +## Kilo + +Kilo n’écrit que `provider.opencodex` dans le premier fichier global existant sous `~/.config/kilo` (`XDG_CONFIG_HOME` déplace ce répertoire ; `kilo.jsonc` est créé si aucun candidat n’existe). Si un autre fichier candidat définit aussi `provider.opencodex`, l’état signale un conflit et Appliquer refuse. Les autres clés restent inchangées. Appliquer réécrit tout le fichier ; commentaires et virgules finales ne sont pas conservés. Sélectionnez `opencodex/` dans Kilo. + +Désactiver peut retirer le bloc appartenant à OpenCodex du fichier enregistré même si un autre candidat est en conflit ou ne peut pas être analysé ; cet autre fichier reste intact. + +```bash +ocx integration client enable --client kilo +``` + +## Factory Droid + +Factory Droid utilise `~/.factory/settings.json` (`%USERPROFILE%\.factory\settings.json` sous Windows). Activez explicitement l’intégration avec `ocx integration client enable --client droid`, puis choisissez un modèle personnalisé dans `/model`. Les entrées gérées n’utilisent pas de clé et fonctionnent uniquement en boucle locale. La désactivation supprime ces entrées ; l’annulation restaure les octets sauvegardés. Si l’ancien `config.json` contient des entrées OpenCodex ou si `settings.local.json` remplace `customModels`, résolvez ce conflit avant l’activation. Consultez la [documentation Factory BYOK](https://docs.factory.ai/model-independence/byok). diff --git a/docs-site/src/content/docs/fr/guides/macos-menu-bar.md b/docs-site/src/content/docs/fr/guides/macos-menu-bar.md new file mode 100644 index 0000000000..6c7caf5234 --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/macos-menu-bar.md @@ -0,0 +1,58 @@ +--- +title: Application de barre de menus macOS +description: Utilisez la zone de notification macOS, le panneau natif d’utilisation et le widget de l’application de bureau OpenCodex. +--- + +L’élément de barre de menus macOS fait partie de l’application de bureau OpenCodex. Il affiche l’utilisation du proxy local et ouvre un panneau natif d’utilisation. La même application comprend aussi le tableau de bord et une extension WidgetKit. Consultez le [guide de l’application de bureau](/fr/guides/desktop-app/) pour l’installation sur les autres plateformes. + +## Installation + +Téléchargez `OpenCodex--macos.dmg` depuis la [dernière version](https://github.com/lidge-jun/opencodex/releases). Ouvrez le DMG et faites glisser `OpenCodex.app` vers Applications. L’application de bureau nécessite macOS 13 ou une version ultérieure ; son widget nécessite macOS 14 ou une version ultérieure. + +## Premier lancement + +Les versions publiées de `OpenCodex.app` sont signées avec un Developer ID, utilisent l’environnement d’exécution renforcé et sont notariées par Apple, avec le ticket joint à l’application. Au premier lancement, macOS ne demande normalement que la confirmation habituelle pour une application téléchargée sur Internet. S’il la bloque tout de même, ouvrez **System Settings → Privacy & Security** et choisissez **Open Anyway** pour OpenCodex. Les applications que vous compilez vous-même reçoivent une signature ad hoc ; consultez [Compilation depuis les sources](#compilation-depuis-les-sources). + +À l’ouverture, l’application affiche sa progression de démarrage dans une fenêtre. Elle active **Start at Login** une fois lors du premier lancement ; vous pouvez désactiver ce réglage depuis le menu de la zone de notification. Lors des lancements suivants par l’élément de connexion, la fenêtre reste masquée tandis que l’icône demeure disponible. + +## Barre de menus et panneau d’utilisation + +Le titre de la barre de menus affiche par défaut le total de jetons du jour. Dans les réglages **Menu bar & widget** du tableau de bord, vous pouvez choisir les requêtes, les jetons, le coût estimé, le quota ou l’icône seule. + +Utilisez **Show Usage** dans le menu de la zone de notification pour ouvrir le panneau natif. Selon vos réglages d’affichage, il présente les totaux du jour et des 30 derniers jours, un graphique d’utilisation, une liste de modèles et les limites des fournisseurs et des comptes. Les totaux comprennent les jetons et les requêtes, ainsi que le coût estimé lorsqu’il est activé. Les lignes de quota indiquent leur fenêtre, leur pourcentage et l’heure de réinitialisation. Les mesures manquantes apparaissent sous la forme `—`, et une utilisation partielle est signalée comme incomplète. + +Le panneau comporte les contrôles **Refresh**, **Dashboard** et **Settings**. **Dashboard** ouvre la vue d’utilisation dans la fenêtre de bureau ; **Settings** y ouvre les réglages du composant associé. Le menu propose également **Open Dashboard**, **Open in Browser**, **Start at Login**, **Stop proxy**, **Check for Updates…**, l’élément **Install update** lorsqu’une mise à jour est disponible et **Quit**. **Stop proxy** reste affiché, mais n’est activé que si l’application a elle-même démarré le proxy ; un proxy démarré séparément continue de fonctionner. Fermer la fenêtre ou utiliser Command-Q masque l’application lorsque son icône est disponible ; utilisez **Quit** dans le menu pour la quitter. + +Le bouton de mise à jour du tableau de bord ouvre la page de mise à jour de l’application ; elle vérifie et installe la même mise à jour signée que le menu de la zone de notification. + +Le titre est actualisé toutes les 60 secondes. Tant que le panneau natif est ouvert, ses données sont également actualisées toutes les 60 secondes ; **Refresh** demande une mise à jour immédiate. + +## Widget + +Sur macOS 14 ou une version ultérieure, ouvrez une fois OpenCodex.app, puis cliquez avec la touche Contrôle sur une zone vide du bureau, choisissez **Edit Widgets**, recherchez **OpenCodex** et ajoutez la taille souhaitée. Selon leur taille, les widgets affichent différentes combinaisons de l’état du proxy, des jetons et requêtes du jour, du coût estimé, des quotas et d’un graphique d’utilisation. L’extension lit un instantané local écrit par l’application de bureau ; il contient des données d’affichage, sans clés API ni données brutes de compte. L’application actualise l’instantané du widget tous les cinq cycles de 60 secondes de l’icône, soit environ toutes les cinq minutes lorsque le proxy est connecté. WidgetKit demande aussi une nouvelle chronologie après cinq minutes. + +## Connexion au proxy + +L’application de bureau demande à son CLI intégré d’exécuter `ocx resolve --json`. Elle se connecte à un proxy local existant et accessible, ou ne démarre son environnement d’exécution intégré que si le CLI établit qu’aucun environnement n’écoute. Si la découverte reste incertaine, le démarrage signale le problème au lieu de lancer un second proxy. L’application communique avec le port trouvé sur `127.0.0.1`. + +Pour les requêtes de gestion, l’application essaie d’abord sans jeton. Si le proxy répond HTTP 401, elle réessaie avec `OPENCODEX_ADMIN_AUTH_TOKEN` provenant de son environnement ou avec le fichier `admin-api-token` du répertoire de configuration trouvé. Elle n’utilise pas le trousseau macOS pour ce jeton. L’enveloppe de bureau ne peut pas se connecter à un proxy lié uniquement à une adresse qu’elle ne peut pas joindre en loopback. + +## Compilation depuis les sources + +Sur macOS 13 ou une version ultérieure, avec Bun, Rust et les outils Swift/Xcode pour macOS, compilez le tableau de bord depuis la racine du dépôt, puis exécutez les commandes de bureau depuis `desktop/` : + +```bash +bun install +bun run build:gui +cd desktop +bun install +bun run prepare-sidecar +bun run prepare-widget +bun run build:local +``` + +`build:local` produit l’application locale et le DMG sans exiger de clé de signature pour l’outil de mise à jour Tauri. Une commande directe `bunx tauri build` exige `TAURI_SIGNING_PRIVATE_KEY`, car elle produit également un artefact de mise à jour. La compilation du widget utilise une signature ad hoc sauf si `MACOS_SIGN_IDENTITY` est défini, et les paquets de bureau locaux sont également signés ad hoc. L’application fonctionne, mais macOS n’enregistre pas une extension de widget signée ad hoc ; une compilation locale n’affiche donc généralement aucun widget OpenCodex. `build:local` signe toujours l’application ad hoc : définir seulement `MACOS_SIGN_IDENTITY` ne suffit pas. Le widget n’est enregistré que si l’application et l’extension sont toutes deux signées par la même équipe Developer ID, comme dans la version publiée. Utilisez une version publiée si vous avez besoin du widget. + +## Désinstallation + +Désactivez **Start at Login** dans le menu si vous l’aviez activé, puis déplacez `OpenCodex.app` d’Applications vers la Corbeille. Cela supprime le CLI intégré et l’extension du widget, mais ne supprime ni l’état `$OPENCODEX_HOME` du proxy ni un service `ocx` installé séparément. L’application de bureau écrit également un identifiant d’installation et des marqueurs d’élément de connexion dans son répertoire de configuration, ainsi qu’un instantané du widget sous `~/Library/Containers/com.opencodex.desktop.widget/Data/Library/Application Support/OpenCodex/snapshot.json` ; déplacer l’application vers la Corbeille ne supprime pas ces fichiers. diff --git a/docs-site/src/content/docs/fr/guides/native-main-profiles.md b/docs-site/src/content/docs/fr/guides/native-main-profiles.md new file mode 100644 index 0000000000..c893e6275f --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/native-main-profiles.md @@ -0,0 +1,89 @@ +--- +title: Profils de connexion principale native +description: Gérez les profils de connexion Codex native enregistrés séparément du routage OpenCodex Pool. +--- + +## La connexion native ne sélectionne pas le Pool + +Dans **Codex Set → Multi-auth**, ouvrez **Manage main login** dans le panneau +**Native main login**, juste sous la carte du compte principal. Le même panneau +figure à côté de cette carte dans l'espace des comptes fournisseurs. Il gère la +connexion physique de Codex, et non le compte choisi pour la prochaine requête +Pool. Il n'ajoute pas d'onglet Integrations et ne remplace pas les contrôles Pool. + +Le **Effective CODEX_HOME** affiché appartient au serveur OpenCodex. Avec un +tableau de bord distant, il peut s'agir d'un autre ordinateur que celui du +navigateur. **Registered active profile** désigne le propriétaire enregistré +dans le magasin chiffré des profils ; le serveur vérifie la connexion physique +avant de basculer. Une connexion modifiée hors d'OpenCodex peut ainsi produire +une erreur de propriétaire non concordant, au lieu d'être écrasée en silence. + +## Enregistrer et basculer + +Utilisez **Save current as profile** pour enregistrer la connexion existante de +l'application. Si la connexion active est déjà enregistrée, cette opération met +à jour son libellé ; elle n'inscrit pas un nouveau compte. Le changement de +profil exige le magasin d'identifiants par fichiers pris en charge et un +trousseau du système d'exploitation disponible. Les codes de diagnostic +expliquent pourquoi les contrôles sont indisponibles ; ne contournez pas une +erreur de trousseau ou de propriété en copiant des identifiants Pool dans le +fichier de connexion native. Le panneau est désactivé pendant la réauthentification +native existante de la carte principale. + +Choisissez **Switch** à côté d'un profil enregistré inactif. Vérifiez le libellé +cible et le répertoire personnel côté serveur, arrêtez Codex natif utilisant ce +répertoire, puis cochez la confirmation d'arrêt et validez. Aucun changement +d'identifiants n'est envoyé avant confirmation. Le panneau relit le répertoire, +le propriétaire actif et l'état de récupération juste avant l'envoi. Si cet +état a changé, examinez-le et confirmez de nouveau. Le backend existant reste +la référence pour les verrous, la vérification des processus, la fin des +requêtes en cours, l'activation et le retour arrière. + +Après la réussite, suivez l'indication de redémarrage affichée et rouvrez Codex +natif avec ce répertoire. Le panneau relit l'état des profils et actualise le +contrôleur de comptes existant. Il ne déclenche aucune mutation de sélection ou +de configuration Pool et ne modifie ni les clés fournisseur, ni les tâches, ni +l'historique. Le backend continue de rapprocher l'identité native `__main__`, +comme dans le parcours CLI. + +## Récupération et profils précédents + +Il s'agit d'opérations distinctes : + +- **Recover interrupted change** rapproche une transaction backend inachevée. + **Restore pending transaction** demande son retour arrière. Les deux exigent + une confirmation d'arrêt distincte. Ces contrôles restent disponibles quand + une liste de profils endommagée est illisible, mais que les diagnostics + signalent une récupération en attente. +- **Return to previously displayed**, après un changement réussi, sélectionne + le profil affiché avant le changement selon le parcours normal avec + confirmation. Ce raccourci n'existe que dans la mémoire de la page, pour le + répertoire et le propriétaire actif attendus ; il disparaît si la page est + rechargée ou si le proxy ou le propriétaire change. L'API ne renvoie pas la + source de la transaction : ce n'est donc pas un journal d'annulation vérifié + par le serveur. Un autre opérateur peut avoir changé la connexion entre la + lecture préalable et votre bascule. Après un rechargement, sélectionnez + directement le profil enregistré souhaité. + +Une erreur réseau ne prouve pas qu'une écriture a échoué ou été annulée. Après +une mutation envoyée, même si la réponse se perd, le panneau relit l'état du +serveur et ne réessaie jamais automatiquement. Si l'actualisation échoue, un +changement réussi n'est pas annoncé comme annulé. Actualisez et consultez les +diagnostics avant toute autre opération. La commande `ocx account main doctor` +fournit des diagnostics côté serveur. + +## Périmètre de cette phase + +Ce panneau liste, enregistre, active et récupère les profils existants via la +frontière `/api/native-main-profiles`. Il ne lance aucun processus de connexion +et n'expose aucun jeton d'écriture de préparation. L'ajout d'une autre connexion +native reste du ressort de la commande CLI `ocx account main add` ; l'inscription +depuis le navigateur est un suivi distinct de l'issue #3417. La +réauthentification de l'emplacement principal actuel suit un autre parcours, +que ce panneau ne remplace pas. + +Les données des profils ne sont pas écrites dans le stockage du navigateur. Le +client ne projette que les champs publics, affiche des codes d'erreur autorisés +plutôt que les messages bruts du serveur et utilise le mécanisme de requêtes +authentifiées existant de l'application. L'authentification de gestion, les +contrôles de session GUI/CSRF et l'admission des routes du backend restent inchangés. diff --git a/docs-site/src/content/docs/fr/guides/opencode.md b/docs-site/src/content/docs/fr/guides/opencode.md index 6bb2da7e29..a645fb882b 100644 --- a/docs-site/src/content/docs/fr/guides/opencode.md +++ b/docs-site/src/content/docs/fr/guides/opencode.md @@ -131,10 +131,7 @@ Rien à annuler — aucun fichier de configuration généré n'est écrit sous ` `limit.context` n’est écrit que lorsque le catalogue fournit une fenêtre de contexte faisant autorité. Dans le cas contraire, le bloc `limit` entier est omis et opencode conserve ses propres valeurs par défaut. -Le schéma d’opencode rejette un bloc `limit` qui contient `context` sans `output`. Comme le catalogue ne fournit -aucune limite de sortie faisant autorité par modèle, opencodex émet également un budget `output` de `32000`, limité -à la fenêtre de contexte afin qu’un modèle à petit contexte ne reçoive jamais `output > context`. Cette valeur sert -uniquement à satisfaire le schéma ; elle ne prétend pas représenter la véritable limite d’un modèle particulier. +La limite de sortie utilise le maximum connu du modèle dans le catalogue ou les métadonnées générées. La valeur `32000` ne sert que de repli si cette limite est inconnue. La limite est toujours plafonnée à la fenêtre de contexte, et les limites connues inférieures à `32000` sont conservées. Le bloc fournisseur `opencodex` est régénéré à chaque lancement, donc des ajustements par modèle y sont apportés ne survivra pas. Conservez plutôt les entrées personnalisées sous votre propre clé de fournisseur. diff --git a/docs-site/src/content/docs/fr/guides/pi.md b/docs-site/src/content/docs/fr/guides/pi.md index 030d91679c..8afed1dd5a 100644 --- a/docs-site/src/content/docs/fr/guides/pi.md +++ b/docs-site/src/content/docs/fr/guides/pi.md @@ -28,7 +28,8 @@ d’exportation de la variable d’environnement et le nombre de modèles dotés "api": "openai-completions", "apiKey": "$OPENCODEX_API_KEY", "compat": { - "sendSessionAffinityHeaders": true + "sendSessionAffinityHeaders": true, + "supportsDeveloperRole": false }, "models": [ { @@ -36,7 +37,7 @@ d’exportation de la variable d’environnement et le nombre de modèles dotés "name": "Claude Opus 5 (anthropic)", "input": ["text"], "contextWindow": 200000, - "maxTokens": 32000 + "maxTokens": 128000 } ] } @@ -46,6 +47,8 @@ d’exportation de la variable d’environnement et le nombre de modèles dotés Les fournisseurs Pi générés activent `compat.sendSessionAffinityHeaders`. Conservez ce réglage lors de la fusion ou de la modification manuelle du fournisseur : Pi transmet un identifiant de session stable, dont OpenCodex dérive l’affinité pour la destination canonique OpenCode Go. Pi peut omettre cet identifiant lorsque `cacheRetention` vaut `none`. +Les fournisseurs Pi générés définissent aussi `compat.supportsDeveloperRole` à `false` : Pi envoie alors son prompt système avec le rôle `system` au lieu de `developer`. OpenCodex transmet les rôles Chat Completions tels quels, et plusieurs amonts compatibles OpenAI refusent `developer` avec une erreur 400 ; tous acceptent `system`. + Les identifiants de modèle sont les sélecteurs canoniques du proxy : les modèles routés apparaissent donc sous la forme `provider/model` (`anthropic/claude-opus-5`) et les slugs natifs OpenAI restent sans préfixe (`gpt-5.6-sol`). Le `name` suffixe — `(anthropic)`, `(native)`, `(routed)` — permet de distinguer, dans le sélecteur de Pi, deux modèles de même nom @@ -103,9 +106,7 @@ refuse de démarrer sans jeton — voir [Accès à distance](/fr/reference/confi faisant autorité. Dans le cas contraire, les deux champs sont omis pour ce modèle et Pi applique ses propres valeurs par défaut ; `ocx export` affiche le nombre de lignes concernées. -`maxTokens` est un budget de `32000` destiné à satisfaire le schéma. Il est plafonné à la fenêtre de contexte, de sorte qu’un -modèle doté d’un petit contexte ne reçoive jamais davantage de sortie que de contexte. Cette valeur ne constitue pas une affirmation sur la -limite maximale réelle d’un modèle donné. +La limite de sortie utilise le maximum connu du modèle dans le catalogue ou les métadonnées générées. La valeur `32000` ne sert que de repli si cette limite est inconnue. La limite est toujours plafonnée à la fenêtre de contexte, et les limites connues inférieures à `32000` sont conservées. Le champ `cost` est volontairement absent. Il exige les quatre champs de prix, alors qu’OpenCodex ne possède aucune donnée tarifaire pour les modèles routés ; émettre des zéros reviendrait à affirmer que tous les diff --git a/docs-site/src/content/docs/fr/guides/providers.md b/docs-site/src/content/docs/fr/guides/providers.md index 7e9115e003..641b9a9373 100644 --- a/docs-site/src/content/docs/fr/guides/providers.md +++ b/docs-site/src/content/docs/fr/guides/providers.md @@ -94,7 +94,7 @@ Le catalogue du transfert ChatGPT ajoute également les identifiants non qualifi ## 2. Connexion au compte (OAuth) -Huit préréglages de fournisseurs utilisent une connexion OAuth. GitHub Copilot s'y ajoute au moyen d'un pont +Des préréglages de fournisseurs peuvent utiliser une connexion au compte, y compris GitHub Copilot au moyen d'un pont expérimental et non officiel reposant sur un flux d'autorisation d'appareil. opencodex enregistre leurs identifiants dans `~/.opencodex/auth.json` et les actualise automatiquement. La CLI de connexion accepte également `ocx login codex`, qui n'est pas l'un des fournisseurs ci-dessus : la commande est routée vers la @@ -121,7 +121,8 @@ ocx logout | --- | --- | --- | --- | | `xai` | `openai-chat` | `https://cli-chat-proxy.grok.com/v1` | OAuth utilise la passerelle d'abonnement Grok CLI distincte. Le remplacement par clé API utilise `https://api.x.ai/v1` et peut injecter Priority Processing. Catalogue Grok découvert en direct en priorité ; `grok-4.5` est le modèle de repli par défaut. | | `anthropic` | `anthropic` | `https://api.anthropic.com` | Modèles Claude ; liste des modèles récupérée en direct depuis `/v1/models`. | -| `kimi` | `openai-chat` | `https://api.kimi.com/coding/v1` | Modèles de programmation Kimi K2.7/K2.6/K2.5. | +| `kimi` | `openai-chat` | `https://api.kimi.com/coding/v1` | Modèles Kimi Code. L'alias `kimi-for-coding` pointe actuellement vers K2.8 Preview (contexte de 1 M de tokens, raisonnement `low`/`high`/`max`, texte et images). `k3-256k` offre une limite fixe de 256 K. | +| `kimi-responses` | `openai-responses` | `https://api.kimi.com/coding/v1` | Réutilise la connexion OAuth de `kimi` avec les mêmes modèles sur le protocole Responses. Le contenu du raisonnement reste chiffré côté serveur ; les appels d'outils et leurs résultats restent visibles. | | `nous` | `openai-chat` | `https://inference-api.nousresearch.com/v1` | Passerelle d'abonnement Nous Research (le même service en amont que celui utilisé par Hermes Agent). Connexion par autorisation d'appareil auprès de `portal.nousresearch.com` ; le jeton d'accès est le JWT d'inférence envoyé avec chaque requête. Le catalogue mixte de modèles payants et `:free` (`tencent/hy3:free`, `stepfun/step-3.7-flash:free`, ...) est découvert en direct pour le compte connecté. Les jetons d'actualisation sont à usage unique et renouvelés à chaque actualisation. | | `kiro` | `kiro` | `https://runtime.us-east-1.kiro.dev` | La connexion initiale importe la session de l'installation locale de `kiro-cli`, déjà authentifiée (sous Unix, installez avec `curl -fsSL https://cli.kiro.dev/install` | `bash`; sous Windows PowerShell, utilisez `irm 'https://cli.kiro.dev/install.ps1'` | `iex`; puis exécutez `kiro-cli login`). **Ajouter un compte** déconnecte `kiro-cli`, lance une nouvelle connexion dans le navigateur qui change le compte utilisé par `kiro-cli`, puis enregistre les métadonnées propres au profil. Les comptes OpenCodex existants sont préservés ; une annulation ou un échec restaure la session `kiro-cli` précédente. | | `google-antigravity` | `google` | `https://daily-cloudcode-pa.googleapis.com` | Google OAuth avec le protocole Cloud Code Assist. La découverte en direct utilise le point de terminaison CCA authentifié `v1internal:fetchAvailableModels` et publie les modèles d'agent accessibles au compte connecté ; le catalogue maintenu reste la solution de repli. | @@ -160,6 +161,14 @@ est nécessaire pour améliorer le taux de succès du cache du Coding Plan ; les restent. Si un service en amont explicitement activé rejette ce champ, opencodex ne le retire pas avant de réessayer et ne modifie pas la configuration enregistrée. Tous les autres fournisseurs le refusent par défaut. +Les prix de `k3`, `k3[1m]` et `k3-256k` pour `kimi`, `kimi-code` et `kimi-responses` +sont des estimations fondées sur les [tarifs de l'API](https://platform.kimi.ai/docs/pricing/chat), +avec l'écriture en cache par défaut de cinq minutes. Ils ne représentent ni la facturation ni +les quotas du Code Plan : la variante 1 M consomme environ deux fois le quota de `k3-256k`. +L'alias `kimi-for-coding` pointe désormais vers K2.8 Preview ; son ancien tarif K2.7 n'est plus +utilisé. Son coût reste inconnu sans `modelCosts` fourni par l'utilisateur, et les règles de routage +qui excluent les coûts inconnus peuvent donc l'écarter. + Vous pouvez également démarrer OAuth à partir du [tableau de bord Web](/fr/guides/web-dashboard/). ### Plusieurs comptes OAuth @@ -261,6 +270,9 @@ la source et la ligne du jeton : Sous Windows, l'importation recherche `%LOCALAPPDATA%\Kiro-Cli\data.sqlite3`. La connexion forcée ou par **Ajouter un compte** nécessite également le binaire local de la CLI : opencodex consulte d'abord `PATH`, puis se rabat sur `%LOCALAPPDATA%\Kiro-Cli\kiro-cli.exe` et `C:\Program Files\Kiro-Cli\kiro-cli.exe`. +Si aucun de ces dossiers ne contient `kiro-cli.exe`, un `kiro.exe` placé dans ces deux mêmes dossiers +`Kiro-Cli` est utilisé. opencodex n'exécute jamais un `kiro` ou `kiro.exe` trouvé dans le `PATH` +ou dans les répertoires bin partagés de macOS/Linux : installez ou liez la CLI sous le nom `kiro-cli`. Après une importation réussie, opencodex conserve les informations d'identification importées dans `~/.opencodex/auth.json`. @@ -283,7 +295,7 @@ existante n'est pas concernée. ## 3. Catalogue des clés API -opencodex fournit 96 préréglages intégrés : 80 à clé, 12 OAuth, trois locaux et un préréglage par défaut de +opencodex fournit 100 préréglages intégrés : 83 à clé, 13 OAuth, trois locaux et un préréglage par défaut de transfert ChatGPT. Dans le tableau de bord, le sélecteur **Ajouter un fournisseur** ouvre le tableau de bord du fournisseur à clé, valide la clé et l'enregistre ; la validation dépend du fournisseur. Parmi les entrées notables : @@ -376,9 +388,13 @@ OpenCode Zen obtenue sur [opencode.ai/auth](https://opencode.ai/auth). Si OpenCo tiers pour le niveau sans clé, opencodex pourra le suivre ; d'ici là, le préréglage sert à documenter la restriction. Conditions en amont : [opencode.ai/docs/zen](https://opencode.ai/docs/zen/). -La plupart utilisent l'adaptateur `openai-chat` avec une clé Bearer ; quelques fournisseurs qui n'exposent -qu'un point de terminaison compatible Anthropic, comme **Xiaomi MiMo**, emploient l'adaptateur `anthropic` -(`x-api-key`). Volcengine Coding Plan et Agent Plan utilisent leur point de terminaison Responses natif par `openai-responses`. Lors des continuations d'outils validées sur Ark Coding Plan, renvoyer l'élément `reasoning` retourné par le tour précédent provoque `400 InvalidParameter` ; le préréglage Coding Plan retire donc ces éléments avant de transmettre l'entrée de continuation. Cela perd l'état de raisonnement de ce tour et se désactive avec `dropResponsesReasoningItems: false`. Une configuration Coding Plan déjà enregistrée en `openai-chat` n'est pas réécrite et reste sur Chat : pour basculer, passez `adapter` à `openai-responses` et `responsesPath` à `/responses`, ou supprimez puis rajoutez le préréglage. +La plupart utilisent l'adaptateur `openai-chat` avec une clé Bearer ; les préréglages compatibles +Anthropic, comme **Xiaomi MiMo** (`xiaomi`), utilisent l'adaptateur `anthropic` (`x-api-key`). Xiaomi propose +aussi un préréglage OpenAI Chat, `xiaomi-mimo`, et un préréglage avec forfait de jetons, `mimo`. Tous trois +utilisent MiMo V2.6 par défaut (`mimo-v2.6-pro`, ou `mimo-v2.6-flash` pour `xiaomi-mimo`). Xiaomi retirera +`mimo-v2.5` et `mimo-v2.5-pro` le 2026-10-21 sans redirection : changez toute valeur V2.5 par défaut +enregistrée avant cette date ; opencodex ne la modifiera pas à votre place. +Volcengine Coding Plan et Agent Plan utilisent leur point de terminaison Responses natif par `openai-responses`. Lors des continuations d'outils validées sur Ark Coding Plan, renvoyer l'élément `reasoning` retourné par le tour précédent provoque `400 InvalidParameter` ; le préréglage Coding Plan retire donc ces éléments avant de transmettre l'entrée de continuation. Cela perd l'état de raisonnement de ce tour et se désactive avec `dropResponsesReasoningItems: false`. Une configuration Coding Plan déjà enregistrée en `openai-chat` n'est pas réécrite et reste sur Chat : pour basculer, passez `adapter` à `openai-responses` et `responsesPath` à `/responses`, ou supprimez puis rajoutez le préréglage. Le préréglage DeepSeek intégré route également `deepseek-v4-flash` par son point de terminaison Responses natif et conserve le streaming SSE en amont. Si ce modèle termine tous les éléments de sortie mais omet l'événement Responses final, opencodex applique une réparation après un délai de grâce de cinq secondes, limitée à ce @@ -436,14 +452,20 @@ prise en charge des outils d'agent. Créez un jeton de service dans la [console et copiez la clé d'inférence de Vultr depuis la vue d'ensemble de l'abonnement dans la [console Vultr](https://my.vultr.com). -**Découverte Command Code.** Le préréglage lit la liste `/provider/v1/models` de Command Code depuis l'hôte -fixe de l'API Provider, préserve les identifiants natifs du fournisseur et limite la découverte à 256 KiB et -256 lignes brutes. `ocx login command-code` prend en charge OAuth par connexion dans le navigateur, avec -importation facultative des identifiants locaux depuis `~/.commandcode/auth.json` pour les utilisateurs de la -CLI Command Code. Le catalogue, propre au compte, provient du point de terminaison de découverte authentifié -après la connexion. Les requêtes de chat du préréglage Provider-API `commandcode` utilisent la clé Bearer active -configurée ; le préréglage OAuth `command-code` utilise le jeton Bearer du compte enregistré pour la découverte -authentifiée et les requêtes de chat. Créez des clés Provider-API dans +**Découverte Command Code.** Le préréglage lit la liste `/provider/v1/models` de Command Code depuis +l'hôte fixe de l'API Provider, conserve les identifiants natifs et limite la découverte à 256 KiB et +256 lignes brutes. `ocx login command-code` permet une connexion OAuth dans le navigateur, avec importation +facultative des identifiants de la CLI depuis `~/.commandcode/auth.json` pour ses utilisateurs actuels. Le +catalogue des modèles est propre au compte et provient du point de terminaison de découverte authentifié après +la connexion. Le préréglage Provider-API (`commandcode`) envoie la clé active configurée : la plupart des +identifiants de modèle utilisent Chat Completions avec un en-tête Bearer, tandis que les identifiants `claude-*` +utilisent Anthropic Messages avec `x-api-key`, car Command Code ne les sert que sur `/provider/v1/messages`. +Un fournisseur qui reprend le nom `commandcode` pour un autre point de terminaison conserve son propre +protocole. Le préréglage OAuth (`command-code`) utilise le jeton Bearer du compte enregistré pour la découverte +authentifiée et diffuse la génération depuis `/alpha/generate` au format NDJSON. Le balisage d'appel d'outil +MiMo renvoyé comme texte par la passerelle est supprimé lorsqu'il fait double emploi avec un appel réel. Sur +les modèles MiMo, un appel complet à un outil déclaré, sans équivalent natif, n'est rétabli qu'après une fin +sans erreur ; si le tour est interrompu ou filtré, le balisage reste du texte. Créez des clés Provider-API dans [Command Code Studio](https://commandcode.ai/studio/). **Quota Command Code.** Le tableau de bord et `ocx account refresh` sondent les fenêtres @@ -622,12 +644,13 @@ Cursor est géré séparément comme adaptateur expérimental. `adapter: "cursor le sélecteur **Ajouter un fournisseur** du tableau de bord comme entrée expérimentale de la configuration locale, avec les métadonnées du catalogue statique de repli de Cursor. Lorsqu'un jeton d'accès Cursor est configuré, opencodex utilise le transport HTTP/2 direct de Cursor. Sa liste de repli intégrée comprend `gpt-5.6-sol` / -`terra` / `luna` (contexte de 1M), les variantes ordinaires et Fast de Grok 4.5 et 4.6 (500K), ainsi que -`kimi-k3` (262K) ; la découverte en direct détermine celles qui restent visibles pour le compte. Grok 4.6 expose -`low` / `medium` / `high` / `xhigh` sous les deux formes, tandis que 4.5 s'arrête à `high`. Les requêtes Fast -envoient le modèle Grok de base correspondant avec des paramètres `effort` et `fast=true` `requested_model` -distincts ; les identifiants aplatis `cursor-grok-{version}-{effort}-fast` servent uniquement à la découverte et -à la sélection. Cursor ne fournit Kimi K3 qu'avec des identifiants de protocole suffixés par l'effort ; +`terra` / `luna` (contexte de 1M), les variantes ordinaires et Fast de Grok 4.5, 4.6 et 4.7 (500K), ainsi que +`kimi-k3` (262K) ; la découverte en direct détermine celles qui restent visibles pour le compte. Grok 4.6 et 4.7 exposent +`low` / `medium` / `high` / `xhigh` sous les deux formes, tandis que 4.5 s'arrête à `high`. Pour Grok 4.5 et 4.6, +les requêtes Fast envoient le modèle de base avec des paramètres `effort` et `fast=true` distincts dans +`requested_model` ; leurs identifiants aplatis `cursor-grok-{version}-{effort}-fast` servent uniquement à la découverte +et à la sélection. Grok 4.7 figure sans préfixe `cursor-` et envoie directement `grok-4.7-{effort}-fast`. +Cursor ne fournit Kimi K3 qu'avec des identifiants de protocole suffixés par l'effort ; `cursor/kimi-k3` expose donc une échelle `low` / `high` / `max` avec `max` par défaut, conformément à la valeur par défaut documentée de l'API du modèle. L'exécution native read/write/delete/ls/grep/shell/fetch pilotée par le serveur Cursor est désactivée par défaut, car elle contourne le parcours d'approbation et le bac à sable de diff --git a/docs-site/src/content/docs/fr/guides/remote-hub.md b/docs-site/src/content/docs/fr/guides/remote-hub.md index 881a3fc9c1..abda45d604 100644 --- a/docs-site/src/content/docs/fr/guides/remote-hub.md +++ b/docs-site/src/content/docs/fr/guides/remote-hub.md @@ -3,6 +3,8 @@ title: Déploiement Remote Hub description: Déployer un hub opencodex avec une gestion locale, Tailscale Serve et OAuth sans interface locale. --- +Pour les liaisons SSH entre machines, consultez [Liaison distante](/fr/guides/remote-link/). + Un hub conserve les identifiants fournisseur, le catalogue et l’usage sur un hôte. Les clients authentifiés appellent directement son plan de données. Le plan de gestion est distinct : son écoute facultative reste sur `127.0.0.1` et ne sert que le tableau de bord et `/api/*`. Elle ne sert jamais `/v1/*`, `/healthz`, `/readyz` ni WebSocket. Ne publiez pas le port `10101` et n’utilisez pas Tailscale Funnel. ## Rôles, connexion et sécurité @@ -18,6 +20,7 @@ ocx sync Les diagnostics de disponibilité lisibles par un humain affichent les caractères de contrôle des valeurs du catalogue sous forme d’échappements hexadécimaux visibles, aussi bien à la première connexion que lorsque `ocx sync` refuse un catalogue de hub actualisé. Le statut JSON conserve la valeur de diagnostic d’origine. La clé client est écrite dans le fichier privé `service-api-token`, jamais dans `config.json`. En mode connecté, l’usage provient du hub et est filtré par `apiKeyId`; après déconnexion, il provient du stockage local. Il n’existe aucune réplication entre les deux. +`ocx service uninstall` supprime le service local mais conserve une clé existante lorsque le client est connecté, que ses métadonnées de connexion sont invalides ou non concordantes, ou qu’un marqueur de connexion en attente correspond à la clé actuelle. Un marqueur valide pour une clé plus ancienne ne conserve pas une clé de service sans rapport. Si un marqueur est peu sûr, malformé ou illisible, le nettoyage du jeton ne peut pas être vérifié ; la commande avertit au lieu d’affirmer que la clé a été conservée. Utilisez `ocx disconnect` pour supprimer la clé locale et l’état d’un client connecté. Le jeton admin permet la gestion ordinaire mais ne peut jamais créer une session de consentement. Les actions de consentement exigent une `gui-session`, une Origin correspondante et un jeton CSRF. `Tailscale-User-Login` n’est fiable que sur l’entrée de gestion dédiée; renseignez les identités exactes dans `remoteGui.allowedTailscaleUsers`. diff --git a/docs-site/src/content/docs/fr/guides/remote-link.md b/docs-site/src/content/docs/fr/guides/remote-link.md new file mode 100644 index 0000000000..6a619f8d5d --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/remote-link.md @@ -0,0 +1,77 @@ +--- +title: Liaison distante +description: Connecter un ordinateur OpenCodex Home à un ordinateur Child avec SSH. +--- + +Une liaison entre machines connecte un ordinateur OpenCodex **Home** à un ordinateur **Child** avec SSH. Home sert le trafic de Child par le tunnel SSH, et les deux ordinateurs gardent leur service OpenCodex local sur le port `10100`. Le tableau de bord transmet la clé de liaison propre à Child par SSH, sans vous demander de saisir un jeton. + +## Conditions requises + +- Home peut se connecter à Child avec une clé OpenSSH. +- Pour une liaison initiée par Child, Child peut se connecter à Home avec une clé OpenSSH (la connexion par mot de passe n’est pas prise en charge). +- OpenCodex 2.66.0 ou ultérieur est installé sur Child (et sur Home pour une liaison initiée par Child). +- Les deux ordinateurs utilisent macOS ou Linux. +- Le tableau de bord qui lance la liaison est ouvert sur cet ordinateur lui-même (navigateur ou application de bureau, installation autonome) ou via une session Hub appairée. + +SSH par mot de passe et Windows restent hors du flux actuel. Une liaison peut être lancée des deux côtés : depuis Home, comme décrit ci-dessous, ou depuis Child, comme décrit dans la section « Connecter cet ordinateur comme Child ». + +## Ajouter un Child depuis `#remote` + +1. Ouvrez le tableau de bord sur `#remote` et activez Remote Link. +2. Choisissez **Home**, puis **Continue**. La liste des hôtes SSH s’ouvre. +3. Choisissez un hôte parmi les candidats SSH, ou saisissez un alias de configuration SSH. +4. Lancez le test de connexion et comparez l’empreinte proposée avec celle de l’ordinateur visé. Cette comparaison aide à détecter un mauvais hôte ou une clé d’hôte modifiée avant que SSH ne lui fasse confiance. +5. Confirmez l’empreinte, puis connectez Child. + +Le tableau de bord ne demande pas de saisir un jeton. Il sonde d’abord l’hôte et ne peut appliquer la liaison qu’après votre confirmation explicite de l’empreinte. + +## Connecter cet ordinateur comme Child + +Sur l’ordinateur qui doit utiliser les fournisseurs de Home : + +1. Ouvrez le tableau de bord sur `#remote` et activez Remote Link. +2. Choisissez **Child**. La liste des hôtes SSH s’ouvre. +3. Choisissez l’hôte SSH de Home, lancez le test de connexion, puis comparez et confirmez son empreinte d’hôte. +4. Lisez l’avertissement et choisissez **Connect as Child**. + +La connexion redémarre OpenCodex sur cet ordinateur. Les tours Codex déjà en cours se terminent d’abord, et les nouvelles requêtes peuvent échouer pendant une minute au plus pendant le redémarrage. Le tableau de bord se recharge ensuite de lui-même et affiche la liaison Child. Codex continue d’utiliser `http://127.0.0.1:/v1` sur cet ordinateur, sans jeton ni variable d’environnement à définir : l’OpenCodex local relaie chaque requête vers Home, qui y répond avec ses propres fournisseurs et comptes. + +Le rôle **Child** n’est disponible que lorsque OpenCodex tourne sur son port configuré, car Child redémarre exactement sur ce port. Si le tableau de bord indique qu’OpenCodex ne tourne pas sur son port configuré, redémarrez-le d’abord sur ce port. + +## État de la liaison + +- **Connected** signifie que le tunnel SSH est prêt et que Child peut utiliser la liaison Home. +- **Reconnecting** signifie que le tunnel est réessayé. Les requêtes peuvent temporairement renvoyer `503` avec `Retry-After`. Sur un Child connecté depuis son propre tableau de bord, une requête attend d’abord jusqu’à 15 secondes le retour du tunnel. +- **Failed** signifie que la liaison nécessite une intervention. Vérifiez l’authentification SSH, la clé d’hôte confirmée, la redirection ou le délai indiqué. Un Child connecté depuis son propre tableau de bord réessaie de lui-même après une mise en veille, une panne ou un redémarrage : environ une fois par minute après un délai dépassé ou une erreur de redirection, et toutes les cinq minutes après une erreur d’authentification. Une clé d’hôte modifiée n’est jamais réessayée. + +Une liaison en échec ne bascule pas silencieusement vers un fournisseur local. + +## Supprimer un Child + +Sélectionnez **Disconnect** pour Child et confirmez son alias. Home arrête le tunnel, révoque la clé de liaison de Child et supprime l’enregistrement enregistré. + +Si Home ne peut pas joindre Child pour exécuter la déconnexion, choisissez **Remove here only**. Cela supprime le tunnel, la clé et l’enregistrement locaux. Connectez-vous ensuite à Child et exécutez : + +```bash +ocx disconnect +``` + +Pour déconnecter une liaison initiée par Child, exécutez `ocx disconnect` sur Child. La commande déconnecte le tunnel client et révoque la liaison sur Home via SSH. Si cette révocation échoue, elle affiche : `Home revoke failed; run ocx link revoke --link-id on the home.` + +## Sécurité + +Child utilise les fournisseurs et les identifiants de fournisseur de l’ordinateur Home via la liaison. Home crée une clé distincte pour chaque Child ; la suppression de la liaison révoque cette clé. Comparez l’empreinte de l’hôte avant de confirmer afin de ne pas accepter par erreur une mauvaise machine ou une clé modifiée. Les sessions du tableau de bord émises depuis une identité Tailscale ne peuvent pas gérer les liaisons. Sur Child, la clé reste dans OpenCodex : les identifiants que Codex ou Claude Code envoient sur Child ne sont pas transmis à Home, et tout programme de Child qui atteint `127.0.0.1:` utilise Home sans clé, avec la même confiance locale qu’une installation autonome. Les pages web d’autres sites sont refusées. + +## Référence CLI + +```text +ocx link port [--json] +ocx link issue --alias --tunnel-port [--json] +ocx link status [--json] +ocx link revoke --link-id [--json] +``` + +## Guides associés + +- [Déploiement Remote Hub](/fr/guides/remote-hub/) +- [Remote Workspace](/fr/guides/remote-workspace/) diff --git a/docs-site/src/content/docs/fr/guides/remote-workspace.md b/docs-site/src/content/docs/fr/guides/remote-workspace.md new file mode 100644 index 0000000000..707c61bfa1 --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/remote-workspace.md @@ -0,0 +1,236 @@ +--- +title: Espace de travail distant +description: Conservez Codex, Claude Code, Pi et leurs connexions sur un même OCX Hub, tandis que des ordinateurs équipés seulement d'OCX fournissent l'espace de travail et l'environnement de compilation. +--- + +Pour les liaisons SSH entre machines, consultez [Liaison distante](/fr/guides/remote-link/). + +Remote Workspace permet à un OpenCodex Hub d'exécuter vos agents de programmation tandis qu'un +autre ordinateur fournit les fichiers du projet, les commandes, les tests et la puissance de +calcul. Un téléphone ou un troisième ordinateur peut piloter la session depuis le tableau de +bord du Hub. + +```text +Phone browser -> Computer 1 OCX Hub -> encrypted channel -> Computer 2 OCX Executor + Codex / Claude / Pi project and commands + logins and sessions no coding CLI login +``` + +L'Executor n'a besoin que d'OpenCodex. Il n'a besoin ni de Codex, ni de Claude Code, ni de Pi, +ni d'une connexion ChatGPT, ni d'une clé API de fournisseur. Il ouvre une connexion WebSocket +sortante vers le Hub : aucun port public ni transfert de port du routeur n'est nécessaire. + +:::caution[Base expérimentale] +Remote Workspace est facultatif et n'est pas déployé en production. Linux offre les outils de +fichiers et, sous conditions, l'exécution de commandes avec bubblewrap. Windows et macOS +n'offrent que les outils de fichiers : leurs assistants natifs officiels rejettent les requêtes +de sonde et de commande. Les commandes Windows restent non prises en charge jusqu'à ce qu'un +responsable vérifié du cycle de vie puisse conserver la capacité de nettoyage pendant une +annulation. L'absence de prise en charge des commandes ne provoque jamais leur exécution sur le +Hub. +::: + +## Configurer le Hub + +L'ordinateur 1 détient toutes les connexions aux agents de programmation et toutes les sessions +de modèle. Installez et connectez-y les agents voulus, puis démarrez OpenCodex comme Hub : + +```bash +ocx config set runtimeRole hub +OCX_REMOTE_WORKSPACE_ENABLED=1 ocx start +ocx gui +``` + +Définissez `OCX_REMOTE_WORKSPACE_ENABLED=1` sur le processus du Hub lui-même. Le définir +seulement pour une commande du tableau de bord n'active pas un service déjà en cours. Sans +activation explicite, le Hub renvoie un état désactivé, sans créer de clés d'espace de travail +ni sonder les environnements d'agents de programmation. + +Utilisez un déploiement HTTPS authentifié si vous ouvrez le tableau de bord depuis un téléphone +ou un autre ordinateur. Consultez [Déploiement Remote Hub](/fr/guides/remote-hub/) pour le modèle +pris en charge d'entrée de gestion et de Tailscale. Ne publiez pas un port de tableau de bord +local sans authentification. + +Remote Workspace avec Codex utilise les profils d'autorisation actuels d'App Server. Si la +configuration Codex sélectionnée sur le Hub définit encore `sandbox_mode` ou +`sandbox_workspace_write`, le tableau de bord signale Codex comme indisponible plutôt que de +le lancer avec une frontière affaiblie. Migrez ce profil Codex avant d'utiliser la fonction ; +ne configurez pas simultanément l'ancien bac à sable et un profil d'autorisation. + +## Associer un Executor + +1. Ouvrez **Remote Workspace** dans le tableau de bord du Hub. +2. Sélectionnez **Create pairing code**. +3. Sur l'ordinateur 2, placez-vous dans le répertoire du projet à exposer. +4. Copiez la commande **Linux / macOS terminal** ou **Windows PowerShell** générée pour cet + ordinateur. Elle associe le répertoire actuel et maintient + `ocx remote-workspace agent` connecté dans ce terminal. + +Le parcours manuel équivalent est : + +```bash +cd /path/to/project +printf '%s\n' 'ONE-TIME-CODE' | ocx remote-workspace pair 'https://your-hub.example' \ + --pairing-code-stdin --root "$PWD" +ocx remote-workspace agent +``` + +Sous Windows PowerShell, utilisez la commande affichée dans le tableau de bord. Sa forme +manuelle équivalente est : + +```powershell +$pairingCode = 'ONE-TIME-CODE' +$pairingCode | ocx remote-workspace pair 'https://your-hub.example' ` + --pairing-code-stdin --root (Get-Location).Path +if ($LASTEXITCODE -eq 0) { ocx remote-workspace agent } +``` + +L'exécutable OCX Bun actuel est ajouté automatiquement comme fichier unique en lecture seule +au bac à sable Linux. Si le projet a besoin d'une chaîne d'outils installée par l'utilisateur +hors des chemins système, associez-la explicitement sans exposer le reste du répertoire +personnel : + +```bash +printf '%s\n' 'ONE-TIME-CODE' | ocx remote-workspace pair 'https://your-hub.example' \ + --pairing-code-stdin --root "$PWD" \ + --toolchain-root "$HOME/.nvm/versions/node/v24/bin" +``` + +Le code source de l'assistant natif est fourni pour examen. Le compiler n'active pas les +commandes Windows ou macOS dans cette version. `--executor-helper` reste un sélecteur +d'assistant examiné ; la présence d'un binaire ou d'un chemin configuré ne prouve pas la prise +en charge des commandes. + +Le code à usage unique est lu depuis l'entrée standard, pas depuis les arguments de ligne de +commande. L'association crée une clé locale de signature d'appareil et un jeton porteur propre +à cet appareil. Le Hub ne conserve que son empreinte et ne reçoit jamais le chemin réel de +l'Executor. Arrêtez l'agent au premier plan avec Ctrl+C ; un nouveau lancement reconnecte le +même appareil. + +Vérifiez l'inscription locale sans afficher de secrets : + +```bash +ocx remote-workspace status +``` + +## Démarrer une session de programmation distante + +Dans le tableau de bord, choisissez : + +1. l'ordinateur connecté ; +2. un dossier d'espace de travail approuvé localement ; +3. Codex, Claude Code ou Pi depuis le Hub ; et +4. un mode d'accès. + +**Read only** est le mode par défaut ; il permet de lister les répertoires et de lire les +fichiers. L'option d'écriture apparaît sous le nom **Edit files and run commands** uniquement +si l'Executor a réussi la sonde du bac à sable de commandes ; sinon, elle s'appelle +**Edit files only**. Le tableau de bord montre deux emplacements distincts pour indiquer que +le modèle et la connexion restent sur le Hub, tandis que les opérations sur l'espace de travail +s'exécutent sur l'ordinateur sélectionné. + +Envoyez des invites depuis le tableau de bord du Hub sur l'ordinateur 1, l'ordinateur 3 ou un +téléphone. La session ne peut pas changer silencieusement d'ordinateur ou de dossier. Si +l'Executor se déconnecte, elle passe à l'état **Executor offline** et n'utilise jamais le +système de fichiers du Hub en repli. + +L'envoi d'une invite reçoit aussitôt un accusé d'acceptation ; le tableau de bord interroge la +session pour suivre sa progression et son achèvement. Si l'accusé se perd, le brouillon reste +visible avec un avis indiquant que l'envoi est incertain. Vérifiez la progression de la session +avant de renvoyer l'invite : le tableau de bord ne réessaie jamais automatiquement. + +**Stop** reste disponible pendant l'exécution d'une invite. Il interrompt le tour de l'agent +sur le Hub, annule une commande Executor active et empêche une réponse tardive de rouvrir la +session arrêtée. + +## Redémarrage et reconnexion + +Le Hub conserve des métadonnées de session limitées et un petit instantané des événements +récents. Après son redémarrage, une session inachevée attend son Executor d'origine. Une fois +cet appareil reconnecté, l'invite suivante reprend le fil Codex, la session Claude Code ou +l'identifiant de session Pi d'origine. + +Claude Code crée son historique durable après la première invite terminée. Si le Hub s'arrête +avant qu'une nouvelle session Claude ait terminé une invite, aucune conversation ne peut être +reprise : démarrez plutôt une nouvelle session. + +Une modification du manifeste de capacités n'affaiblit pas silencieusement une session +existante. Démarrez une nouvelle session si l'Executor perd le confinement des commandes ou si +ses outils disponibles changent. La révocation d'un ordinateur ferme sa connexion et arrête les +sessions qui y sont liées. + +## Frontières de sécurité + +- Les identifiants fournisseur et l'historique des agents de programmation restent sur le Hub. +- Les clés privées de l'Executor, son jeton d'appareil et ses vrais chemins racine restent dans + son état OCX accessible uniquement au propriétaire. +- Les échecs de code d'association sont limités par pair observé par le noyau sur chaque + écouteur. Dix codes erronés en dix minutes renvoient un `429` générique avec + `Retry-After` ; le Hub ne conserve que des empreintes limitées et expirantes de ces + identités source. Les utilisateurs de Tailscale Serve partagent la limite de boucle locale + de l'écouteur de gestion, car un appelant local direct pourrait usurper son en-tête + d'identité. +- Chaque session de travail utilise une négociation ECDH P-256 éphémère signée avec Ed25519 + et des messages AES-256-GCM ordonnés. +- Une connexion n'apparaît comme active qu'après l'accord des deux parties sur son manifeste + actuel de capacités. +- Une reconnexion peut retirer une capacité si le bac à sable local est indisponible, mais + n'ajoute jamais de capacité hors de l'autorisation enregistrée lors de l'association. +- Chaque requête est liée à un fil de modèle, un appareil, une racine, un mode d'accès et un + ensemble de capacités. +- Les chemins sont relatifs, canonisés et limités ; les sorties par lien symbolique, + jonction ou répertoire parent sont refusées. Les noms d'appareil Windows, les flux de + données alternatifs et les alias terminés par un point ou un espace sont interdits. +- Les opérations de l'Executor sont sérialisées ; les identités des fichiers ouverts sont + revérifiées et les empreintes d'écriture contrôlées juste avant le remplacement atomique. + Remplacer une racine approuvée impose de la réassocier, et les racines des chaînes d'outils + sont revalidées avant chaque commande. +- Les lectures et écritures refusent les fichiers liés par liens physiques. Avant toute + exécution de commande, OCX examine au plus 250 000 entrées de l'espace de travail et + désactive les commandes si une entrée autre qu'un répertoire possède plusieurs liens : un + bac à sable de chemins ne peut pas prouver que l'autre nom de cet inode se trouve dans la + racine approuvée. +- Sous Linux, les commandes passent par bubblewrap avec un seul espace de travail inscriptible, + un environnement effacé, des espaces de noms de processus privés, l'exécutable OCX Bun actuel + comme fichier unique en lecture seule, une sortie et une durée limitées, et un réseau + désactivé par défaut. Les tests de confinement dédiés exigent un environnement hébergé + explicitement configuré ; une suite générique verte ne prouve pas leur exécution. +- macOS n'annonce que les outils de fichiers. Un groupe de processus ne peut retenir un + descendant après un appel à `setsid()`, et importer un profil système Apple Seatbelt étendu + seulement pour lancer une commande exposerait une autorité sans rapport sur les services de + l'hôte. L'assistant natif rejette donc sa sonde et les commandes directes tant qu'OCX ne + dispose pas d'un responsable de confinement des descendants précis et révocable. +- Les requêtes de commandes natives Windows et macOS échouent de façon sûre. Les tests de + refus direct par l'assistant ne doivent pas être confondus avec des preuves de confinement + fonctionnel des commandes ; leur acceptation sous Windows reste à réaliser. +- L'assistant natif épinglé doit se trouver hors de tout espace de travail inscriptible + approuvé. OCX le vérifie avant d'annoncer la prise en charge des commandes et juste avant + chacune, afin que le code du projet ne puisse pas remplacer le binaire chargé d'appliquer + le bac à sable suivant. +- Arrêter une session annule une commande Executor active et nettoie le processus de modèle + du Hub ainsi que le pont d'outils en boucle locale. Windows arrête l'arborescence du + processus wrapper npm possédé, sans laisser son enfant Node actif ; Linux et macOS forcent + l'arrêt d'une CLI seulement si elle ignore la période d'arrêt gracieux. + +Le Hub voit volontairement les invites et les réponses du modèle, puisqu'il exécute l'agent. +Le chiffrement de bout en bout protège les charges RPC de l'Executor. Le Hub associé est +autorisé à choisir des racines approuvées via WSS authentifié ; il peut lire sa propre +conversation avec le modèle. + +## Périmètre actuel + +Remote Workspace ne copie ni ne synchronise les identifiants vers d'autres ordinateurs. Cette +fonction est distincte du routage des fournisseurs de Remote Hub et de tout futur produit de +calcul hébergé ou Super Sync. Une sortie en production exige encore un empaquetage signé de +l'assistant Windows, des preuves CI natives sur les binaires exacts, une revue indépendante +d'un mainteneur et un essai réel sur trois ordinateurs. + +## API d'acceptation des invites + +`POST /api/remote-workspace/sessions/:id/prompt` renvoie HTTP 202 avec l'instantané de session +acceptée. Son identifiant de session et sa séquence d'événements monotone identifient cet +instantané ; 202 ne signifie pas que le tour du modèle est terminé. Interrogez +`GET /api/remote-workspace/sessions` pour les événements ultérieurs et l'état final. La +reconnexion et la reprise de l'environnement restent occupées pendant ce tour. Si l'accusé +de réception se perd, l'acceptation reste incertaine : les clients doivent interroger la +session avant de décider d'envoyer de nouveau l'invite. diff --git a/docs-site/src/content/docs/fr/guides/response-inspection.md b/docs-site/src/content/docs/fr/guides/response-inspection.md new file mode 100644 index 0000000000..d4849c219f --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/response-inspection.md @@ -0,0 +1,40 @@ +--- +title: Inspection des réponses et réponses volumineuses +description: Interaction entre la conservation limitée des diagnostics, l'inspection des flux et la livraison des réponses. +--- + +OpenCodex limite la taille des diagnostics de réponse conservés sans appliquer +cette limite aux octets livrés au client. Les autres limites des fournisseurs, +des requêtes et du transport s'appliquent indépendamment. + +## Réponses JSON et erreurs ordinaires + +L'inspection JSON conserve au plus 32 MiB d'octets source. Si le corps dépasse +cette limite, la journalisation abandonne sa copie conservée et continue de +transmettre la réponse originale. Elle n'interprète pas un préfixe tronqué comme +des métadonnées fiables d'usage ou de modèle. L'usage déjà fourni par une autre +voie fiable est préservé ; un usage manquant n'est pas remplacé par un zéro +inventé. Les diagnostics d'erreur ordinaires non JSON ne conservent que les +premiers 8 KiB et passent par la logique de masquage existante. + +Le client reçoit les blocs au fil de sa lecture, sans attendre que l'inspection +diagnostique du corps entier s'achève. Un échec de lecture est enregistré comme +502 et une annulation comme 499 dans l'historique des requêtes ; ces résultats +diagnostiques ne réécrivent pas les en-têtes HTTP déjà envoyés. La journalisation +est finalisée une seule fois. + +## Réponses en flux + +L'inspection SSE native se met en pause lorsqu'elle prend trop d'avance sur la +consommation du client. Sa marge est de 32 MiB, plus le surcoût des blocs source +et de la prélecture native ; ce n'est ni une limite à la taille totale de la +réponse, ni un plafond sur toute la mémoire du processus. Une réponse plus +longue reste inspectée jusqu'à son événement de fin réel, y compris l'usage +final et l'état de continuation. + +Après la déconnexion du client, la vidange limitée existante peut encore +observer une fin tardive pendant au plus 15 secondes ou 32 MiB d'inspection +supplémentaire. Un arrêt forcé est différent : il abandonne les candidats +inachevés au lieu de les enregistrer comme réponses terminées. Le choix du +transport existant et les limites mémoire de WebSocket restent inchangés. Aucun +nouveau paramètre de configuration n'est nécessaire. diff --git a/docs-site/src/content/docs/fr/guides/sidecars.md b/docs-site/src/content/docs/fr/guides/sidecars.md index 367b98d9c3..1fcc91cfec 100644 --- a/docs-site/src/content/docs/fr/guides/sidecars.md +++ b/docs-site/src/content/docs/fr/guides/sidecars.md @@ -164,6 +164,16 @@ le délai d'attente et la limite précédemment choisis. les clés omises inchangées. `timeoutMs` utilise les limites entières de l'environnement d'exécution (1–2147483647 ms). +La carte du service auxiliaire de recherche web reprend la même forme de contrôle : la première +ligne du sélecteur de modèle est **Désactivé (Off)**. La désactivation arrête l'interception de +`web_search` par OpenCodex et l'intégration Codex écrit `web_search = "disabled"` dans +`~/.codex/config.toml`, car Codex continue sinon d'annoncer son propre outil hébergé +`web_search` natif, ce qu'il faut lorsqu'un serveur de recherche MCP doit être le seul chemin +de recherche. La réactivation supprime cette ligne et rétablit la ligne racine `web_search` +écrite par l'opérateur, enregistrée dans le journal Codex. L'écriture exige un +`~/.codex/config.toml` géré (`ocx sync`) ; la carte du tableau de bord vous avertit +lorsqu'elle n'a pas eu lieu et `ocx agent sidecar web --enabled off` indique si elle a réussi. + Vous pouvez toujours définir `enabled: false` dans `config.json` si vous préférez modifier le fichier directement. La recherche et la description d'images avec OAuth Anthropic réutilisent les identifiants Claude Code existants du magasin d'empreintes précédent. Testez néanmoins ce comportement avec le diff --git a/docs-site/src/content/docs/fr/guides/subagent-v1-default.md b/docs-site/src/content/docs/fr/guides/subagent-v1-default.md new file mode 100644 index 0000000000..cafcd8b6a3 --- /dev/null +++ b/docs-site/src/content/docs/fr/guides/subagent-v1-default.md @@ -0,0 +1,150 @@ +--- +title: Pourquoi v1 est la surface de sous-agent par défaut +description: Ce que bloque la limite des tâches chiffrées en v2, pourquoi OpenCodex livre désormais v1 et comment utiliser malgré tout v2. +--- + +OpenCodex s'installe avec la surface de sous-agent réglée sur **v1**. Les pages +Dashboard, Models et Subagents demandent toutes confirmation avant de passer à +**base** ou **v2** et renvoient vers cette page. La CLI ne demande rien. + +La raison est précise : en v2, une tâche confiée par un modèle ChatGPT natif à +un modèle routé est illisible pour ce dernier. C'est le cas de délégation le +plus courant — un parent GPT qui lance un enfant Grok, Claude ou GLM — et en +v2, il échoue systématiquement. + +## Ce que vous voyez alors + +La création de l'agent est rejetée au lieu de produire silencieusement une +tâche enfant vide : + +```json +{ + "error": { + "code": "unreadable_encrypted_agent_task", + "message": "Routed V2 worker task is encrypted for the native ChatGPT backend and cannot be read by the selected provider. Use plaintext V2 agent-message delivery or select a native ChatGPT model." + } +} +``` + +Le serveur renvoie HTTP 400 sans jamais renvoyer le texte chiffré. Cet échec +fermé est volontaire : transmettre une charge illisible donnerait à l'enfant +une consigne vide et conduirait à une réponse erronée présentée avec assurance. + +## Pourquoi cela se produit + +![Comparaison de la même délégation sur deux voies. En v1, un parent ChatGPT envoie une tâche en texte clair via OpenCodex ; elle franchit la frontière fournisseur et l'enfant routé la lit. En v2, le parent envoie encrypted_content créé par le backend ChatGPT ; OpenCodex ne peut pas le déchiffrer. La tâche s'arrête donc à la frontière fournisseur et la requête échoue avec unreadable_encrypted_agent_task.](../../../../assets/subagent-v2-encrypted-task.svg) + +En v1, le parent émet la tâche de l'enfant en texte clair. OpenCodex la lit et +l'achemine ; l'enfant routé reçoit une consigne exploitable. + +En v2, le parent émet la tâche sous forme de `encrypted_content`, produit par +le backend ChatGPT. La clé reste dans ce backend. OpenCodex ne l'a jamais eue : +le proxy ne peut donc rien déchiffrer ni réécrire. La valeur est réellement du +texte chiffré, et non du texte clair caché derrière un indicateur. La limite +est structurelle, pas un défaut de configuration, et aucun réglage du proxy +ne peut y remédier. + +Trois configurations restent possibles, ce qui éclaire la forme de l'échec : + +| Topologie | v1 | v2 | +| --- | --- | --- | +| Parent ChatGPT vers enfant routé | fonctionne | **échoue par défaut** ; voir le relais optionnel ci-dessous | +| Parent routé vers enfant routé | fonctionne | fonctionne | +| Parent ChatGPT vers enfant ChatGPT | fonctionne | fonctionne — le backend peut déchiffrer ce qu'il a produit | + +Le backend peut toujours lire son propre texte chiffré. Seul le franchissement +de la frontière échoue. + +## Est-ce corrigé en amont ? + +Pas encore, du moins pas pour la partie décisive. Le projet amont a fusionné +[openai/codex#35845](https://github.com/openai/codex/pull/35845), qui ajoute +la prise en charge des messages de collaboration en texte clair, mais du côté +de la *réception*. Il traite du texte clair déjà produit ; il ne fait pas en +sorte qu'un parent OpenAI le produise. + +L'envoi reste un problème ouvert : +[#36376](https://github.com/openai/codex/issues/36376), reproduit sur les CLI +0.146 à 0.151 sous Windows, macOS et Linux, et +[#37197](https://github.com/openai/codex/issues/37197), qui nomme directement +la pièce manquante : une règle de livraison côté envoi. Aucun des deux n'a +d'engagement d'un mainteneur ni de date prévue. + +OpenCodex a consigné la conséquence dans [#92](https://github.com/lidge-jun/opencodex/issues/92), +fermée sans suite prévue : rien dans ce dépôt ne peut la corriger. L'issue +renvoie donc aux travaux en amont, plutôt qu'à une tâche en attente ici. + +## Fonctionnement actuel des trois modes + +| Mode | Surface | Quand le choisir | +| --- | --- | --- | +| **v1** (par défaut) | Chaque modèle annonce les outils de création classiques avec espace de noms. Une création peut désigner directement un autre modèle. | Si vous déléguez entre fournisseurs. C'est le réglage livré par défaut. | +| **base** | Épinglages des modèles en amont : Sol et Terra utilisent v2, Luna utilise v1, les modèles non épinglés suivent le propre indicateur de Codex. | Si vous voulez la surface prévue par Codex pour chaque modèle et ne déléguez qu'au sein d'un fournisseur. | +| **v2** | Chaque modèle annonce les outils concurrents sans espace de noms. | Si vous voulez le nouveau modèle de sessions concurrentes et que parent et enfant restent du même côté de la frontière. | + +base figure en deuxième position, car ses épinglages placent Sol et Terra — +les deux modèles depuis lesquels on délègue le plus souvent — en v2. base +n'est pas un compromis pour ce problème : une création ChatGPT vers un modèle +routé s'y comporte comme en v2. + +## Si vous avez déjà choisi base ou v2 + +Rien n'a été modifié pour vous. La mise à niveau vers une version qui livre ce +réglage par défaut ne réécrit pas un paramètre existant ; le tableau de bord +affiche l'avis une fois et attend votre réponse. + +- **Continue** conserve le mode actuel et cesse de poser la question. +- **Switch to v1** active v1 et cesse de poser la question. + +Chaque réponse est enregistrée et l'avis ne revient pas. Si vous fermez +l'avis sans répondre, il réapparaît à la prochaine ouverture du tableau de bord. + +Les changements de mode s'appliquent aux **nouvelles** sessions Codex. Lancez +une nouvelle session après votre choix. Si un hôte App durable affiche encore +l'ancienne surface, exécutez `ocx sync` et redémarrez cette surface Codex. + +## Si vous voulez tout de même v2 + +Voici quatre possibilités, dans l'ordre où la plupart des utilisateurs +devraient les essayer : + +1. **Keep ChatGPT on v1.** Dans v2, l'option `keepNativeChatGptOnV1` laisse + Sol et Terra sur la surface v1 pour qu'ils puissent toujours lancer Grok ou + Claude, tandis que les parents routés utilisent v2. C'est la solution la + plus proche d'une combinaison des deux. +2. **Delegate within one provider.** Un parent routé lançant un enfant routé + transmet du texte clair en v2 et fonctionne normalement. +3. **Trust a direct key-auth Responses relay.** Un fournisseur explicitement + marqué `allowEncryptedV2AgentTasks: true` reçoit la charge opaque au lieu + de l'erreur 400. Réservez ce réglage aux destinations dont vous savez + qu'elles peuvent consommer cette charge. +4. **Enable `agentTaskRecovery`.** Fonction expérimentale désactivée par + défaut. Elle récupère via le backend ChatGPT les éléments chiffrés + illisibles `NEW_TASK`, `MESSAGE`, `FOLLOWUP_TASK` et `FINAL_ANSWER`, au + prix de quota, de latence et d'une dépendance à un comportement non + documenté ; la récupération combo reste limitée aux tours des enfants + lancés, et les fragments de jetons découpés restent non pris en charge. + +Consultez [Surface des sous-agents](/fr/guides/sub-agent-surface/) pour le +fonctionnement détaillé de chaque option et +[Configuration des agents](/fr/reference/configuration/agents/) pour les réglages. + +## Quand cette page deviendra inutile + +Lorsqu'une version en amont permettra à un parent ChatGPT natif d'émettre la +tâche d'un enfant routé en texte clair, la raison de ce réglage par défaut +disparaîtra. Le mode par défaut reviendra alors à base, la confirmation ne +s'affichera plus et cette page relèvera de l'historique plutôt que du conseil. + +## Changer de mode + +Dashboard, Models et Subagents proposent le même sélecteur v1/base/v2 et +demandent tous confirmation avant base ou v2. Depuis la CLI : + +```bash +ocx v2 status +ocx v2 mode v1 +``` + +La CLI ne demande pas de confirmation. C'est le même paramètre : choisissez-le +en tenant compte de ce que décrit cette page. diff --git a/docs-site/src/content/docs/fr/guides/web-dashboard.md b/docs-site/src/content/docs/fr/guides/web-dashboard.md index 1e4c90f506..159a972b2d 100644 --- a/docs-site/src/content/docs/fr/guides/web-dashboard.md +++ b/docs-site/src/content/docs/fr/guides/web-dashboard.md @@ -37,6 +37,29 @@ remplissage automatique. Le tableau de bord lui-même ne conserve le jeton qu'en dans `localStorage` ni dans `sessionStorage` ; son enregistrement dépend entièrement du navigateur ou du gestionnaire de mots de passe. +## Barre de résumé des quotas + +Une ligne de résumé en haut de chaque page, sauf la page Sécurité au démarrage, indique +l'utilisation actuelle des quotas de chaque fournisseur, par exemple +`OpenAI 31% | Claude 54% | xAI 12% | Google 8%`. Elle lit les mêmes rapports de quotas que l'espace +fournisseur (`GET /api/provider-quotas`, toutes les 60 secondes tant que l'onglet est visible) et ne +force jamais d'actualisation en amont. + +- Chaque étiquette affiche la fenêtre signalée prioritaire : d'abord hebdomadaire, puis mensuelle, + puis 5 heures, puis une fenêtre nommée par le fournisseur ou des crédits prépayés. +- Une étiquette passe en ambre à 70 % d'utilisation et en rouge à 90 %. +- Survolez une étiquette ou donnez-lui le focus au clavier pour voir toutes les fenêtres signalées + avec leur heure de réinitialisation et l'heure de la lecture. Sur un écran tactile, le premier + appui affiche ces détails. +- Cliquez sur une étiquette (ou appuyez une seconde fois) pour ouvrir l'onglet Comptes de ce + fournisseur dans Fournisseurs, où ses comptes ou clés API sont gérés. +- La barre reste toujours sur une seule ligne. Quand les étiquettes ne tiennent pas, faites-la + défiler horizontalement ou utilisez les boutons « et » à chaque extrémité. +- Les fournisseurs qui ne signalent aucune fenêtre de quota sont omis. La barre est masquée quand + aucun fournisseur n'en signale. +- Le bord droit indique quand le tableau de bord a lu les rapports pour la dernière fois. Il passe en + ambre lorsque la dernière lecture a échoué et que la lecture précédente est encore affichée. + ## Fonctions disponibles | Zone | Fonction | @@ -44,7 +67,7 @@ gestionnaire de mots de passe. | **Résumé du tableau de bord** | Mode multi-agent, état en ligne, version, durée de fonctionnement, nombre de fournisseurs, total de jetons sur 30 jours, fournisseurs actifs et modèles natifs/routés disponibles. | | **Délégation de sous-agent** | Choisissez un modèle natif ou routé et, facultativement, un effort de raisonnement partagés entre les consignes de délégation OpenCodex et l'option distincte de valeurs par défaut natives. Il ne s'agit pas d'un routeur par création de sous-agent côté proxy ; voir ci-dessous. | | **Services auxiliaires** | Choisissez le modèle et l'effort de recherche web, ainsi que le modèle de description visuelle. Les modifications s'appliquent à la requête suivante. | -| **Maintenance** | Resynchronisez le catalogue de modèles Codex, examinez les avertissements de contournement par une configuration locale au projet, recherchez la dernière version stable ou préliminaire et lancez une mise à jour avec redémarrage facultatif du proxy. | +| **Maintenance** | Resynchronisez le catalogue de modèles Codex, examinez les avertissements de contournement par une configuration locale au projet, recherchez la dernière version stable ou préliminaire et lancez une mise à jour avec redémarrage facultatif du proxy. Dans le shell de bureau, son entrée de mise à jour ouvre la page de mise à jour native au lieu d’exécuter la mise à jour du paquet. | | **Sécurité au démarrage** | Vérifiez si le routage Codex injecté résiste à un redémarrage, avec des états distincts pour le service et le lanceur intermédiaire, ainsi que les commandes de réparation exactes. | | **Zone de notification Windows** | Installez au niveau de l'utilisateur un contrôleur lancé à la connexion pour démarrer, arrêter ou redémarrer le proxy en un clic, ouvrir le tableau de bord et consulter l'état. Ce contrôleur n'est pas un service de redémarrage du proxy. | | **Démarrage automatique de Codex** | Autorisez un lanceur intermédiaire Codex déjà installé à exécuter `ocx ensure`. Ce commutateur n'installe ni lanceur ni service d'arrière-plan. | diff --git a/docs-site/src/content/docs/fr/reference/adapters.md b/docs-site/src/content/docs/fr/reference/adapters.md index 9535fe02f4..d45e12b48a 100644 --- a/docs-site/src/content/docs/fr/reference/adapters.md +++ b/docs-site/src/content/docs/fr/reference/adapters.md @@ -95,6 +95,7 @@ Avec l’authentification `key`, [`retryOn429`](/fr/reference/configuration/) s - Convertit les messages en blocs de contenu Anthropic (texte, image base64, `tool_use`, `thinking`). - **Calcul du raisonnement étendu :** Anthropic exige `max_tokens > thinking.budget_tokens`. L’adaptateur associe l’effort de raisonnement à un budget (minimal 1024 … max 32000), calcule ensuite une valeur sûre de `max_tokens` avec une marge pour la sortie et **supprime `temperature`/`top_p`** lorsque le raisonnement est activé, car Anthropic les interdit dans ce cas. +- **Affichage du raisonnement adaptatif :** les modèles à raisonnement adaptatif (Opus 4.7+, Sonnet 5, Fable) reçoivent `thinking.display: "summarized"`, de sorte qu’une longue réflexion arrive aux clients Chat et Responses sous forme de deltas de raisonnement au lieu de minutes de heartbeats. Une requête qui masque le résumé (`reasoning.summary: "none"`) conserve la valeur par défaut du fournisseur. - **Sortie structurée :** les requêtes Responses `text.format` et Chat Completions `response_format` dont le type est `type: "json_schema"` deviennent `output_config.format` dans Anthropic. Le format est fusionné avec une configuration de sortie de raisonnement adaptatif existante, tout en préservant un `output_config.effort` compatible. Les requêtes Anthropic Messages routées conservent ce même format lors de la traduction OAuth stockée. L’adaptateur reproduit le sous-ensemble de JSON Schema pris en charge par le SDK TypeScript d’Anthropic : les contraintes non prises en charge sont déplacées dans `description` à titre d’instructions pour le modèle, `oneOf` devient `anyOf` et les schémas d’objet reçoivent `additionalProperties: false`. Une racine `$ref` conserve le `$defs` adjacent afin que la référence locale reste résoluble. Les champs d’enveloppe OpenAI tels que le `name` du schéma, la `description` de l’enveloppe et `strict` ne font pas partie du protocole Anthropic. Le mode objet JSON sans schéma n’a pas d’équivalent Anthropic et n’est pas traduit. - Envoie toujours `anthropic-version: 2023-06-01`. Diffuse `content_block_delta` (`text_delta`, `thinking_delta`, le compatible `reasoning_delta`, `input_json_delta`). Le décodeur SSE conserve l’état des événements d’un fragment reçu à l’autre et accepte un événement terminal `message_stop` sans saut de ligne final. - Pour les tours Responses routés vers Anthropic avec des outils clients, une garde terminale bornée détecte le cas hautement probable où l’utilisateur a demandé une action, mais où Claude termine en affirmant l’avoir exécutée sans appeler d’outil. Elle effectue au plus une continuation interne ; les réponses normales, les demandes de précision, les tours qui utilisent un outil et les réponses incomplètes au niveau du transport ne sont pas relancés automatiquement. @@ -113,10 +114,19 @@ Avec l’authentification `key`, [`retryOn429`](/fr/reference/configuration/) s **Cibles :** le service Amazon CodeWhisperer Streaming `GenerateAssistantResponse` utilisé par Kiro (`https://runtime.{region}.kiro.dev/`). **Authentification :** jeton d’accès OAuth Kiro en Bearer, accompagné des métadonnées region/profile issues de l’identifiant Kiro. +Après l’admission d’une requête, le proxy lit en arrière-plan la liste des modèles du compte sur le +service de gestion régional. Les modèles observés complètent la liste fournie ; une réponse vide ou +inconnue conserve la dernière liste valide. Cette liste guide seulement la préférence entre comptes +admissibles : un identifiant de modèle inconnu est toujours transmis. Les limites d’entrée signalées +informent la fenêtre de contexte, avec la limite fournie comme repli si les preuves sont partielles. +Les comptes inactifs n’ont pas encore nécessairement de liste observée. + - Construit le `conversationState` de Kiro, mappe les outils Codex et leurs résultats, puis envoie les blocs d’image pris en charge par le protocole Kiro. - Décode `application/vnd.amazon.eventstream`, reconstruit les événements de texte, de raisonnement et d’outil, détecte les données JSON d’outil tronquées et estime l’utilisation, car le service en amont ne renvoie aucun nombre de jetons. -- Utilise à l’identique le `baseUrl` configuré lorsqu’il est personnalisé. Une URL canonique `runtime.{region}.kiro.dev` suit la région d’API de l’identifiant importé ; seule cette forme canonique peut faire l’objet d’un unique repli borné vers `q.{region}.amazonaws.com` après un échec de point de terminaison, de signature, de DNS ou de connexion. -- Gère la récupération après réinitialisation de connexion lorsqu’un rejeu est sûr, cet unique repli de point de terminaison admissible, une actualisation OAuth suivie d’un rejeu après une réponse HTTP 401, ainsi qu’une récupération bornée pour les réponses Kiro 429 transitoires. Un délai de récupération partagé et une seule sonde après ce délai empêchent les requêtes concurrentes d’épuiser des budgets de nouvelle tentative indépendants ; les dépassements fermes de quota et les erreurs de service ordinaires ne sont pas rejoués. + Les jetons restent estimés, mais les crédits de `meteringEvent` sont mesurés dans `providerCredits` : la dernière valeur d'une réponse est retenue, puis les envois physiques facturés séparément sont additionnés. Aucun crédit n'est déduit des jetons. +- Utilise à l’identique le `baseUrl` configuré lorsqu’il est personnalisé. Une URL canonique `runtime.{region}.kiro.dev` suit la région d’API de l’identifiant importé ; seule cette forme canonique peut faire l’objet d’un unique repli borné vers `q.{region}.amazonaws.com` après un échec de point de terminaison, de signature, de DNS ou de connexion, ou une réponse HTTP 502/503/504 reçue avant toute sortie. +- Gère la récupération après réinitialisation de connexion lorsqu’un rejeu est sûr, cet unique repli de point de terminaison admissible, une actualisation OAuth suivie d’un rejeu après une réponse HTTP 401, ainsi qu’une récupération bornée pour les réponses Kiro 429 transitoires. Un délai de récupération partagé et une seule sonde après ce délai empêchent les requêtes concurrentes d’épuiser des budgets de nouvelle tentative indépendants ; un quota épuisé n’est pas réessayé sur le même compte, et les autres erreurs de service ne sont pas rejouées. Tous les envois Kiro utilisent la sortie réseau configurée ; un délai d’en-tête dépassé renvoie 504, l’annulation du client arrête la requête et les erreurs HTTP 5xx finales affichent un texte public fixe. +- Avec deux comptes enregistrés, un refus de débit refroidit brièvement le compte concerné. Un refus mensuel confirmé (HTTP 400 ou 429) l’exclut jusqu’à la réinitialisation observée ou l’expiration des données ; une suspension confirmée (HTTP 403) le met temporairement en quarantaine, mais un 403 ordinaire ne déclenche pas de rotation. La rotation après refus reste active lorsque la préférence proactive est désactivée. Le choix d’un autre compte avant le premier envoi exige une préférence proactive, où le réglage du fournisseur prime sur le réglage global. Un tour terminé par le même compte efface un ancien verdict d’épuisement. - Son analyseur hors flux consomme le même flux d’événements pour la boucle de recherche Web. ### Sémantique d’achèvement @@ -166,6 +176,7 @@ conserve les instructions de réflexion bornées existantes, car ce niveau natif **Authentification :** `key` au moyen de l’en-tête `api-key` (et non Bearer). - Délègue la construction de la requête au relais Responses, vérifie que `baseUrl` ne contient aucun espace réservé de modèle non résolu et remplace `Authorization` par `api-key`. L’URL configurée cible directement l’API Responses v1 d’Azure ; l’adaptateur n’ajoute donc pas `api-version`. +- Partage la récupération Responses pour l’état de raisonnement produit par un autre fournisseur : après un `400 invalid_encrypted_content`, il renvoie la requête une seule fois sans cet état (contenu chiffré et identifiant `rs_…` de l’élément de raisonnement). ## Utilitaires d’image (`image.ts`) diff --git a/docs-site/src/content/docs/fr/reference/cli.md b/docs-site/src/content/docs/fr/reference/cli.md index c736d44c0a..ff64a5510f 100644 --- a/docs-site/src/content/docs/fr/reference/cli.md +++ b/docs-site/src/content/docs/fr/reference/cli.md @@ -23,6 +23,12 @@ Pour observer une installation Windows x64, consultez [`attest`](/fr/reference/c L’affichage d’une liste ou d’un état est l’action par défaut lorsqu’il n’y a aucune ambiguïté. Utilisez `--json` pour obtenir des instantanés structurés et `ocx observe logs --follow --jsonl` pour suivre un flux de journaux de requêtes. Le thème, la langue, la navigation et les autres états purement visuels du navigateur n’ont pas d’équivalent dans la CLI. La configuration de Cloudflare Tunnel ne fait pas partie de cet ensemble de commandes. +## Plafond des sondes de disponibilité + +`ocx health`, `ocx status`, `ocx account *`, `ocx login codex` et `ocx ready` trouvent le proxy en cours d'exécution grâce à une courte sonde : 750 ms par tentative par défaut, 1500 ms avec nouvelles tentatives pour les décisions d'arrêt et de démarrage. Si une couche de sécurité (filtre de contenu, extension réseau de type EDR) ajoute un coût fixe à chaque connexion loopback, ces plafonds peuvent expirer avant qu'un proxy sain réponde. + +Définissez `OCX_PROBE_TIMEOUT_MS` pour relever les plafonds, par exemple `OCX_PROBE_TIMEOUT_MS=5000 ocx status`. La valeur est un nombre entier de millisecondes entre 1 et 30000. Elle ne peut que relever : le défaut de 750 ms et les budgets d'arrêt/démarrage de 1500 ms gardent leur plancher, donc `1000` n'allonge que la sonde par défaut. Une valeur absente, vide, fractionnaire, négative, nulle ou supérieure est ignorée. + ## Codes de sortie et confirmation Une commande réussie renvoie le code 0. Une syntaxe non valide, une commande ou une ressource inconnue, l’échec d’une opération d’API ou l’indisponibilité d’un service requis produit un code non nul. Plus précisément, `ocx health` renvoie 0 uniquement lorsque le proxy est sain, et 1 dans le cas contraire ; cette commande peut donc servir de sonde de service. Les scripts doivent tester le code de sortie plutôt que d’analyser le texte destiné aux utilisateurs. diff --git a/docs-site/src/content/docs/fr/reference/cli/agents.md b/docs-site/src/content/docs/fr/reference/cli/agents.md index 75e2777cec..f9db378b53 100644 --- a/docs-site/src/content/docs/fr/reference/cli/agents.md +++ b/docs-site/src/content/docs/fr/reference/cli/agents.md @@ -15,8 +15,18 @@ les modes de surface, la délégation, l'effort et le comportement de repli s'em ```bash ocx agent subagents set ark/model-a,openai/gpt-5.5 +ocx agent sidecar web --enabled off ``` +`--enabled off` est le même interrupteur que la ligne **Désactivé (Off)** du tableau de bord : +OpenCodex cesse d'exécuter le service auxiliaire et l'intégration Codex écrit +`web_search = "disabled"` dans `~/.codex/config.toml`, ce qui permet à un serveur de +recherche MCP d'être le seul chemin de recherche. `--enabled on` supprime à nouveau cette ligne. +Lorsque l'enregistrement déplace réellement l'interrupteur, la commande signale l'écriture côté Codex +qu'elle a déclenchée (`codexWebSearch` avec `--json`, une ligne `Codex config:` +sinon) et renvoie vers `ocx sync` quand elle n'a pas pu avoir lieu. L'option fonctionne aussi +pour `vision`. + ### `ocx v2 |threads |mode-hint >` Gérez l'indicateur de fonctionnalité Codex `multi_agent_v2` et le mode surface multi-agents à trois états. @@ -166,7 +176,7 @@ Gérez et appliquez la clôture du modèle Grok Build. ## Exportation de la configuration client -### `ocx export --client ` +### `ocx export --client ` Imprimez une configuration client connectée au proxy en cours d'exécution. La commande sérialise le bloc fournisseur `opencodex` — URL de base, liste de modèles et référence d’identifiant du client @@ -177,7 +187,7 @@ les modèles Codex peuvent actuellement voir. | Option | Actions | | --- | --- | -| `--client ` | Requis. Sélectionne le dialecte de configuration client. | +| `--client ` | Requis. Sélectionne le dialecte de configuration client. | | `--json` | Imprimez le document généré en tant que JSON sur la sortie standard pour les scripts. Il s'agit de JSON même lorsque le format natif du client sélectionné est YAML, TOML ou JSON5. | | `--out ` | Écrivez le format de configuration natif du client dans ``. Refuse de remplacer un fichier existant. | | `--force` | Autoriser `--out` à remplacer un fichier existant. | @@ -210,6 +220,8 @@ propres valeurs par défaut à ces lignes. | `aside` | `~/.aside/u//models.json` pour le compte que le fichier `accounts.json` d'Aside désigne comme courant ; un manifeste illisible est refusé plutôt que de retomber sur un compte | `aside-models.json` | aucun — espace réservé de bouclage | | `raycast` | `~/.config/raycast/ai/providers.yaml`, sur macOS comme sur Windows (Raycast n'honore pas `XDG_CONFIG_HOME`) | `raycast-providers.yaml` | aucun — bouclage uniquement, aucune entrée `api_keys` n'est écrite | | `omo` | `~/.omo/agent/models.json` (`OMO_CODING_AGENT_DIR`, puis `SENPI_CODING_AGENT_DIR`, puis `PI_CODING_AGENT_DIR` l'emportent dans cet ordre une fois définis ; une valeur relative est refusée) | `omo-models.json` | aucun — espace réservé de bouclage | +| `kilo` | premier fichier existant parmi `kilo.jsonc`, `kilo.json`, `opencode.jsonc`, `opencode.json` ou `config.json` sous `~/.config/kilo` (`XDG_CONFIG_HOME` déplace ce répertoire) ; utilise `kilo.jsonc` si aucun n'existe | `kilo.jsonc` | `OPENCODEX_KILO_API_KEY` | +| `droid` | `~/.factory/settings.json` (`%USERPROFILE%\.factory\settings.json` on Windows) | `factory-settings.json` | boucle locale uniquement ; aucune variable d’environnement | L'exportation Raycast est un document `providers.yaml` autonome contenant un seul élément `id: opencodex` dans la séquence `providers` : `name: OpenCodex`, l'URL de base `/v1` du proxy et chaque modèle routé avec diff --git a/docs-site/src/content/docs/fr/reference/cli/lifecycle.md b/docs-site/src/content/docs/fr/reference/cli/lifecycle.md index f59ebe9916..59d5376606 100644 --- a/docs-site/src/content/docs/fr/reference/cli/lifecycle.md +++ b/docs-site/src/content/docs/fr/reference/cli/lifecycle.md @@ -15,7 +15,7 @@ Assistant de configuration interactif (`setup` est un alias de `init`). Il deman ### `ocx start [--port ] [--socks5 [host:port] | --socks5-off]` -Démarre le serveur proxy, de préférence sur le port `10100`. La commande écrit l’état du PID et du port d’exécution, et refuse de démarrer une deuxième instance active. Lorsque le port préféré est occupé, `start` interroge le processus qui l’occupe puis s’arrête dans tous les cas : elle refuse de démarrer si un processus opencodex y répond et signale sinon que le processus est inconnu. Elle ne déplace jamais l’écouteur vers un autre port d’elle-même, car cela laisserait le premier proxy en cours d’exécution et redirigerait Codex vers le second. Un autre `--port` explicite est également refusé avec le même `OPENCODEX_HOME`, car les modes d’observation et de plafond écrivent tous deux dans le même journal de dépenses. Utilisez un `OPENCODEX_HOME` distinct pour une instance sœur indépendante ; `port: 0` ne sépare que l’attribution du port, pas l’état. Au démarrage, elle synchronise dans le catalogue Codex les modèles de chaque fournisseur. À l’arrêt, elle rétablit le fonctionnement natif de Codex, sauf si le proxy a été lancé comme service géré (`OCX_SERVICE=1`). +Démarre le serveur proxy, de préférence sur le port `10100`. La commande écrit l’état du PID et du port d’exécution, et refuse de démarrer une deuxième instance active. Lorsque le port préféré est occupé, `start` interroge le processus qui l’occupe puis s’arrête dans tous les cas : elle refuse de démarrer si un processus opencodex y répond et signale sinon que le processus est inconnu. Elle ne déplace jamais l’écouteur vers un autre port d’elle-même, car cela laisserait le premier proxy en cours d’exécution et redirigerait Codex vers le second. Un autre `--port` explicite est également refusé avec le même `OPENCODEX_HOME`, car les modes d’observation et de plafond écrivent tous deux dans le même journal de dépenses. Utilisez un `OPENCODEX_HOME` distinct pour une instance sœur indépendante ; `port: 0` ne sépare que l’attribution du port, pas l’état. Au démarrage, elle synchronise dans le catalogue Codex les modèles de chaque fournisseur. À l’arrêt, elle rétablit le fonctionnement natif de Codex, sauf si le proxy a été lancé comme service géré (`OCX_SERVICE=1`). Une instance sœur démarrée à côté d’un proxy déjà actif ne fait ni l’un ni l’autre, même lorsqu’elle est arrêtée avec `ocx stop` ou par un signal : elle ne sert que les requêtes directes sur son propre port, et Codex, Grok et Claude restent dirigés vers le proxy qui tournait déjà. `--socks5` (par défaut `127.0.0.1:10808`) enregistre l’URL SOCKS5 dans `config.proxy` et achemine les requêtes HTTP(S) sortantes dans un véritable tunnel SOCKS5. `--socks5-off` supprime uniquement @@ -200,6 +200,31 @@ aux contrôles de santé en cas de contention CPU : la zone de notification affi Après la mise à jour, exécutez `ocx service repair` pour migrer cette priorité enregistrée et redémarrer le service. Une confirmation UAC peut être nécessaire. Une priorité déjà normale ou haute ne déclenche pas, à elle seule, de réenregistrement. +Sous Linux, l’unité systemd invoque le premier fichier `ocx` ordinaire et exécutable trouvé dans `PATH` +au moment de l’installation, plutôt que les chemins Bun et CLI à l’intérieur de l’arborescence du paquet +installé. Les gestionnaires de versions comme **mise** et **asdf** installent dans un répertoire +versionné et suppriment l’ancien lors d’une mise à niveau ; leur shim stable permet à l’unité de +continuer à résoudre. Les checkouts de source sans lanceur `ocx` conservent la forme directe Bun + CLI. +Un `OPENCODEX_BUN_PATH` de confiance choisi avant le démarrage de Bun est conservé à travers le shim ; +les chemins du Bun embarqué dans le paquet sont redécouverts après les mises à niveau. + +Sous macOS, launchd utilise à la place les chemins Bun et CLI propres au paquet choisis lors de +l’installation ou de la réparation. Cela empêche un shim PATH mutable de recevoir le jeton d’API du +service et l’environnement de proxy configuré lors d’un redémarrage ultérieur. Après la mise à niveau +d’une installation gérée par un gestionnaire de versions, exécutez `ocx service repair` pour +actualiser ces chemins avant de redémarrer le service. + +Les définitions installées avant ce changement portent encore les anciens chemins versionnés et ne +peuvent pas migrer d’elles-mêmes — une fois l’ancien exécutable supprimé, aucun code opencodex ne +s’exécute pour le réparer. Exécutez `ocx service repair` une fois après la mise à niveau. Les +démarrages du service Linux suivent alors le lanceur ; la réparation macOS écrit les nouveaux chemins +du paquet dans la définition launchd. Un proxy déjà en cours d’exécution n’est pas remplacé par une +mise à niveau externe : lorsque la CLI installée est plus récente que le proxy en cours, exécutez +`ocx service restart` pour que la nouvelle version serve. Sous macOS, `repair` ne suffit pas dans ce +cas : la définition n’a pas changé, et une réparation qui ne change rien ne recharge rien. Si c’est le +proxy qui est plus récent, vérifiez l’installation de la CLI et le `PATH` comme décrit sous +[`ocx status`](#ocx-status---json). + | Sous-commande | Action | | --- | --- | | aucune | Installe et démarre le service s’il est absent ; sinon, applique `repair` au service existant. Une définition Task Scheduler Windows saine est réutilisée ; une définition obsolète peut être réenregistrée et nécessiter une élévation. | @@ -304,6 +329,7 @@ Utilisez `ocx service` pour maintenir un proxy d’arrière-plan toujours actif, Installe et contrôle l’icône OpenCodex dans la zone de notification Windows. Elle démarre à l’ouverture de session et fournit des commandes du proxy accessibles en un clic. `start` et `stop` contrôlent uniquement l’icône ; utilisez son menu pour contrôler le proxy. `--no-start` s’applique à `install` et installe l’icône sans la lancer immédiatement. Obsolète : l’application OpenCodex fournit la zone de notification sous Windows, macOS et Linux ; `ocx tray` reste disponible pour les installations sans l’application de bureau. +Lorsqu'une version plus récente du paquet est connue, la zone de notification ajoute un point bleu à l'icône en ligne, d'avertissement ou hors ligne et affiche **Update available**. Elle vérifie le badge mis en cache localement environ une fois par minute ; les résultats obsolètes ou indisponibles retirent le point. L'élément ouvre le tableau de bord, où vous pouvez lancer la mise à jour du paquet. L'installation automatique est désactivée. ## Tableau de bord @@ -317,6 +343,10 @@ Ouvre le [tableau de bord Web](/fr/guides/web-dashboard/) à l’adresse `http:/ ### `ocx update [--tag latest|preview]` +Lorsque OpenCodex est installé avec mise, cette commande échoue avant d'arrêter le proxy ou de modifier les fichiers du paquet et affiche `mise upgrade ` avec l'alias mise local vérifié. La vérification des mises à jour reste disponible et signale une gestion externe. Des métadonnées de propriété mise illisibles ou incohérentes bloquent aussi toute modification sans deviner le nom de l'outil, et `--tag preview` ne change jamais la sélection configurée dans mise. + +Sous Linux, un service en arrière-plan dont le lanceur enregistré est le lanceur de paquet de mise (`/latest/node_modules/.bin/ocx`, et non un shim mise) suit `mise upgrade` tout seul : une dizaine de secondes après la stabilisation de la nouvelle version, il draine les requêtes actives et redémarre sur celle-ci, et il se rétablit de la même façon si mise supprime plus tard la version qu'il exécutait. Sous macOS, pour un service installé via un shim mise et pour un proxy au premier plan, redémarrez-le vous-même après la mise à jour (sous macOS, d'abord `ocx service repair`). + Met à jour opencodex depuis npm. Les installations stables utilisent `@latest` ; les préversions restent sur `@preview`, sauf si vous indiquez `--tag latest|preview`. La commande détecte un dépôt de sources et vous invite alors à exécuter `git pull && bun install`. Elle ne fait rien si la version la plus récente correspondant à cette balise est déjà installée. Avant tout arrêt, les installations npm effectuent sous Unix un contrôle borné de la propriété et de l’accès au cache. Les liens symboliques imbriqués sont examinés avec `lstat`, sans être suivis ; Windows ignore explicitement ce contrôle propre à Unix. En cas d’échec, l’opération s’interrompt tandis que l’icône et le proxy fonctionnent encore. Le proxy actif est ensuite arrêté avant le remplacement des fichiers. Un service installé est reconstruit et redémarré automatiquement ; pour une installation au premier plan, la commande indique `ocx start` comme étape suivante. Avant leur conservation, les enregistrements de mise à jour du tableau de bord masquent les chemins de profil et de cache ainsi que les valeurs UID/GID. diff --git a/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md b/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md index b6ff94de9f..91b7433099 100644 --- a/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md @@ -105,24 +105,36 @@ Répertoriez et changez de compte de fournisseur et de pools de clés API via le la surface est : ```text -Usage: ocx account ... +Usage: ocx account ... list [provider] Codex account pool, OAuth accounts and API keys (identifiers shown masked as the API returns them). +history openai [--limit <1-200>] Recent routing decisions for one Codex pool account. current Show the active account or key. -use Switch the active credential; 'main' selects the Codex App login. +use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection unless an account carries that id. +clear Clear the manual Codex account selection unconditionally. refresh Force-refresh Codex or provider quota reports. auto-switch Control the Codex pool threshold. -priority [first|earlier|normal|later|last|-100..100|reset] Selection order; omit the value to read it. -remove --yes Remove a stored account or key after an existence check. +alias Set or clear an account's display name; '-' clears it. +pause Hold an account out of automatic selection. +resume Return a paused account to automatic selection. +pause-exhausted Pause every account whose quota is spent. +clear-cooldown Drop a cooldown the proxy set after an upstream failure. +strategy [] Stratégie du pool ; least-loaded est réservé à Kiro. +sticky [<1-100>] Requests a bound thread keeps on one account; omit the value to read it. +priority [first|earlier|normal|later|last|-100..100|reset] Selection order; omit the value to read it. +remove --yes Remove a stored account or key after an existence check. add-key [--label