diff --git a/README.md b/README.md index 907a481db91..ec815156577 100644 --- a/README.md +++ b/README.md @@ -363,12 +363,17 @@ network failures, or invalid decisions fail open to the first currently eligible cancellation still cancels the request. Automated tests use a mocked TypeSafe endpoint and do not validate a live JEV account. +A JEV Combo can instead ask a self-hosted decision model, such as Ollama's keyless `tev1`: add a +`jev-decision` provider whose `baseUrl` is the full `/v1/systemone` endpoint and set the Combo's +`decisionProvider` to it. TypeSafe credentials are never sent there. Details: +[System One-compatible server](https://opencodex.me/guides/combos/#system-one-compatible-server). + ## Providers & adapters OpenAI (ChatGPT login or API key), Anthropic, Google Gemini, xAI, Kimi, Azure OpenAI, Ollama (local + Cloud), Cursor (experimental), and every OpenAI-compatible endpoint — plus DeepSeek, -Groq, OpenRouter, Together, Fireworks, Cerebras, Mistral, Hugging Face, NVIDIA NIM, MiniMax, +Groq, OpenRouter, OpenGateway, Together, Fireworks, Cerebras, Mistral, Hugging Face, NVIDIA NIM, MiniMax, Qwen Cloud, Qoder Global and CN (official PAT + CLI), SiliconFlow, and more. Full list: `ocx init` or the [provider docs](https://opencodex.me/guides/providers/). diff --git a/bin/ocx.mjs b/bin/ocx.mjs index 406748d8e1e..2018c0bd0d0 100755 --- a/bin/ocx.mjs +++ b/bin/ocx.mjs @@ -47,6 +47,7 @@ import { checkRegistryPackageIntegrity } from "../src/update/registry-integrity. import { hasPendingTeardownIn } from "../src/config/pending-teardown-names.mjs"; import { npmCachePreflightFailureMessage, + resolveNpmCachePath, runNpmCachePreflight, } from "../src/update/npm-cache-preflight.mjs"; import { handoffWindowsTrayForUpdate, planWindowsTrayUpdate } from "../src/update/tray-update-plan.mjs"; @@ -271,12 +272,21 @@ function runPackageManagerSelfUpdate(manager) { console.log(`Verified ${PKG}@${latest} integrity metadata ${integrity.integrity.slice(0, 24)}…`); } + // The cache root is resolved once, with the environment staging uses, then checked and pinned: + // the stage installs with exactly the root this pre-flight inspected (#6288). + let npmCachePath; if (manager === "npm") { - const cachePreflight = runNpmCachePreflight(); + const npmCache = resolveNpmCachePath({ env: unprivilegedOwnershipMutationEnvironment(process.env) }); + // Windows skipped this gate before #6288: an unresolvable npm cache path keeps that behavior + // there (no check, no pin) and only a confirmed broken root aborts the update. + const cachePreflight = npmCache.ok + ? runNpmCachePreflight({ cachePath: npmCache.path }) + : process.platform === "win32" ? { ok: true, reason: "windows_skip" } : npmCache; if (!cachePreflight.ok) { console.error(`opencodex: ${npmCachePreflightFailureMessage(cachePreflight.reason)}. Aborting before stopping the proxy.`); process.exit(1); } + npmCachePath = npmCache.path; } // Remember whether a background service manages the proxy BEFORE stopping — `ocx stop` @@ -737,6 +747,7 @@ function runPackageManagerSelfUpdate(manager) { pkgName: PKG, targetVersion: latest || undefined, tag, + cachePath: npmCachePath, runNpm: (args) => { const invocation = npmInvocation(args); if (!invocation) return { status: 1 }; diff --git a/desktop/src-tauri/Cargo.lock b/desktop/src-tauri/Cargo.lock index b5b02e5d216..f2bcecdab82 100644 --- a/desktop/src-tauri/Cargo.lock +++ b/desktop/src-tauri/Cargo.lock @@ -2645,7 +2645,7 @@ dependencies = [ [[package]] name = "opencodex-desktop" -version = "2.75.0-preview.20261001" +version = "2.76.0-preview.20261003" dependencies = [ "base64 0.22.1", "dbus", diff --git a/desktop/src-tauri/Cargo.toml b/desktop/src-tauri/Cargo.toml index 9354e8fb2dd..29d8bc8bf23 100644 --- a/desktop/src-tauri/Cargo.toml +++ b/desktop/src-tauri/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "opencodex-desktop" -version = "2.75.0-preview.20261001" +version = "2.76.0-preview.20261003" description = "OpenCodex desktop shell" authors = ["OpenCodex contributors"] license = "MIT" diff --git a/desktop/src-tauri/src/provider_icons.rs b/desktop/src-tauri/src/provider_icons.rs index 750bd5cb7b1..045d9e8f16f 100644 --- a/desktop/src-tauri/src/provider_icons.rs +++ b/desktop/src-tauri/src/provider_icons.rs @@ -67,6 +67,7 @@ const ALIASES: &[(&str, &str)] = &[ ("opencode-go", "opencode.svg"), ("opencode-zen", "opencode.svg"), ("openrouter", "openrouter-color.svg"), + ("opengateway", "opengateway.svg"), ("opper", "opper.svg"), ("qianfan", "qianfan-color.svg"), ("qoder", "qoder.svg"), @@ -128,6 +129,7 @@ fn paint(file: &str) -> &'static str { | "novita.svg" | "ollama-color.svg" | "opencode.svg" + | "opengateway.svg" | "opper.svg" | "packycode.svg" | "siliconflow.svg" @@ -189,6 +191,7 @@ fn svg(file: &str) -> Option<&'static str> { "openai.svg" => svg!("openai.svg"), "opencode.svg" => svg!("opencode.svg"), "openrouter-color.svg" => svg!("openrouter-color.svg"), + "opengateway.svg" => svg!("opengateway.svg"), "opper.svg" => svg!("opper.svg"), "orcarouter.svg" => svg!("orcarouter.svg"), "packycode.svg" => svg!("packycode.svg"), diff --git a/desktop/src-tauri/src/startup.rs b/desktop/src-tauri/src/startup.rs index 85c71455264..0a8e5572d7e 100644 --- a/desktop/src-tauri/src/startup.rs +++ b/desktop/src-tauri/src/startup.rs @@ -526,10 +526,14 @@ impl Startup { /// Consume the decision and publish the extended ceiling in the same critical section, so /// the guard's next expiry check sees either a pending/answered prompt or the new deadline, /// never the gap between them. - fn resolve_consent(&self, deadline: Instant) { + fn resolve_consent(&self, mut deadline: Instant, approved: bool) -> Instant { + if approved && cfg!(target_os = "windows") { + deadline += Duration::from_secs(60); + } let mut live = self.live(); live.deadline = deadline; live.consent = ConsentState::Idle; + deadline } fn set_deadline(&self, deadline: Instant) { @@ -933,7 +937,7 @@ async fn run(app: &AppHandle, started: Instant) { deadline += asked.elapsed(); // The extension and the clear are one critical section: the guard sees // either a prompt still pending or the moved ceiling, never the gap. - startup.resolve_consent(deadline); + deadline = startup.resolve_consent(deadline, approved); if !approved { attach_as_guest( app, @@ -2482,13 +2486,36 @@ mod tests { Expiry::Blocked )); // Once the run publishes the moved ceiling the guard waits on it instead of firing. - startup.resolve_consent(tokio::time::Instant::now() + Duration::from_secs(60)); + startup.resolve_consent(tokio::time::Instant::now() + Duration::from_secs(60), false); assert!(matches!( startup.expire_run(tokio::time::Instant::now(), 1, "expired".to_owned()), Expiry::Waiting(_) )); } + #[test] + fn approved_windows_takeover_has_one_bounded_extended_deadline() { + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + let deadline = Instant::now() - Duration::from_secs(5); + assert_eq!(startup.resolve_consent(deadline, false), deadline); + let extended = startup.resolve_consent(deadline, true); + if cfg!(target_os = "windows") { + assert_eq!(extended - deadline, Duration::from_secs(60)); + assert!(matches!( + startup.expire_run(deadline - DEADLINE, 1, "expired".to_owned()), + Expiry::Waiting(_) + )); + } else { + assert_eq!(extended, deadline); + } + startup.resolve_consent(Instant::now() - Duration::from_secs(95), true); + assert!(matches!( + startup.expire_run(deadline - DEADLINE, 1, "expired".to_owned()), + Expiry::Fired(_) + )); + } + #[test] fn a_terminal_run_posts_no_prompt() { // The other half of the race: the failure already landed, so the ask path must not diff --git a/desktop/src-tauri/tauri.conf.json b/desktop/src-tauri/tauri.conf.json index 9d9354015bc..efa74d1b5df 100644 --- a/desktop/src-tauri/tauri.conf.json +++ b/desktop/src-tauri/tauri.conf.json @@ -1,7 +1,7 @@ { "$schema": "https://schema.tauri.app/config/2", "productName": "OpenCodex", - "version": "2.75.0-preview.20261001", + "version": "2.76.0-preview.20261003", "identifier": "com.opencodex.desktop", "build": { "frontendDist": "../ui", diff --git a/devlog/_fin/261001_omo_lazycodex_carry/000_plan.md b/devlog/_fin/261001_omo_lazycodex_carry/000_plan.md new file mode 100644 index 00000000000..c75a98d7e01 --- /dev/null +++ b/devlog/_fin/261001_omo_lazycodex_carry/000_plan.md @@ -0,0 +1,62 @@ +# 261001 omo (Codex / LazyCodex) carry series + +LilMGenius opened three stacked PRs that let opencodex manage the model of each Codex agent role installed by LazyCodex (#6262), size roles and auto-assign models (#6269), and suggest a delegation model on the Subagents page (#6274). They could not land as fork PRs: the readiness gate needs the author's local-validation attestation, fork CI never ran its test shards, and `hygiene` flagged an unsponsored management-API surface. This unit lands the same work as three maintainer carry PRs in dependency order, each crediting the author, each merged only after its exact head passes hosted CI. + +## Loop spec + +- **Loop archetype:** satisfy-spec (verifier defines done: exact-head required CI green, then squash merge). +- **Trigger:** maintainer request (2026-10-01) to research omo and merge LilMGenius's work via cxc-loop. +- **Goal:** #6262, #6269 and #6274 content on `dev`, originals closed with credit. +- **Non-goals:** no behavior changes beyond the contributor heads; no Pi omo or OpenCode omo changes; no #6348 or other JEV work; no release or promotion. **No local test suites or typecheck** (maintainer instruction); hosted CI is the only execution evidence. +- **Verifier:** `gh pr checks --required` on the exact head SHA (reads every required job of that head, including test shards, typecheck, structure, privacy, file-size ratchet, hygiene, enforce-target). Local: `git diff --check` and `git merge-tree --write-tree origin/dev HEAD` (textual union only). +- **Stop condition:** all three carries merged and originals closed, or a blocker that needs maintainer direction. +- **Memory artifact:** this unit (000–030 docs) and the session goalplan `lilmgenius-omo-codex-lazycodex-role-model-series`. +- **Expected terminal outcomes:** DONE (three merges); BLOCKED (CI failure needing design change, union conflict that changes behavior); NEEDS_HUMAN (security-review objection). +- **Escalation condition:** a required job fails for a reason that is not a mechanical carry fix; a behavior decision beyond the contributor heads; a security-review objection. Shared gate 4 compatibility repairs (including a byte-for-byte ratchet move) are pre-authorized and do not escalate. +- **Resource bounds:** none set by the user beyond the host goal. + +## Research + +- **omo variants** (maintainer review 2026-09-30 01:47 on #6262): Pi omo (senpi, `~/.omo/agent`, existing omo tab), OpenCode omo (oh-my-opencode, untouched), Codex omo (LazyCodex). The series is now scoped to LazyCodex; Pi omo has no diff against `dev`. +- **LazyCodex detection verified against LazyCodex source** (`~/developer/codex/161_lazycodex`): the installer writes `lazycodex-install.json` into the plugin root (`plugins/omo/scripts/install-flow.mjs:4` `INSTALL_SNAPSHOT_FILE`, `plugins/omo/dist/cli/index.js:99259`); the plugin is installed with `codex plugin add omo@sisyphuslabs` (README). `src/clients/lazycodex.ts` requires `plugins."omo@sisyphuslabs".enabled = true` and a receipt under `plugins/cache/sisyphuslabs/omo//`, matching the Codex plugin cache layout. +- **Open review threads** re-checked at heads 6262 `ada7ec14b1`, 6269 `6c97ca9a4f`, 6274 `b180fcec0a` (architect D2–D4): unreadable `omo.jsonc` (handled by `readOmoRoleModels` → `unreadable`), escaped quoted keys (decoded, undecodable refused), multiline closing quotes (scanner consumes them), mirror retry (`retryMirror`), DelegationSuggest live region (mounted polite region). All fixed with regressions; threads are resolved on landing with a pointer. +- **Topology:** strictly stacked 6262 ⊂ 6269 ⊂ 6274, merge-base `961a4b569` (26 behind `dev` at `0328373fb8`); `git merge-tree` of 6274 vs `dev` is clean. Layer sizes: L1 18 commits / 39 files, L2 19 / 43, L3 12 / 38. All 49 commits authored by `LilMGenius `. +- **Ratchet:** no touched file is in `tests/fixtures/file-size-baseline.json` caps; i18n catalogs are exempt. Uncapped files fail at 2,000 lines; `src/server/management/agent-settings-routes.ts` reaches 1,941 at 6274 (watch in wp3). + +## Work-phase map (dependency order) + +| WP | Doc | Content | Depends | +|---|---|---|---| +| wp0 | this unit | docs-only roadmap | — | +| wp1 | 010 | carry `961a4b569..6262`, PR, CI, merge | wp0 | +| wp2 | 020 | carry `6262..6269` onto merged dev, PR, CI, merge | wp1 | +| wp3 | 030 | carry `6269..6274` onto merged dev, PR, CI, merge, close originals | wp2 | + +## Decisions (architect Godel `01a0f664-d0ea-7000-894a-50e8a319b548`) + +- D1 sequential layer cherry-pick onto landed dev — **accepted**: each PR diff is its own layer; author identity stays on every commit; squash body adds `Co-authored-by: LilMGenius `. +- D2/D3/D4 findings already fixed — **accepted**; no extra code; resolve threads with pointers on landing. +- D5 union gates (ratchet 2,000-line ceiling, test-layout maps, i18n parity, structure ownership, route registry) — **accepted**; enforced by hosted CI, watched by hand at carry time. +- D6 preserve recent dev changes in shared files — **accepted**; conflicts resolved by keeping both sides, never wholesale contributor tree. +- D7 merge-dev-into-contributor-head alternative — **rejected**: post-squash parent overlap makes later layers re-carry earlier commits. +- D8 #6274 scope is general Codex delegation, not LazyCodex-only — **accepted as intended**: the author states it configures Codex delegation defaults; the maintainer asked to merge the whole series. +- Reflection: see 001_reflection.md. + +## Security review note + +## Gates shared by every carry (r2, after architect reflection) + +1. **Pinned inputs.** Replay immutable full-SHA ranges only: wp1 `961a4b569512bd568106c4ea67a21a346e002779..ada7ec14b1b089f461e43e208d54750591fd26f9`, wp2 `ada7ec14b1b089f461e43e208d54750591fd26f9..6c97ca9a4fdec6c59901fa115be341d9cc590a15`, wp3 `6c97ca9a4fdec6c59901fa115be341d9cc590a15..b180fcec0a4c3ff63b9f230acc79e5f029bb6cea`. Before replay assert `git rev-parse refs/omo/` still equals the upper SHA; a moved contributor head stops the carry for re-plan. +2. **Fresh base.** `git fetch origin dev` immediately before branching; wp2/wp3 assert `git merge-base --is-ancestor origin/dev`. +3. **Conflict reconciliation.** Keep both sides semantically: one entry per key in i18n catalogs, layout maps and route registry; no duplicated registrations; never take the contributor file wholesale. +4. **Compatibility repairs are allowed** when hosted CI fails for a mechanical union reason (test-layout registration, ratchet 2,000-line ceiling via a byte-for-byte move into a sibling module, i18n key parity, structure ownership). Such a repair is a separate commit named "fix(carry): …", listed in the PR body, and does not change behavior. Anything else escalates. +5. **CI receipt.** Record in the PR body/comment and in this unit: head SHA, base SHA, `gh pr checks --required` output with every required job `pass`, and a coverage assertion that the expected jobs actually ran and succeeded at that head — every `test` shard, `typecheck`, structure, privacy scan, file-size ratchet, `hygiene`, `enforce-target`, and GUI lint/tests when `gui/` changed; a job that is absent, skipped or cancelled fails this gate. Also record the CI run id and attempt for the `pull_request` event at that head. Immediately before merge, re-read the head SHA and required checks; merge with `gh pr merge --squash --match-head-commit `. +6. **Publication.** PR template fully filled; GUI screenshots linked from the contributor's existing pr-assets (no image committed to the branch); squash body ends with `Co-authored-by: LilMGenius `; maintainer-integration record (MAINTAINERS.md "maintainer integration": actor lidge-jun admin, exact-head CI evidence) posted as a PR comment. +7. **Security review.** Before merge, an independent read-only security reviewer (fresh subagent, not the architect) reviews the exact carry head for its surfaces — management-API route admission and sibling guard, file writes under `$CODEX_HOME/agents` and `~/.omo/omo.jsonc` (path validation, atomicity, no secret logging), and the loopback self chat-completion (admission header, response bounds). Its verdict (PASS / FINDINGS) is bound to the head SHA and recorded in the PR comment and in this unit. FINDINGS block merge until fixed or explicitly dispositioned; an unresolved objection is NEEDS_HUMAN. Owner authorization (2026-10-01) covers the merge decision, not the review. +8. **D8 evidence.** #6274 author comment 2026-09-30T07:35: "Delegation suggest configures Codex delegation defaults, so it stays unscoped from the omo variants"; the maintainer's 2026-10-01 instruction covers the whole series. +9. **Maintainer-objection gate.** The originals carry two `CHANGES_REQUESTED` reviews by lidge-jun (2026-09-30 01:44 stack split, withdrawn by the 01:47 review; 01:47 omo-variant scoping). The author addressed the variant scoping in `e9ea34661`, `7c81640d0`, `b91537a75`, `d7a4f91a7` (#6262) and the matching #6269/#6274 commits. When each carry PR opens, dismiss both reviews on its original with a message citing those commits and the carry PR; immediately before each merge run `scripts/ci/assert-mergeable-review.sh --maintainer-integration ` and require exit 0. Any other maintainer objection is NEEDS_HUMAN. +10. **Base binding.** `pull_request` CI tests the merge of head and base at trigger time. Immediately before merge: `git fetch origin dev`; if `origin/dev` moved past the base recorded in the CI receipt, merge `origin/dev` into the carry branch (no force push), push, and repeat gates 5, 7 and 9 on the new head; for gate 7 the security reviewer inspects the incoming dev delta and re-attests the new head (a short "no security impact" verdict bound to that SHA suffices). Merge only when the receipt base equals current `origin/dev`. + +## Security review note (surfaces) + +Surfaces: management API routes (`/api/codex-agent-roles`, auto-assign, `/api/injection-model/suggest`), file writes under `$CODEX_HOME/agents` and `~/.omo/omo.jsonc`, and a loopback self chat-completion (`src/lib/local-chat-completion.ts`). No credential storage or OAuth change. Sibling instances refuse writes. Each carry PR states this and requests the MAINTAINERS.md security review explicitly. diff --git a/devlog/_fin/261001_omo_lazycodex_carry/010_wp1_role_model_picker.md b/devlog/_fin/261001_omo_lazycodex_carry/010_wp1_role_model_picker.md new file mode 100644 index 00000000000..dfc97ea1c3e --- /dev/null +++ b/devlog/_fin/261001_omo_lazycodex_carry/010_wp1_role_model_picker.md @@ -0,0 +1,23 @@ +# wp1 — carry #6262 (LazyCodex role model picker) + +Source: LilMGenius:feat/codex-role-models head `ada7ec14b1`, layer `961a4b569..ada7ec14b1` (18 commits, 39 files, +1636/-31). + +## Steps + +1. `git switch -c codex/omo-lazycodex-role-models origin/dev` (this branch also carries this unit's docs). +2. Assert `git rev-parse refs/omo/6262` = `ada7ec14b1b089f461e43e208d54750591fd26f9`, then `git cherry-pick 961a4b569512bd568106c4ea67a21a346e002779..ada7ec14b1b089f461e43e208d54750591fd26f9` oldest first (gate 1). On conflict reconcile per gate 3: `dev` additions in i18n catalogs, `scripts/test-layout/layout.json`, `tests/fixtures/test-layout-expected.json`, `structure/clients/integrations.md`; `src/lib/jsonc.ts` (the series only modifies it; it exists unchanged on `origin/dev` and at the pinned base, blob `dda3c392`. An earlier accidental trial cherry-pick in the main checkout hit a modify/delete conflict only because that checkout's local `dev` was stale at `9177663665`; the trial was aborted and the checkout restored). +3. `git diff --check origin/dev...HEAD`; `git merge-tree --write-tree origin/dev HEAD` must be clean. +4. Push; open PR to `dev` titled `feat(codex): pick the model for each LazyCodex agent role (carry #6262)` with the PR template filled, the three-variant table and screenshots from #6262 (LilMGenius pr-assets SHA `d0a29e1c` links), "Carried from #6262 by @LilMGenius", `Co-authored-by: LilMGenius `, Verification "local suites not run (maintainer instruction); hosted CI at exact head", security-review request, and the maintainer-integration record. +5. Wait for every required check on the exact head; fix mechanical carry failures only (escalate otherwise). +6. Squash-merge with the co-author trailer; resolve #6262 open threads with a pointer; leave #6262 open until wp3 closes all three. + +## Files (layer L1) + +docs-site guides/integrations.md, reference/cli/agents.md; gui i18n ×10, main.tsx, pages/Integrations.tsx, pages/integrations/LazyCodexRoleModels.tsx, styles/lazycodex-role-models.css, tests/lazycodex-role-models.test.tsx; scripts/test-layout/layout.json; skills/ocx/references/01_management_surface.md; src/cli/agent.ts, cli/capabilities.ts, clients/lazycodex.ts, clients/omo-role-models.ts, codex/agent-role-models.ts, codex/prompt-layers/toml-edit.ts, codex/subagent-model-fallback.ts, lib/jsonc.ts, server/management-api.ts, server/management/codex-agent-role-routes.ts, route-registry.ts, sibling-guard.ts; structure/clients/integrations.md, structure/subagents.md; tests cli-headless-parity, lazycodex-detection, omo-role-models, codex-agent-role-models, codex-agent-role-routes; tests/fixtures/test-layout-expected.json. + +## Acceptance + +- Exact-head required checks all `pass` (`gh pr checks --required`). +- PR diff equals layer L1 plus this unit's docs and any listed "fix(carry)" compatibility commits (no #6269/#6274 files). All shared gates (1–10) in 000_plan.md apply. +- Co-author trailer present in the squash commit on `dev`. +- Conditional paths are covered by the carried tests: no LazyCodex → GET returns no roles and PUT 409 (`tests/server/codex-agent-role-routes.test.ts`); unreadable mirror → `unreadable` (`tests/clients/omo-role-models.test.ts`); escaped/multiline keys (`tests/routing/codex-agent-role-models.test.ts`). diff --git a/devlog/_fin/261001_omo_lazycodex_carry/020_wp2_role_auto_assign.md b/devlog/_fin/261001_omo_lazycodex_carry/020_wp2_role_auto_assign.md new file mode 100644 index 00000000000..406e266fa58 --- /dev/null +++ b/devlog/_fin/261001_omo_lazycodex_carry/020_wp2_role_auto_assign.md @@ -0,0 +1,26 @@ +# wp2 — carry #6269 (role auto-assign) + +Source: LilMGenius:feat/codex-role-auto-assign head `6c97ca9a4f`, layer `ada7ec14b1..6c97ca9a4f` (19 commits, 43 files, +2206/-174). + +## Steps + +1. After wp1 merges: `git fetch origin dev && git switch -c codex/omo-lazycodex-role-auto-assign origin/dev`. +2. Assert `git rev-parse refs/omo/6269` = `6c97ca9a4fdec6c59901fa115be341d9cc590a15` and the wp1 squash SHA is an ancestor of `origin/dev`, then `git cherry-pick ada7ec14b1b089f461e43e208d54750591fd26f9..6c97ca9a4fdec6c59901fa115be341d9cc590a15` (gates 1–3), reconciling in particular `src/server/management/agent-settings-routes.ts` (dev native picker-order acceptance), `src/config/schema/config-schema.ts` and `src/types/config.ts` (dev Codex-credit additions). +3. Line-count check for uncapped files near 2,000 (`agent-settings-routes.ts`). + +**Re-verification at P (2026-10-01, dev `a6114b62ed`).** wp1 landed as #6366 with a maintainer security fix (`70f696593c`): `writeCodexAgentRoleModel` validates the role TOML before and after the edit (`invalid_role_file`), and the route answers a fixed `write_failed` message. A probe replay showed the second L2 commit `a0e3ac8d39` (role `model_reasoning_effort` write) conflicts with that fix in `src/codex/agent-role-models.ts` and `tests/routing/codex-agent-role-models.test.ts`. Resolution (gate 3, union, no behavior dropped from either side): +- `AgentRoleModelErrorCode` keeps `invalid_effort` and `invalid_role_file` (plus the existing codes). +- `writeCodexAgentRoleModel`: `assertValidRoleToml(role, before, "before")` → `withModel = setTomlRootModel(...)` → `after = effort === undefined ? withModel : setTomlRootReasoningEffort(withModel, validateAgentRoleEffort(effort))` → unchanged check → `assertValidRoleToml(role, after, "after")` → write. +- Tests: keep the invalid-TOML refusal test and both new effort tests. +Later L2 commits are replayed after this resolution; any further conflict in these two files follows the same union rule. Because the effort path now also passes the after-validation, gate 7 re-review covers the effort writer. +4. PR `feat(codex): auto-assign LazyCodex role models by sizing each role (carry #6269)`; same template/credit/security/integration records; screenshots from #6269 body. +5. Exact-head CI, squash merge, resolve threads. + +## Files (layer L2) + +As listed by `git diff --name-only refs/omo/6262 refs/omo/6269`: role-auto-assign/sizing modules, local-chat-completion, codex-role-auto-assign routes, LazyCodexRoleAutoAssign.tsx, config schema/types (`codexRoleTiers`), i18n ×10, docs, tests (codex-role-auto-assign, codex-role-sizing, local-chat-completion, codex-role-auto-assign-routes, lazycodex-role-auto-assign GUI), layout maps. + +## Acceptance + +- Exact-head required checks pass; diff equals L2 plus listed "fix(carry)" commits. All shared gates (1–10) apply; branch only after wp1 squash SHA is an ancestor of origin/dev. +- Without LazyCodex, `POST /api/codex-agent-roles/auto-assign` returns 409 before calling the sizing model (`tests/server/codex-role-auto-assign-routes.test.ts`); oversized sizing response is cut while streaming (`tests/lib/local-chat-completion.test.ts`). diff --git a/devlog/_fin/261001_omo_lazycodex_carry/030_wp3_delegation_suggest.md b/devlog/_fin/261001_omo_lazycodex_carry/030_wp3_delegation_suggest.md new file mode 100644 index 00000000000..831261e4bce --- /dev/null +++ b/devlog/_fin/261001_omo_lazycodex_carry/030_wp3_delegation_suggest.md @@ -0,0 +1,23 @@ +# wp3 — carry #6274 (delegation suggest) and close the originals + +Source: LilMGenius:feat/subagent-default-sizing head `b180fcec0a4c3ff63b9f230acc79e5f029bb6cea`, layer `6c97ca9a4fdec6c59901fa115be341d9cc590a15..b180fcec0a4c3ff63b9f230acc79e5f029bb6cea` (12 commits, 38 files, +961/-46). + +## Steps + +1. After wp2 merges: `git fetch origin dev`; assert wp2's squash SHA is an ancestor of `origin/dev` and `git rev-parse refs/omo/6274` = `b180fcec0a4c3ff63b9f230acc79e5f029bb6cea` (gate 1, gate 2). Branch `codex/omo-delegation-suggest` from `origin/dev`; `git cherry-pick 6c97ca9a4fdec6c59901fa115be341d9cc590a15..b180fcec0a4c3ff63b9f230acc79e5f029bb6cea`. +2. Ratchet watch: `src/server/management/agent-settings-routes.ts` must stay under 2,000 lines after the union; if not, apply shared gate 4 (byte-for-byte move of the suggest handler into a sibling module, separate "fix(carry)" commit). +3. PR `feat(subagents): suggest a delegation model by sizing the work (carry #6274)` with shared gates 5–7: template, screenshots from #6274's pr-assets, co-author trailer, security review, maintainer-integration record. Scope note (D8, gate 8): applies to Codex delegation defaults generally, not only LazyCodex. +4. Exact-head CI receipt with coverage assertion (gate 5), then `gh pr merge --squash --match-head-commit `. +5. Close #6262, #6269, #6274 with a comment linking each carry PR and merge SHA, thanking @LilMGenius; resolve remaining threads with pointers. + +## Acceptance + +**Re-verification at P (2026-10-01, dev `da13a02727` = #6367).** Pinned `refs/omo/6274` is still `b180fcec0a`. The replay `6c97ca9a4f..b180fcec0a` on `codex/omo-delegation-suggest` hit one conflict in `gui/src/pages/integrations/LazyCodexRoleAutoAssign.tsx`. #6274 moves `TIER_LABEL`/`EFFORT_LABEL` into `sizing-labels.ts`, while #6367 hoisted `alreadySet` to module scope at the same spot. Resolved by taking #6274's move and keeping the module-scope `alreadySet` (gate 3). The rest applied cleanly (12 commits). + +Gate 4 compatibility repair `881bb588f4` "fix(carry): keep sizing failure text out of delegation suggest responses": #6274's delegated-work suggest path reuses the sizing completion and still echoed raw error text, the same leak class the #6367 security review found. It now goes through `publicSizingError`, with a regression in `tests/codex-integration/injection-model-suggest-routes.test.ts`. `agent-settings-routes.ts` is 1,943 lines, under the 2,000 ceiling. React Doctor shows 0 diagnostics on the changed GUI files, and the privacy scan passes. + + +- Exact-head required checks pass with the gate 5 coverage assertion; diff equals L3 plus any listed "fix(carry)" compatibility commits. +- The polite live region stays mounted and announces running and completion text (`gui/tests/subagents-delegation-suggest.test.tsx`). +- `--apply` skips an already-matching setting and the suggest route is read-only until `PUT /api/injection-model` (`tests/codex-integration/injection-model-suggest-routes.test.ts`). +- Originals closed with links; goalplan criteria c-1 to c-4 met with evidence. diff --git a/devlog/_fin/261001_omo_lazycodex_carry/090_outcome.md b/devlog/_fin/261001_omo_lazycodex_carry/090_outcome.md new file mode 100644 index 00000000000..810a52260a0 --- /dev/null +++ b/devlog/_fin/261001_omo_lazycodex_carry/090_outcome.md @@ -0,0 +1,26 @@ +# 090 — outcome + +All three layers of LilMGenius's omo (Codex / LazyCodex) series are on `dev`. Each landed as a maintainer carry PR with `Co-authored-by: LilMGenius `, in dependency order, and each was verified only by hosted CI at its exact head. The owner instructed that no local test suites be run. + +| WP | Carry PR | Source | Squash on dev | CI run (exact head) | Security review | +|---|---|---|---|---|---| +| wp1 | #6366 | #6262 `ada7ec14b1` | `a6114b62ed` | 36834055042 @ `70f696593c` | FINDINGS (2) → fixed → PASS | +| wp2 | #6367 | #6269 `6c97ca9a4f` | `da13a02727` | 36861822233 @ `c721b63cb0` | FINDINGS (1) → fixed → PASS | +| wp3 | #6389 | #6274 `b180fcec0a` | `de8afe2e86` | 36863310320 @ `e29b86290c` | PASS, carry repair accepted | + +## What changed beyond the contributor heads + +- `70f696593c` (wp1): the role TOML must parse both before and after a model edit, otherwise the write is refused with `invalid_role_file`. Other write failures answer a fixed `write_failed` message with no path or UID. +- `30ae36a149` (wp2): auto-assign sizing failures now return only a category or a bare HTTP status (`publicSizingError`). The same commit clears two React Doctor warnings. A later fixture fix (`c721b63cb0`) satisfies the privacy scan. +- `881bb588f4` (wp3): the same sanitisation for delegation suggest. +- Union resolutions: wp2 `a0e3ac8d39` (the effort writer) was merged with the wp1 TOML validation. wp3 merged the label move with the module-scope `alreadySet`. + +## What did not go to plan + +- The first CI runs on wp2 failed twice, on React Doctor and then on the privacy scan. Both failures came from maintainer edits, not contributor code. Running React Doctor and the privacy scan statically before pushing would have caught them. +- An accidental trial cherry-pick in the main checkout ran against a stale local `dev`. It was aborted with no residue. +- The wp3 replay and repair happened before A instead of inside B. The FSM refused B→C for having no source delta, so this `_fin` move is wp3's B delta. + +## Originals + +#6262 (already closed), #6269 and #6274 were closed with links and credit. My `CHANGES_REQUESTED` reviews on all three were dismissed, citing the author's fixing commits. diff --git a/devlog/_plan/261001_jev_decision_routing/000_plan.md b/devlog/_plan/261001_jev_decision_routing/000_plan.md new file mode 100644 index 00000000000..cb37ab267cc --- /dev/null +++ b/devlog/_plan/261001_jev_decision_routing/000_plan.md @@ -0,0 +1,38 @@ +# 261001 JEV decision routing + +Status: open. Loop session `01a0f5d4-89c4-7973-9b32-3cadcb4f0929`, branch `codex/jev-decision-routing` from `origin/dev` `64294638a6`. + +A `strategy: "jev"` combo asks a decision service which target and reasoning effort should take the next call. Today that service is always a System One endpoint. This unit turns the transport into a pluggable **decision backend** so the same state and candidate list can be answered either by a System One server or by any model opencodex already routes. + +## Sources carried + +| Source | Author | What is carried | What is not | +|---|---|---|---| +| #6302 | SeongwoongCho | Commits `d60975ee1b`..`304bd4aaee` (self-hosted `jev-decision` rows, `decisionProvider`, `decisionTimeoutMs`, cleartext loopback opt-in, GUI select, save/PATCH guards) cherry-picked with authorship. CLI partial-update fixes `2740904285` and `af1b035e7a`, adapted. | Level mode, quota signals, quota warmer, configurable wording (`4b5c636db2` and later). They go back to the author as a rebase onto this PR. | +| #6185 | yxr1995-maker | Pure discovery helpers (`decision-discovery.ts`: model hint, `/systemone` endpoint derivation, dedupe) and a read-only discovery endpoint; regression cases proving environment TypeSafe keys never reach another destination. | `Decisions.tsx` page and sidebar tab, automatic destination scan, global `JEV_MODEL` override, `jev-opencode` preset (MAINTAINERS.md: a new preset is a credential-destination change needing primary-source evidence; OpenCode zen is documented as an ordinary row), adopt endpoint that copies credentials. | +| #6275 | codingbooo | Its custom-endpoint regression, adapted to a non-`jev` row id (a `jev` row stays pinned to TypeSafe by design). | The `isConfiguredJev` retargeting of the `jev` row. | + +All three authors get `Co-authored-by` trailers on the squash. + +## Decisions + +- **D1 config.** Combo fields: `decisionProvider` (existing), `decisionModel` (new, opencodex route string), `decisionTimeoutMs` (existing). The backend is derived, never stored: `decisionModel` set → `model`; `decisionProvider` set and not `jev` → `systemone`; otherwise → `typesafe`. Setting both is a config error. Existing combos need no change. +- **D2 backend seam.** `resolveJevDecision` becomes a dispatcher over two backends that share one runtime envelope (candidate bounds, state check, deadline, abort handling, fail-open): System One (unchanged bytes on the TypeSafe path) and Model (prompt → JSON `{"choice": ""}`). Level/quota modes can later add a question builder on top of the same envelope. +- **D3 model invocation.** The pure backend takes an injected `invoke(model, prompt, signal)`. Server glue in `src/server/responses/jev-model-invoke.ts` runs a fresh internal `/v1/responses` turn through the dispatcher's `handleResponses` with its own send budget, its own turn lease (`tryAdmitTurn`), a detached log context whose spend tracker is settled, the parent's admission, and no parent history/tools/session headers. The response is read through `readBoundedResponseBytes` (64 KiB) and `createSseInspector`. +- **D4 recursion.** A decision model that resolves (canonical `combo/` or alias, after synthetic selector stripping) to its own combo or to any `strategy: "jev"` combo is refused by `comboConfigIssues` against the prospective combo map. At runtime an `internalDecisionCall` flag refuses combo reentry defensively and fails open. +- **D5 stats.** `PersistedJevDecisionV1` gains optional `backend`; old rows keep parsing and land in an `unknown` bucket. Aggregates add per-backend count and average latency. +- **D6 surfaces.** CLI `ocx combo set --decision-model `; management PUT/POST combos accept `decisionModel`; `POST /api/combos/decision-test` probes an unsaved decision configuration with synthetic candidates; `GET /api/combos/decision-discovery` lists configured System One rows and catalog rows that look like decision models with the derived endpoint. +- **D7 dashboard.** A `Decision method` section in the combo detail panel and add dialog (TypeSafe / System One-compatible server / opencodex model, timeout, Test, recent stats). No separate page. + +## Work-phase map + +| Doc | Work-phase | Depends on | Closes with | +|---|---|---|---| +| 010_wp1_backend_interface.md | wp1 backend seam, System One backend, stats backend kind, carried tests | — | focused routing/usage tests, typecheck | +| 020_wp2_model_backend_surfaces.md | wp2 model backend, server glue, validation, CLI/API, decision test, discovery | wp1 | focused routing/server/cli tests, typecheck | +| 030_wp3_dashboard_docs.md | wp3 GUI section, i18n, docs-site, structure docs | wp2 | gui tests, lint:i18n, build, structure:check | +| 040_wp4_landing.md | wp4 full validation, PR, CI, source PR closure | wp3 | exact-head CI green, PRs commented | + +## Out of scope + +Merging to `dev`, releases, promotions, level/quota mode, new credential stores, provider presets. diff --git a/devlog/_plan/261001_jev_decision_routing/010_wp1_backend_interface.md b/devlog/_plan/261001_jev_decision_routing/010_wp1_backend_interface.md new file mode 100644 index 00000000000..13dd86044c7 --- /dev/null +++ b/devlog/_plan/261001_jev_decision_routing/010_wp1_backend_interface.md @@ -0,0 +1,50 @@ +# 010 wp1: decision backend seam and System One backend + +Revalidate against the tree at the start of wp1; line numbers below are from `80efc0e373`. + +## Files + +### MODIFY `src/combos/jev.ts` + +- Export `type JevDecisionBackend = "typesafe" | "systemone" | "model"`. +- `JevDecision` gains `backend: JevDecisionBackend`. +- Extract the shared envelope into exported helpers (stay in this file to keep imports flat): + - `jevDecisionTimeoutMs(value?: number): number` — bounds check now inline in `resolveJevDecision`. + - `jevDecisionPreflight(options): { failed?: gate; state?: Record }` — empty candidates → `no_choices`; `candidatesFitRequestBounds` → `invalid`; `buildJevState` + `hasJevDecisionState` → `no_state`. + - `jevRouteChoiceKeys(candidates): string[]` exporting the allowlist (wrapper over `candidateOptions`), and `jevRouteOption(candidates, key)` returning `{targetKey, effort}`. +- Rename the System One transport body to `resolveJevSystemOneDecision(options, endpoint)`; behavior identical. Byte-for-byte TypeSafe request (`model`, `state`, `questions`) keeps the existing body-equality test green. +- `resolveJevDecision(options)` becomes the dispatcher: `options.decisionModel` → model backend (wp2; until then not reachable), else System One. Every returned decision carries `backend`: `decisionProvider` absent or `jev` → `typesafe`, otherwise `systemone`. +- `ResolveJevDecisionOptions` gains `decisionModel?: string` and `invokeModel?: JevModelInvoke` (type declared in wp2's module; in wp1 declare the type here and re-export). + +### MODIFY `src/usage/jev-stats.ts` + +- `PersistedJevDecisionV1.backend?: JevDecisionBackend`. +- `normalizePersistedJevDecision`: copy `backend` only when it is one of the three values; anything else is dropped without rejecting the row. +- `JevStatsResponse` gains `backends: Array<{ backend: JevDecisionBackend | "unknown"; decisions: number; applied: number; averageLatencyMs: number | null }>` (fixed order typesafe, systemone, model, unknown; zero rows omitted). +- `StreamingJevStatsAccumulator`: per-backend counters in `add`, deep copy in `clone`, projection in `summarize`. + +### MODIFY `src/server/responses/core-combo.ts` (JEV block ~517-575) + +- Persist `backend: decision.backend` in `logCtx.jevDecision` and in the debug line. Exception path: backend derived via exported `jevDecisionBackendFor(combo)`. + +### MODIFY `src/combos/index.ts` + +- Re-export the new helpers/types. + +### NEW `src/server/management/decision-discovery.ts` (from #6185, pure) + +- `DECISION_MODEL_HINT`, `isDecisionModelCandidate(row, query)`, `systemOneEndpoint(baseUrl)`, `uniqueDiscoveryCandidates(rows, query)` carried as written in #6185 `5ffb4c1ba9`. + +## Tests + +- MODIFY `tests/routing/jev-decision.test.ts`: assert `backend` on apply and fail-open decisions for TypeSafe and a self-hosted row; add #6275's custom endpoint regression adapted to row id `custom-decider` over HTTPS (custom URL, model, own bearer, no TypeSafe env key). +- NEW `tests/routing/jev-decision-destination.test.ts` (cases from #6185): `TYPESAFE_API_KEY`/`JEV_API_KEY` set in env never appear in headers sent to a self-hosted row; a self-hosted row whose apiKey references those env names is unusable (`missing_key`, no send); a `jev` row with a foreign baseUrl still posts to TypeSafe. +- NEW `tests/server/decision-discovery.test.ts` (from #6185): hint, query override, endpoint derivation, dedupe. +- MODIFY `tests/usage/jev-stats*.test.ts` (locate): old row without backend parses; invalid backend dropped; backend buckets after clone+add. +- Register new test files in `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`. + +## Accept + +- `bun test tests/routing/jev-decision.test.ts tests/routing/jev-decision-destination.test.ts tests/routing/jev-decision-provider-combo.test.ts tests/server/server-jev-combo-e2e.test.ts tests/server/decision-discovery.test.ts` + usage jev-stats tests pass. +- TypeSafe body-equality test unchanged and green (activation: TypeSafe env key set, default combo). +- `bun run typecheck` 0. diff --git a/devlog/_plan/261001_jev_decision_routing/020_wp2_model_backend_surfaces.md b/devlog/_plan/261001_jev_decision_routing/020_wp2_model_backend_surfaces.md new file mode 100644 index 00000000000..4238356d0c1 --- /dev/null +++ b/devlog/_plan/261001_jev_decision_routing/020_wp2_model_backend_surfaces.md @@ -0,0 +1,83 @@ +# 020 wp2: model backend, validation, CLI and management surfaces + +## NEW `src/combos/jev-model-backend.ts` (pure, no server imports) + +```ts +export type JevModelInvoke = (request: { model: string; instructions: string; input: string; signal: AbortSignal }) + => Promise<{ text: string; usage?: Record }>; +export const JEV_MODEL_INSTRUCTIONS: string; // fixed system text: pick exactly one key, reply {"choice":""} only +export function buildJevModelPrompt(state: Record, candidates: readonly JevCandidate[]): string; +export function parseJevModelChoice(text: string, allowed: ReadonlySet): string; // throws on anything else +export async function resolveJevModelDecision(options: ResolveJevDecisionOptions & { decisionModel: string; invokeModel: JevModelInvoke }): Promise; +``` + +- Prompt = JSON text `{ state, options: { ":": "" , ... } }` reusing `buildJevState` and the same option keys as System One (`candidateOptions`). Max 64 options, request ≤ 64 KiB (UTF-8, prompt + instructions) else `invalid`. Fewer than 1 option → `no_choices`. +- Parser: trim; strip one surrounding ```json fence; `JSON.parse`; require object with string `choice` in allowlist; also accept a bare quoted string key. Text > 4 KiB → `malformed`. Unknown key → `invalid`. Non-JSON → `malformed`. +- Errors map: invoke throws `JevModelInvokeError{gate:"http"|"network"|"malformed"|"missing_key"}` → that gate; deadline → `timeout`; caller abort rethrown by identity. `backend: "model"`. Usage keys normalized like System One (`inputTokens/outputTokens`). + +## NEW `src/server/responses/jev-model-invoke.ts` + +`createJevModelInvoker({ req, config, options, handleResponses }): JevModelInvoke`: +- Body `{ model, stream: true, store: false, instructions, input: [{role:"user",content:[{type:"input_text",text}]}], tools: [] }`. +- Headers: `content-type` plus `authorization` only when the parent carried it (credential ownership stays with core-auth). No session/thread/turn headers. +- Options: `{ abortSignal: signal, admission: options.admission, codexAuthPolicy: options.codexAuthPolicy, sendBudget: createInferenceSendBudget(req, childLog), turnAdmissionLease: lease, internalDecisionCall: true }`; lease from `tryAdmitTurn()`, missing lease → throw gate `network` (fail-open); released in `finally`. +- `childLog = { model, provider: "unknown", inboundProtocol: "responses" }`; settle its spend tracker in `finally` with measured usage. +- Read: non-2xx → `http` (cancel body); `readBoundedResponseBytes(64 KiB)` oversize → `malformed`; SSE via `createSseInspector` completed object, JSON via fatal decode; status must be `completed`; extract `output[].content[].type==="output_text"` text. + +## MODIFY `src/server/responses/core-options.ts`, `request-prepare.ts` + +- `HandleResponsesOptions.internalDecisionCall?: boolean`. +- In request preparation: when set, skip shadow interception and refuse a route that resolves to a combo (return 400 `invalid_request_error`), so a decision call can never recurse. + +## MODIFY `src/server/responses/core-combo.ts` + +- Pass `decisionModel` and `invokeModel: createJevModelInvoker({ req, config, options, handleResponses: requestDispatchers.handleResponses })` into `resolveJevDecision`. + +## Validation + +- `src/types/config.ts` `OcxComboConfig.decisionModel?: string | null`; `src/combos/types.ts` `NormalizedComboConfig.decisionModel?`, normalize trims. +- `comboConfigIssues` gains `options.combos?` (prospective map). Rules: non-empty ≤ 512 chars; only with strategy `jev`; not together with a non-null `decisionProvider`; strip synthetic selector (`parseSyntheticRowId`) then `resolveComboId` — reject when it is this combo (by id or any of its aliases in the prospective entry) or any combo whose strategy is `jev`. +- Config schema (`config-schema.ts` ~720) passes the full combos map. Management PUT/POST passes `nextCombos` and revalidates every other combo's `decisionModel` when a combo changes strategy/alias. +- Save-time: `decisionModelRouteError(config, comboId, model)` in `src/server/management/decision-model-validation.ts` uses `previewRouteModel`; refuses unroutable models and `jev-decision` adapter rows. +- `combo-routes.ts` omission preservation for `decisionModel`; selecting one selector with explicit null clears the other. +- `provider-id-rewrite.ts` rewrites the provider prefix of `decisionModel`; `comboDependsOnProvider` counts it (DELETE guard). + +## CLI `src/cli/combo.ts` + +- `--decision-model `; reject with `--decision-provider` non-`-`; setting one sends `null` for the other. Carry the partial-update fixes (`2740904285`, `af1b035e7a`): `set` without `--targets` GETs and merges the existing row, filtering null listing fields; drop JEV fields when strategy changes away from `jev`. + +## Management + +- NEW route `POST /api/combos/decision-test` in `combo-routes.ts` (registered in route-registry with read-only mutation metadata): body `{ decisionProvider?, decisionModel?, decisionTimeoutMs? }`; validates like save; runs `resolveJevDecision` with two synthetic candidates (`probe/a:low`, `probe/a:high`) and a fixed probe task; returns `{ ok, backend, gate, latencyMs }`. Model backend uses `createJevModelInvoker` with a synthetic local Request. +- NEW route `GET /api/combos/decision-discovery?q=`: `{ configured: [{ id, url, model, usable, issue? }], discovered: [{ provider, model, endpoint }] }` from config rows and the model catalog via the #6185 helpers. No probe, no adoption. + +## Tests + +- NEW `tests/routing/jev-model-backend.test.ts`: prompt shape and bounds; parser (fenced, bare, unknown, oversize, non-JSON); fail-open gates; timeout; caller abort rethrow; backend tag. +- NEW `tests/server/server-jev-model-decision-e2e.test.ts`: loopback Bun.serve upstream for an ordinary provider; jev combo with `decisionModel` → decision request reaches that provider without parent tools/history and the chosen target serves the turn; decision provider returning garbage → fail-open to first target; separate send budget (parent target still sends). +- MODIFY combo validation tests: self id, self alias in same PUT, other jev combo alias, non-jev combo allowed, both selectors rejected, strategy change of referenced combo rejected. +- CLI test: `--decision-model`, conflict, partial update keeps targets. +- decision-test route test for both backends (mocked post / invoker). + +## Accept + +Focused files above + existing JEV suites + `tests/lab/core-lab-boundary.test.ts` + `bun run typecheck`. + + +## Reflection amendments (architect MISALIGNED → folded) + +- R1 recursion: the runtime guard refuses only self/JEV combos (non-JEV combos are legitimate decision models). With `internalDecisionCall`, request preparation skips shadow interception at both sites and memory-model rewriting, and refuses a resolved JEV combo before the `handleComboResponses` dispatch; the flag rides into combo children through options. +- R2 auth: forward `authorization` together with `chatgpt-account-id` when the parent carried them; never session/thread/turn headers. +- R3 helpers: `jevDecisionBackendFor(combo: { decisionProvider?: string; decisionModel?: string }): JevDecisionBackend` exported from `src/combos/jev.ts`. `class JevModelInvokeError extends Error { gate: "http" | "network" | "malformed" | "missing_key" }` exported from `src/combos/jev-model-backend.ts`. Validation stays import-free: `comboConfigIssues` options gain `normalizeDecisionModel?: (model: string) => string`; management and config schema inject a wrapper over `parseSyntheticRowId(id, config)`; default is identity. +- R4 prompt: `src/combos/jev.ts` exports `jevRouteOptions(candidates): Array<{ key; targetKey; effort; description }>` (wraps `candidateOptions` + `criterionDescription`); both backends use it. Bounds: ≤ 64 candidates (existing) and ≤ 64 expanded options for the model backend. +- R5 bounds: the invoker checks the complete serialized Responses body ≤ 64 KiB, passes the deadline-combined signal to `readBoundedResponseBytes`, cancels non-2xx bodies, rejects `error`/non-completed, extracts only `message` → `output_text`. Tests cover lease release and spend settlement. +- R6 field chains: decisionModel — CLI/add dialog → PUT → `comboConfigIssues`/normalize (sparse) → config write/reload → GET `/api/combos` → GUI `parseComboList` → `executeComboResponses`. backend — `resolveJevDecision` → `logCtx.jevDecision` → request-log persist (`normalizePersistedJevDecision` in request-log and usage/log) → hydration → accumulator add/clone → `/api/usage?jev=1` → GUI `JevStatsResponse` type and panel. A round-trip test exists for each chain; `jevAutoDraft` keeps no decisionModel. + + +## Audit amendments (A round 1 FAIL → folded) + +- B1 credentials (supersedes R2). The decision request carries **no** caller credential: no `authorization`, `chatgpt-account-id`, `x-api-key` or proxy admission header is copied, and the invoker passes `callerDirectAuth: null`, `openAiSidecarAuth: null`, `nativeCallerAuth: null` explicitly so `handleResponses` captures nothing. Only the parent's typed `admission` (scope enforcement) rides along. The decision model therefore runs on credentials configured on provider rows, as the design requires ("reuse existing provider credentials; no new credentials"). A route that would need the caller's own bearer (keyless Cursor, caller-owned ChatGPT forward) fails and the decision fails open; docs say so. `request-prepare`/`core-auth` treat `internalDecisionCall` as a credential-domain boundary (no caller-credential restoration). Negative tests: parent opaque bearer, proxy admission secret, and ChatGPT bearer+account id are absent from the upstream request of a decision model routed to a custom loopback provider; a keyless caller-auth provider yields fail-open with no send of the parent bearer. +- B2 byte proof. Before any refactor in wp1, generate `tests/fixtures/jev-typesafe-request-golden.json` from the current implementation (fixed body, candidates incl. operator note, deterministic) and add `tests/routing/jev-typesafe-golden.test.ts` comparing the exact serialized request string. Commit it first; it must stay green across wp1/wp2. +- B3 prospective config. Selector normalization and route preview run against `{ ...config, combos: nextCombos }` in management and against the loaded map in config schema. Tests: same-PUT alias then `` and synthetic `--fast` selector referencing self → rejected; changing a referenced combo to `strategy: "jev"` → rejected with the referencing combo named. +- B4 landing gates. wp4 runs full `bun run test` (resource exception only if documented with focused coverage), both layout guards, file-size ratchet, `tests/lab/core-lab-boundary.test.ts`, and a dispatched **security review** of the final diff (credential forwarding, outbound policy, env keys) recorded in the PR. Every new root-suite test is registered in `scripts/test-layout/layout.json` and `tests/fixtures/test-layout-expected.json`. +- Notes folded: docs describe cleartext opt-in as loopback **and literal private LAN addresses**; send budget created from the original parent `req` (no thread header forwarded upstream); `POST /api/combos/decision-test` registered as mutating; runtime recursion refusal and fail-open tested independently of save-time validation. diff --git a/devlog/_plan/261001_jev_decision_routing/030_wp3_dashboard_docs.md b/devlog/_plan/261001_jev_decision_routing/030_wp3_dashboard_docs.md new file mode 100644 index 00000000000..ffbe07ef508 --- /dev/null +++ b/devlog/_plan/261001_jev_decision_routing/030_wp3_dashboard_docs.md @@ -0,0 +1,58 @@ +# 030 wp3: dashboard, i18n, docs + +## GUI + +- NEW `gui/src/components/combo-workspace-jev-decision.tsx`: `ComboJevDecisionSection` replaces `JevDecisionFields` (moved out of `combo-workspace-controls.tsx`). Radio group: TypeSafe / System One-compatible server / opencodex model. Server: existing provider select (with discovery hints from `/api/combos/decision-discovery`). Model: select over a separate `decisionModels` inventory (enabled routable models; jev combos and the combo itself excluded). Shared timeout input. `Test` button → `POST /api/combos/decision-test` with the draft, shows ok/gate/latency, cancels on selection change. Saved combos show a compact recent summary from `/api/usage?jev=1&comboId=&range=7d` (decisions, applied %, avg latency, per-backend). +- MODIFY `gui/src/combo-workspace-data.ts`: `ComboItem.decisionModel`; `parseComboList`, `draftEquals`, `toPutBody` (explicit null for the inactive selector), `validateComboDraft` (model required for model method, not self/jev), `jevDecisionSummary` shows the model. +- MODIFY `gui/src/jev-decision-service.ts`: `jevDecisionModelOptions(...)`, `jevDecisionMethod(item)`. +- MODIFY detail panel / add modal / ComboWorkspace / Combos page to pass `apiBase`, `decisionModels`, combos. +- MODIFY `gui/src/components/jev-stats-panel.tsx`: per-backend row. +- i18n: new keys in `en.ts` and the other nine locales; generalize "decision service" wording where it now covers models. +- Tests: `gui/tests/jev-decision-fields.test.tsx`, `tests/gui/combo-workspace-jev-decision.test.ts`, `gui/tests/jev-stats-panel.test.tsx`. + +## Docs + +- `docs-site/src/content/docs/guides/combos.md`: JEV section → "Decision method" with the three backends, model example (`ollama/qwen3:4b`), OpenCode zen as a System One row example, recursion rule, timeout, stats, CLI/API. +- `docs-site/.../reference/configuration/routing.md` field table: `decisionModel`. CLI reference `--decision-model`. +- `structure/providers-and-adapters.md` JEV contract, `structure/runtime.md` combo dispatch, management/usage owners for the new routes and backend aggregation; `bun run structure:check`. +- `skills/ocx` surface regenerate if the capability registry changes. + +## Accept + +`cd gui && bun test tests && bun run lint && bun run lint:i18n && bun run build`; root `tests/gui`; `bun run structure:check`; `bun run skill:surface:check`; screenshot of the section captured for the PR. + + +## Resume plan (2026-10-01, fork session 01a0f655) + +State at resume: wp1 and wp2 are committed (`a4dd92baac`, `36c444fb5c`). The wp3 GUI section, data helpers, overview/stats wiring, ten-locale keys, and the combos guide rewrite are present but uncommitted. The maintainer forbids local test suites and typechecks for this lane, so the wp3 accept list moves to hosted CI on the wp4 head; nothing below is run locally. + +Remaining wp3 work: + +1. `combo-workspace-jev-decision.tsx`: restore the accessibility contract the removed `JevDecisionFields` carried. The System One select gets `aria-describedby` pointing at a `-decision-provider-hint` paragraph (service hint plus base URL) and, when the stored row is unusable, at a `-decision-provider-issue` paragraph. The model input gets its own `-decision-model-hint`. The method hint keeps the TypeSafe default text and the empty-server text. +2. `gui/tests/jev-decision-fields.test.tsx`: move the four existing cases to the method buttons (System One select lists only server rows; switching to TypeSafe is a method click; an unusable deep link lands on the TypeSafe method with no select), and add a model-method case: pick a route, save sends `decisionModel` with `decisionProvider: null`, the combo itself and JEV combos are absent from the route list. +3. `tests/gui/combo-workspace-jev-decision.test.ts`: `decisionModel` parse/PUT round trip, provider/model mutual exclusion in `toPutBody`, `invalidDecisionModel` for empty, self, and JEV routes, `jevDecisionMethod`, and `jevDecisionModelOptions` exclusions. +4. `gui/tests/jev-stats-panel.test.tsx`: a per-backend row renders when the payload carries `backends` and is absent for an older server. +5. Docs: `reference/cli/agents.md` gains `--decision-model ` and `ocx combo test`; `structure/providers-and-adapters.md` JEV section names the backend seam (`jev-dispatch.ts`, `jev-model-backend.ts`, `jev-decision-contract.ts`, `server/responses/jev-model-invoke.ts`), the recursion refusal, the decision-test/discovery routes, and replaces the "no JEV-only editor" sentence with the Decision method section. +6. Commit wp3 in two commits (GUI + tests, docs + structure) on `codex/jev-decision-routing`. + +Verification moves to wp4: exact-head hosted CI (GUI tests, lint, i18n, build, structure:check, privacy scan, full suite shards). + +### Audit round 1 amendments (VERDICT: FAIL → plan amended) + +7. `tests/gui/combo-workspace-jev-decision.test.ts` summary assertions (`jevDecisionSummary` exact objects) gain `model: null`, plus a model-backed summary case. +8. `JEV_BACKEND_LABEL_KEYS` moves out of the TSX component into `gui/src/jev-decision-service.ts` (oxlint `only-export-components` rejects a non-primitive constant export beside a component); the component and `jev-stats-panel.tsx` import it from there. +9. `jevDecisionModelForbidden` strips trailing synthetic selectors before comparing, mirroring `normalizeDecisionModelSelector`: one trailing `--fast` and one trailing `--` (`none|minimal|low|medium|high|xhigh|max|ultra`), in either order. The server stays authoritative; the GUI check only prevents an obviously refused save. Regression cases cover `combo/self--fast` and a JEV alias with `--high`. +10. `reference/configuration/routing.md`: add the `decisionModel` row (mutually exclusive with `decisionProvider`, recursion refusal), reword the credential sentence so only the TypeSafe method needs the `jev` key, and repoint the link to `/guides/combos/#decision-method`. +11. Scope change, recorded: the section ships without discovery hints and without its own recent summary. Recent statistics, including the new per-backend table, stay in the detail panel's Stats tab (`JevStatsPanel`), and discovery stays an API/CLI surface (`GET /api/combos/decision-discovery`). Neither is required by the maintainer brief; both can follow as a GUI-only change. + +### Audit round 2 amendment + +9 (revised). Exact selectors win first, as on the server (`fast-row.ts` known-id guard, `effort-row.ts`): when the route equals some combo's `combo/`, alias, or model exactly, only that combo decides (refused if it is the edited combo or a JEV combo) and nothing is stripped. Otherwise strip exactly one trailing synthetic marker, either `--fast` or `--`, never both (the server refuses composed Fast+effort selectors), and re-check. Regression cases: a non-JEV combo literally named `self--fast` stays allowed while `self` is JEV; `combo/self--fast` and `--high` are refused; `combo/x--fast--high` is not stripped twice. + +### Audit round 3 amendment + +9 (final). The GUI check is conservative: it may only refuse what the server certainly refuses. A route is never stripped when it exactly equals any known selector, meaning any combo's `combo/`, alias, or model, or any catalog route the section receives (physical model ids, namespaced ids, and model aliases as listed by `/api/models`). Only an unknown route ending in exactly one synthetic marker is stripped once and re-checked against combos. Regression cases add a catalog model literally named `router--high` that stays allowed while `router` is a JEV alias. Anything the GUI cannot decide is left to the server's save-time `decisionModelRouteError`, whose message the dashboard already surfaces. + +### Audit round 4 amendment + +9 (final, simplified). The GUI does no suffix stripping at all. It refuses only an exact match: the route equals the edited combo's `combo/`, alias, or model, or the same of any JEV combo. Every suffixed or otherwise uncertain selector is left to the server's save-time `decisionModelRouteError`, which knows whether Fast and effort parsing are active, and the dashboard surfaces that error on save. The GUI can therefore never refuse more than the server. Regression cases: exact self and exact JEV alias are refused in the GUI; `combo/self--fast`, `router--high` (with `router` a JEV alias), and a catalog model named `router--high` all pass the GUI check. diff --git a/devlog/_plan/261001_jev_decision_routing/040_wp4_landing.md b/devlog/_plan/261001_jev_decision_routing/040_wp4_landing.md new file mode 100644 index 00000000000..7e0dfd3118d --- /dev/null +++ b/devlog/_plan/261001_jev_decision_routing/040_wp4_landing.md @@ -0,0 +1,9 @@ +# 040 wp4: landing + +1. `bun run typecheck`, full `bun run test` (documented resource exception only), `bun run privacy:scan`, `bun run structure:check`, both test-layout guards, file-size ratchet, core-lab boundary, GUI gates. +1a. Dispatch an independent security review of the final diff (credential forwarding, outbound policy, env keys); record its verdict in the PR Verification section. +2. Push `codex/jev-decision-routing`; upload screenshot to `pr-assets` branch, link by commit SHA. +3. `gh pr create --base dev` with every template section; `Co-authored-by` trailers for SeongwoongCho, yxr1995-maker, codingbooo in the description; `Closes #6268` style link (manual close note since base is dev). +4. Wait for required CI on the exact head SHA; fix and repush on failure. +5. After green: comment+close #6185, #6275, #6302 with link and thanks; comment on #6348 (already closed) asking for a rebase of level/quota mode onto the new PR; comment on #6268 linking the PR. +6. Report PR URL, head SHA, CI run URL, closed PRs, residual risks. diff --git a/devlog/_plan/261001_quota_send_lock_split/000_plan.md b/devlog/_plan/261001_quota_send_lock_split/000_plan.md new file mode 100644 index 00000000000..4ecc1dcdaee --- /dev/null +++ b/devlog/_plan/261001_quota_send_lock_split/000_plan.md @@ -0,0 +1,94 @@ +# Quota send-lock contributor split: roadmap + +Status: roadmap locked by the docs-only cycle; implementation runs one PABCD cycle per PR. +Date: 2026-10-01. Baseline: `origin/dev` `64294638a69e25ca0c7a4e2102e2349973161f71`. + +## Problem + +The Codex desktop composer disables Send when the signed-in ChatGPT quota reports +exhausted, even for models opencodex routes to independently funded providers +([#6196](https://github.com/lidge-jun/opencodex/issues/6196), duplicate +[#4878](https://github.com/lidge-jun/opencodex/issues/4878)). Two contributor PRs +address it: + +- [#5947](https://github.com/lidge-jun/opencodex/pull/5947) (lcxhh521, head + `61947c04d9`, +7317 / 54 files) mixes three mechanisms: (a) a local-CA TLS + intercept of chatgpt.com traffic (send-unblock), (b) a PAC fallback entry so the + intercept survives opencodex stopping, (c) an app-server shim: a `CODEX_CLI_PATH` + launcher that execs the bundled `codex app-server` and filters only its stdout, + clearing the plain-quota gate in `account/rateLimits/read` and + `account/rateLimits/updated`. +- [#5879](https://github.com/lidge-jun/opencodex/pull/5879) (MateuszJuszczyk, head + `9a5b6d9d72`) toggles `codexDesktopAuthless` automatically on quota exhaustion. + +## Evidence + +- #6196 (TooSpace, 2026-09-27..29): with (a)+(b) deployed, the loopback listener saw + **zero established connections over ~20 h**; the bundled app-server owns the + chatgpt.com sockets, so the intercept never sees the gate reads. +- #6196 (TooSpace, 2026-10-01 10:47 Asia/Taipei): Plus account, 5-hour window 100%, + weekly 51%, Desktop 26.928.21956 / bundled CLI 0.159.2: launching through the shim + launcher re-enabled Send before the window reset; `dynamic_app_tools_peer_rejected` + did not recur. Ingwannu's note: the run had intercept + shim + restart together, + so it does not isolate the shim as sole cause. +- `devlog/_plan/260928_macos_quota_gate/000_design.md` (maintainer design) prefers + upstream provider-aware admission and rejects CA installation and quota rewriting. + The split PRs below are explicitly experimental/opt-in candidates the maintainers + may close against that decision. + +## Existing defence and its gaps + +`codexMainAccountHardLock` (default on since #5694, `src/codex/main-account-hard-lock.ts`) +stops opencodex from sending main-account traffic at 98% on any governing window. +Gaps: usage outside opencodex, observation lag (a long turn crosses 98%), the small +Plus 5-hour window, and `unknown` state admitting. + +## PR map (dependency order) + +| Doc | Work-phase | Branch | Base | Credit | +| --- | --- | --- | --- | --- | +| 010 | wp1 hard-lock hardening | `codex/main-hard-lock-window-thresholds` | `dev` | none (maintainer code) | +| 020 | wp2 app-server shim | `codex/chatgpt-app-server-shim` | `dev` | Co-authored-by: lcxhh521 | +| 030 | wp3 local-CA send-unblock intercept | `codex/chatgpt-send-unblock-intercept` | wp2 branch (stacked) | Co-authored-by: lcxhh521 | +| 040 | wp4 PAC fallback | `codex/chatgpt-pac-fallback` | wp3 branch (stacked) | Co-authored-by: lcxhh521 | +| 050 | wp5 #5879 disposition | none (rationale only) | - | - | +| 060 | wp6 closeout | - | - | - | + +Stacking reason: wp2, wp3 and wp4 all extend the same new `ocx chatgpt` command and +`chatgptDesktop` config block. Opening them independently against `dev` would +make the second one to land conflict on `src/cli/chatgpt-command.ts`, the config +type/validator, the CLI registry, the guide and `structure/clients/chatgpt-desktop.md`. +The PAC fallback has no function without the TLS listener (its CONNECT entry only +splices into that listener), so wp4 cannot target `dev` alone. The shim has no +dependency on either and is the piece with field evidence, so it sits at the bottom. +Stack lifecycle: + +- Parent merged (squash): rebase the child with + `git rebase --onto origin/dev `, retarget it to `dev`, and + wait for fresh CI on the new head. Cascade the same rebase to every descendant + (wp4 onto the new wp3 head) and wait for fresh CI on each. Land bottom-up only. +- Parent closed: its children carry its commits and dependencies, so they close with + it unless a maintainer asks for a reconstruction on a different base. The children + are themselves "may close" candidates, so this is the expected path if the shim is + rejected. + +wp1 is independent of all three. + +## Common constraints + +- New branches from latest `origin/dev`; `codex/` prefix; no git config, no GIT_* env. +- Respect `tests/fixtures/file-size-baseline.json`; unlisted files stay below 2000 lines. +- New test files: register in `scripts/test-layout/layout.json` `explicit` and + `tests/fixtures/test-layout-expected.json`. +- `src/router.ts`, `src/server/lifecycle.ts`, `src/server/responses/core.ts` must not + reach new optional modules (core-lab boundary). +- GUI strings in all ten locale catalogs; GUI PRs need a screenshot on `pr-assets`. +- Verification per PR: `bun run typecheck`, focused tests, `bun run test:changed`, + `bun run structure:check`, `bun run privacy:scan`, `bun run skill:surface:check` when + the CLI changes, the core-lab boundary test, and the full `bun run test` (or the + documented resource exception recorded in the PR), then required CI green on the + exact pushed head. +- wp2-wp4 touch process exec and credential interception: PR descriptions request + explicit security review per MAINTAINERS.md. +- Merge/close decisions belong to maintainers; this unit opens PRs only. + diff --git a/devlog/_plan/261001_quota_send_lock_split/010_hard_lock_hardening.md b/devlog/_plan/261001_quota_send_lock_split/010_hard_lock_hardening.md new file mode 100644 index 00000000000..0136468734c --- /dev/null +++ b/devlog/_plan/261001_quota_send_lock_split/010_hard_lock_hardening.md @@ -0,0 +1,141 @@ +# 010 — wp1: window-aware hard-lock thresholds and outside-usage warning + +Branch `codex/main-hard-lock-window-thresholds` from `origin/dev`. Maintainer code. + +## Behaviour + +1. The 5-hour (short) window locks at a lower percentage than the weekly/monthly + (long) windows. Defaults: short **90**, long **98** (unchanged). Both configurable. +2. When the main account's usage rises while opencodex has not used the main + account, the dashboard and `ocx status` show a warning: usage is being consumed + outside opencodex, so the lock may not prevent exhaustion. + +## Config (NEW key) + +`src/types/config.ts` next to `codexMainAccountHardLock`: + +```ts +/** Per-window lock thresholds in percent (#6196). short: 5h window, default 90; long: weekly/monthly, default 98. */ +codexMainAccountHardLockThresholds?: { short?: number; long?: number }; +``` + +`src/config/schema/config-schema.ts`: object of two optional integers in +`[MAIN_ACCOUNT_HARD_LOCK_MIN_PERCENT, 100]`, `.catch(undefined)` per field so a +malformed disk value degrades to defaults (same pattern as the boolean at :188). + +`PUT /api/settings` (`src/server/management/config-routes.ts`): accept the object, +reject non-integer/out-of-range values and `short > long` with 400 before any write, +include it in the rollback snapshot; GET returns the effective thresholds in the +existing hard-lock projection. + +## Constants (MODIFY `src/codex/quota-types.ts`) + +```ts +export const MAIN_ACCOUNT_HARD_LOCK_PERCENT = 98; // long default, unchanged export +export const MAIN_ACCOUNT_HARD_LOCK_SHORT_PERCENT = 90; // NEW short default +export const MAIN_ACCOUNT_HARD_LOCK_MIN_PERCENT = 80; // NEW lowest configurable value +``` + +## Lock (MODIFY `src/codex/main-account-hard-lock.ts`) + +- NEW `resolveMainAccountHardLockThresholds(config): { short: number; long: number }` + — defaults above, clamps invalid in-memory values back to defaults, forces + `short <= long`. +- `WindowReading` gains `kind: "short" | "long"`; `governingWindows()` tags the 5h + reading short, weekly and monthly-only long. +- `getMainAccountHardLockStatus()` compares each window with its own threshold and + returns `thresholds` plus `window: "short" | "long"` of the first blocking reading. +- `PolicyConfig` adds `codexMainAccountHardLockThresholds`. + +## Evidence retention (MODIFY `src/codex/quota.ts`) + +The three policy-evidence retention guards (`assignCarriedShort` ~:263, +`preserveKnownWeekly` ~:374, `preserveKnownShort` ~:407) compare against 98. With a +configurable threshold a reading between the threshold and 98 could be dropped by an +elapsed reset or partial update and silently unlock. Change the comparison to +`MAIN_ACCOUNT_HARD_LOCK_MIN_PERCENT` so any reading that can block under some +allowed configuration is retained until a fresh lower reading arrives. Retaining a +non-blocking reading only keeps a stale percentage visible; it never blocks. + +## Error copy (MODIFY `src/codex/auth-context.ts` ~:517) + +`CodexMainAccountHardLockError` message stops naming 98: "Codex main account is +blocked by the main-account quota policy (5h ≥ X%, weekly ≥ Y%)." The constructor +takes the thresholds from the status. Update tests that assert the old string. + +## Outside-usage detector (NEW `src/codex/main-account-external-usage.ts`) + +Process-local, no persistence, no timers. It keeps its own baseline of fresh readings +and never reads the merged policy snapshot, which can contain carried windows. + +```ts +export const EXTERNAL_USAGE_QUIET_MS = 30 * 60_000; +export const EXTERNAL_USAGE_TTL_MS = 6 * 60 * 60_000; +type FreshWindow = { kind: "short" | "long"; percent: number; resetAtMs: number }; +export function noteMainAccountActivity(now = Date.now()): void; +/** Only windows with a validated percent AND a known reset in this observation. */ +export function observeMainAccountUsage(identityKey: string, windows: FreshWindow[], now = Date.now()): void; +export function forgetMainAccountUsage(): void; +export function getMainAccountExternalUsageWarning(identityKey: string | undefined, now = Date.now()): + { window: "short" | "long"; fromPercent: number; toPercent: number; observedAt: number } | undefined; +export function resetMainAccountExternalUsageForTests(): void; +``` + +Rule: the module holds `{ identityKey, perWindow: { percent, resetAtMs, observedAt } }`. +A new `identityKey` replaces the baseline and drops any warning. For a fresh window +with the same kind and the same normalized `resetAtMs` (both known; a missing reset +never proves the same episode), a rise of at least 1 point with no main-account +activity noted since `baseline.observedAt - EXTERNAL_USAGE_QUIET_MS` records a warning. +A changed reset rebaselines that window and clears its warning. Warnings expire after +`EXTERNAL_USAGE_TTL_MS` or once the recorded `resetAtMs` passes, whichever is first; the getter returns nothing for a different identity. +Over-counting activity only suppresses warnings. A single turn longer than the quiet +margin can still cause a false positive, so the copy says "possible usage outside +opencodex". + +Hooks: + +- `src/codex/auth-context.ts` `assertMainAccountPolicy()` (:703) runs on every + main-credential attach (main-pool, substituted main, caller bearer matching observed + main) and on some admission checks before refusal. Call `noteMainAccountActivity()` + there before the `!config` early return; the extra calls only suppress warnings. +- `src/codex/quota.ts` `setAccountQuotaFromParsed()` (~:318): when `isMain`, + `mainWriter` is present and `policyQuota` (this call's validated observation) is + non-null and carries usage, build `FreshWindow`s from `policyQuota` only (short: + `shortPercent` + `shortResetAt`; long: `weeklyPercent` + `weeklyResetAt`, or monthly + when `monthlyIsPrimaryWindow`), normalize with `resetAtToMs`, and call + `observeMainAccountUsage(mainWriter.identityKey, windows)`. +- `clearAccountQuota()` (main or all) calls `forgetMainAccountUsage()`. +- Readers pass `getObservedMainQuotaIdentityKey()` to the getter. + +## Surfaces + +- `src/codex/auth-api/account-list.ts` (~:374): the main-account hard-lock DTO gains + `thresholds` and optional `externalUsage`. +- GUI `gui/src/components/MainAccountHardLockSetting.tsx` and + `codex-account-pool-main-card.tsx`: copy uses the effective thresholds; amber notice + when `externalUsage` is present. Locale keys in all ten catalogs; replace fixed-98 + strings (`en.ts:2312-2325`). +- `ocx status`: `src/oauth/health.ts` `fetchCodexHealthFromLiveProxy()` already reads + `/api/codex-auth/accounts`; project the main hard-lock state, thresholds and + external-usage warning, and print one line in `src/cli/status-oauth.ts` (human) and + include it in JSON. Keep `src/cli/index.ts` (1979 lines) untouched. + +## Tests (NEW sibling files; register in both layout files) + +- `tests/codex-integration/main-account-hard-lock-thresholds.test.ts`: short 90 blocks + at 90 with weekly 30; weekly 95 admits by default and blocks with long 95; + configured short/long; invalid config falls back; retention of a 92% short reading + across an elapsed reset. +- `tests/codex-integration/main-account-external-usage.test.ts`: rise without + activity warns; activity inside the quiet margin suppresses; reset change clears; + identity change clears; expiry. +- `tests/config/settings-main-account-hard-lock.test.ts` (MODIFY, 119 lines): PUT + validation and rollback for thresholds. +- GUI test for the warning notice next to the existing hard-lock setting tests. + +## Docs + +`docs-site/src/content/docs/reference/cli/providers-accounts.md` (+ Korean), +`docs-site/src/content/docs/reference/configuration/server.md` key row, +`structure/providers/openai-accounts.md`, `structure/config.md`. + diff --git a/devlog/_plan/261001_quota_send_lock_split/020_app_server_shim.md b/devlog/_plan/261001_quota_send_lock_split/020_app_server_shim.md new file mode 100644 index 00000000000..f1e99800dba --- /dev/null +++ b/devlog/_plan/261001_quota_send_lock_split/020_app_server_shim.md @@ -0,0 +1,94 @@ +# 020 — wp2: app-server shim (experimental, macOS, default off) + +Branch `codex/chatgpt-app-server-shim` from `origin/dev`. Code from #5947 with +`Co-authored-by: lcxhh521 ` (use the author's +commit email from `git log pr/5947` if public). + +## Scope + +Only mechanism (c). No listener, CA, PAC, launch watcher, or `unblockSend`. + +## Files + +NEW `src/chatgpt/app-server-shim/`: + +- `gate-rewrite.ts` — pure copy of #5947 `rewrite.ts` `isRecord`, `PLAIN_QUOTA_REACHED_TYPE`, + `hasNonQuotaBlock`, `unlockRateLimitGate` (pr/5947 rewrite.ts:40-47, 157-246). + No conversation/endpoint/SSE code. +- `app-server-rewrite.ts` — #5947 file, import from `./gate-rewrite`. +- `filter.ts` — #5947 `app-server-shim.ts` line filter, without its `import.meta.main` + entry. NEW `runChatgptAppServerFilter({ selfTest })`: with `selfTest` it rewrites a + built-in fixture line, checks the result and exits 0; otherwise it filters stdin to + stdout. If the rewrite machinery throws outside the per-line guard, it switches to + raw byte passthrough for the rest of the stream. +- Entry point: hidden `ocx internal chatgpt-app-server-filter [--self-test]` in + `src/cli/internal-command.ts` (hidden commands stay out of the registry and skill + surface by design). This works for source installs and compiled standalone builds, + because the launcher re-enters the current CLI through `process.execPath` + + `selfLaunchArgv()` (`src/lib/self-launch-argv.ts`) instead of pointing Bun at a `.ts` + file that a compiled build does not ship. +- `launcher.ts` — `buildChatgptShimLauncher(argv, real)`, `writeChatgptShimLauncher()`, + `chatgptShimLauncherPath()` (adapted from #5947 runtime.ts:123-161). Launcher body: + +```bash +#!/bin/bash +# opencodex (experimental): ChatGPT app-server stdout passes through the quota-gate filter. +REAL='' +FILTER=('' [''] internal chatgpt-app-server-filter) +if [ "$(uname -s)" = "Darwin" ] && [ -x "${FILTER[0]}" ] \ + && "${FILTER[@]}" --self-test >/dev/null 2>&1; then + exec "$REAL" "$@" > >(exec "${FILTER[@]}") +fi +exec "$REAL" "$@" +``` + + Guarantee, stated exactly: a failed precondition or a failed self-test runs the + original binary with untouched stdout. A filter that passes the self-test and then + dies mid-session closes the pipe. Expected (not yet validated against the bundled + app-server): the server gets SIGPIPE or a write error and Desktop respawns it through + the same launcher. The filter's passthrough mode limits this to an exit/crash case. + +NEW `src/cli/chatgpt-command.ts` — `ocx chatgpt launch | restore | status` (macOS only; +other platforms print "macOS only" and exit 1). `launch` refuses unless +`chatgptDesktop.appServerShim === true`, writes the launcher, quits the app if +running, and relaunches with `open -a ChatGPT --env CODEX_CLI_PATH=`. +`restore` relaunches without the variable and deletes the launcher. `status` prints +`app-server shim (experimental): on|off`, launcher presence and whether a running +app carries the variable. All help text says "experimental". + +MODIFY `src/cli/dispatch.ts`, `src/cli/help.ts`, `src/cli/registry.ts` +plus `src/cli/capabilities.ts` and the regenerated skill surface (`bun run skill:surface`). + +Config: `src/types/config.ts` `chatgptDesktop?: { appServerShim?: boolean }`; +`leaf-validators.ts` strict object, `config-schema.ts` registration with +`.catch(undefined)`; `src/config/diagnostics.ts` warns when set on non-macOS. + +## Tests + +NEW `tests/chatgpt-unblock/app-server-shim.test.ts`: gate rewrite cases from #5947 +`unblock-app-server-shim.test.ts` (re-pointed to the new modules), self-test exit code, +passthrough after an internal rewrite failure. +NEW `tests/chatgpt-unblock/app-server-shim-launcher.test.ts`: launcher text for source +argv and compiled argv (`selfLaunchArgv` with `isStandaloneExecutable`); executing the +generated script on a stub `REAL` with (a) missing runtime, (b) failing self-test → +stub sees untouched stdout, (c) passing self-test → stub output filtered, (d) passing +self-test then filter exits → stub terminates instead of hanging (mock behaviour only). +NEW `tests/chatgpt-unblock/chatgpt-desktop-config.test.ts`: absent/false/malformed → off, +strict write rejection. CLI registry/capabilities parity via existing tests after +`bun run skill:surface`. Register new files in both layout files. + +## Docs + +`docs-site/src/content/docs/guides/chatgpt-desktop.md` (English, marked experimental, +security notes: generated executable, `CODEX_CLI_PATH`, quota gate fields rewritten +while displayed usage stays honest), sidebar entry in `astro.config.mjs`; +`structure/clients/chatgpt-desktop.md` + `structure/manifest.json` ownership of +`src/chatgpt/`; `structure/INDEX.md` via `bun run structure:index`. + +## PR description must state + +The 260928 maintainer design rejects rewriting quota gate data; this PR is offered +as an opt-in experiment because it is the only path with field evidence on a real +exhausted Plus account (#6196, 2026-10-01), and that run also had the intercept +enabled. + diff --git a/devlog/_plan/261001_quota_send_lock_split/030_send_unblock_intercept.md b/devlog/_plan/261001_quota_send_lock_split/030_send_unblock_intercept.md new file mode 100644 index 00000000000..a614ccc979b --- /dev/null +++ b/devlog/_plan/261001_quota_send_lock_split/030_send_unblock_intercept.md @@ -0,0 +1,41 @@ +# 030 — wp3: local-CA send-unblock intercept (candidate) + +Branch `codex/chatgpt-send-unblock-intercept`, stacked on wp2. Code from #5947, +`Co-authored-by: lcxhh521`. A candidate maintainers may close. + +## Files (from pr/5947 at 61947c04d9) + +Whole files: `src/chatgpt/desktop-unblock/{listener,ca-trust,ws-frame,ws-relay,ws-upstream,launch-watcher,runtime,rewrite}.ts`, +`src/server/index/chatgpt-unblock-lifecycle.ts`, `src/server/index/optional-listeners.ts` hunk, +`src/lib/socks5-handshake.ts` + `src/lib/socks5-fetch.ts` (shared SOCKS5 refactor), +`tests/lab/core-lab-boundary.test.ts` hunk. + +Removed while extracting: + +- PAC: runtime.ts:32,49-50,84-121,166-169,177-181,204-221; launch-watcher PAC branches + (102,114-116,146-165,177-179,190-203,239-242,271-277,472-475); CLI PAC status. +- Shim duplicates: runtime.ts:44-45,123-161,222-231 and watcher shim branches; the + shim lives in wp2. `rewrite.ts` imports the gate helpers from + `src/chatgpt/app-server-shim/gate-rewrite.ts` instead of redefining them. + +Config: `chatgptDesktop.unblockSend`, `chatgptDesktop.port` added to wp2's block. +CLI: `ocx chatgpt launch|restore|status|install-watcher` gain the intercept mode. + +Tests: `rewrite`, `unblock-ca-trust`, `unblock-listener`, `unblock-ws-frame`, +`unblock-ws-relay`, `unblock-runtime` (non-PAC cases), `unblock-launch-script` +(non-PAC, non-shim), `unblock-watcher-install`, `unblock-config-boundary`, +`tests/lib/socks5-handshake.test.ts`. Register in both layout files. + +Docs: guide sections for the intercept in English (+ locales from #5947 where they do +not describe PAC/shim), structure updates, transports inventory SOCKS5 row. + +## PR description must state + +- Security cost: issues a local CA and asks the user to run + `security add-trusted-cert` into the login keychain trust store; the loopback + listener terminates TLS for chatgpt.com and relays the account's credentials. +- #6196 evidence: on current Desktop builds the gate reads come from the bundled + app-server; the listener saw zero established connections in ~20 h. +- Conflicts with the 260928 maintainer design; offered only so the decision can be + made on a reviewable diff. + diff --git a/devlog/_plan/261001_quota_send_lock_split/040_pac_fallback.md b/devlog/_plan/261001_quota_send_lock_split/040_pac_fallback.md new file mode 100644 index 00000000000..97cda739ea6 --- /dev/null +++ b/devlog/_plan/261001_quota_send_lock_split/040_pac_fallback.md @@ -0,0 +1,19 @@ +# 040 — wp4: PAC fallback (candidate) + +Branch `codex/chatgpt-pac-fallback`, stacked on wp3. Code from #5947, +`Co-authored-by: lcxhh521`. A candidate maintainers may close. + +Adds `src/chatgpt/desktop-unblock/{pac,entry-proxy}.ts`, restores the PAC hunks +removed in 030 (runtime, launch-watcher, lifecycle logging, CLI status), config +`chatgptDesktop.pacFallback`, tests `unblock-pac`, `unblock-entry-proxy`, PAC cases of +`unblock-runtime`, `unblock-launch-script` (314-325, 373-426), `unblock-watcher-install` +(142), and the guide's PAC section (English 92-120 plus locales). + +End state: wp2+wp3+wp4 tree equals #5947's feature set with the shim relocated. + +## PR description must state + +Same security cost as 030 (requires the trusted local CA) plus: the PAC embeds the +system proxy/PAC configuration captured at start, and #6196 measured zero traffic on +both the TLS listener and the CONNECT entry. + diff --git a/devlog/_plan/261001_quota_send_lock_split/050_pr5879_disposition.md b/devlog/_plan/261001_quota_send_lock_split/050_pr5879_disposition.md new file mode 100644 index 00000000000..492342ae63b --- /dev/null +++ b/devlog/_plan/261001_quota_send_lock_split/050_pr5879_disposition.md @@ -0,0 +1,26 @@ +# 050 — wp5: #5879 auto desktop-authless disposition + +Decision: **no carried PR; rationale only.** #5879 is closed with credit after the +split PRs are up. + +Reasons (read-only review against `64294638`): + +1. Requirement conflict: #6196's reporter requires the signed-in account to stay in + the UI. Authless removes the account plane (usage, Fast, plugin/thread identity), + and the 260928 design rejects `requires_openai_auth = false` as the fix. +2. Disruption: every transition calls `performCodexRestart()`, restarting Desktop and + app servers from a background sweep without user consent. +3. Recovery proof is unsound on today's API: `refreshCodexQuotaForActivation()` + returns silently for unavailable leases/reauth/missing credentials and + `isCodexQuotaExhausted(null)` is false, so unknown state releases authless. +4. Persists before injecting, restarts even when injection fails, and the next sweep + sees the stored state and never retries; overwrites the manual + `codexDesktopAuthless` preference. +5. Overlap: wp1 lowers the short-window threshold so opencodex traffic leaves the + main account earlier, and wp2 addresses the client gate directly without leaving + the signed-in mode. + +A revised proposal would need authoritative recovery evidence, a separate automatic +state that never overwrites the manual preference, retry of failed application, and +deferred restarts. + diff --git a/devlog/_plan/261001_quota_send_lock_split/060_closeout.md b/devlog/_plan/261001_quota_send_lock_split/060_closeout.md new file mode 100644 index 00000000000..410a0bd4eb6 --- /dev/null +++ b/devlog/_plan/261001_quota_send_lock_split/060_closeout.md @@ -0,0 +1,12 @@ +# 060 — wp6: closeout + +After every split PR is open and its required CI is green on the exact head: + +1. Comment on #5947: links to wp2/wp3/wp4 PRs, which part went where, the + `Co-authored-by` trailer, thanks; then close. +2. Comment on #5879: link to 050 rationale (in the wp1 PR), credit, thanks; close. +3. Comment on #4878 linking #6196 and the split PRs; close as duplicate. +4. Report PR links, head SHAs, CI runs, closed PRs and residual risks. + +No merges, releases or promotions. + diff --git a/devlog/_plan/261001_zed_hosted_uayor_carry/010_finish_carry.md b/devlog/_plan/261001_zed_hosted_uayor_carry/010_finish_carry.md new file mode 100644 index 00000000000..a531b8813de --- /dev/null +++ b/devlog/_plan/261001_zed_hosted_uayor_carry/010_finish_carry.md @@ -0,0 +1,73 @@ +# 010 — Finish the Zed Hosted provider carry (#5912) + +Unit: carry andrew05060414's experimental Zed Hosted AI provider (#5912) onto `dev` as a +maintainer PR. Maintainer decision (2026-10-01): adopt as use-at-your-own-risk (UAYOR). + +## State at entry + +Branch `codex/zed-hosted-ai-carry`, 5 commits ahead of `origin/dev`, 0 behind, already pushed +at `03e9dce881`. Carried: native RSA callback login (`src/oauth/zed.ts`), account-scoped +short-lived LLM token exchange with expiry refresh (`src/providers/zed.ts`), live roster, the +`/completions` envelope adapter for Anthropic / Google / OpenAI Responses / xAI Chat +(`src/adapters/zed.ts`), UAYOR wording in the dashboard ToS-risk set (`gui/src/oauth-tos-risk.ts`) +and in `docs-site` guides/providers, reference/adapters, reference/configuration/providers. + +## Remaining diff (this cycle) + +1. `src/providers/zed.ts` — `scrubZedCredentials()` removes the account access token and the + user id from upstream error text before it becomes an `Error` message, on the two calls that + send the account credential (account lookup, LLM token exchange). `responseJson` takes the + credentials. The `/completions` path carries only the short-lived LLM token and keeps the + generic `redactSecretString`. +2. `tests/providers/zed-hosted-provider.test.ts` — regression: a 401 body echoing both values + yields a message that names the step and contains neither value. Trailing newline restored. +3. Commit with `Co-authored-by` for andrew05060414; push (no force). +4. Open the maintainer PR to `dev` with every template section, `Co-authored-by`, the UAYOR + framing, an explicit security-review request (OAuth / credential path per MAINTAINERS.md), + and real-account UAT stated as not verified. + +## Verification + +Local test and typecheck runs are banned for this lane (maintainer instruction). Evidence is +the exact-head hosted CI of the PR; queued, skipped, or cancelled checks are not passes. +Failures are read with `gh run view --log-failed`, fixed, and pushed again. + +## Closeout + +When exact-head CI is green, close #5912 with a link to the new PR and thanks. Do not merge. + +## Reflection + +The remaining risk is reviewer-side, not mechanical: Zed's service terms. That is why the PR +asks for security review and states UAYOR instead of claiming support. No change to core +request paths (`src/router.ts`, `src/server/lifecycle.ts`, `src/server/responses/core.ts`). + +## Audit round 1 (FAIL) — folded into this plan + +Reviewer (gpt-6.1-sol, read-only) found three P1 blockers; all are accepted. + +5. **Rejection boundaries.** `fetchZedAuthenticatedUser` and `fetchZedLlmToken` can reject from + `fetchFn` or from the body read inside `responseJson` with a message that echoes the account + credential, bypassing `scrubZedCredentials`. Add `scrubZedRejection(error, credentials)`, which + decides only on content: if the error's message (or `String(error)` for a non-Error) contains + neither the account token nor the user id, rethrow the original by identity, so a clean abort + keeps its identity. If it contains either, throw a replacement that never carries the value: a + `DOMException` with the same `name` and the scrubbed message when the original is a + `DOMException` (so `AbortError`/`TimeoutError` cancellation semantics survive), otherwise an + `Error` with the scrubbed message and the same `name`; numeric `status` is copied over and + the original is dropped as `cause` (its message would leak). The `await fetchFn(...)` and + `await responseJson(...)` calls of both functions sit inside the `try` whose `catch` applies + it, and so does the `/completions` fetch inside `zedLlmFetch`. Regression cases: fetch + rejection echoing both values; body-read rejection (a stream that errors mid-read) echoing them; + numeric `status` preserved; a credential-bearing `AbortError` comes back as an `AbortError` + without the values; a clean `AbortError` is rethrown by identity. +6. **File-size ratchet.** `src/server/management/provider-routes.ts` is 1998 lines on `dev` with no + baseline cap; the carried 22-line Zed probe takes it to 2020, past the 2000 NEW_OVERSIZED + threshold. Move the probe into `src/server/management/zed-provider-probe.ts` + (`probeZedProvider(prov, apiKey, accountId)`) and call it with a single dynamic-import line, so + the file ends at 1999 lines and Zed code stays off the module graph until a Zed probe runs. +7. **Preset counts.** The registry grows from 100 to 101 presets (Zed is OAuth; key-based stays 83). + Update the anchored sentence in all eight provider guides (total 101, OAuth 14), all eight + quickstarts (101), and `structure/ops/docs-and-release.md` (101 total, 14 OAuth), as checked by + `tests/ci-workflows/docs-provider-preset-counts.test.ts`. Union risk: another open lane (Mirasim) + also adds a preset; whichever lands second must recount, and the PR says so. diff --git a/devlog/_plan/261002_claude_cli_picker/010_plan.md b/devlog/_plan/261002_claude_cli_picker/010_plan.md new file mode 100644 index 00000000000..9509f86c892 --- /dev/null +++ b/devlog/_plan/261002_claude_cli_picker/010_plan.md @@ -0,0 +1,65 @@ +# 261002 Claude Code CLI first-party picker — plan (010) + +## Problem + +With `claudeCode.cliFirstParty` on, the standalone `claude` CLI talks to OpenCodex's intercept +proxy, but its `/model` picker only lists Anthropic models. Claude Desktop's picker mode already +injects opencodex routes into Desktop's claude.ai bootstrap catalog; the CLI has no equivalent. + +## Evidence (live probe, Claude Code 2.1.287, first-party OAuth, throwaway capture proxy in .tmp/) + +- The CLI builds `/model` from `GET https://api.anthropic.com/api/organizations//model_selector/cc` + (`/api/model_selector/cc` for token/api-key auth). Body: `{model_selector_state, model_selector_config: + [{id:"cc", models:[…Desktop-shaped rows…]}]}`. User-Agent `claude-cli/ (external, cli)`. +- Debug log: `[servedCatalog] primary: … the served list replaces the compiled picker`. +- Rows appended to the cc surface appear in `/model` and are selectable when the id is Claude-shaped: + `claude-opus-4-8-20260919`, `claude-opus-4-8-p1ab`, `claude-opus-4-8-k3x`, `…[1m]` all worked. + `gpt-6-luna`, `ocx-claude-native--gpt-6-sol`, `ocx-claude-xai--grok-4.7` render + "Update Claude Code to use this model" (unselectable). So the Desktop-picker `ocx-claude-*` ids cannot + be reused; the Desktop 3P registry ids (`activeDesktop3pAlias`) can. +- The CLI caches the served catalog ~1 h in `/cache/model-catalog/--cc.json` + and does not refetch while fresh, so a toggle needs that cache invalidated. +- When the served catalog flag is off, the CLI falls back to the compiled picker plus + `additional_model_options` from `GET /api/claude_cli/bootstrap` (UA `claude-code/`), shape + `{model, name, description, disabled_reason?}`. + +## Diff-level plan + +1. `src/claude/intercept/picker-bootstrap.ts`: `PickerModelEntry.description?`; `injectPickerModels(body, models, + explain?, options?: {surfaces?, strip?})` — defaults unchanged for Desktop (`ccd`,`code`); when an entry carries + a description it is written after stripping. +2. New `src/claude/intercept/cli-catalog.ts` (no heavy imports): `cliCatalogKind(method, path)`, + `rewriteCliCatalogBody(kind, text, models)` (cc surface via injectPickerModels with surfaces `["cc"]` and extra + strip `notice`, `selection_notice`; bootstrap appends `additional_model_options` rows not already present), + `rewriteCliCatalogResponse(response, kind, models)` (2xx JSON only, size cap, drops content-length/etag, + untouched bytes on any failure), `cliCatalogEligible(kind, userAgent, desired)` (model_selector: client's own + intent via interceptRouteFor; bootstrap: `desired.cli` and not a Desktop entrypoint), + `invalidateClaudeCodeServedCatalog(claudeDir)` (unlink `*-cc.json` only). +3. `src/claude/intercept/picker-models.ts`: extract shared candidate/profile rendering; add + `buildCliPickerModels(input)`: skip real Anthropic routes, alias = `activeDesktop3pAlias`, keep only when + `resolveDesktop3pAlias(alias) === route` (never advertise an id the router cannot decode), 1M marker as in + Desktop, name = label, description `opencodex · `. `createPickerModelSnapshot(load, path, build?)` and the + persisted-snapshot parser accept the optional description. +4. `src/claude/intercept/listener.ts`: optional `cliCatalog(req, kind)` hook; for a matching GET, relay upstream + (same upstream choice as today) and rewrite when models are returned. Messages dispatch unchanged. +5. `src/claude/intercept/runtime.ts`: when picker routes and desired clients are wired, build a CLI snapshot + (`claude-intercept/cli-picker-models.json`), lazily refreshed (stale 5 min; first request waits ≤2.5 s), and pass + the hook. No timers, no discovery unless an eligible catalog request arrives. +6. `src/server/management/agent-settings-routes.ts`: after a successful `cliFirstParty` change, invalidate the CLI + served-catalog cache (best effort) so the next `claude` launch refetches through the proxy. +7. Tests: new `tests/claude-integration/claude-cli-picker.test.ts` (+ layout.json and test-layout-expected.json). +8. Docs: `docs-site/.../guides/claude-code.md` paragraph; `structure/runtime.md` sentence on the catalog rewrite. + +## Verification + +Focused tests, `bun run typecheck`, `bun run test:changed`, `structure:check`, `privacy:scan`; live: dev server +from this branch (isolated OPENCODEX_HOME with cliFirstParty + a provider that forwards to the running ocx), +`~/.claude/settings.json` env applied through the real apply path, plain `claude` in tmux → `/model` lists routes +→ pick one → prompt answered; settings and catalog cache restored afterwards. + +## Risks + +- Claude Code changes the catalog contract: rewrite is fail-open (unchanged bytes). +- Aliases depend on the Desktop 3P registry built at startup; unregistered routes are omitted, not guessed. +- UA classification is a hint, not a boundary (same as the Messages split). + diff --git a/devlog/_plan/261002_claude_cli_picker/020_audit.md b/devlog/_plan/261002_claude_cli_picker/020_audit.md new file mode 100644 index 00000000000..a34e0f0e734 --- /dev/null +++ b/devlog/_plan/261002_claude_cli_picker/020_audit.md @@ -0,0 +1,18 @@ +# 020 Audit fold + +Reviewer verdict: NEAR-PASS (independent subagent, read-only). Folded: + +1. Registry readiness — new `src/claude/desktop-3p-startup.ts`: `initDesktop3pRegistry(config)` (deduped, startup inputs incl. + entitlements) replaces the inline block in `src/cli/index.ts`; the CLI loader awaits it when the registry is empty. + Served rows are re-filtered against the live registry at response time. +2. Bootstrap — the listener consults the catalog hook before the relay-native shortcut; bootstrap eligibility is + `desired.cli` only (its `claude-code/` UA carries no entrypoint); upstream stays the route-selected one + (`CLAUDE_INTERCEPT_UPSTREAM` for unknown clients). +3. cc rewrite only for clients classified `cli` with `desired.cli`; Desktop entrypoints untouched (test). +4. Cache invalidation lives in `settings.ts` apply/remove (on change) and in `reconcileClaudeFirstPartySettings` (every ok + reconcile), covering the toggle route, ensure, disable and rollback paths. agent-settings-routes.ts is not touched. +5. Logs carry kind/status/counts only. 8. CLI hook wired from `loadPickerRoutes`, independent of the Desktop picker. +10. Desktop bootstrap output regression test. 13. Risk: `additionalModelOptionsCache` in `~/.claude.json` keeps + bootstrap rows until the CLI's next bootstrap fetch (every launch). +Not folded: 7 (error text) — documented instead. + diff --git a/devlog/_plan/261002_claude_ux/010_roadmap.md b/devlog/_plan/261002_claude_ux/010_roadmap.md new file mode 100644 index 00000000000..e35e74d8a48 --- /dev/null +++ b/devlog/_plan/261002_claude_ux/010_roadmap.md @@ -0,0 +1,27 @@ +# 261002 Claude UX — roadmap (010) + +Three outcomes, four work-phases (wp1 = this roadmap). + +| wp | Unit | Doc | Branch | +| --- | --- | --- | --- | +| wp2 | Start the Claude intercept pair on demand instead of asking for a restart | 020 | `codex/claude-intercept-on-demand` | +| wp3 | Top-level Claude page under Codex: Account / Code / Desktop / Settings | 030 | `codex/gui-claude-page` | +| wp4 | Local OpenCodex.app rebuild from merged dev, install handed to the user | 040 | none | + +## Evidence + +- `startClaudeIntercept` (src/claude/intercept/runtime.ts) runs once, from `createClaudeInterceptLifecycle().start` in + src/server/index/optional-listeners.ts at `startServer`. It resolves `null` when Claude routing is off + (`claudeCode.enabled === false`), the intercept is disabled, the role is `client`, or the public port is ephemeral, and a + bind failure only warns. Nothing starts it later, so `getClaudeInterceptState()` stays `null` for the life of the process. +- The CLI first-party toggle (`PUT /api/claude-code {cliFirstParty}`, agent-settings-routes.ts:1557) then refuses with + `intercept_unavailable` ("…restart needed"), and the Desktop panel shows `claudeDesktop.firstParty.proxyStopped`: + "로컬 프록시 127.0.0.1:{port}가 실행 중이 아님 — OpenCodex 재시작". `claude.firstParty.disabled` also tells the user to restart. + The live user config has `claudeCode.enabled: false`, `desktopMode: first-party`, which reproduces it. +- The Desktop picker controller and runtime are created inside `startClaudeIntercept` only, so the picker is also absent. + +## Order + +wp2 and wp3 are independent and run in parallel lanes (separate worktrees). wp3 rebases after wp2 merges if both touch +ClaudeCode/ClaudeDesktop copy. wp4 needs both merged. + diff --git a/devlog/_plan/261002_claude_ux/020_intercept_on_demand.md b/devlog/_plan/261002_claude_ux/020_intercept_on_demand.md new file mode 100644 index 00000000000..39f15bdef99 --- /dev/null +++ b/devlog/_plan/261002_claude_ux/020_intercept_on_demand.md @@ -0,0 +1,31 @@ +# 020 Intercept on demand (wp2) + +## Behaviour + +- A serialized, idempotent `ensureClaudeIntercept()` starts the intercept pair (CONNECT proxy, TLS listener, CLI catalog hook, + Desktop picker runtime/controller when wired) in the running process using the persisted config, reusing the options the + lifecycle got at `startServer` (dispatch, ports, picker routes). A second call while one is in flight awaits the same start; + a call when already bound returns the bound state. +- Result is a discriminated outcome: `{ok:true,state}` or `{ok:false, reason}` with reason + `disabled` (Claude routing off or `claudeCode.intercept.enabled=false`), `client_role`, `ephemeral_port`, + `port_in_use` (with the port), `failed` (message without secrets). No reason means "restart". +- Callers: the `cliFirstParty` toggle and every Desktop first-party/picker apply route call it before refusing; turning Claude + routing on (`PUT /api/native-integrations/claude {enabled:true}`) calls it; a new `POST /api/claude-intercept/start` + (dashboard session) lets the GUI retry explicitly. Status payloads gain `interceptReason` so the GUI can say why. +- Turning routing off does not tear the pair down (existing relay-native behaviour stays). +- The startup path keeps its synchronous contract (AGENTS.md: no await between `Bun.serve` and lab activation); the lifecycle + only records its options and a starter, it does not change the startup order. + +## GUI + +- Replace every "restart OpenCodex" instruction for the intercept (ko/en/... `claudeDesktop.firstParty.proxyStopped`, + `claude.firstParty.disabled`, `claude.firstParty.refusal.interceptUnavailable`) with the concrete reason plus a + "Start" action that calls the new endpoint, and a port hint for `port_in_use` + (`claudeCode.intercept.port`). All locales updated. + +## Tests + +Lifecycle: ensure starts when startup returned null, is idempotent and serialized, maps EADDRINUSE to `port_in_use`. +Route: `cliFirstParty:true` with no bound intercept starts it and succeeds; disabled routing returns `disabled`. +Desktop status reports `interceptReason`. File-size: keep agent-settings-routes.ts under 2000 lines (move to a sibling). + diff --git a/devlog/_plan/261002_claude_ux/030_claude_page.md b/devlog/_plan/261002_claude_ux/030_claude_page.md new file mode 100644 index 00000000000..e5dd4b2da5d --- /dev/null +++ b/devlog/_plan/261002_claude_ux/030_claude_page.md @@ -0,0 +1,26 @@ +# 030 Claude page (wp3) + +## IA + +Sidebar: Dashboard, Codex, **Claude** (new, directly under Codex), Providers, … Integrations keeps its other clients; the Claude +tab inside Integrations is removed and its hashes redirect: `#integrations/claude` → `#claude/code`, +`#integrations/claude/desktop` → `#claude/desktop`. + +Page shaped like Codex Set (page-tabs, hash-routed, panels mounted on first visit and kept mounted while inactive, keyboard arrows): + +| Tab | Hash | Content | +| --- | --- | --- | +| Account | `#claude` / `#claude/account` | Anthropic accounts: the existing provider auth + AnthropicAccountPoolSettings (pool on/off, auto-switch threshold), reset grants | +| Code | `#claude/code` | existing ClaudeCode page (CLI first-party, models, sidecars) | +| Desktop | `#claude/desktop` | existing ClaudeDesktop page (mode, picker, bindings) | +| Settings | `#claude/settings` | Claude routing on/off, intercept status/port and start action (from wp2 when merged), compatibility mode, inject agents, auto-context — only settings that exist today, no new behaviour | + +## Design read + +Dashboard tool surface (D4, variance 3, motion 1): reuse existing page-tabs, panels and tokens; no new visual language. +One primary action per tab. Empty/disabled states explain why and link to the action. + +## Checks + +lint:gui, build:gui, GUI tests for tab routing/redirects, i18n completeness (all locales), screenshots of each tab in light/dark +for the PR and a UI review by an inherited-model subagent. diff --git a/devlog/_plan/261002_claude_ux/040_app_rebuild.md b/devlog/_plan/261002_claude_ux/040_app_rebuild.md new file mode 100644 index 00000000000..7894db481d4 --- /dev/null +++ b/devlog/_plan/261002_claude_ux/040_app_rebuild.md @@ -0,0 +1,6 @@ +# 040 Local app rebuild (wp4) + +From the merged dev head: `bun install`, `(cd gui && bun install && bun run build)`, `cd desktop && bun run prepare-sidecar && bun run prepare-widget && bun run build:local`. +Output: `desktop/src-tauri/target/release/bundle/macos/OpenCodex.app` and dmg. Do not replace /Applications/OpenCodex.app or +restart the running proxy; hand the path to the user to install. Unsigned local build: Gatekeeper may need right-click → Open. + diff --git a/devlog/_plan/261002_claude_ux/050_audit.md b/devlog/_plan/261002_claude_ux/050_audit.md new file mode 100644 index 00000000000..9847a9836c2 --- /dev/null +++ b/devlog/_plan/261002_claude_ux/050_audit.md @@ -0,0 +1,24 @@ +# 050 Audit fold (wp1) + +Independent inherited-model reviewer: NEAR-PASS. Folded into 020/030 as binding requirements: + +1. `ensureClaudeIntercept` is a method of `ClaudeInterceptLifecycle` (claude-intercept-lifecycle.ts), so a later start goes + through the same dispatch wrapper (`listener ??= requestServer`) and `ownsListener` classifies it as the intercept ingress; + expose it through `OptionalListenerSet` and the existing `linkListener: () => optionalListeners` seam (index.ts gains no lines). +2. One `inflight` promise; `stopped` flag refuses ensure after stop; `stop()` awaits the in-flight start; ensure awaits a + pending startup start first and only starts a new pair if it resolved to null. +3. Record the last start outcome: pure precheck (disabled / client role / ephemeral port) plus bind error mapping + (EADDRINUSE on the CONNECT proxy → `port_in_use` with port). Picker proxy bind failure stays non-fatal and is reported as + `pickerReason`. +4. Port mismatch (bound port ≠ configured) gets reason `port_mismatch`: rebind is out of scope; the copy explains the bound port + is in use for this run and how to change it, never "restart". +5. Use startServer's live config object; recompute `observeClaudeDesktopMode` on each ensure. +6. Security: ensure (POST /api/claude-intercept/start and the triggering PUTs) is refused on the hub-management ingress and for + data-plane API keys; same gate everywhere; tests prove it. +7. GUI routing list for wp3: app-routing.ts (Page union, page list, `#claude` redirect inverted), App.tsx title map + NAV, + `nav.claude` and all copy in all locales, INTEGRATION_TAB_HASHES old hashes kept as redirects, integration-tabs.ts:34, + overview-clients.ts:318/367, api-surface-cards.tsx:20, Integrations.tsx:20, Claude.tsx:10-11. + +Notes kept: agent-settings-routes.ts sibling move up front (1943/2000); a later start may run picker CA drain/keychain → GUI +pending state; routing off keeps the pair bound (relay-native). + diff --git a/docs-site/astro.config.mjs b/docs-site/astro.config.mjs index 0c5f250128f..568c060ad93 100644 --- a/docs-site/astro.config.mjs +++ b/docs-site/astro.config.mjs @@ -108,6 +108,7 @@ export default defineConfig({ { label: "Grok Build", translations: { fr: "Grok Build", ko: "Grok Build", "zh-CN": "Grok Build", "zh-TW": "Grok Build", ru: "Grok Build", ja: "Grok Build", tr: "Grok Build" }, slug: "guides/grok-build" }, { label: "opencode", translations: { fr: "opencode", ko: "opencode", "zh-CN": "opencode", "zh-TW": "opencode", ru: "opencode", ja: "opencode", tr: "opencode" }, slug: "guides/opencode" }, { label: "Pi", translations: { fr: "Pi", ko: "Pi", "zh-CN": "Pi", "zh-TW": "Pi", ru: "Pi", ja: "Pi", tr: "Pi" }, slug: "guides/pi" }, + { label: "ChatGPT Desktop (experimental)", slug: "guides/chatgpt-desktop" }, { label: "Integrations", translations: { fr: "Intégrations", ko: "연동", "zh-CN": "集成", "zh-TW": "整合", ru: "Интеграции", ja: "連携", tr: "Entegrasyonlar" }, slug: "guides/integrations" }, { label: "MiniMax clients", translations: { fr: "Clients MiniMax", ko: "MiniMax 클라이언트", "zh-CN": "MiniMax 客户端", "zh-TW": "MiniMax 客戶端", ru: "Клиенты MiniMax", ja: "MiniMax クライアント", tr: "MiniMax İstemcileri" }, slug: "guides/minimax" }, { label: "Sidecars: Web Search & Vision", translations: { fr: "Services auxiliaires : recherche web et vision", ko: "사이드카: 웹 검색 & 비전", "zh-CN": "边车:网络搜索与视觉", "zh-TW": "邊車:網路搜尋與視覺", ru: "Сайдкары: веб-поиск и зрение", ja: "サイドカー: ウェブ検索 & ビジョン", tr: "Sidecar'lar: Web Arama ve Görme" }, slug: "guides/sidecars" }, @@ -173,6 +174,7 @@ export default defineConfig({ { label: "Disk Usage from Temp Files", translations: { fr: "Espace disque et fichiers temporaires", ko: "임시 파일 디스크 사용량", "zh-CN": "临时文件磁盘占用", "zh-TW": "暫存檔磁碟用量", ru: "Использование диска временными файлами", ja: "一時ファイルのディスク使用量", tr: "Geçici Dosya Disk Kullanımı" }, slug: "troubleshooting/disk-usage-temp-files" }, { label: "Codex Cannot Sign In or Load", translations: { fr: "Codex ne peut pas se connecter", ko: "Codex 로그인 불가", "zh-CN": "Codex 无法登录", "zh-TW": "Codex 無法登入", ru: "Codex не может войти", ja: "Codex にサインインできない", tr: "Codex Oturum Açamıyor" }, slug: "troubleshooting/codex-cannot-sign-in" }, { label: "Update Failed on Windows", translations: { fr: "Échec de la mise à jour sous Windows", ko: "Windows에서 업데이트 실패", "zh-CN": "Windows 上更新失败", "zh-TW": "Windows 上更新失敗", ru: "Сбой обновления в Windows", ja: "Windows で更新に失敗する", tr: "Windows'ta Güncelleme Başarısız" }, slug: "troubleshooting/update-failed" }, + { label: "Spend Ledger Refused in a Synced Folder", translations: { fr: "Registre de dépenses refusé dans un dossier synchronisé", ko: "동기화 폴더에서 지출 원장 거부", "zh-CN": "同步文件夹中的支出账本被拒绝", "zh-TW": "同步資料夾中的支出帳本被拒絕", ru: "Отказ журнала расходов в синхронизируемой папке", ja: "同期フォルダーで支出台帳が拒否される", tr: "Eşitlenen Klasörde Harcama Defteri Reddi" }, slug: "troubleshooting/spend-ledger-synced-folder" }, ], }, { label: "Contributing", translations: { fr: "Contribuer", ko: "기여하기", "zh-CN": "贡献", "zh-TW": "貢獻", ru: "Как внести вклад", ja: "コントリビュート", tr: "Katkıda Bulunma" }, slug: "contributing" }, diff --git a/docs-site/src/content/docs/fr/getting-started/quickstart.md b/docs-site/src/content/docs/fr/getting-started/quickstart.md index 3cf3a91b39c..8d39e2502ab 100644 --- a/docs-site/src/content/docs/fr/getting-started/quickstart.md +++ b/docs-site/src/content/docs/fr/getting-started/quickstart.md @@ -13,7 +13,7 @@ ocx init `ocx init` vous accompagne dans les étapes suivantes : -1. **Choix d’un fournisseur** — sélectionnez l’un des 100 préréglages intégrés au registre, ou `custom` pour saisir une +1. **Choix d’un fournisseur** — sélectionnez l’un des 102 préréglages intégrés au registre, ou `custom` pour saisir une URL de base et un adaptateur. 2. **Clé API** — collez une clé ou référencez une variable d’environnement telle que `${ANTHROPIC_API_KEY}`. 3. **Modèle par défaut** — pour les fournisseurs clés, locaux et personnalisés, acceptez le préréglage ou saisissez un identifiant de modèle. diff --git a/docs-site/src/content/docs/fr/guides/claude-code.md b/docs-site/src/content/docs/fr/guides/claude-code.md index e2f9b6ece52..3c66bdecaff 100644 --- a/docs-site/src/content/docs/fr/guides/claude-code.md +++ b/docs-site/src/content/docs/fr/guides/claude-code.md @@ -19,9 +19,17 @@ choisit la plus faible utilisation connue dans la fenêtre configurée par `anth (`five-hour` par défaut, `weekly` ou `max-utilization`) lorsqu'elle dépasse `autoSwitchThreshold` ; `round-robin` répartit les sessions uniformément (`stickyLimit`, `1` par défaut) ; `fill-first` utilise le compte actif jusqu'à un délai de récupération, une réauthentification ou le seuil, puis passe au suivant. Cette fonction est -**désactivée par défaut**, affiche un avertissement dans l'interface et n'a pas été éprouvée en production. -Anthropic peut restreindre les comptes dont l'activité ressemble à une rotation automatisée ; la rotation ne -protège pas contre l'application des règles du fournisseur. +**désactivée par défaut** et reste expérimentale. + +Le tableau de bord indique les conditions prévues : des abonnements qui vous appartiennent ou que vous êtes +autorisé à utiliser, le client Claude Code officiel et une personne qui supervise la session. Anthropic n'a pas +approuvé le regroupement automatique de comptes, des comptes d'une même organisation peuvent partager un quota +(un compte de plus n'ajoute alors pas forcément de capacité), et changer de compte ne protège pas contre +l'application des règles du fournisseur. OpenCodex n'envoie aucune requête de maintien à chaud (keep-warm) et, +par défaut, ne rafraîchit pas les jetons Claude et ne lit pas l'usage en arrière-plan : l'usage est lu quand le +tableau de bord, l'app de la barre des menus ou une commande `ocx` le demande. Les seuils sont des préférences +de sélection, pas des plafonds d'usage ou de facturation. Ces indications ne constituent pas un avis juridique ; +consultez les conditions actuelles d'Anthropic. Comportement lorsque cette option est activée : diff --git a/docs-site/src/content/docs/fr/guides/providers.md b/docs-site/src/content/docs/fr/guides/providers.md index 641b9a93735..7d168123544 100644 --- a/docs-site/src/content/docs/fr/guides/providers.md +++ b/docs-site/src/content/docs/fr/guides/providers.md @@ -295,7 +295,7 @@ existante n'est pas concernée. ## 3. Catalogue des clés API -opencodex fournit 100 préréglages intégrés : 83 à clé, 13 OAuth, trois locaux et un préréglage par défaut de +opencodex fournit 102 préréglages intégrés : 84 à clé, 14 OAuth, trois locaux et un préréglage par défaut de transfert ChatGPT. Dans le tableau de bord, le sélecteur **Ajouter un fournisseur** ouvre le tableau de bord du fournisseur à clé, valide la clé et l'enregistre ; la validation dépend du fournisseur. Parmi les entrées notables : @@ -358,10 +358,20 @@ promotionnels de Cline ne sont accessibles que dans l'IDE ou la CLI Cline, pas p | Xiaomi MiMo | `https://api.xiaomimimo.com/anthropic` | | Xiaomi MiMo (OpenAI Chat) | `https://api.xiaomimimo.com/v1` | | Kilo | `https://api.kilo.ai/api/gateway` | +| OpenGateway | `https://apis.opengateway.ai/v1` | | GitLab Duo | `https://cloud.gitlab.com/ai/v1/proxy/openai/v1` | | Cloudflare AI Gateway | `https://gateway.ai.cloudflare.com/v1/{account-id}/{gateway}/anthropic` | | …et plus encore | opencode zen, Vercel AI Gateway, Venice, NanoGPT, Synthetic, Qianfan, Alibaba, Parallel, ZenMux, LiteLLM | +**OpenGateway** est une passerelle compatible OpenAI exploitée par Sionic AI à +`https://apis.opengateway.ai/v1`. Son catalogue public compte environ 80 modèles actifs +(vérifiés le 2026-10-02). Le préréglage actualise automatiquement la liste via le +`GET /v1/models` public et conserve les modèles Chat Completions actifs (ainsi que `openai/o3-pro`, réservé à Responses et routé vers Responses). Les modèles servis +par Sionic, `deepseek/deepseek-v4.1-flash-ultrafast` et `z-ai/glm-5.3-flash-ultrafast`, +sont affichés en premier. Créez une clé dans le [tableau de bord OpenGateway](https://opengateway.ai/api-keys), +puis lancez `ocx provider add opengateway` ou sélectionnez **OpenGateway** dans le tableau de bord. +Les requêtes chat utilisent la clé Bearer configurée ; la liste publique ne valide pas cette clé. + **OpenCode Zen** (`opencode-zen`) et le préréglage sans clé **OpenCode Free** utilisent tous deux `https://opencode.ai/zen/v1`. Sur cette passerelle, les modèles gratuits atteignent souvent une limite de rafale sur une courte fenêtre, d'environ 15 à 20 requêtes par minute (mesure de la communauté ; OpenCode ne publie diff --git a/docs-site/src/content/docs/fr/reference/configuration/providers.md b/docs-site/src/content/docs/fr/reference/configuration/providers.md index 7b285f3d9bd..b939aff7146 100644 --- a/docs-site/src/content/docs/fr/reference/configuration/providers.md +++ b/docs-site/src/content/docs/fr/reference/configuration/providers.md @@ -245,8 +245,7 @@ rotation automatique peut déclencher des restrictions du fournisseur. | `anthropicAccountPool.quotaWindow?` | `"five-hour" \| "weekly" \| "max-utilization"` | `"five-hour"` | Barre d'utilisation signalée par le fournisseur, mise en cache et utilisée pour la sélection selon l'utilisation. `five-hour` conserve le comportement actuel. `weekly` utilise la barre hebdomadaire et ignore les comptes dont la barre sur 5 heures est épuisée tant qu'un autre compte admissible reste disponible, mais y revient si aucun autre ne reste. `max-utilization` utilise la valeur connue la plus élevée et peut donc employer la barre sur 5 heures avant que la barre hebdomadaire soit disponible ; si aucune n'est connue, le compte suit l'ordre des utilisations inconnues. Les utilisations connues précèdent les inconnues, mais si tous les comptes admissibles sont inconnus, la sélection en renvoie tout de même un dans leur ordre admissible. Après le départage documenté par la plus faible utilisation sur 5 heures, une égalité exacte conserve cet ordre. Une session saine avec affinité n'est pas rééquilibrée de manière proactive. Pour l'affectation des nouvelles sessions et la reprise du routage après un remplacement admissible à la suite d'un 429, `quota` classe directement les candidats admissibles avec cette fenêtre ; `fill-first` avance dans un ordre stable selon le seuil et les règles d'épuisement de cette fenêtre ; `round-robin` l'ignore. Le délai de récupération, les limites de basculement et l'éligibilité de réauthentification restent des états locaux distincts. Les barres hebdomadaires ne sont connues qu'après leur interrogation dans la page Fournisseurs du tableau de bord. | | `anthropicAccountPool.stickyLimit?` | `number` | `1` | Liaisons de nouvelle session réussies conservées sur une sélection à tour de rôle. Portée 1–100. | -Lorsque cette option est activée, un 429 enregistre une temporisation bornée à partir de `Retry-After` ou d'un délai de repli, puis peut -faire basculer la requête vers un autre compte. L'affinité est locale au processus et de taille bornée. Les erreurs de renouvellement du jeton suivent les règles de réauthentification existantes. Un 403 confirmé lié à un abonnement ou à la facturation du compte permet un basculement avant la sortie, avec une temporisation selon `Retry-After` ou de dix minutes par défaut. Un refus d’autorisation ordinaire ne déclenche pas de basculement. Si tous les comptes admissibles sont en temporisation, les clients reçoivent un 429 accompagné de +Seul un 429 confirmant le rejet d'un quota partagé de cinq heures ou hebdomadaire déclenche la temporisation et le basculement du compte. Une limitation temporaire suspend son admission en conservant l'affinité : au plus un bref nouvel essai sur le même compte et un détour vers un autre compte admissible par requête. Sans en-tête probant, un 429 permet un seul bref nouvel essai sur le même compte, sans refroidir le compte ni inventer Retry-After. Le comportement par défaut avec un seul compte reste identique. Un rejet propre à Fable laisse Sonnet admissible ; choix manuel et affinité vérifient aussi les quotas partagés et ceux de la famille demandée. Les observations passives de famille expirent après trente minutes ou à leur réinitialisation connue et sont revalidées par une seule requête à la fois. Les seuils restent des préférences souples avec le repli existant lorsque tous les candidats sont épuisés, sans plafond strict d'utilisation ou de facturation. L'affinité est locale au processus et de taille bornée. Les erreurs de renouvellement du jeton suivent les règles de réauthentification existantes. Un 403 confirmé lié à un abonnement ou à la facturation du compte permet un basculement avant la sortie, avec une temporisation selon `Retry-After` ou de dix minutes par défaut. Un refus d’autorisation ordinaire ne déclenche pas de basculement. Si tous les comptes admissibles sont en temporisation, les clients reçoivent un 429 accompagné de `Retry-After` lorsqu'il est connu, et non une erreur d'authentification. :::caution[Expérimental] @@ -385,6 +384,8 @@ acceptez le contournement des autorisations Codex et de la sémantique du bac à Grok 4.7 propose le mode Fast sur OAuth, avec `low` / `medium` / `high` / `xhigh` et une fenêtre de 500 000 jetons. Son [tarif xAI](https://docs.x.ai/developers/models/grok-4.7) standard par million de jetons est de 2,00 $ en entrée, 0,50 $ en entrée mise en cache et 6,00 $ en sortie ; à partir de 200 000 jetons de contexte, les tarifs sont de 4,00 $ / 1,00 $ / 12,00 $. +Sans configuration explicite de `fastWire` pour le fournisseur, une clé API opencodex limitée par `allowedModels` doit autoriser `xai/grok-4.7-build-fast` (ou son identifiant sans préfixe) pour les requêtes Fast via OAuth. Autoriser uniquement `xai/grok-4.7` ne donne pas accès à cette variante Fast. Une clé limitée au modèle Fast peut utiliser cette variante ; les requêtes ordinaires ou Fast désactivé exigent toujours `xai/grok-4.7`. Si un `fastWire` explicite est configuré, autorisez le modèle réellement envoyé : par exemple, un wire de type `service-tier` conserve `xai/grok-4.7` et exige la permission pour ce modèle. Les restrictions de fournisseur restent applicables. + ## Routage des fournisseurs OpenRouter OpenRouter peut servir un modèle au moyen de plusieurs fournisseurs d'inférence. `openRouterRouting` maintient les diff --git a/docs-site/src/content/docs/getting-started/quickstart.md b/docs-site/src/content/docs/getting-started/quickstart.md index d49783ed4ec..05913d24b03 100644 --- a/docs-site/src/content/docs/getting-started/quickstart.md +++ b/docs-site/src/content/docs/getting-started/quickstart.md @@ -18,7 +18,7 @@ ocx init `ocx init` walks you through: -1. **Pick a provider** — choose one of the 100 built-in registry presets or `custom` to type a base +1. **Pick a provider** — choose one of the 102 built-in registry presets or `custom` to type a base URL and adapter. 2. **API key** — paste a key, or reference an environment variable like `${ANTHROPIC_API_KEY}`. 3. **Default model** — for key, local, and custom providers, accept the preset or enter a model id. diff --git a/docs-site/src/content/docs/guides/chatgpt-desktop.md b/docs-site/src/content/docs/guides/chatgpt-desktop.md new file mode 100644 index 00000000000..dfa3f62725b --- /dev/null +++ b/docs-site/src/content/docs/guides/chatgpt-desktop.md @@ -0,0 +1,112 @@ +--- +title: ChatGPT Desktop app-server shim (experimental) +description: An opt-in macOS experiment that rewrites plain-quota gate fields on the bundled app-server stdout pipe. +--- + +This experiment is **macOS only and off by default**. It filters the bundled ChatGPT +app-server's JSON-RPC stdout to open known plain-quota gates. It does not increase +an account's quota or make an upstream service accept a request it refuses. + +Enable it in your OpenCodex `config.json`: + +```json +{ + "chatgptDesktop": { "appServerShim": true } +} +``` + +Then run: + +```bash +ocx chatgpt launch +ocx chatgpt status +``` + +`launch` creates an executable launcher under the OpenCodex config directory, +quits ChatGPT if it is running, and relaunches it with +`open -a --env CODEX_CLI_PATH=`. The app is found by its bundle +identifier, `com.openai.codex`, so an install in `~/Applications` or on another +volume works, and another app that shares the "ChatGPT" name is never quit or +opened. Save ongoing work first: this +restarts the app. It does not require a running OpenCodex proxy. + +To remove the launcher and relaunch without the override: + +```bash +ocx chatgpt restore +``` + +Restore leaves the config flag as configured. Set `chatgptDesktop.appServerShim` +to `false` or remove it to disable future explicit shim launches. Normal launches +from Dock or Spotlight do not apply the shim automatically. + +If ChatGPT is not installed (no `com.openai.codex` bundle is found), `restore` +only removes the launcher: it cannot relaunch anything and exits with an error. + +## Rewrite boundary + +Only `account/rateLimits/updated` notifications and responses whose top-level +result contains `rateLimits`, `rateLimitsByLimitId`, or `ordinaryUsageAllowed` +are eligible. Plain `rate_limit_reached` markers are cleared; known quota gate +flags (`allowed`, `limit_reached` / `limitReached`, `ordinaryUsageAllowed`) are +opened only with plain-quota evidence (a cleared plain reached type or a window at +100%). A flag closed for a reason the payload does not show stays closed. Workspace, +credit, unknown reached-type and spend-control +restrictions keep the usage gate closed. + +Displayed usage stays honest: percentages, reset times, window durations, plan +information and other display fields stay as received. Unrelated JSON-RPC +messages, nested tool output, conversation send-block metadata and malformed +lines pass through. Only changed lines are serialized again; other bytes retain +their original encoding and line endings. Stdin, stderr and the real binary's +exit status retain their direct connection to the app. + +## Executable and environment security + +The generated launcher has mode `0755` and embeds the current OpenCodex executable +and, for source installs, the CLI entry path. `CODEX_CLI_PATH` tells ChatGPT to +execute this launcher instead of its bundled binary directly. Keep the launcher, +its config directory, and the OpenCodex installation under your control: changing +these executable paths changes code the app runs. The launcher still `exec`s the +bundled binary of the discovered bundle; if that bundle has no app-server binary, +`launch` refuses instead of writing a launcher. + +Before writing the launcher, `launch` also checks that the bundle and its +app-server binary are owned by you or root, are not writable by group or others, +and pass strict code-signature verification under OpenAI's team ID +(`2DC432GLL2`). A bundle that fails any of these is refused, so a copy placed by +another account cannot be made to run inside your session. The launcher file is +written to a temporary file and renamed into place; an existing symbolic link at +that path is replaced, not followed. + +This integration installs no certificate, network listener, PAC, or background +watcher. It does not log the app's messages or environment. Status reports whether +the running ChatGPT bundle process carries the expected launcher override. + +## Failure behavior and known limits + +When the platform is not macOS, the OpenCodex runtime is missing, or the filter +self-test fails, the launcher runs the original binary with untouched stdout. A +missing bundled app-server binary is the exception: there is nothing to fall back +to, so the launcher exits with an error (see below). +A filter that passes the self-test and then dies mid-session closes the pipe. +What the bundled app-server does after that has not been verified; it may get +SIGPIPE or a write error and be respawned by Desktop through the same launcher. +The filter's passthrough mode limits this to an exit/crash case: a rewrite exception +passes its line through, and an unexpected rewrite-machinery failure switches the +remaining stream to raw bytes. + +The experiment depends on the bundled binary path, the app honoring +`CODEX_CLI_PATH`, and current RPC field shapes. Updates may change these. A moved +or removed OpenCodex installation fails the launcher preflight and runs the +original binary. Run `ocx chatgpt launch` again after relocating the installation. +If an app update moves or removes the bundled app-server binary itself, the +launcher cannot start it: it prints a message naming `ocx chatgpt launch` and +`ocx chatgpt restore` on stderr and exits, and Desktop cannot start its +app-server until you run one of them. A single output line longer than 8 MiB is +passed through unparsed rather than buffered. + +This standalone shim does not rewrite conversation metadata or route model calls. +Other app gates or upstream refusals can still prevent sending. Evidence reported +on an exhausted Plus account also used an intercept, so it does not establish +that this shim alone resolves every desktop send lock. diff --git a/docs-site/src/content/docs/guides/claude-code.md b/docs-site/src/content/docs/guides/claude-code.md index 88eae2b4083..1a7953a8df0 100644 --- a/docs-site/src/content/docs/guides/claude-code.md +++ b/docs-site/src/content/docs/guides/claude-code.md @@ -28,21 +28,34 @@ for the exact model and recovery scope. ## Claude OAuth account pool (experimental) +Native Anthropic subscription passthrough retains the upstream `anthropic-ratelimit-*` +response headers on streaming, JSON and upstream-error responses, so compatible Claude Code +statusLine consumers can read the provider's quota observations. Missing observations are +not filled with invented values. This relay does not change account selection or retry behavior. + You can log in multiple Claude accounts via the Providers dashboard (`ocx login anthropic` / add-account). By default every request uses the **active** account only. An **experimental, opt-in** Claude account pool (`anthropicAccountPool.enabled`) adds sticky session affinity and usage-aware new-session selection across those OAuth accounts. It does **not** gate 429 failover: with two or more usable accounts stored, a rate-limited request moves -to another account whether the pool is on or off, and that cannot be switched off. For **new** +to another account whether the pool is on or off, and the toggle does not switch that off; pause +an account to keep it out of failover. For **new** sessions, `anthropicAccountPool.strategy` selects among eligible accounts: `quota` (default) picks the lowest known usage in the window set by `quotaWindow` (`five-hour` by default, or `weekly` / `max-utilization`) when above `autoSwitchThreshold`; `round-robin` spreads evenly (`stickyLimit`, default `1`); `fill-first` drains the active account until cooldown, -reauthentication, or threshold, then advances. It is **off by default**, shows a GUI warning, -and is not battle-tested — Anthropic may restrict accounts that look like automated rotation; -rotation does not protect against provider enforcement. +reauthentication, or threshold, then advances. It is **off by default** and experimental. + +The dashboard lists the conditions the pool is meant for: subscriptions you own or are authorized +to use, the genuine Claude Code client, and a person supervising the session. Anthropic has not +endorsed automated account pooling, accounts in one organization may share a quota (so another +account may not add capacity), and switching accounts does not protect against provider +enforcement. OpenCodex sends no keep-warm requests and, by default, neither refreshes Claude tokens +nor reads usage in the background: usage is read when the dashboard, the menu bar app, or an `ocx` +command asks for it. Thresholds are selection preferences, not usage or billing caps. This is +product guidance, not legal advice; check Anthropic's current terms. To bind a model to particular stored Claude accounts, add ordered `anthropicAccountPool.routes` rules while the pool is enabled. Each rule has a safe `name`, a full case-sensitive `match` glob, an `accounts` array of stored account IDs, and optional `fallback` (default `false`). The first matching rule limits active, manual, affinity, strategy and 429 recovery picks to its accounts. A healthy affinity outside that rule is ignored for this request but kept for other models; the routed choice does not overwrite it. Without fallback, an empty route returns a local 401, or 429 with `Retry-After` when all its declared accounts are cooling, before contacting Anthropic. The client response does not name the route; the proxy log records `route:#`, where `n` is the rule’s 1-based position. `fallback: true` uses the ordinary pool only when the route has no eligible account, including its ordinary fill-first successor order; if its stored accounts are all cooling, the returned 429 uses the earliest cooldown across that expanded pool, even if a saved route account has been removed. An unmatched model follows the existing pool policy; disabling the pool leaves saved rules inactive and restores active-account and presence-driven 429 behavior. A rule is an operator allowlist, not proof of model entitlement. @@ -118,7 +131,7 @@ native `claude` binary instead, so the command stays useful with routing off: | Where routing is off | What happens | | --- | --- | | `claudeCode.enabled: false` in config | Native launch, with a notice that routing is disabled | -| The running proxy reports `enabled: false` from `GET /api/claude-code` | Native launch, with a notice to restart the service after re-enabling | +| The running proxy reports `enabled: false` from `GET /api/claude-code` | Native launch; re-enabling starts interception on demand | | `claudeCode.enabled` absent or `true` | Routed through the proxy, unchanged | Only an explicit `false` triggers the fallback, so a proxy predating the field stays routed. A @@ -198,7 +211,7 @@ working. OpenCodex only writes two variables into the `env` block of `~/.claude/ } ``` -Claude Desktop first-party routes its Code tab and subagents through OpenCodex. The standalone Claude Code CLI has a separate first-party switch. Both clients read the same `~/.claude/settings.json` proxy and CA settings: if only one switch is on, the other client still transits the local proxy, where TLS terminates, but its Messages requests relay to Anthropic unchanged. Other Anthropic paths relay unchanged and unrelated hosts remain blind tunnels. +Claude Desktop first-party routes its Code tab and subagents through OpenCodex. The standalone Claude Code CLI has a separate first-party switch. Both clients read the same `~/.claude/settings.json` proxy and CA settings: if only one switch is on, the other client still transits the local proxy, where TLS terminates, but its Messages requests relay to Anthropic unchanged. Other Anthropic paths relay unchanged and unrelated hosts remain blind tunnels. With the CLI switch on, the standalone `claude` also lists routed models in `/model`; see [Model picker in the CLI](#model-picker-in-the-cli). :::note[Windows system proxy (Clash, v2rayN, corporate proxies)] When a Windows system proxy is on, Claude Desktop hands it to the Code tab as `HTTPS_PROXY`, and @@ -293,11 +306,12 @@ next request; Desktop does not need a restart. Turn on the CLI switch in Claude → Code, or run `ocx claude config set --first-party on`; use `off` to disable it. The switch is immediate and refuses `{enabled:false, cliFirstParty:true}` before any field is saved; it may also refuse to turn on if the local intercept is unavailable, the CA cannot be prepared, settings cannot be read, or a foreign proxy setting owns the keys. Off persists even when the intercept is unavailable; disabling Claude routing alone leaves an owned settings env untouched. For fully native terminal traffic with only Desktop first-party on, set `NO_PROXY='*'` in the shell. This still carries the first-party account risk stated above. Disabling Claude routing leaves the owned settings env untouched. While the bound listener still runs, every Messages request relays unchanged; after it stops, plain `claude` cannot connect until OpenCodex runs or Desktop/CLI first-party is turned off. Native `ocx claude` sets `NO_PROXY=*` only for an owned env without a foreign inherited `HTTPS_PROXY`/`https_proxy`. With a foreign proxy it preserves that value and warns that the settings-owned intercept still applies; turn Desktop/CLI first-party off or unset the setting. -The UI distinguishes uncertainty about whether settings still point at its proxy (unknown), a token-bearing opencodex proxy with a foreign CA (foreign: fix HTTPS_PROXY / NODE_EXTRA_CA_CERTS manually), and a tokenless loopback proxy beside a foreign CA (local: ownership is unconfirmed; remove HTTPS_PROXY if unused). With matching applied settings and a bound listener but Claude routing off, disabled means requests relay unchanged until restart; turn first-party off to remove settings. An owned URL with no listener is stopped; a bound listener with an owned CA but mismatched port or token is broken even when routing is off. With an intent on, stopped or broken plus ineligible interception displays routingOff: Claude routing or the intercept is off, or this machine is a client of another opencodex hub; enable interception on this machine or turn first-party off to remove the settings. Only when interception is eligible does stopped advise starting opencodex and broken advise `ocx ensure` or restart. CLI intent with no proxy is not applied; one intent with a live proxy gets the shared-relay notice; any remaining proxy with neither intent is residual, unless unknown, foreign, or local takes precedence. +The UI distinguishes uncertainty about whether settings still point at its proxy (unknown), a token-bearing opencodex proxy with a foreign CA (foreign: fix HTTPS_PROXY / NODE_EXTRA_CA_CERTS manually), and a tokenless loopback proxy beside a foreign CA (local: ownership is unconfirmed; remove HTTPS_PROXY if unused). With matching applied settings and a bound listener but Claude routing off, disabled means requests relay unchanged while it is bound; turn first-party off to remove settings. An owned URL with no listener is stopped; a bound listener with an owned CA but mismatched port or token is broken even when routing is off. With an intent on, stopped or broken plus ineligible interception displays routingOff: Claude routing or the intercept is off, or this machine is a client of another opencodex hub; enable interception on this machine or turn first-party off to remove the settings. When interception is stopped, use **Start interception** or `ocx claude intercept start` to retry within the running OpenCodex service. Saving first-party settings also starts it automatically. A refusal reports disabled routing, client role, an ephemeral public port, an occupied port, or a startup failure. For an occupied port, free it or set `claudeCode.intercept.port`. If a pair is already bound on a different port, the response names both ports; use the bound port or restore the configured value. CLI intent with no proxy is not applied; one intent with a live proxy gets the shared-relay notice; any remaining proxy with neither intent is residual, unless unknown, foreign, or local takes precedence. -- Model discovery (`/model` → "From gateway") is not available; Claude Code only queries - `GET /v1/models` on a configured gateway. Bind a built-in Anthropic model id to a route - (`ocx claude desktop bind`, above), use `modelMap`, or type an alias directly. +- Gateway discovery (`/model` → "From gateway") is not used, because Claude Code only queries + `GET /v1/models` on a configured gateway. Routed models reach the picker another way; see + [Model picker in the CLI](#model-picker-in-the-cli). You can also bind a built-in Anthropic + model id to a route (`ocx claude desktop bind`, above) or use `modelMap`. - `ANTHROPIC_SMALL_FAST_MODEL` and `CLAUDE_CODE_SUBAGENT_MODEL` are chosen by the CLI before the request is sent; set them in `settings.json` yourself if a sidecar or subagent should use a mapped id. @@ -307,6 +321,23 @@ The UI distinguishes uncertainty about whether settings still point at its proxy - Claude Code honours `HTTPS_PROXY`/`NODE_EXTRA_CA_CERTS` as documented for corporate proxies; a CLI release that stops doing so would stop routing, not break login. +#### Model picker in the CLI + +With the CLI first-party switch on and Claude routing enabled, start a new `claude` and open +`/model`. Your opencodex models appear next to the Claude models from your account, labelled +like "Grok 4.7 (xai)" and described with their route (`opencodex · xai/grok-4.7`). Picking one +routes that session to the model, just like a binding. + +- Only new `claude` sessions pick up the list. Restart any session that was already running when + you turned the switch on or off. +- Claude Code saves the list for about an hour. Turning the switch on or off, and `ocx ensure`, + clear that saved copy, so the next `claude` launch fetches a fresh one. If the routed models + are missing (the first launch after OpenCodex starts can miss them while models are still being + discovered), run `ocx ensure` and restart `claude`. +- The rows use Claude-style ids (for example `claude-opus-4-8-…`), because the CLI only offers + ids shaped like Claude models. Claude Desktop's Code tab is unaffected; it keeps using + [first-party bindings](#use-opencodex-models-from-the-desktop-code-tab-first-party-bindings). + ## Claude Desktop profile (gateway mode) The profile below is written only in gateway mode. Claude Desktop uses a separate profile from Claude Code. Open **Claude → Desktop** in the diff --git a/docs-site/src/content/docs/guides/codex-integration.md b/docs-site/src/content/docs/guides/codex-integration.md index 3ca45b32568..e0b148927d2 100644 --- a/docs-site/src/content/docs/guides/codex-integration.md +++ b/docs-site/src/content/docs/guides/codex-integration.md @@ -225,6 +225,27 @@ caller, including one that posts them directly, and a model turn records no hist opencodex decides this by reading Codex own config itself, so the switch does not depend on the injected URL or on anything a client sends, and turning it off takes effect without a restart. +### Hosted image results in Codex App + +When a routed Responses provider returns a completed hosted `image_generation_call` +with base64 image data, opencodex saves the validated image under its local +`artifacts/` directory and delivers a final assistant image message to a locally +connected Codex client. The image stays outside the collapsible progress section. +This applies to streaming and non-streaming Responses, without another generation +request or a change to the selected provider. + +When a client replays these generated image messages as assistant history, opencodex +replaces its generated local image links with opaque artifact references before +forwarding that history. This protects the display paths without changing the image +message already shown in the app. It does not redact unrelated user-supplied paths. + +This display compatibility requires loopback admission and a recognized Codex client. +Remote and generic API clients retain the provider's hosted response format. +Partial previews and URL-only results are not rendered by this compatibility layer. +Artifacts use the existing retention limit, so save images you want to keep before +older files are pruned. An invalid image or a failed local write produces a visible +failure message instead of a broken image link. + ### Built-in image generation (`image_gen`) Codex's built-in `image_gen` tool does not go through `/v1/responses` — the codex-rs extension @@ -1004,6 +1025,8 @@ If the new OAuth credential's authenticated usage lookup confirms an exhausted 5 In **Codex Set → Multi-auth**, enable the **Codex credits** switch in the **Codex Auth** header to display each main and pool account’s latest observed credits directly below Week. It is off by default and persists as `showCodexCredits`. The balance is a locale-formatted number, with Unlimited or an overage warning when reported; the bar indicates availability, not a percentage, because no total credit limit is supplied. Hiding credits changes display only, and a new login waits for its own observation. +When an account reaches 100% on a usage window and still holds credits, upstream keeps serving it and draws the balance. OpenCodex does not let that happen by default: an account at 100% is switched out while its weekly or monthly window (only monthly on 30-day plans) or its 5-hour window is full, and used again once that window resets. The order of the other accounts does not change, and when no other account is available, selection finds none rather than spending credits. A request for the main account is refused like a hard-lock refusal until the reset. A full 5-hour window without a reset time holds an account only while that reading is fresh. Allowing the main account does not lift its hard lock (on by default at 98%), which still stops it first; turn the lock off if the main account should spend credits (a lock at 100% still stops it at 100%). To let accounts keep working from their credits, turn on **Use credits** next to the **Codex credits** switch in the Codex Auth header. The switch allows every account. To choose accounts one by one, open an account card's **⋯** menu and use **Use credits after limit** there; the header switch shows a middle position when only some accounts are on. An account allowed to spend carries a **Uses credits** badge on its card. New accounts start off. The choice is stored in `creditCodexAccountIds` and never redeems reset credits; the **Codex credits** display switch only shows balances and never changes routing. + Background revalidation is separate and off by default. It requires Token Guardian, the `openai` provider's `proactive` refresh policy, and `tokenGuardian.codexWarmupEnabled`. It skips accounts awaiting deferred registration validation. ### Cancelling main-account device reauthentication @@ -1395,3 +1418,7 @@ The process exits 0 only if all four live scenarios pass, 1 otherwise, and 2 for invalid arguments or missing credentials. This is a **wire diagnostic**, not an end-to-end Codex App/CLI interface test, live certification or instruction to enable the experimental feature for production work. + +## Streaming line endings + +The shared SSE decoder accepts LF, CRLF and standalone CR line endings, even when a delimiter spans network chunks. This allows compatible providers to stream events without requiring LF-only framing. diff --git a/docs-site/src/content/docs/guides/combos.md b/docs-site/src/content/docs/guides/combos.md index 90fba7f99d6..e68510d4718 100644 --- a/docs-site/src/content/docs/guides/combos.md +++ b/docs-site/src/content/docs/guides/combos.md @@ -207,12 +207,23 @@ order. Weights and `stickyLimit` do not affect this strategy. This ranking and provider exclusion before dispatch require fresh model-inference limits that apply to the current single API key as a whole. OAuth/current-account summaries, caller-forward routes, multiple keys, and snapshots with changed credentials or destinations are display-only for this early decision. The same applies when `Authorization`, `x-api-key`, or `x-goog-api-key` headers override credentials; search-only and MCP-only windows are excluded. If no eligible target has an applicable reset, configuration order wins. Account selection and retries still enforce their normal limits. -### JEV: decision-guided first pick +### Decision method -`jev` asks [TypeSafe JEV](https://console.typesafe.ai) to choose the first eligible target and a -compatible reasoning effort for the current request. It is opt-in: adding the TypeSafe credential -does not change existing models, aliases, defaults, or Combo behavior. A JEV-backed model appears -only after you create a Combo whose strategy is `jev`. +`strategy: "jev"` asks a decision backend to choose the first eligible target and a compatible +reasoning effort for the current request. It is opt-in: create a JEV Combo and select that Combo +to use it. Adding a decision credential does not change existing models, aliases, or defaults. + +| Method | Combo selection | Backend reported in stats | +| --- | --- | --- | +| TypeSafe (default) | Omit both selectors, or set `decisionProvider: "jev"`. | `typesafe` | +| System One-compatible server | Set `decisionProvider` to a configured `jev-decision` row id other than `jev`. | `systemone` | +| opencodex model | Set `decisionModel` to an ordinary opencodex route, such as `ollama/qwen3:4b`. | `model` | + +`decisionProvider` and `decisionModel` are mutually exclusive. The backend is derived from those +fields; there is no stored `backend` setting. Every method uses the same eligible target and effort +allowlist, bounded decision state, timeout, cancellation, and fail-open policy. + +#### TypeSafe default The quickest setup is: @@ -263,13 +274,179 @@ the standard provider-derived alias `JEV_API_KEY` printed by `ocx provider add`. } ``` -OpenCodex sends one bounded decision request to the fixed +With the default method, OpenCodex sends one bounded decision request to the fixed `https://api.typesafe.ai/v1/systemone` endpoint with model `jev-latest`. Only currently eligible configured targets are offered. JEV chooses the target and effort together; the effort is still constrained by that target's advertised ladder. JEV is not asked again if the selected target has a retryable failure—the existing Combo cooldown and fallback loop continues through the remaining configured targets. +#### System One-compatible server + +A JEV Combo can ask a hosted or self-hosted server that implements the System One wire contract. +For example, an Ollama server with System One support can serve `tev1` at +`POST /v1/systemone` without an API key. Add a provider row with +`adapter: "jev-decision"` whose `baseUrl` is the **full** decision endpoint, then name it in the +Combo's `decisionProvider`: + +```json +{ + "providers": { + "ollama-tev1": { + "adapter": "jev-decision", + "baseUrl": "http://127.0.0.1:11434/v1/systemone", + "allowPrivateNetwork": true, + "defaultModel": "tev1:4b", + "liveModels": false + } + }, + "combos": { + "jev-local": { + "strategy": "jev", + "decisionProvider": "ollama-tev1", + "decisionTimeoutMs": 60000, + "reasoningEffortMode": "adaptive", + "targets": [ + { "provider": "openai", "model": "gpt-6-astra" }, + { "provider": "openai", "model": "gpt-5.6-sol" }, + { "provider": "openai", "model": "gpt-5.6-luna" } + ] + } + } +} +``` + +- The row's `baseUrl` must be the full decision endpoint and its path must end in `/systemone`. + The decision model is `defaultModel`, else the first `models` entry; a row with neither is + treated as unusable and fails open without a request (`jev-latest` is TypeSafe's model and is + never sent to a self-hosted host). The row is a decision service only: it is never published as a + routable model and cannot be a Combo target. +- A loopback or LAN endpoint needs `allowPrivateNetwork: true` set explicitly on that row. Use a + literal loopback, RFC 1918, or IPv6 ULA address for plain `http:`; every resolved address must stay + in the allowed set and no outbound proxy may apply (add the + host to `NO_PROXY`). Every other destination must use HTTPS. Redirects still fail open. +- Only the row's own `apiKey` is sent, and only when it is set; a keyless row sends no + `Authorization` header. A row whose `apiKey` references `${TYPESAFE_API_KEY}`/`${JEV_API_KEY}` or + another provider's keychain entry is refused as unusable, so `TYPESAFE_API_KEY`, `JEV_API_KEY`, and + the `jev` row's key are never sent to a self-hosted endpoint. The `jev` id itself (explicit or + omitted) always means the TypeSafe endpoint with model `jev-latest`. +- Self-hosted services receive each target/effort option as a plain description string (for + example `Target openai/gpt-5.6-sol (provider openai, model gpt-5.6-sol) with low reasoning + effort.`), because Ollama accepts only string or `null` option descriptions. TypeSafe keeps + receiving the structured option objects. +- OpenCodex offers 2–26 options to a System One-compatible row. Each target contributes one option + per offered reasoning effort; use per-target `reasoningEfforts` to trim them. Fewer than 2 or + more than 26 options fail open without a request. +- A cold model load can take tens of seconds, and an aborted decision request makes Ollama abandon the + load. Pre-warm the model and keep it resident (`OLLAMA_KEEP_ALIVE=-1`, or `keep_alive`), and raise + `decisionTimeoutMs` (1000–120000 ms, default 4000) when the service is slower than four seconds. + Every timeout or error still fails open to the first eligible target. + +The provider's **Test connection** sends a bounded probe decision to its endpoint. +**Create JEV Auto** on a System One provider (keyless rows included) prefills that row as the +decision provider. Disabled rows, endpoints not ending in `/systemone`, and rows without a model +are shown with a reason and cannot be picked. + +[Laya MLX](https://github.com/mizorewww/laya-mlx) runs typed decisions on Apple Silicon. To use it +with this method, expose it through a System One-compatible HTTP wrapper, then configure a row like +`ollama-tev1` above with the wrapper's full `/systemone` URL and accepted model id as `defaultModel`. +The MLX Python runtime alone is not an HTTP decision endpoint. For a loopback HTTP wrapper, keep +`allowPrivateNetwork: true` and use its literal loopback address. + +For a hosted example, a contributor reported the following Zen System One endpoint working in +[#6185](https://github.com/lidge-jun/opencodex/pull/6185). Configure it manually as an ordinary row; +this is not an OpenCodex registry preset or a guarantee of current availability: + +```json +{ + "providers": { + "zen-decision": { + "adapter": "jev-decision", + "baseUrl": "https://opencode.ai/zen/v1/systemone", + "defaultModel": "jev-1.13-free", + "apiKey": "${ZEN_DECISION_API_KEY}", + "liveModels": false + } + }, + "combos": { + "jev-zen": { + "strategy": "jev", + "decisionProvider": "zen-decision", + "targets": [ + { "provider": "openai", "model": "gpt-6-astra" }, + { "provider": "openai", "model": "gpt-5.6-luna" } + ] + } + } +} +``` + +Use that endpoint's own key if authentication is required; omit `apiKey` only if the endpoint +accepts keyless calls. Never reuse `TYPESAFE_API_KEY` or `JEV_API_KEY` for it. + +#### opencodex model + +An ordinary chat or Responses model can make the same routing choice without implementing +`/systemone`. Configure its inference provider as usual and set `decisionModel` to its route: + +```json +{ + "combos": { + "jev-chat": { + "strategy": "jev", + "decisionModel": "ollama/qwen3:4b", + "decisionTimeoutMs": 60000, + "targets": [ + { "provider": "openai", "model": "gpt-6-astra" }, + { "provider": "openai", "model": "gpt-5.6-luna" } + ] + } + } +} +``` + +Other examples are `deepseek/deepseek-v4-flash`, `openai/gpt-5.6-luna`, or +`opencode-zen/` for a configured Zen inference model. These are ordinary route strings, +not decision-service presets; the provider and model must be available on your installation. +For `openai/gpt-5.6-luna`, configure credentials the internal call can use, such as a stored Codex +pool login. A caller-owned ChatGPT forward login is insufficient. + +The model receives fixed router instructions and one JSON prompt: + +```json +{ + "state": { "user_task": "Review the requested change" }, + "options": { "": "Target description and allowed reasoning effort" } +} +``` + +`state` is the bounded evidence described below; `options` maps generated option keys to plain +descriptions. The reply contract is `{"choice":""}`, naming exactly one offered option. +The instructions treat state as evidence, prefer lower resource use among adequate options, and +require JSON only. An invalid or unlisted choice, malformed response, or failed call fails open. + +The internal decision turn carries **no caller credential**, caller headers, tools, session, or +conversation history. It must use credentials stored for the selected provider, or a genuinely +keyless local provider such as Ollama. Caller-auth-only configurations, such as Cursor without a +stored credential or a caller-owned ChatGPT forward login, cannot serve as decision models and +fail open. Selecting a route is not proof that its credential is usable; test it before relying on it. + +A non-JEV Combo can be used as `decisionModel`, but the decision route cannot name this Combo or +any Combo with `strategy: "jev"`. This applies to canonical `combo/` selectors and aliases, +including selectors with effort or Fast variants. Save-time validation rejects recursion and the +runtime also blocks JEV reentry defensively. The decision turn has its own send budget and turn +lease; it is not logged as a separate request, and reported decision usage belongs to the parent +request's `jevDecision`. + +#### Shared limits, state, and statistics + +`decisionTimeoutMs` applies to all three methods: an integer from 1000 to 120000 ms, default 4000. +Requests and responses are bounded to 64 KiB. The candidate list is bounded to 64 targets; the +model method also caps target/effort options at 64, response text at 4096 characters, and the +decision turn's output at 1024 tokens. The +System One row's 2–26 option limit applies as described above. A timeout or limit failure uses the +first currently eligible target; it never expands the allowlist or retries the decision. + For each JEV target, **Models → Combos → Config** has an optional **Additional model notes for JEV** field (up to 512 characters; line breaks and tabs are allowed, other control characters are rejected). It is stored as `targets[].modelProfile` in the combo config. The built-in target profile remains in the trusted `instructions.model_profiles`; a non-empty note is @@ -278,7 +455,7 @@ than replaces that built-in profile. Notes can describe operator-specific contex allowances; do not confuse subscription allowances with public per-token API pricing. Blank notes are ignored. Operator notes are evidence for the decision, not commands, and cannot expand the target allowlist or reasoning-effort limits. Only put information there that may be disclosed to -TypeSafe. +the selected decision backend. Each logical model call is decided on its own; there is no per-conversation pin. Consecutive turns of one session can therefore land on different targets, and every switch starts a cold provider prompt @@ -288,16 +465,16 @@ from JEV under `cooldownWaitPolicy: "before-last-resort"` while any normal targe offered only when nothing else is reachable. The decision boundary fails open when the key is missing, no safe task/tool/image decision state is -available, the four-second decision deadline expires, the service redirects or returns an error, or +available, the configured decision deadline expires, the service redirects or returns an error, or the response is malformed or selects an unlisted choice. In those cases OpenCodex uses the first currently eligible target, preferring `medium` when that target supports it. Caller cancellation is different: it cancels the decision and the model request instead of dispatching the fail-open target. The decision state is deliberately bounded: up to 500 characters of the current user task, a 240-character previous-assistant tail, a 520-character latest-tool-output tail, the tool name, and -boolean image/tool signals may be sent to TypeSafe. It excludes the JEV credential, request headers, -raw image bytes, tool arguments, encrypted reasoning, and full conversation history. Do not select -`jev-auto` for content you do not want TypeSafe to process. Recognized OpenCodex machine-context +boolean image/tool signals may be sent to the selected decision backend. It excludes credentials, +request headers, raw image bytes, tool arguments, encrypted reasoning, and full conversation history. +Use a decision method only for content you are willing to send to that backend. Recognized OpenCodex machine-context envelopes are removed from all three text samples, but ordinary assistant and tool-output text is not a secret scanner and may still contain sensitive content. TypeSafe states that Jev is not trained on customer requests, but its terms set no fixed retention period for submitted state and @@ -310,11 +487,13 @@ Automated tests use mocked TypeSafe responses plus a no-key fail-open smoke; a l requires an operator-supplied key and is not run implicitly. After the Combo has served requests, open **Models → Combos → jev-auto → Stats** to inspect JEV's -picks without replacing the normal model picker or Usage page. The tab separates TypeSafe decision +picks without replacing the normal model picker or Usage page. The tab separates backend-reported decision tokens from tokens reported by physical model sends, and shows decision gates, fail-open picks, reasoning efforts, retries/fallbacks, cache tokens, latency, confidence, and per-model totals for 7 days, 30 days, or all available history. Statistics come from the local append-only usage ledger; -they contain the bounded decision metadata described above, not prompts or credentials. +they contain the bounded decision metadata described above, not prompts or credentials. Backend +rows show decision count, applied count, and average latency for `typesafe`, `systemone`, and +`model`; older records without a backend are grouped as `unknown`. ## What happens when a target fails @@ -527,7 +706,9 @@ task workflow. Open the local dashboard and choose **Models → Combos**. The workspace creates, edits, renames, and removes combos, and its target picker excludes disabled models, nested combos, and the credential-only JEV provider. **Create JEV Auto** opens the same Combo editor with an editable decision target template; -an existing `jev-auto` id or alias is reported instead of creating a duplicate. +an existing `jev-auto` id or alias is reported instead of creating a duplicate. A JEV Combo also shows +**Decision service** (TypeSafe JEV or a configured `jev-decision` provider) and **Decision timeout +(ms)**, and the Combos overview lists each JEV Combo's decision service, endpoint, and timeout. Each target also shows a live quota badge: **Available**, **Out of quota**, or **Quota unknown**. The editor blocks Save and Create for quota only when every usable target has a current server-confirmed exhausted inference limit for its configured credential. Display-only account, model, search and MCP quota, or missing or expired routing evidence, does not cause this block. The block expires at the applicable reset or freshness boundary and is rechecked when the page becomes active or visible; Refresh reloads both Combo data and quota. The dashboard editor does not yet expose `cooldownMs` or `waitForCooldownMs`; use the configuration file or management @@ -545,8 +726,9 @@ ocx combo remove --yes ``` `set` also accepts `--strategy`, `--sticky`, `--effort`, `--alias`, `--native-alias`, -`--display-name`, and `--rename-from`. Use `-` as the value of `--effort`, `--alias`, or -`--display-name` to clear that field. `--native-alias` requires a currently supported bare native +`--display-name`, `--decision-provider`, `--decision-timeout`, and `--rename-from`. Use `-` as the +value of `--effort`, `--alias`, `--display-name`, `--decision-provider`, or `--decision-timeout` to +clear that field. The two decision flags apply only to `--strategy jev`. `--native-alias` requires a currently supported bare native model alias and a non-empty display name. `create` and `update` are aliases for `set`; `delete` is an alias for `remove`; and the same subcommands are available under `ocx route combo`. @@ -562,9 +744,14 @@ the request-rate fallback. A stored `cooldownMs` can only be removed by editing `waitForCooldownMs` resets to its default when a `PUT` explicitly sends `0`, because the sparse serializer omits that default. Omission preserves both values and the dashboard does not expose them yet. Omitting `defaultEffortMode`, `reasoningEffortMode`, `imageInput`, or `cooldownWaitPolicy` likewise -keeps the stored value, and a re-sent target without `lastResort` keeps that target's flag (matched by -provider and model). The dashboard always sends `imageInput` and `reasoningEffortMode`, so switching -them back to `auto` or `strict` there still replaces the stored value. +keeps the stored value. For a request that keeps `strategy: "jev"`, the decision method is kept +only when both `decisionProvider` and `decisionModel` are omitted: sending either one replaces the +stored method, so `decisionModel: null` without `decisionProvider` selects TypeSafe. +`decisionTimeoutMs` is kept on its own whenever it is omitted. A different strategy drops all three, +and a re-sent target without `lastResort` keeps that target's flag (matched by +provider and model). The dashboard always sends `imageInput` and `reasoningEffortMode`, and for a JEV +Combo `decisionProvider`, `decisionModel` and `decisionTimeoutMs` (`null` for the default), so switching them back to +the default there still replaces the stored value. For the complete persisted configuration, see [Configuration](/reference/configuration/). @@ -606,6 +793,8 @@ Combos are stored in the top-level `combos` object, keyed by combo id: | `alias` | No | none | Optional trimmed public model id; use the alias rules above. An empty value is stored as no alias. | | `nativeAlias` | No | `false` | Explicitly permit a currently supported bare native `alias` to take routing and catalog precedence. Never inferred from the alias. | | `displayName` | No | none | Bounded display-only catalog label. Required and non-empty when `nativeAlias` is true. | +| `decisionProvider` | No | `"jev"` | JEV only. Provider id of the decision service: `"jev"` (TypeSafe, valid without a provider row; the same as omission) or a configured `adapter: "jev-decision"` row with a `/systemone` `baseUrl`, such as a self-hosted Ollama `tev1`. | +| `decisionTimeoutMs` | No | `4000` | JEV only. Integer from 1000 to 120000: the decision deadline before failing open to the first eligible target. | ## Troubleshooting diff --git a/docs-site/src/content/docs/guides/desktop-app.md b/docs-site/src/content/docs/guides/desktop-app.md index 64f06a5dd2c..03116c3125d 100644 --- a/docs-site/src/content/docs/guides/desktop-app.md +++ b/docs-site/src/content/docs/guides/desktop-app.md @@ -59,6 +59,11 @@ it from the tray or launch the app again. Use the tray's **Open dashboard** or **Open in browser** action to move between the embedded dashboard and your normal browser. The tray also provides update checks. +On Windows, approving **Take over** allows up to 90 seconds of startup work for the existing +runtime to stop safely, ownership to transfer, and the bundled runtime to start. Time spent +deciding at the prompt does not count toward this limit. Keep the app open while it finishes; +if it fails, use **Retry** to resolve the current runtime again. + On macOS, closing the dashboard keeps the app running in the menu bar. Open OpenCodex again from Dock or Finder to restore the dashboard without restarting the proxy. ## Startup safety on macOS diff --git a/docs-site/src/content/docs/guides/integrations.md b/docs-site/src/content/docs/guides/integrations.md index 4cb6ead058c..b4488b926d8 100644 --- a/docs-site/src/content/docs/guides/integrations.md +++ b/docs-site/src/content/docs/guides/integrations.md @@ -21,11 +21,17 @@ file, and removes it again. Seventeen clients work this way, each with a switch: | ZCode | `~/.zcode/v2/config.json` | JSON | on restart | loopback placeholder | | Aside | `~/.aside/u//models.json` | JSON | after fully quitting and reopening Aside | loopback placeholder | | Raycast | `~/.config/raycast/ai/providers.yaml` | YAML | immediately on save — Raycast watches the file | none — loopback only | -| omo | `~/.omo/agent/models.json` | JSON | new sessions | loopback placeholder | +| omo (Pi / senpi) | `~/.omo/agent/models.json` | JSON | new sessions | loopback placeholder | | Cline CLI | `~/.cline/data/settings/providers.json` and sibling `models.json` | JSON pair | after stopping and restarting Cline | loopback placeholder | | Kilo | first existing `kilo.jsonc`, `kilo.json`, `opencode.jsonc`, `opencode.json`, or `config.json` under `~/.config/kilo` | JSONC | new sessions | `OPENCODEX_KILO_API_KEY` | | Factory Droid | `~/.factory/settings.json` (`%USERPROFILE%\.factory\settings.json` on Windows) | JSON | immediately via file watching | none — keyless loopback | +"omo" names three products that share the `~/.omo` folder. The **omo** tab manages Pi-based omo +(the senpi engine) through `~/.omo/agent/models.json`, as in the table above. Codex-based omo +(LazyCodex) gets its own controls on the Codex tab, described in +[omo (Codex / LazyCodex) role models](#omo-codex--lazycodex-role-models). OpenCode-based omo +(oh-my-opencode) keeps its own config under OpenCode; opencodex does not read or write it. + Generated catalogs include only enabled models from each provider selection. This applies to both downloads and managed integrations, including Pi and Aside. The management model list still shows the full roster so you can enable additional models. @@ -521,6 +527,64 @@ key) in the app's API key field. The app sends it as `Authorization: Bearer`, wh `/v1/chat/completions` accepts as proxy admission and never forwards upstream; see the [authentication matrix](/reference/proxy-formats/#authentication-matrix). +## omo (Codex / LazyCodex) role models + +When LazyCodex is installed, the Codex tab shows an **omo (Codex / LazyCodex)** section listing +every Codex agent role found in `$CODEX_HOME/agents/*.toml`, with the model each one is pinned +to. Codex runs a role on that pin no matter which model the parent asks for, so this is where a +role's model is actually decided. LazyCodex counts as installed when the `omo@sisyphuslabs` +Codex plugin is enabled in `$CODEX_HOME/config.toml` and an installed copy under +`$CODEX_HOME/plugins/cache/sisyphuslabs/omo/` carries its `lazycodex-install.json`. A `~/.omo` +folder on its own does not count, because Pi-based omo creates it too. Without LazyCodex the +section is hidden and the command line reports it as not installed. Pick a model on a row and +press Save: + +- opencodex rewrites only the root `model = "..."` line of that role's file. The role's + instructions, comments, and other keys are left exactly as they were. A role with no pin gets + one added near the top of the file. +- The same value is written to `codex.agents..model` in `~/.omo/omo.jsonc`, which + LazyCodex 5.1.1 and later reads. If that file does not exist it is not created. If it contains + comments it is left untouched, because saving would remove them; the tab says so, and you can + set the value there by hand. + +Nothing happens until you press Save; syncing or restarting opencodex never changes a role file. +New Codex sessions pick up the change. The same controls exist on the command line: + +```bash +ocx agent roles +ocx agent roles set explorer xai/grok-4.5 +``` + +### Auto-assign + +Auto-assign is part of omo (Codex / LazyCodex): it sits above the role table in that section and +exists only while LazyCodex is detected. Without it the dashboard shows neither, the API answers +409 `lazycodex_not_detected`, and `ocx agent roles suggest` is refused. + +Auto-assign proposes a model for every role at once. opencodex asks your +default Codex model (the root `model` in Codex `config.toml`) one question: for each role, given its +description and the start of its instructions, which capability tier (fast, standard or frontier) +and how much reasoning (glance, measured, thorough or exhaustive) does it need? That model never +picks a model. opencodex then picks the cheapest model from your picker list that reaches the tier: + +- Models listed under `codexRoleTiers` in the opencodex config (`{ "fast": [...], "standard": [...], "frontier": [...] }`) + have that tier. +- Other models with a known price are ranked by price and split evenly across the three tiers. With only + one or two priced models, the dearest is frontier and the other, if any, is standard. +- Models with no price and no listed tier are never proposed. List them to include them. + +Each proposal shows the model, the tier, the reasoning effort, a one-line reason, and what would move +it up or down. A role the model could not size clearly is shown as not sized, with the reason, and +cannot be applied. Nothing is written until you press Apply on a row or Apply all. Applying uses the +same save as picking by hand, and also rewrites the role's `model_reasoning_effort` when the file already has +one. The effort is placed on the chosen model's own levels: its lowest, its default, one above the +default, or its highest. + +```bash +ocx agent roles suggest +ocx agent roles suggest --model xai/grok-4.5 --apply +``` + ## Kilo Kilo CLI, VS Code, and JetBrains share one global config. This integration writes diff --git a/docs-site/src/content/docs/guides/protocol-paths.md b/docs-site/src/content/docs/guides/protocol-paths.md index fe47c0b3ae1..e5c2b1f38da 100644 --- a/docs-site/src/content/docs/guides/protocol-paths.md +++ b/docs-site/src/content/docs/guides/protocol-paths.md @@ -144,6 +144,11 @@ These paths still use the internal Responses bridge or are not covered: - `previous_response_id`, `store`, `background`, and compaction stay Responses features. - Adapters whose wire is none of the three APIs (Gemini, Kiro, Cursor, and others) are translated through the IR with no feature claims. +- Managed native Messages to first-party Anthropic preserve a coherent observed Claude Code CLI + user agent, session UUID and allowlisted SDK headers across token refresh and account selection. + Credential placement remains proxy-owned and beta allowlisting still applies. Missing or invalid + identity uses the existing compatibility headers; compatible hosts and generated Responses retain + their current behavior. Headers alone do not authorize requests or establish official support. - Native Chat over OAuth is not planned. Native Messages over Anthropic OAuth covers only an unpooled account; a pooled account set stays on the bridge. - The managed native Messages path forwards a caller's `anthropic-beta` values only from a short diff --git a/docs-site/src/content/docs/guides/providers.md b/docs-site/src/content/docs/guides/providers.md index 3608bea9e46..18f52701c82 100644 --- a/docs-site/src/content/docs/guides/providers.md +++ b/docs-site/src/content/docs/guides/providers.md @@ -85,7 +85,7 @@ labels local presets separately; those normally omit both `authMode` and `apiKey | --- | --- | --- | | `key` | Sends your API key (`Authorization: Bearer …`, or `x-api-key` / `api-key` per adapter). The key may be a literal or an `${ENV_VAR}` reference. | Most providers. | | `forward` | Relays **your incoming Codex auth headers** verbatim to the provider — no key stored. This is the ChatGPT-login passthrough. | OpenAI (`openai-responses` adapter). | -| `oauth` | Resolves a stored OAuth access token (auto-refreshed before expiry) and uses it as the bearer key. | xAI, Anthropic, Kimi, Kiro, Google Antigravity, Cursor, Command Code, GitHub Copilot, Nous Portal. | +| `oauth` | Resolves a stored OAuth access token (auto-refreshed before expiry) and uses it as the bearer key. | xAI, Anthropic, Kimi, Kiro, Google Antigravity, Cursor, Zed Hosted AI, Command Code, GitHub Copilot, Nous Portal. | The [`retryOn429`](/reference/configuration/) same-key 429 replay applies only to API-key providers (`authMode: "key"`). OAuth, forward, and local presets are excluded — their @@ -137,7 +137,7 @@ Two exceptions are worth knowing because you can hit them: | OrcaRouter | `ocx login orcarouter-oauth` — consent mints a user-owned, long-lived `sk-orca-…` key, and the request then carries a key | `orcarouter` — the same key pasted by hand | | Meta Muse | `ocx login meta-muse` can import a local Muse Code CLI key or start device login. Meta scopes that credential to its own CLI, so this is an unsupported use: how the calls settle is not observable from the API, and you should treat every call as billable against your account | `meta-model` is the supported path — every call is metered per token, and a Muse Code subscription does not work there | -Cursor, Kiro and Nous Portal are login-only and have no API-key equivalent. Google Antigravity is +Cursor, Kiro, Zed Hosted AI and Nous Portal are login-only and have no API-key equivalent. Google Antigravity is login-only too: `ocx login google-antigravity` signs in with your Google account over the Cloud Code Assist wire, and the `google` preset beside it is the AI Studio Gemini API — a different product reached with its own key, not a key mode for the same login. @@ -186,6 +186,7 @@ ocx login nous # Nous Portal (device grant; free + paid models) ocx login kiro # import kiro-cli credentials (or token fallback) ocx login google-antigravity ocx login cursor # standalone Cursor PKCE login +ocx login zed # Zed native-app callback login (experimental) ocx login command-code # Command Code browser OAuth (or import ~/.commandcode/auth.json) ocx login orcarouter-oauth # OrcaRouter browser consent + PKCE ocx login devin # Cognition/Devin: import Devin CLI credential, else Auth0 browser sign-in @@ -204,6 +205,7 @@ ocx logout | `kiro` | `kiro` | `https://runtime.us-east-1.kiro.dev` | The dashboard Login and Add account buttons offer Builder ID, Google, GitHub device login, or Kiro CLI. Native device login adds an account without signing `kiro-cli` out. The Kiro CLI choice imports or starts a CLI session; its Add account workflow temporarily switches the CLI account and restores it on cancellation or failure. | | `google-antigravity` | `google` | `https://daily-cloudcode-pa.googleapis.com` | Google OAuth over the Cloud Code Assist wire. Live discovery uses CCA's authenticated `v1internal:fetchAvailableModels` endpoint and publishes the agent models available to the signed-in account; the maintained catalog remains the fallback. | | `cursor` | `cursor` | `https://api2.cursor.sh` | Experimental PKCE login, live HTTP/2 transport with an opt-in HTTP/1.1 compatibility path, and account-filtered model discovery. | +| `zed` | `zed` | `https://cloud.zed.dev` | Experimental Zed Hosted AI bridge. Uses Zed's native-app RSA callback, exchanges the account token for a short-lived hosted-inference token, and discovers the account's live roster. | | `orcarouter-oauth` | `openai-chat` | `https://api.orcarouter.ai/v1` | Browser consent and key exchange use `https://www.orcarouter.ai` with S256 PKCE. The returned user-owned `sk-orca-…` API key is stored in the existing credential store and reused until revoked. | | `devin` | `devin` | `https://server.codeium.com` | Experimental unofficial Cognition/Devin bridge. Login first imports the credential the installed Devin CLI already holds (`devin auth login` writes a `devin-session-token` to its own `credentials.toml`); when none is present it opens Auth0 browser sign-in and exchanges the pasted token via Cognition's `RegisterUser` for a long-lived API key. `ocx login devin-cli` remains as a deprecated alias. Models are discovered per account with `GetCascadeModelConfigs`. Account quota comes from `GetUserStatus` under one eight-second request and body deadline: dated daily and weekly windows, plus the monthly prompt and flex credit pool for a credit-billed plan or an unknown strategy with both reset dates absent when a balance field is present. Negative used credits omit the monthly window; valid zero available credits mark it exhausted. A timed-out probe keeps the last good quota. Not shown in the dashboard preset by default. Chat and usage reporting are verified against a live account across three models. | | `github-copilot` | `openai-chat` | `https://api.githubcopilot.com` | Experimental. GitHub device flow + `copilot_internal` exchange (VS Code OAuth client). Requires an active Copilot subscription; not an official third-party API. | @@ -230,6 +232,27 @@ Studio never performs this repair. Native output schemas are outside both policy After a terminal Nous refresh failure, run `ocx login nous` to reauthenticate. +### Zed Hosted AI (experimental) + +:::caution[Unofficial — use at your own risk] +Zed does not provide or endorse this bridge. It reuses your Zed account's hosted-model +entitlement outside the Zed editor, which may be outside Zed's terms of service. Zed may rate +limit, restrict, or suspend the account. Review Zed's current terms before you sign in; the +provider stays off until you add it and run `ocx login zed` yourself, and the dashboard asks +you to acknowledge this risk before it starts the login. +::: + +Run `ocx login zed` and complete Zed's native-app sign-in in the browser. OpenCodex starts a +loopback callback, gives Zed an RSA public key, and keeps the matching private key local while +the callback returns the account identity and encrypted access token. The stored account id and +token are paired on every request; OpenCodex exchanges that credential for Zed's short-lived LLM +token before calling `https://cloud.zed.dev/completions`. + +Zed's live model roster is account-scoped display metadata. The selected model id is forwarded as +provided, and the bridge infers the hosted backend family from the live row or the model name; +model discovery is not an allowlist. The Zed access token has no refresh endpoint, so when Zed +revokes it, run `ocx login zed` again. + For the canonical Kimi Coding Plan presets (`kimi` account login and `kimi-code` API key), opencodex forwards only a caller-supplied stable `prompt_cache_key` to the Chat Completions request (the Responses wire accepts the same field, but opencodex does not send it there today); @@ -467,7 +490,7 @@ The account list marks Kiro accounts excluded from automatic selection with a re ## 3. API-key catalog -opencodex ships 100 built-in presets: 83 key-based, 13 OAuth, three local, and one default +opencodex ships 102 built-in presets: 84 key-based, 14 OAuth, three local, and one default ChatGPT-forward preset. The dashboard's **Add provider** picker opens a key provider's dashboard, validates the key, and stores it; validation is provider-specific. Notable entries: @@ -560,10 +583,20 @@ region-pinned EU routes, is at [opper.ai/models](https://opper.ai/models). Opper | Kilo | `https://api.kilo.ai/api/gateway` | | Opper | `https://api.opper.ai/v3/compat` | | TokenLab | `https://api.tokenlab.sh/v1` | +| OpenGateway | `https://apis.opengateway.ai/v1` | | GitLab Duo | `https://cloud.gitlab.com/ai/v1/proxy/openai/v1` | | Cloudflare AI Gateway | `https://gateway.ai.cloudflare.com/v1/{account-id}/{gateway}/anthropic` | | …and more | opencode zen, Vercel AI Gateway, Venice, NanoGPT, Synthetic, Qianfan, Alibaba, Parallel, ZenMux, LiteLLM | +**OpenGateway** is an OpenAI-compatible gateway operated by Sionic AI at +`https://apis.opengateway.ai/v1`. Its public catalog contains about 80 active models +(as verified on 2026-10-02); the preset automatically refreshes the live model list from +public `GET /v1/models` and keeps active Chat Completions models (plus the Responses-only `openai/o3-pro`, which is pinned to Responses). Sionic-served +`deepseek/deepseek-v4.1-flash-ultrafast` and `z-ai/glm-5.3-flash-ultrafast` are listed first. +Create a key in the [OpenGateway dashboard](https://opengateway.ai/api-keys), then run +`ocx provider add opengateway` or select **OpenGateway** in the dashboard. Chat requests +use the configured Bearer key; the public model list does not validate that key. + **TokenLab** ([sponsor](https://github.com/lidge-jun/opencodex/blob/main/SPONSORS.md)) is an OpenAI-compatible API gateway at [tokenlab.sh](https://tokenlab.sh/r/OPENCODEX), operated by TOKENLAB AI INC. diff --git a/docs-site/src/content/docs/guides/sub-agent-surface.md b/docs-site/src/content/docs/guides/sub-agent-surface.md index fc79bb1fc9e..a0a2eee2970 100644 --- a/docs-site/src/content/docs/guides/sub-agent-surface.md +++ b/docs-site/src/content/docs/guides/sub-agent-surface.md @@ -88,6 +88,13 @@ The dashboard's **Sub-agent delegation** controls three related settings: - `injectionEffort` is the optional `reasoning_effort` to request for that model. - `injectionPrompt` replaces the built-in v2 guidance text. +Not sure which model to delegate to? **Suggest** beside the delegation model asks for a short +description of the work Codex usually hands off, sizes it with one call to your default Codex model, +and proposes the cheapest sufficient model and an effort from the list the picker offers, with the +reasoning and the signals that would move it up or down. Nothing changes until you choose +**Use this**, which saves exactly like picking the model and effort by hand. Tiers come from +`codexRoleTiers` or price, as in [role auto-assign](/guides/integrations/#auto-assign). + `multiAgentGuidanceEnabled` defaults to on and is the master switch for opencodex-authored guidance on both surfaces. Turning it off suppresses both the v2 designation block and v1 proactive text. @@ -243,6 +250,7 @@ Use `ocx agent` for delegation, roster, effort-cap, and fallback settings: ```bash ocx agent status ocx agent injection set --model anthropic/claude-sonnet-5 --effort xhigh +ocx agent injection suggest "read-only searches across the repo" --apply ocx agent subagents set gpt-5.6-sol,anthropic/claude-sonnet-5 ocx agent fallback set gpt-5.6-luna,xai/grok-4.5 --poll-ms 60000 ocx effort set --subagent max diff --git a/docs-site/src/content/docs/guides/web-dashboard.md b/docs-site/src/content/docs/guides/web-dashboard.md index 7af6c4dc4c9..3a4c01848ff 100644 --- a/docs-site/src/content/docs/guides/web-dashboard.md +++ b/docs-site/src/content/docs/guides/web-dashboard.md @@ -71,6 +71,11 @@ host and port over a LAN IP or an alias. ## Dashboard layout +Responses first-output timing includes streamed function arguments and custom-tool input, as well +as text and reasoning. A tool-only turn can therefore have a first-output time even without prose. +Empty deltas and tool-start notifications do not start this timer. It measures the proxy's first +observed output, not the start of hidden model reasoning or exact model decoding throughput. + Overview uses matching status cards and full-width settings rows. On wide screens, labels share one column and model/effort controls share another. On narrower screens, controls move below their labels in the same reading order. Long version labels are shortened visually; hover the version @@ -385,3 +390,9 @@ While browser authentication is pending, the dashboard does not recommend restar ### Usage chart keyboard and touch controls Usage heatmap days have one Tab entry point. Use Up/Down for adjacent days and Left/Right for adjacent weeks. Weekly bars expose the same day details on keyboard focus, pointer hover, or touch. Day labels include the date, request count, and token count; tooltip overlays stay within the viewport. + +### Claude + +The **Claude** sidebar page sits directly below **Codex**. One header and tab strip stay in place while you switch tabs, ordered Account, Code, Desktop, Settings. The Account tab shows what the Anthropic provider's **Accounts** tab shows on **Providers**: login, the browser option, the Claude account roster with switch, pause, remove, and reauthentication, the paste-code field, account pool settings, and quota. Provider-level controls such as the default provider, removal, and the enabled switch stay on **Providers**. When Anthropic is not configured, **Add Anthropic** starts Claude sign-in directly, after the same risk notice the Add provider dialog shows. Opening Claude without a tab selects Account when Anthropic is configured, otherwise Code. Code holds the Claude connection switch; Settings shows that connection as On or Off, interception as Running or Stopped, and the intercept port. When interception is stopped, Settings shows why and offers **Start interception**, which starts it in place without restarting OpenCodex. On Desktop, one status row shows whether Claude Desktop runs the profile and whether it is saved, with Save and Save & apply beside it. Compatibility, agent instructions, and context controls remain on Code. + +Bookmarks select a tab directly: `#claude/account`, `#claude/code`, `#claude/desktop`, and `#claude/settings`. The former `#integrations/claude` and `#integrations/claude/desktop` bookmarks redirect to Code and Desktop. Arrow keys move between tabs; Home and End select the first and last tab. diff --git a/docs-site/src/content/docs/ja/getting-started/quickstart.md b/docs-site/src/content/docs/ja/getting-started/quickstart.md index 3760250bc91..370d2cddbc5 100644 --- a/docs-site/src/content/docs/ja/getting-started/quickstart.md +++ b/docs-site/src/content/docs/ja/getting-started/quickstart.md @@ -18,7 +18,7 @@ ocx init `ocx init` では次の手順を説明します。 -1. **プロバイダーを選択してください** — 100 個の組み込みレジストリプリセットのいずれか、または `custom` を選択してベース URL とアダプターを入力します。 +1. **プロバイダーを選択してください** — 102 個の組み込みレジストリプリセットのいずれか、または `custom` を選択してベース URL とアダプターを入力します。 2. **API キー** — キーを貼り付けるか、`${ANTHROPIC_API_KEY}` のような環境変数を参照します。 3. **デフォルト モデル** — キー、ローカル、カスタム プロバイダーの場合は、プリセットを受け入れるか、モデル ID を入力します。 4. **プロキシ ポート** — デフォルトは `10100` です。 diff --git a/docs-site/src/content/docs/ja/guides/providers.md b/docs-site/src/content/docs/ja/guides/providers.md index 282f7335c05..0d8ec20f6ce 100644 --- a/docs-site/src/content/docs/ja/guides/providers.md +++ b/docs-site/src/content/docs/ja/guides/providers.md @@ -193,7 +193,7 @@ Kiro のログインには Kiro CLI が必要です。Unix では `curl -fsSL ht ## 3. API キーカタログ -opencodex には組み込みプリセットが 100 個含まれています。キー方式 83、OAuth 13、ローカル 3、 +opencodex には組み込みプリセットが 102 個含まれています。キー方式 84、OAuth 14、ローカル 3、 デフォルト ChatGPT 転送プリセット 1 です。ダッシュボードの **Add provider** ピッカーはキー発行ページを開き、 入力したキーを検証した後保存します(検証はプロバイダー固有です)。主な項目は以下のとおりです: @@ -254,10 +254,20 @@ Cline IDE/CLI のみで API からは使えません。`minimax/minimax-m2.5` | Xiaomi MiMo | `https://api.xiaomimimo.com/anthropic` | | Xiaomi MiMo (OpenAI Chat) | `https://api.xiaomimimo.com/v1` | | Kilo | `https://api.kilo.ai/api/gateway` | +| OpenGateway | `https://apis.opengateway.ai/v1` | | GitLab Duo | `https://cloud.gitlab.com/ai/v1/proxy/openai/v1` | | Cloudflare AI Gateway | `https://gateway.ai.cloudflare.com/v1/{account-id}/{gateway}/anthropic` | | …その他多数 | opencode zen、Vercel AI Gateway、Venice、NanoGPT、Synthetic、Qianfan、Alibaba、Parallel、ZenMux、LiteLLM | +**OpenGateway** は Sionic AI が運営する OpenAI 互換ゲートウェイです。Base URL は +`https://apis.opengateway.ai/v1` で、公開カタログには約 80 の active モデルがあります +(2026-10-02 確認)。プリセットは公開 `GET /v1/models` から一覧を自動更新し、active な +Chat Completions モデル(および Responses 専用で Responses にルーティングされる `openai/o3-pro`)を取得します。Sionic が提供する +`deepseek/deepseek-v4.1-flash-ultrafast` と `z-ai/glm-5.3-flash-ultrafast` を先頭に表示します。 +[OpenGateway ダッシュボード](https://opengateway.ai/api-keys)でキーを作成し、 +`ocx provider add opengateway` を実行するか、ダッシュボードで **OpenGateway** を選択してください。 +Chat リクエストは設定済みの Bearer キーを使います。公開一覧はキーの有効性を検証しません。 + **OpenCode Zen**(`opencode-zen`)とキー不要の **OpenCode Free** プリセットは `https://opencode.ai/zen/v1` を共有します。このゲートウェイ上の無料モデルは、しばしばおおよそ毎分 15–20 リクエストの短時間レート制限に当たります(コミュニティ計測。OpenCode は RPM を公表しません)。Zen は `Retry-After` / `X-RateLimit-*` ヘッダーなしの汎用 429 を返すことがあります。これはキー不要デスクトップ枠(`opencode-free` で Big Pickle/無料モデル約 200 回 / 5 時間)とは別です。Zen がそのような 429 で `Retry-After` を省略した場合、opencodex はクライアント向けエラーに案内を足し、合成 `Retry-After` を付けます(上流の `Retry-After` があればそれが優先されます)。同一キーの待機再試行は [`retryOn429`](/ja/reference/configuration/) でオプトインします。 diff --git a/docs-site/src/content/docs/ja/reference/configuration/providers.md b/docs-site/src/content/docs/ja/reference/configuration/providers.md index 18b16e9331e..e7f99b8cacb 100644 --- a/docs-site/src/content/docs/ja/reference/configuration/providers.md +++ b/docs-site/src/content/docs/ja/reference/configuration/providers.md @@ -222,7 +222,7 @@ affinity を維持します。これらの戦略は provider enforcement を回 | `anthropicAccountPool.quotaWindow?` | `"five-hour" \| "weekly" \| "max-utilization"` | `"five-hour"` |使用量ベースのアカウント選択で使う、プロバイダー報告のキャッシュ済み使用率です。`five-hour` は従来の動作を維持します。`weekly` は週次使用量を使い、他に対象アカウントが残る間だけ 5 時間使用量が上限に達したアカウントを除外し、残らない場合はそれらへフォールバックします。`max-utilization` は判明している値のうち最も高いものを使うため、週次使用量が未取得でも 5 時間使用量を利用できます。どちらも不明なら unknown の順位付けに従います。既知の使用量は unknown より先ですが、対象がすべて unknown でも対象順の先頭を選択します。記載した 5 時間使用量による同点判定後も完全に同点なら、対象順を維持します。正常な affinity セッションを先回りして再配置することはありません。新規セッションの割り当てと、対象となる 429 代替後のルーティング復旧では、`quota` はこの期間で対象候補を直接順位付けし、`fill-first` はこの期間のしきい値と上限到達ルールを使って安定順に進み、`round-robin` はこの設定を無視します。クールダウン、フェイルオーバー上限、再認証の適格性は別のローカル状態です。アカウント別の週次使用量は、ダッシュボードのプロバイダーページで取得した後にのみ利用できます。 | | `anthropicAccountPool.stickyLimit?` | `number` | `1` |成功した新しいセッションのバインドは 1 つのラウンドロビン選択で保持されます。範囲は 1 ~ 100。 | -有効にすると、429 レコードは `Retry-After` またはデフォルトのバックオフからの制限されたクールダウンを記録し、リクエスト内でローテーションする可能性があります。アフィニティはプロセスローカルであり、サイズ制限があります。トークン更新の失敗は既存の再認証ルールに従います。確認済みの契約・アカウント請求の 403 は出力前に別アカウントへ切り替え、`Retry-After` または既定の10分間のクールダウンを記録します。一般的な権限拒否では切り替えません。すべての対象となるアカウントが冷却されている場合、クライアントは、既知の場合、認証エラーではなく、`Retry-After` を含む 429 を受け取ります。 +共有の5時間・週次クォータが拒否された429だけがアカウントのクールダウンと切り替えを行います。一時的な速度制限はアフィニティを維持して送信受付だけを停止し、リクエストごとに短い同一アカウント再試行1回と別の対象アカウントへの切り替え1回までです。根拠ヘッダーのない429は同一アカウントで1回だけ再試行し、アカウントを冷却せず、Retry-Afterも生成しません。既定の単一アカウント動作は変わりません。Fable専用の拒否はSonnetを制限しません。手動選択とアフィニティもリクエストモデルの共有・ファミリークォータを確認します。パッシブなファミリー情報は30分または既知のリセットで期限切れとなり、1つの送信リクエストで再確認します。使用量しきい値はソフトな優先設定で、全候補消耗時のフォールバックを維持します。使用量・課金のハード上限ではありません。アフィニティはプロセスローカルであり、サイズ制限があります。トークン更新の失敗は既存の再認証ルールに従います。確認済みの契約・アカウント請求の 403 は出力前に別アカウントへ切り替え、`Retry-After` または既定の10分間のクールダウンを記録します。一般的な権限拒否では切り替えません。すべての対象となるアカウントが冷却されている場合、クライアントは、既知の場合、認証エラーではなく、`Retry-After` を含む 429 を受け取ります。 :::caution[実験的] Anthropic アカウント ポリシーのリスクを理解していない限り、これは無効のままにしてください。不明な場合は、`ocx account use anthropic ` を手動で切り替えることをお勧めします。 @@ -329,6 +329,8 @@ Anthropic アカウント ポリシーのリスクを理解していない限り Grok 4.7 は OAuth で Fast を利用でき、`low` / `medium` / `high` / `xhigh` と 500,000 トークンのコンテキストを提供します。[xAI の標準料金](https://docs.x.ai/developers/models/grok-4.7)は 100 万トークンあたり入力 2.00 ドル、キャッシュ入力 0.50 ドル、出力 6.00 ドルです。コンテキストが 200,000 トークン以上の場合は 4.00 / 1.00 / 12.00 ドルになります。 +プロバイダーの `fastWire` を明示的に設定していない場合、`allowedModels` で制限した opencodex API キーの OAuth Fast リクエストには、`xai/grok-4.7-build-fast`(またはプロバイダー名を除いたモデル ID)の許可が必要です。`xai/grok-4.7` だけの許可では、この Fast モデルを利用できません。Fast モデルだけを許可したキーでも利用できますが、通常のリクエストや Fast が無効な場合には `xai/grok-4.7` の許可が必要です。`fastWire` を明示的に設定した場合は、実際に送信されるモデルを許可してください。例えば、`service-tier` 方式では `xai/grok-4.7` が維持され、そのモデルの許可が必要です。プロバイダーの制限も引き続き適用されます。 + ## OpenRouter プロバイダーのルーティング OpenRouter は、複数の推論プロバイダーを通じて 1 つのモデルを提供できます。 `openRouterRouting` は優先プロバイダーでリクエストを保持します。 `modelOpenRouterRouting` は、正確なモデル ID に置き換えられます。キャッシュのサポート、保持、ヒット率、価格は推論プロバイダーによって異なるため、これはプロンプト キャッシュ アフィニティに役立ちます。 diff --git a/docs-site/src/content/docs/ko/getting-started/quickstart.md b/docs-site/src/content/docs/ko/getting-started/quickstart.md index 7ddac09c0d8..5c5ae7d9fb1 100644 --- a/docs-site/src/content/docs/ko/getting-started/quickstart.md +++ b/docs-site/src/content/docs/ko/getting-started/quickstart.md @@ -18,7 +18,7 @@ ocx init `ocx init`은 다음 과정을 안내합니다: -1. **프로바이더 선택** — 내장 레지스트리 프리셋 100개 중 하나를 고르거나 `custom`을 선택해 base URL과 adapter를 직접 입력합니다. +1. **프로바이더 선택** — 내장 레지스트리 프리셋 102개 중 하나를 고르거나 `custom`을 선택해 base URL과 adapter를 직접 입력합니다. 2. **API 키** — 키를 붙여넣거나 `${ANTHROPIC_API_KEY}` 같은 환경 변수를 참조합니다. 3. **기본 모델** — 키, 로컬, custom 프로바이더에서는 프리셋을 그대로 쓰거나 모델 ID를 직접 입력합니다. 4. **프록시 포트** — 기본값은 `10100`입니다. diff --git a/docs-site/src/content/docs/ko/guides/codex-integration.md b/docs-site/src/content/docs/ko/guides/codex-integration.md index 9eea66614ad..1c069f19f19 100644 --- a/docs-site/src/content/docs/ko/guides/codex-integration.md +++ b/docs-site/src/content/docs/ko/guides/codex-integration.md @@ -404,3 +404,7 @@ opencodex가 managed [background service](/ko/reference/cli/lifecycle/#백그라 ## 메인 계정 재인증 취소 메인 계정의 기기 코드 재인증을 취소할 때 DELETE 요청의 일시적 실패, 네트워크 오류, 알 수 없거나 아직 종료되지 않은 상태의 응답이 발생하면 진행 중인 흐름과 취소 실패 표시를 유지하여 취소를 다시 시도할 수 있게 합니다. 일반적으로는 상태 조회도 계속하므로 로그인이 완료되면 이를 확인할 수 있습니다. 흐름이 `pending` 또는 `committing`일 때 재시도 가능한 취소 실패와 GET 상태 조회의 2xx 이외 HTTP 응답이 겹치면, 응답 도착 순서와 관계없이 서버가 마지막으로 제공한 기기 코드·확인 URL·진행 단계를 유지하거나 복원하여 같은 흐름의 취소를 다시 시도할 수 있게 합니다. GET의 HTTP 실패는 상태 조회를 종료하지만, 두 번째 로그인 POST를 보내지 않고 취소를 다시 시도할 수 있습니다. 종료 상태인 `failed` 응답은 흐름을 해제하고 정규화된 실패 사유를 표시하며, `succeeded` 응답만 로그인 성공을 알립니다. `cancelled`로 확인된 응답은 흐름을 해제하여 새 기기 코드 로그인을 시작할 수 있게 합니다. HTTP 404와 `unknown_flow` 코드가 명확하게 반환된 경우에도 만료된 흐름 ID를 해제하여 새 기기 코드 로그인을 시작할 수 있게 하지만, 로그인 성공이나 취소 확정으로 표시하지 않습니다. 이전 흐름에서 늦게 도착한 POST·GET·DELETE 응답은 새 흐름을 변경하거나 새 흐름의 로그인이 성공했다고 알릴 수 없습니다. + +## 스트리밍 줄바꿈 + +공유 SSE 디코더는 LF, CRLF 및 단독 CR 줄바꿈을 처리하며 네트워크 청크 사이에서 구분자가 나뉘어도 동작합니다. 호환 제공자는 LF 형식으로만 스트림을 만들 필요가 없습니다. diff --git a/docs-site/src/content/docs/ko/guides/providers.md b/docs-site/src/content/docs/ko/guides/providers.md index a3184d953ce..e36b1876e29 100644 --- a/docs-site/src/content/docs/ko/guides/providers.md +++ b/docs-site/src/content/docs/ko/guides/providers.md @@ -190,7 +190,7 @@ Kiro 로그인에는 Kiro CLI가 필요합니다. Unix에서는 `curl -fsSL http ## 3. API 키 카탈로그 -opencodex에는 빌트인 프리셋이 100개 들어 있습니다. 키 방식 83개, OAuth 13개, 로컬 3개, +opencodex에는 빌트인 프리셋이 102개 들어 있습니다. 키 방식 84개, OAuth 14개, 로컬 3개, 기본 ChatGPT 포워드 프리셋 1개입니다. 대시보드의 **Add provider** 선택기는 키 발급 페이지를 열고, 입력한 키를 검증한 뒤 저장합니다(검증은 프로바이더별로 다릅니다). 주요 항목은 다음과 같습니다: @@ -252,10 +252,20 @@ Cline IDE/CLI에서만 제공되며 API로는 사용할 수 없습니다. `minim | Xiaomi MiMo | `https://api.xiaomimimo.com/anthropic` | | Xiaomi MiMo (OpenAI Chat) | `https://api.xiaomimimo.com/v1` | | Kilo | `https://api.kilo.ai/api/gateway` | +| OpenGateway | `https://apis.opengateway.ai/v1` | | GitLab Duo | `https://cloud.gitlab.com/ai/v1/proxy/openai/v1` | | Cloudflare AI Gateway | `https://gateway.ai.cloudflare.com/v1/{account-id}/{gateway}/anthropic` | | …그 외 다수 | opencode zen, Vercel AI Gateway, Venice, NanoGPT, Synthetic, Qianfan, Alibaba, Parallel, ZenMux, LiteLLM | +**OpenGateway**는 Sionic AI가 운영하는 OpenAI 호환 게이트웨이입니다. Base URL은 +`https://apis.opengateway.ai/v1`이며 공개 카탈로그에는 활성 모델이 약 80개 있습니다 +(2026-10-02 확인). 프리셋은 공개 `GET /v1/models`에서 모델 목록을 자동 갱신하고 활성 +Chat Completions 모델(그리고 Responses 전용이라 Responses로 보내는 `openai/o3-pro`)만 가져옵니다. Sionic이 제공하는 +`deepseek/deepseek-v4.1-flash-ultrafast`와 `z-ai/glm-5.3-flash-ultrafast`를 먼저 표시합니다. +[OpenGateway 대시보드](https://opengateway.ai/api-keys)에서 키를 발급한 뒤 +`ocx provider add opengateway`를 실행하거나 대시보드에서 **OpenGateway**를 선택하세요. +Chat 요청은 설정한 Bearer 키를 사용하며 공개 모델 목록은 키의 유효성을 검증하지 않습니다. + **OpenCode Zen**(`opencode-zen`)과 키 없는 **OpenCode Free** 프리셋은 `https://opencode.ai/zen/v1`을 공유합니다. 그 게이트웨이의 무료 모델은 종종 분당 약 15–20회 요청의 짧은 창 속도 제한에 걸립니다(커뮤니티 측정; OpenCode는 RPM을 공개하지 않음). Zen은 `Retry-After` / `X-RateLimit-*` 헤더 없는 일반 429를 반환할 수 있습니다. 이는 키 없는 데스크톱 할당량(`opencode-free`에서 약 5시간당 Big Pickle/무료 모델 200회)과 별개입니다. Zen이 그런 429에서 `Retry-After`를 생략하면 opencodex는 클라이언트 오류에 안내를 더하고 합성 `Retry-After`를 붙입니다(업스트림 `Retry-After`가 있으면 그것이 우선). 동일 키 대기 재시도는 [`retryOn429`](/ko/reference/configuration/)로 선택합니다. diff --git a/docs-site/src/content/docs/ko/reference/cli.md b/docs-site/src/content/docs/ko/reference/cli.md index da0ed2b0248..d3dbd395ad2 100644 --- a/docs-site/src/content/docs/ko/reference/cli.md +++ b/docs-site/src/content/docs/ko/reference/cli.md @@ -40,3 +40,19 @@ Windows x64 설치 관측은 [`attest` 명령](/ko/reference/cli/agents/)을 참 `ocx --version`, `ocx -v`, `ocx version`은 스크립트가 읽기 좋은 한 줄짜리 버전을 출력하고 종료합니다. 일반 도움말에는 두 개의 디스패치 대상이 의도적으로 빠져 있습니다. `__refresh-version [preview]`는 분리된 프로세스에서 업데이트 알림 캐시를 새로 고치고, `__gui-update-worker [latest|preview] [restart]`는 대시보드 업데이트 작업을 실행합니다. 이들은 구현 세부 사항일 뿐이며 안정적인 사용자 명령이 아닙니다. 대시보드는 worker PID를 기록하고, worker가 죽은 활성 작업은 복구하며, PID가 없는 오래된 활성 기록은 10분 뒤 오래된 것으로 취급하고, 살아 있는 worker를 동시 업데이트로부터 보호합니다. + +## Capability argument validation + +`ocx capabilities`는 알 수 없는 인수, 반복된 플래그, 빈 `--route` 값을 종료 코드 64로 거부합니다. 유효한 경로에 등록된 기능이 없으면 종료 코드 4를 반환합니다. + +## Integer option values + +`--limit` 같은 정수 옵션은 안전하게 표현 가능한 십진 정수를 받습니다. `1_000`, `1,000` 같은 숫자 구분자는 허용합니다. 빈 값, 16진수, 지수 표기 및 소수는 요청을 보내기 전에 거부합니다. + +## Windows JSON configuration files + +`ocx config validate `과 `ocx config import --yes`는 선행 BOM이 있는 UTF-8 JSON도 읽습니다. 표준 입력(`-`)에도 적용되므로 Windows PowerShell이나 편집기의 UTF-8 내보내기를 사용할 수 있습니다. UTF-16은 지원하지 않습니다. + +## Default alias listing + +`ocx alias --json`은 `ocx alias list --json`과 같습니다. `--json`은 명시한 alias 작업의 앞이나 뒤에 둘 수 있습니다. diff --git a/docs-site/src/content/docs/ko/reference/cli/agents.md b/docs-site/src/content/docs/ko/reference/cli/agents.md index 1a26b811bf9..38ea7dc4cd8 100644 --- a/docs-site/src/content/docs/ko/reference/cli/agents.md +++ b/docs-site/src/content/docs/ko/reference/cli/agents.md @@ -263,3 +263,15 @@ ocx system codex-cli-update attest --candidate --npm-prefix `가 표시됩니다. 이 값을 `ocx logs explain `에 넣어 라우팅 결정을 확인할 수 있습니다. ID가 없거나 ID에 제어 문자가 있으면 조회 키를 바꾸어 표시하지 않고 이 항목을 생략합니다. JSON과 JSONL은 원래 ID와 형식을 유지합니다. + +## Routing profile lookup status + +`ocx route policy show `는 프로필이 없으면 종료 코드 4를 반환합니다. 명령 인수가 빠졌거나 잘못되면 2를 반환하므로 스크립트에서 두 경우를 구별할 수 있습니다. + +## Upstream error details + +업스트림 오류 응답에 여러 메시지 필드가 있으면 기존 우선순위에서 처음 발견한 비어 있지 않은 문자열을 사용합니다. 빈 값이나 잘못된 형식의 필드 때문에 유효한 후순위 진단이 사라지지 않습니다. diff --git a/docs-site/src/content/docs/ko/reference/cli/lifecycle.md b/docs-site/src/content/docs/ko/reference/cli/lifecycle.md index 0d3f98f1010..dc6d38b8ee3 100644 --- a/docs-site/src/content/docs/ko/reference/cli/lifecycle.md +++ b/docs-site/src/content/docs/ko/reference/cli/lifecycle.md @@ -457,3 +457,7 @@ npm에 게시하면 사용할 수 있게 됩니다. ## Remote Hub 클라이언트 라이프사이클 `ocx connect --pairing-code-stdin`, `ocx connect status`, `ocx sync`, `ocx connect rotate --pairing-code-stdin`을 사용합니다. `ocx disconnect`는 오프라인에서도 로컬 상태를 복원하지만 허브 키는 폐기하지 않습니다. 연결 중에는 `ocx connect revoke --admin-token-stdin`으로 저장된 `apiKeyId`를 폐기할 수 있고, 연결을 끊은 뒤에는 허브의 **Integrations → API Keys**를 사용해야 합니다. 비밀값은 stdin으로만 전달하고 argv에 넣지 마세요. + +## Setup port validation + +`ocx init`의 포트는 1–65535 범위의 십진 정수입니다. Enter만 누르면 10100을 사용합니다. `0`, `10100oops`, `1.5` 같은 잘못된 값은 임의로 바꾸거나 자르지 않고, 오류를 알린 뒤 포트를 다시 묻습니다. diff --git a/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md b/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md index 12ddcad4aac..6a0a272b8d1 100644 --- a/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md @@ -74,9 +74,9 @@ ocx login anthropic ## 계정과 키 풀 -### 메인 계정 98% 보호 +### 메인 계정 사용량 보호 -**Codex 설정 → 다중 인증 → 고급 설정**에서 Ultra Fast 옆의 **메인 계정 98% 차단**이 +**Codex 설정 → 다중 인증 → 고급 설정**에서 Ultra Fast 옆의 **메인 계정 차단**이 기본으로 켜져 있습니다. 끄면 바로 적용되고, 다시 켤 때는 영향 안내가 먼저 나오며 취소하면 설정은 바뀌지 않습니다. 고급 설정을 접어도 메인 계정 카드에 보호 상태와 사용량 확인 필요 여부, 현재 차단 여부가 표시됩니다. @@ -88,10 +88,23 @@ ocx login anthropic Reserve입니다. 차단 중에는 그 메인 계정의 Reserve를 활성화할 수 없으므로, 메인 계정을 끝까지 쓰고 Reserve로 넘어가려면 스위치를 끄세요. -**5h 창과 주간 창은 각각 따로 차단합니다.** 둘 중 하나만 98%에 닿아도, 다른 창에 여유가 +**5h 창과 주간 창은 각각 따로 차단합니다.** 5h 창이 90%, 주간 창이 98%에 닿으면 다른 창에 여유가 있어도 바로 차단합니다. 월간 전용 계정은 월간을 기준으로 합니다. 차단한 창이 모두 새 사용률 -98% 미만(0% 리셋 포함)을 보고하면 자동으로 차단을 풀고, 스위치는 켜 둡니다. 이후 다시 98%가 -되면 차단합니다. 5h 수치를 읽지 못해도 주간 차단은 그대로 유지됩니다. +각 기준 미만(0% 리셋 포함)을 보고하면 자동으로 차단을 풀고, 스위치는 켜 둡니다. 이후 다시 각 기준에 +닿으면 차단합니다. 5h 수치를 읽지 못해도 주간 차단은 그대로 유지됩니다. +`config.json`의 `codexMainAccountHardLockThresholds`에 `short`와 `long`을 설정할 수 있습니다. +기본값은 `{ "short": 90, "long": 98 }`입니다. 둘 다 80~100 범위의 정수여야 하고, +`short`는 `long`보다 클 수 없습니다. 생략한 값은 기본값을 사용합니다. 설정 API도 같은 객체를 +받으며 `mainAccountHardLock.thresholds`에 실제 적용값을 반환합니다. + +같은 리셋 구간에서 최신 사용량이 1%포인트 이상 늘었는데 이 프록시를 통한 메인 계정 활동이 +최근에 없었다면 대시보드와 `ocx status`에 **opencodex 밖에서 사용량이 늘었을 수 있음**을 표시합니다. +오래 실행한 턴도 같은 현상을 만들 수 있어 외부 사용의 확정 증거는 아닙니다. 외부 사용 때문에 +차단해도 한도를 소진할 수 있습니다. 알림은 프로세스 메모리에만 유지되며 계정이나 리셋 구간이 +바뀌면 지워지고, 6시간 경과 또는 관측한 리셋 시각 중 이른 시점에 만료됩니다. +`ocx status --json`은 실행 중인 관리 API에서 읽은 상태, 적용 기준, 차단 창 종류와 선택적 +`externalUsage` 알림을 `mainAccountHardLock`에 담습니다. + 값이 빠진 응답을 0%로 보지 않으며, 이미 확인한 차단 수치를 누락된 응답만으로 지우지도 않습니다. 예정된 리셋 시간이 지났다는 이유만으로 풀지는 않습니다. 차단 중에는 1분 주기 점검이 알려진 차단 창의 리셋 시각까지 기다린 뒤 계정의 실제 사용량을 확인합니다. 이후에도 차단 상태이거나 @@ -105,7 +118,7 @@ Pool 모드에서 사용량 조회의 `--refresh`는 캐시 유효기간을 무 1차 창이 명시적 `null`이고 2차 창에 유효한 주간 사용량이 있어도 같습니다. 어느 경우든 2차·3차 창은 명시적 `null`이거나 기간이 24시간 이상이고 유효한 사용량 수치가 있어야 합니다. 파서의 단기·장기 구분 기준을 따르므로 주간·월간뿐 아니라 하루짜리 창도 해당합니다. -현재 창에는 동일한 98% 기준을 적용합니다. 이 판단은 응답 한 건의 정보에 의존하며 연속 관측을 +현재 장기 창에는 설정된 장기 기준(기본 98%)을 적용합니다. 이 판단은 응답 한 건의 정보에 의존하며 연속 관측을 요구하지 않습니다. 창 필드가 하나라도 생략되었거나, 창의 기간을 모르거나, 응답 헤더만 일부 도착한 경우에는 이전 차단을 해제하지 않습니다. 모든 창이 `null`이고 크레딧만 있거나, 보조 월간 수치만 있는 응답도 차단을 풀지 않습니다. @@ -125,7 +138,7 @@ Direct 모드의 공급자 사용량 보고서에서도 공유 상태에 반영 계속 사용할 수 있습니다. 보호 기능이 켜져 있으면 소유권이 확인된 시작 과정에서 native 프로필 복구와 정리를 마친 뒤 -메인 인증정보의 메모리 내 식별 연결을 복원하므로, 저장된 98% 차단이 재시작 후에도 유지됩니다. +메인 인증정보의 메모리 내 식별 연결을 복원하므로, 저장된 정책 차단이 재시작 후에도 유지됩니다. 연결을 준비하는 동안 호출자 인증정보를 쓰는 Direct, 메인 계정 지정, 메인 fallback, 메인 pin 요청은 잠시 503을 받을 수 있고, 저장된 Pool 계정은 그동안에도 그대로 쓸 수 있습니다. 이 초기화를 위해 다른 서비스 소유이거나 소유권이 미확인인 홈의 인증정보를 읽지는 않습니다. @@ -161,7 +174,7 @@ Reserve가 활성화되지 않을 수 있습니다. 스위치를 끄면 원래 요청을 거절합니다. 다른 계정이나 일반 Luna로 몰래 바꾸지 않습니다. 일반 사용량 조회는 기존 허용을 취소할 수 있지만 새로 허용하지는 않습니다. -전체 쿨다운, 일시정지, 재인증, 98% 하드락은 여전히 적용됩니다. 소진된 메인 계정에서 Reserve를 +전체 쿨다운, 일시정지, 재인증, 메인 계정 하드락은 여전히 적용됩니다. 소진된 메인 계정에서 Reserve를 쓰려면 하드락을 꺼야 하지만, 껐다고 서버의 사용 권한이 생기지는 않습니다. 이 호환 경로는 대화와 대화 압축용입니다. 이미지 설명·웹 검색 보조 모델이나 독립 검색 릴레이에 Reserve를 지정하는 용도는 지원하지 않으므로, 그 기능에는 다른 모델을 선택하세요. diff --git a/docs-site/src/content/docs/ko/reference/configuration/providers.md b/docs-site/src/content/docs/ko/reference/configuration/providers.md index 3a003a48b5a..1f15a246edc 100644 --- a/docs-site/src/content/docs/ko/reference/configuration/providers.md +++ b/docs-site/src/content/docs/ko/reference/configuration/providers.md @@ -224,7 +224,7 @@ affinity 초기화 뒤의 기존 작업도 포함될 수 있습니다. 출력 | `anthropicAccountPool.quotaWindow?` | `"five-hour" \| "weekly" \| "max-utilization"` | `"five-hour"` | 사용량 기반 계정 선택에 사용하는, 공급자가 보고한 캐시 사용률 막대입니다. `five-hour`는 기존 동작을 유지합니다. `weekly`는 주간 막대를 사용하며 다른 사용 가능한 계정이 남아 있을 때만 5시간 막대가 소진된 계정을 건너뛰고, 아무 계정도 남지 않으면 해당 계정으로 폴백합니다. `max-utilization`은 알려진 값 중 가장 높은 값을 사용하므로 주간 사용량을 알기 전에도 5시간 사용량을 쓸 수 있고, 둘 다 모르면 unknown 순서를 따릅니다. 알려진 사용량은 unknown보다 앞서지만, 사용 가능한 계정이 모두 unknown이어도 사용 가능한 순서의 계정을 선택합니다. 앞서 설명한 5시간 사용량 동점 판정 뒤에도 완전히 같으면 사용 가능한 순서를 유지합니다. 정상 affinity 세션을 선제적으로 재배치하지 않습니다. 새 세션 배정과 가능한 429 대체 이후 라우팅 복구에서 `quota`는 이 창으로 사용 가능한 후보의 순위를 직접 매기고, `fill-first`는 이 창의 임계값과 소진 규칙에 따라 안정 순서로 이동하며, `round-robin`은 이 설정을 무시합니다. 쿨다운, failover 한도, 재인증 가능 여부는 별도의 로컬 상태로 유지됩니다. 계정별 주간 막대는 대시보드의 프로바이더 페이지에서 조회한 뒤에만 알 수 있습니다. | | `anthropicAccountPool.stickyLimit?` | `number` | `1` | 성공한 새 세션 결속이 한 번의 라운드로빈 선택에 유지되는 횟수입니다. 범위는 1–100입니다. | -활성화되면 429 레코드가 `Retry-After` 또는 기본 backoff에서 제한된 쿨다운을 기록하고, 요청 안에서 회전할 수 있습니다. 결속은 프로세스 로컬이며 크기가 제한됩니다. 토큰 갱신 실패는 기존 재인증 규칙을 따릅니다. 확인된 구독 또는 계정 결제 403은 출력 전에 다른 계정으로 전환하며 `Retry-After` 또는 기본 10분 동안 냉각합니다. 일반 권한 거부는 전환하지 않습니다. 적격한 계정이 모두 쿨다운 중이면, 클라이언트는 인증 오류가 아니라 알려진 경우 `Retry-After`가 포함된 429를 받습니다. +공유 5시간/주간 할당량의 거부가 확인된 429만 계정 쿨다운과 전환을 유발합니다. 일시적인 속도 제한은 affinity를 유지하며 해당 계정의 전송만 잠시 중지합니다. 요청마다 짧은 동일 계정 재시도 한 번과 적격한 다른 계정 전환 한 번까지만 허용됩니다. 근거 헤더가 없는 429는 동일 계정에서 한 번만 짧게 재시도하며 계정 상태를 냉각하지 않고 Retry-After를 만들어 붙이지 않습니다. 기본 단일 계정 동작은 유지됩니다. Fable 전용 거부는 Sonnet을 제한하지 않습니다. 수동 선택과 affinity도 요청 모델의 공유/계열 할당량을 확인합니다. 수동 조회가 아닌 응답 헤더의 계열 근거는 30분 또는 알려진 리셋 시각에 만료되며 요청 하나로 재확인합니다. 사용량 임계값은 소프트 선호이고 모든 후보가 소진되면 기존 폴백을 유지합니다. 사용량 또는 결제 하드 캡이 아닙니다. 결속은 프로세스 로컬이며 크기가 제한됩니다. 토큰 갱신 실패는 기존 재인증 규칙을 따릅니다. 확인된 구독 또는 계정 결제 403은 출력 전에 다른 계정으로 전환하며 `Retry-After` 또는 기본 10분 동안 냉각합니다. 일반 권한 거부는 전환하지 않습니다. 적격한 계정이 모두 쿨다운 중이면, 클라이언트는 인증 오류가 아니라 알려진 경우 `Retry-After`가 포함된 429를 받습니다. :::caution[실험적 기능] Anthropic 계정 정책 위험을 이해하지 못한다면 이 기능은 꺼두십시오. 확신이 없으면 수동 `ocx account use anthropic ` 전환을 우선하십시오. @@ -328,6 +328,8 @@ Cursor 서버 주도 로컬 도구는 기본값으로 비활성화됩니다. Cod Grok 4.7은 OAuth에서 Fast를 지원하며, `low` / `medium` / `high` / `xhigh`와 500,000토큰 컨텍스트 창을 제공합니다. [xAI 표준 요금](https://docs.x.ai/developers/models/grok-4.7)은 100만 토큰당 입력 $2.00, 캐시 입력 $0.50, 출력 $6.00이며, 컨텍스트가 200,000토큰 이상이면 각각 $4.00 / $1.00 / $12.00입니다. +공급자 `fastWire`를 명시적으로 설정하지 않았다면, `allowedModels`로 제한한 opencodex API 키의 OAuth Fast 요청에는 `xai/grok-4.7-build-fast`(또는 공급자 접두사 없는 모델 ID)의 허용이 필요합니다. `xai/grok-4.7`만 허용하면 이 Fast 모델을 사용할 수 없습니다. Fast 모델만 허용한 키도 사용할 수 있지만, 일반 요청이나 Fast가 비활성화된 요청에는 `xai/grok-4.7` 허용이 필요합니다. `fastWire`를 명시적으로 설정했다면 실제 전송되는 모델을 허용해야 합니다. 예를 들어 `service-tier` 방식은 `xai/grok-4.7`을 유지하므로 해당 모델의 허용이 필요합니다. 공급자 제한도 계속 적용됩니다. + ## OpenRouter 공급자 라우팅 OpenRouter는 하나의 모델을 여러 추론 공급자로 제공할 수 있습니다. `openRouterRouting`은 요청을 선호하는 공급자에 유지하고, `modelOpenRouterRouting`은 정확한 모델 id에 대해 이를 대체합니다. 캐시 지원, 유지 시간, 히트율, 가격이 추론 공급자마다 다르기 때문에 프롬프트 캐시 결속에 유용합니다. diff --git a/docs-site/src/content/docs/reference/adapters.md b/docs-site/src/content/docs/reference/adapters.md index 1dae3494998..130e9e91e27 100644 --- a/docs-site/src/content/docs/reference/adapters.md +++ b/docs-site/src/content/docs/reference/adapters.md @@ -655,6 +655,20 @@ quietly switching tier. Resolution never moves to another family. When the account catalog is unavailable, the effort is appended to the id instead. +## `zed` + +**Targets:** Zed Hosted AI's `POST /completions` endpoint at `cloud.zed.dev`. +**Auth:** Zed native-app account identity plus access token, exchanged for a short-lived LLM token. + +- Uses the native RSA callback login (`ocx login zed`) and pairs the returned `user_id` with the + access token for account-scoped user and organization lookups. +- Wraps the existing Anthropic Messages, Google Gemini, OpenAI Responses, and xAI Chat builders + inside Zed's provider envelope, then unwraps Zed's NDJSON/SSE events back into `AdapterEvent`. +- Fetches a bounded, account-scoped live model roster for display and provider-family inference; + the roster is not a model allowlist, so a caller-supplied model id is still forwarded. +- Experimental, unofficial, and not endorsed by Zed. Using it may break Zed's terms of service + and can get the Zed account limited or suspended; that risk is the user's to accept. + ## `azure-openai` (alias: `azure`) **Targets:** **Azure OpenAI**. Wraps `openai-responses` (so also `passthrough: true`). diff --git a/docs-site/src/content/docs/reference/cli.md b/docs-site/src/content/docs/reference/cli.md index dbfcd400066..2cf78314711 100644 --- a/docs-site/src/content/docs/reference/cli.md +++ b/docs-site/src/content/docs/reference/cli.md @@ -156,3 +156,19 @@ refreshes the update-notification cache in a detached process, and implementation details, not stable user-facing commands. The dashboard records the worker PID, recovers an active job whose worker died, treats older PID-less active records as stale after ten minutes, and protects a live worker from concurrent updates. + +## Capability argument validation + +`ocx capabilities` rejects unknown arguments, repeated flags and blank `--route` values with exit 64. A valid route with no declared capability exits 4. + +## Integer option values + +Integer options such as `--limit` require decimal whole numbers within JavaScript safe-integer bounds. Digit separators such as `1_000` and `1,000` are accepted. Empty values, hexadecimal, exponent notation and fractions are rejected before a request is sent. + +## Windows JSON configuration files + +`ocx config validate ` and `ocx config import --yes` accept UTF-8 JSON with or without a leading BOM, including stdin (`-`). This supports UTF-8 exports from Windows PowerShell and editors. UTF-16 input is not accepted. + +## Default alias listing + +`ocx alias --json` is equivalent to `ocx alias list --json`. The output flag can precede or follow an explicit alias action. diff --git a/docs-site/src/content/docs/reference/cli/agents.md b/docs-site/src/content/docs/reference/cli/agents.md index 933e2fc6baa..727a5699fc8 100644 --- a/docs-site/src/content/docs/reference/cli/agents.md +++ b/docs-site/src/content/docs/reference/cli/agents.md @@ -7,7 +7,7 @@ These commands control agent policy and routing, inspect the live proxy, and con ## Agent policy -### `ocx agent ...` +### `ocx agent ...` Manage the headless multi-agent roster, effort caps, prompt injection, fallback, and sidecar settings. Use `status` for the current policy. See [Sub-agent surfaces](/guides/sub-agent-surface/) for how @@ -17,6 +17,28 @@ surface modes, delegation, effort, and fallback behavior fit together. ocx agent subagents set ark/model-a,openai/gpt-5.5 ``` +`ocx agent roles` is for omo (Codex / LazyCodex). It lists each Codex agent role in +`$CODEX_HOME/agents` with its model pin and whether `~/.omo/omo.jsonc` can be updated, or says +LazyCodex is not installed, in which case `set` is refused. `ocx agent roles set ` rewrites only +that role's root `model` line and mirrors the value into omo.jsonc at +`codex.agents..model`. A missing omo.jsonc, or one containing comments, is left unchanged +and the command says so. See [omo (Codex / LazyCodex) role models](/guides/integrations/#omo-codex--lazycodex-role-models). + +```bash +ocx agent roles set explorer xai/grok-4.5 +``` + +`ocx agent roles suggest` is omo (Codex / LazyCodex) only, refused like `set` when LazyCodex is not +installed. It sizes every role with one call to the default Codex model (or `--model`) and +prints a proposed model and effort per role without writing anything. `--apply` writes every proposal +through the same write as `set`, skipping and naming the roles whose model and effort already match. +See [Auto-assign](/guides/integrations/#auto-assign). + +`ocx agent injection suggest ` does the same for the delegation model: it sizes the described +work, proposes the cheapest sufficient model and an effort from the delegation picker's list, and writes +nothing unless `--apply` is given, which saves through the same write as `injection set`. See +[Delegation model and effort](/guides/sub-agent-surface/#delegation-model-and-effort). + `ocx agent sidecar web --list` and `ocx agent sidecar vision --list` print the models the server currently offers for each sidecar — the exact filtered set the dashboard picker shows (picker-visible rows plus the login-entitled Luna/Haiku auth slots, intersected with executor @@ -130,6 +152,18 @@ unqualified bare OpenAI model id. Bare `gpt-5.6-*` native aliases use Codex Pool Account-qualified OpenAI routes remain distinct, while provider-qualified routes such as `openai-apikey/gpt-5.6-*` use their configured API key and never fall through to the native alias. Read the safety and visibility contract in the guide before enabling the compatibility pair. +With `--strategy jev` only, the decision method is chosen by one of two mutually exclusive flags. +`--decision-provider ` names a configured `jev-decision` row (for example a self-hosted +Ollama `tev1`) as a System One-compatible server. `--decision-model ` names an ordinary +opencodex route (for example `ollama/qwen3:4b`) that answers the same choice as JSON; it cannot be +this combo or any JEV combo. Omitting both uses TypeSafe. `--decision-timeout ` sets the +decision deadline (1000–120000, default 4000); `-` clears any of the three. + +`ocx combo test [--combo ] [--decision-provider | --decision-model ] +[--decision-timeout ]` sends one synthetic decision probe through a saved combo's method or an +unsaved selection and reports the gate, backend, and latency; it may spend one decision call. +`ocx combo discover [--query ]` lists configured System One rows and catalog models that look +like decision models, with the derived endpoint. See [Combos](/guides/combos/) for routing behavior and configuration guidance. @@ -451,3 +485,15 @@ backup can restore the configuration; store exported files as secrets. `ocx usage` reads the connected hub with this client's enrolled data key. Human output identifies the hub source and client-key scope; `--json` returns the same scoped data. Range, surface, provider/model filters and custom `--since`/`--until` bounds remain available. Account breakdowns and other clients' records are not shared. An old or unavailable hub produces an explicit error instead of substituting local usage; upgrade the hub if it does not support this read. The read-only data-plane endpoint is `GET /v1/usage`, using `x-opencodex-api-key` with a configured client key. Environment-wide and admin keys are refused. It accepts `range`, `surface`, `provider`, `model`, `since`, and `until`; unknown/repeated options and caller-selected key IDs are rejected. Oversized skipped rows retain the explicit incomplete-history warning. + +## Explain a listed request + +Human `ocx logs` output includes `id=`. Pass that value to `ocx logs explain ` to inspect routing decisions. Rows without an ID or with control characters in their ID omit the field instead of displaying a different lookup key. JSON and JSONL output retain their existing schema. + +## Routing profile lookup status + +`ocx route policy show ` exits 4 when the profile does not exist. Missing or invalid command arguments exit 2. Scripts can distinguish a missing profile from incorrect usage. + +## Upstream error details + +When an upstream error envelope contains several message fields, OpenCodex uses the first nonblank string in its established priority order. Empty or malformed fields no longer hide a valid fallback diagnostic. diff --git a/docs-site/src/content/docs/reference/cli/lifecycle.md b/docs-site/src/content/docs/reference/cli/lifecycle.md index 64c9eedd31e..64d990df618 100644 --- a/docs-site/src/content/docs/reference/cli/lifecycle.md +++ b/docs-site/src/content/docs/reference/cli/lifecycle.md @@ -149,6 +149,10 @@ current port holder and retry the restart after the conflict is resolved. Idempotently ensure a background proxy is running, then sync its live model catalog. If `codexAutoStart` is `false`, it prints that autostart is disabled and does nothing. +With a validated connected-client configuration, `ensure` succeeds without starting a local +provider proxy after reconciling the client journal. This does not probe or certify the remote +hub's availability. Invalid or mismatched client state still fails. + ### `ocx restore [back]` · `ocx eject [back]` Restore native Codex **without** stopping the proxy — strips the injected config lines and routed @@ -231,6 +235,10 @@ section, so the two commands should agree on restart protection. If you are diag compare the reported live startup verdict with the local service details rather than treating the shell probe as more authoritative. +The `clients=pending-restart(...)` diagnostic lists Codex CLI clients that predate the routing +injection. On macOS, Electron renderer, utility, and crashpad helpers under Codex.app's framework +are excluded from that client list, including helpers whose executable paths contain spaces. + Human output also includes an **OAuth health** block after the OAuth logins summary: `OAuth health: ok` when every known account is healthy, or `OAuth health: warning` with one redacted line per non-healthy account (provider, masked account id, status such as reauthentication required, rate or @@ -497,6 +505,10 @@ CPU contention, making the tray report Offline even while the process is alive. run `ocx service repair` to migrate that registered priority and restart the service. This migration may request UAC approval; a priority already set to normal or high does not itself trigger replacement. +The Windows wrapper supports locale dates containing parentheses, including Korean and Japanese +date formats. After upgrading, run `ocx service repair` to replace an older generated wrapper +that exits before launching Bun on those locales. + The Windows wrapper verifies its baked Bun runtime and CLI entry before every start attempt. If an interrupted package update removed either file, it logs one `installation is incomplete` message and stops instead of retrying the same missing executable every five seconds. Reinstall opencodex, then @@ -842,3 +854,7 @@ publishes them to npm. ## Remote Hub client lifecycle Use `ocx connect --pairing-code-stdin`, `ocx connect status`, `ocx sync`, and `ocx connect rotate --pairing-code-stdin`. The initial catalog download fails after five seconds without incoming bytes, but active transfers may run longer; use `--catalog-timeout ` (1–120) to override that inactivity window. `ocx disconnect` restores local state offline and does not revoke the hub key. While connected only, `ocx connect revoke --admin-token-stdin` revokes the persisted `apiKeyId`; after disconnect use the hub's **Integrations → API Keys** page. Secrets are stdin-only and never belong in argv. + +## Setup port validation + +`ocx init` accepts TCP ports 1–65535 as decimal whole numbers. Press Enter to use 10100. Invalid input such as `0`, `10100oops` or `1.5` is never silently replaced or truncated; setup reports it and asks for the port again. diff --git a/docs-site/src/content/docs/reference/cli/providers-accounts.md b/docs-site/src/content/docs/reference/cli/providers-accounts.md index 49cf4d0f763..839bc7161b4 100644 --- a/docs-site/src/content/docs/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/reference/cli/providers-accounts.md @@ -137,9 +137,9 @@ Remove the stored OAuth credential for a provider. ## Accounts and key pools -### Main-account 98% protection +### Main-account quota protection -In **Codex settings → Multi-auth → Advanced settings**, **Block main account at 98%** +In **Codex settings → Multi-auth → Advanced settings**, **Block main account** is on by default beside Ultra Fast. Switching it off applies immediately; turning it back on first shows the consequences, and cancelling does not change the setting. The main-account card shows monitoring, unknown usage, or a current policy block even when Advanced settings is closed. @@ -151,10 +151,23 @@ account usable ([#5694](https://github.com/lidge-jun/opencodex/issues/5694)). Th Reserve: while the block is in force, Reserve on that main account cannot activate. To let the main account run to exhaustion and hand over to Reserve, turn the switch off. -The **5h window and the weekly window each block on their own**: either one reaching 98% blocks +The **5h window and the weekly window each block on their own**: the 5h window reaching 90% or the weekly window reaching 98% blocks immediately, even while the other still has headroom. Monthly-only accounts use their monthly window. The block releases automatically, with the switch still on, once every blocking window -reports a fresh reading below 98% (a 0% reset counts); the next 98% observation blocks again. +reports a fresh reading below its threshold (a 0% reset counts); the next observation at the threshold blocks again. +Configure `codexMainAccountHardLockThresholds` in `config.json` with optional `short` and `long` +percentages, for example `{ "short": 90, "long": 98 }`. Both must be integers from 80 through 100, +and `short` must not exceed `long`; omitted fields use the defaults. The settings API accepts the +same object and returns the effective values in `mainAccountHardLock.thresholds`. + +The dashboard and `ocx status` show **possible usage outside opencodex** when fresh readings rise +by at least one point in the same reset episode without recent main-account activity through this +proxy. This is an advisory observation, not proof: a long-running turn can also cause it. The lock +may not prevent exhaustion from outside traffic. The process-local notice clears on identity or +reset changes and expires after six hours or at the observed reset, whichever comes first. +`ocx status --json` exposes the state, effective thresholds, blocking window kind, and optional +`externalUsage` warning under `mainAccountHardLock` when live management evidence is available. + An unreadable 5h reading cannot hide a weekly block. Unknown usage does not fabricate a zero, and a missing reading does not erase an already measured blocking tuple. A predicted reset time alone does not unlock it. While blocked, the minute sweep waits for the latest known blocking reset, then checks owned usage. @@ -168,7 +181,7 @@ its primary window explicitly lasts **at least 24 hours** and reports valid usag explicitly `null` and a measured secondary supplies the weekly usage. In either case, secondary/tertiary windows must be explicitly `null` or explicitly last at least 24 hours and report valid usage. This follows the parser's short/long boundary, so a one-day window qualifies as well as weekly/monthly windows. The current window still uses the same -98% threshold. This relies on the single reported snapshot; repeated observations are not required. +configured long-window threshold (98% by default). This relies on the single reported snapshot; repeated observations are not required. Any omitted window field, an unknown duration, or partial response headers cannot clear a previous block. All-null responses carrying only credits and supplementary monthly-only readings also cannot establish recovery. The proxy checks the stored credential again before applying a delayed response. An unreadable file @@ -190,7 +203,7 @@ keyring credentials, and traffic outside the proxy can still spend quota. Added providers remain available. With protection on, an owned startup restores the main credential's in-memory identity -binding after native-profile recovery and cleanup, so a persisted 98% block survives a restart. +binding after native-profile recovery and cleanup, so a persisted policy block survives a restart. Caller-owned Direct, exact-main, main-fallback, and main-pin requests can briefly receive 503 while that binding is pending; healthy stored Pool accounts stay eligible throughout. No credential is read from a foreign or unconfirmed service home for this initialization. @@ -224,7 +237,7 @@ Each compatibility request checks a credential-bound server authorization, cache requires ordinary usage to be disallowed, the Luna Reserve banner, and exactly one allowed Reserve bucket. Missing, denied, stale or mismatched evidence refuses the request; it does not switch accounts or silently use ordinary Luna. Passive usage can revoke authorization but cannot create it. -Global cooldown, pause, reauthentication and the 98% hard lock still apply. Turn the hard lock off if +Global cooldown, pause, reauthentication and the main-account hard lock still apply. Turn the hard lock off if you want to use Reserve on an exhausted main account; doing so does not grant server entitlement. This compatibility path supports conversation requests and compaction, not Reserve as a vision or web-search helper or a standalone search-relay model. Choose another model for those helpers. @@ -799,3 +812,6 @@ ocx account routes anthropic --clear ``` The file is limited to 64 KiB. The server validates each route and stores it under `anthropicAccountPool.routes`; writes require the running proxy. Use stored account IDs from `ocx account list anthropic --json`. Rules only affect the enabled pool and never claim that an account is entitled to a model. Request logs identify a matched rule as `route:#` (1-based list position), without its operator name. + +Vision and web-search helpers match their own model independently and can fail locally when their +strict route has no eligible account. See [Anthropic helper account routing](/reference/configuration/providers/#anthropicaccountpool-experimental). diff --git a/docs-site/src/content/docs/reference/configuration/agents.md b/docs-site/src/content/docs/reference/configuration/agents.md index 0680b278d31..ed106c598e0 100644 --- a/docs-site/src/content/docs/reference/configuration/agents.md +++ b/docs-site/src/content/docs/reference/configuration/agents.md @@ -56,6 +56,7 @@ still depends on upstream support for your account. | `subagentModelFallback?` | `string[]` | `[]` | Priority-ordered global fallback models for spawned child turns. | | `subagentModelFallbackByModel?` | `Record` | `{}` | Per-primary-model fallback chains, keyed by the requested primary model id. This is the supported home for per-role fallback metadata; `model_fallback` inside Codex agent TOML makes Codex 0.146+ skip the role (#1190). | | `subagentModelFallbackPollMs?` | `number` | `60000` | Availability-probe cache interval. Values below 1000 ms fall back to the default. | +| `codexRoleTiers?` | `{ fast?, standard?, frontier?: string[] }` | — | Capability tier of named models for [role auto-assign](/guides/integrations/#auto-assign). Listed models take that tier; other models are tiered by price, and unpriced unlisted models are never proposed. | | `effortCap?` | `string` | — | Hard ceiling for qualifying v2 main turns and marked spawned-child turns. Accepts `low` through `ultra`. | | `subagentEffortCap?` | `string` | — | Additional ceiling for spawned-child turns only. When both caps apply, the lower wins. | | `plaintextV2AgentMessages?` | `boolean` | — (unset) | Experimental opt-in. It runs only when explicitly set to `true` and asks eligible native ChatGPT v2 parents to emit `spawn_agent`, `send_message`, and `followup_task` message arguments as plaintext. See [Plaintext v2 agent messages](#plaintext-v2-agent-messages). | diff --git a/docs-site/src/content/docs/reference/configuration/providers.md b/docs-site/src/content/docs/reference/configuration/providers.md index 33371d1104f..04eafa67712 100644 --- a/docs-site/src/content/docs/reference/configuration/providers.md +++ b/docs-site/src/content/docs/reference/configuration/providers.md @@ -45,6 +45,7 @@ separate. Full request URLs such as `/api/v1/responses` are not provider base UR | `contextCapValue?` | `number` | `350000` | Default used on first enable. A later enable restores the selected provider value. Updating the global value with `setAll: true` changes enabled caps only; `setAll: true` without a value enables all configured providers at the current global value. | | `codexAccounts?` | `CodexAccount[]` | `[]` | ChatGPT/Codex pool account metadata managed by Codex Auth. Secrets live separately in `codex-accounts.json`. | | `pausedCodexAccountIds?` | `string[]` | `[]` | Accounts excluded from Pool selection until resumed, including the main `__main__` account when paused. | +| `creditCodexAccountIds?` | `string[]` | `[]` | Accounts allowed to keep serving from ChatGPT credits after a usage limit, including the main `__main__` account. Upstream does not refuse an account that holds credits at 100%; it serves the request and draws the balance. Spending is opt-in: an account not listed here is skipped by selection while one of its usage windows (weekly or monthly, only monthly on 30-day plans, or the 5-hour window) reads 100% with its reset still ahead, and returns once that reset passes. A weekly or monthly reading without a reset time does not hold the account; a 5-hour reading at 100% without a reset holds it only while that reading is still fresh. When no other account is available, automatic selection finds none rather than spending credits. An unlisted `__main__` is also refused, like a hard-lock refusal, when a request names it or carries its credential. Listing `__main__` does not lift the main-account hard lock (on by default at 98%), which still stops the main login first; turn the lock off to let the main account spend credits (a lock at 100% still stops it at 100%). New accounts start unlisted. Managed by the **Use credits** switch in the Codex Auth header and the **Use credits after limit** switch in each account card's **⋯** menu. | | `codexQuotaAutoRefresh?` | `Record` | `{}` | Per-Codex-login-account opt-in for automatic `fiveHour` and `weekly` window activation in Pool mode; Direct mode does not run this worker. In Providers/Codex Auth **Advanced settings**, one control enables or disables both supported windows across all current main and added accounts. New accounts are not opted in automatically. Enable skips windows absent from live WHAM data; disable also clears stale enabled windows. The UI reuses granular `/api/settings` writes, reconciles partial failures, and retries the original ON/OFF intent without replacing unrelated settings or completed reset markers. The API still rejects enabling an unavailable window with HTTP 409. At a reported reset time, opencodex sends one minimal non-stored Codex message using that account's quota and persists the activated timestamp. This does not apply to API-key providers. | | `codexAccountNamespaces?` | `Record` | — | Optional map from an arbitrary public model selector to a stored Codex account target. When account-qualified picker rows are enabled, each selector whose target is present adds separate `/` rows to the Codex picker; each row uses only that account. With any selector active, bare native rows are hidden in the picker, but their ids remain routable and listed by raw `/v1/models` unless explicitly disabled. | | `codexAccountPickerEnabled?` | `boolean` | off when the map is empty | Controls whether eligible `codexAccountNamespaces` mappings generate account-qualified Codex picker rows. `true` allows mapped rows to appear. If omitted with a non-empty map, it is treated as enabled for backward compatibility; if the map is empty, it is off. `false` hides generated rows and restores bare native picker rows without deleting mappings or disabling exact `/` routing. | @@ -212,6 +213,7 @@ Providers can expose a built-in shorthand, such as `agy` for `google-antigravity | `noProxy?` | `string \| string[]` | Destinations this provider reaches directly, using `NO_PROXY` host-pattern syntax. A match bypasses both this provider's own proxy and an inherited global proxy. | | `requestPacing?` | `{ enabled, requestsPerMinute?, minIntervalMs?, maxConcurrentRequests?, models? }` | Optional client-side outbound request-start pacing, separate from upstream usage, billing, and rate-limit indicators. RPM is converted to an even interval; `minIntervalMs` may impose a longer interval. `maxConcurrentRequests` is a positive integer cap on in-flight requests. A provider or model rule may use the concurrency cap alone; provider limits apply across all models, while `models` entries use exact upstream model IDs (for example `nvidia/llama-3.1-nemotron-ultra-253b-v1`) and can only add delay or narrow concurrency. Queue waits do not consume the upstream response-header timeout. HTTP and explicit adapter `fetchResponse`/`runTurn` dispatches are covered. A concurrency-capped canonical Responses WebSocket turn uses HTTP/SSE so its lease can be released when the response body completes, errors, or is cancelled. For `runTurn` adapters, including Cursor, the cap counts active turns rather than physical sends: RunSSE and BidiAppend may overlap within one turn, while another turn waits. Follow-up sends still obey start intervals. | | `upstreamHttpVersion?` | `"auto" \| "http1.1" \| "h1" \| "http2" \| "h2"` | Pin the HTTP version used for upstream requests to this provider. Defaults to `auto`, which lets Bun negotiate. An explicit pin requires an HTTPS target and fails locally when it cannot be honored. Set `http1.1` when a provider's HTTP/2 SSE stream stalls instead of delivering events — the symptom is a long-running streaming request that produces nothing and eventually times out. For Cursor, `http1.1`/`h1` selects its `RunSSE` + `BidiAppend` compatibility transport for inference and also pins live model discovery. Management `POST`/`PATCH` accept `null` to clear it back to `auto`. | +| `tlsProfile?` | `"antigravity-browser"` | **Use at your own risk.** Opt-in browser-like TLS handshake (through the optional `wreq-js` dependency) for the canonical `google-antigravity` OAuth provider. It changes only how the connection looks on the wire; it is not an official Google client, and it does not change what Google's terms allow. Google can still detect, rate-limit, suspend, or ban the account you signed in with, and you alone carry that risk. It is off by default and accepted only with the Google adapter, Cloud Code Assist mode, and Google's canonical HTTPS Antigravity hosts. Redirects stay manual. The provider's own `proxy`/`noProxy` route is carried by the TLS transport; a route it cannot keep fails the request instead of leaving by another path, which includes any direct route while `HTTP_PROXY`, `HTTPS_PROXY`, or `ALL_PROXY` is set. `GET /api/providers` reports the profile state as `pending`, `active`, or `failed`. Omitting the field keeps the normal Bun transport and never loads the dependency. | | `responsesPath?` | `string` | Relative resource path for key-auth `openai-responses` requests. It must start with `/` and contain no scheme, query, or fragment. | | `chatCompletionsPath?` | `string` | Relative resource path for `openai-chat` requests, the mirror of `responsesPath` and subject to the same shape rules. Needed when one upstream serves Chat Completions and Responses under different prefixes: a per-model wire override changes the adapter and leaves `baseUrl` alone, so without this an opted-in Chat request would be sent to the Responses base. Z.AI is the shipped example. | | `allowEncryptedV2AgentTasks?` | `boolean` | Disabled by default. Trust a direct key-auth `openai-responses` provider to consume or relay opaque encrypted V2 sub-agent tasks unchanged. Eligible routes skip `agentTaskRecovery`; all other routes keep the existing recovery or fail-closed behavior. OpenCodex does not decrypt, translate, or recover tasks sent through this opt-in. | @@ -729,6 +731,15 @@ record `grok-4.7-build-fast` as the wire model. API-key mode is unchanged: build public API, so Grok 4.7 Fast there still means priority processing. An explicit `xai/grok-4.7-build-fast` selection from an earlier configuration keeps working. +When no explicit provider `fastWire` is configured, an opencodex API key restricted by +`allowedModels` must permit `xai/grok-4.7-build-fast` (or its bare model id) for OAuth Fast. +Allowing only `xai/grok-4.7` does not authorize this Fast model variant. A key allowing only +the Fast wire model can use it; plain requests or disabled Fast still require `xai/grok-4.7`. +With an explicit provider `fastWire`, permit the actual wire model instead. For example, a +`service-tier` wire keeps `xai/grok-4.7` and requires permission for that model. Provider +restrictions continue to apply. +These rules apply to Responses, Chat Completions, Messages, and routed compaction requests. + xAI charges Priority Processing at 2× the standard token price for input, output, cached, and reasoning tokens; cache discounts are applied before the multiplier. Cost estimates use that premium only when xAI's response confirms `service_tier: "priority"`. A missing or unparsed response tier is @@ -863,8 +874,10 @@ Rotation does not protect against provider enforcement; multi-account use may vi ### `anthropicAccountPool` (experimental) This opt-in pools multiple Anthropic OAuth accounts already stored in `auth.json`. It is off by -default and not battle-tested. Accounts in the same organization may share quota, and automated -rotation may trigger provider restrictions. +default and experimental, and Anthropic has not endorsed automated account pooling. Accounts in the +same organization may share quota, and switching accounts does not protect against provider +restrictions. See the [Claude Code guide](/guides/claude-code/#claude-oauth-account-pool-experimental) +for the subscription and client conditions the pool is meant for. | Key | Type | Default | Description | | --- | --- | --- | --- | @@ -875,11 +888,27 @@ rotation may trigger provider restrictions. | `anthropicAccountPool.stickyLimit?` | `number` | `1` | Successful new-session binds retained on one round-robin selection. Range 1–100. | | `anthropicAccountPool.routes?` | `{name, match, accounts, fallback?}[]` | — | Ordered model routes for the enabled Anthropic OAuth pool. `match` is a full, case-sensitive model ID glob (`*` and `?`); first match wins. `accounts` contains stored account IDs, not aliases. Eligible accounts are limited to the route for initial selection and 429 retry. Without fallback, an empty route returns a local 401, or 429 with route-scoped `Retry-After` if all its declared accounts are cooling. With `fallback: true`, an all-cooling ordinary pool returns 429 with its earliest usable cooldown, even if a saved route account was removed; the client response never names the route; the proxy log uses `route:#` for the rule’s 1-based position. `fallback: true` widens only when no routed account is eligible; fill-first then uses ordinary pool order. No matching rule retains normal selection; disabling the pool makes saved routes inactive. Invalid rules fail validated writes and prevent routed dispatch until corrected. `null` clears routes through the settings API. | -When enabled, 429 records a cooldown and may rotate within the request. The cooldown length comes -from a usable `Retry-After`, otherwise from the latest valid reset time among rate-limit windows -Anthropic reports as `rejected`, including weekly windows. Valid upstream deadlines are not -shortened to a fixed cooldown ceiling; non-finite or unrepresentable deadlines are ignored. -A refusal with no usable deadline falls back to a 60-second default backoff. Affinity is process-local +Anthropic OAuth vision and web-search helpers match these routes using each helper's own model, +independently of the main request model. Each helper authenticates with its routed account, which +can differ from the globally active account. If a strict helper route has no eligible account, the +helper fails locally before any provider request; it does not silently use an account outside the +route. When the main request works but image description or web search fails, check the helper's +configured model and the accounts eligible for that model's route. + +Shared-quota 429s (rejected shared 5h/7d windows) cool the serving account and may recover +on an eligible sibling. Retry-After wins, otherwise the latest valid rejected reset is used, +with a 60-second default when no deadline is usable. A transient rate throttle pauses only +that account's admission, preserves affinity, and permits one short same-account retry plus +at most one sibling detour per request. A headerless 429 permits at most one short retry on +the same account, leaves account health intact, and adds no synthetic Retry-After. These +retries share the physical-send budget and stop on cancellation or committed streamed output. +Default single-account behavior is unchanged. The requested model's shared 5-hour/weekly and +family weekly evidence determine its pool admission. A Fable-only rejection keeps Sonnet +eligible on the same account. Passive Fable evidence ages out after thirty minutes or its +known reset and is revalidated by one serving request at a time. Usage thresholds remain +soft preferences with an all-drained fallback; these controls are not hard usage or billing caps. +Active usage probes preserve absent family windows unless the response authoritatively enumerates limits. +Affinity is process-local and size-bounded. Token-refresh credential failures retain the existing reauthentication policy. Classified pre-output account-entitlement/billing 403s clear affinity and cool the account for `Retry-After`, or ten minutes by default, before trying an eligible replacement. Generic or request-level 403s remain terminal; see [Claude account recovery](/guides/claude-code/). If all eligible accounts are cooling, clients receive 429 with `Retry-After` when known, not an authentication error. @@ -1141,6 +1170,33 @@ multi-user host. Leave local exec off unless every data-plane caller is trusted accept bypassing Codex approval and sandbox semantics. ::: +## Zed provider (`adapter: "zed"`) + +The Zed Hosted AI bridge is experimental and login-only. Run `ocx login zed` before using it; the +login stores the Zed account identity with its native-app access token in the normal OAuth store. + +```json +{ + "providers": { + "zed": { + "adapter": "zed", + "baseUrl": "https://cloud.zed.dev", + "authMode": "oauth", + "defaultModel": "auto" + } + } +} +``` + +The bridge obtains a short-lived hosted-inference token and sends `POST /completions`. The live +`/models` roster is used for account-specific picker metadata only: arbitrary model ids remain +forwardable, with the backend family inferred from the live provider field or model name. + +:::caution[Unofficial — use at your own risk] +Zed does not provide or endorse this integration, and it may be outside Zed's terms of service. +Zed may limit or suspend an account that uses it. The provider is never enabled by default. +::: + ## OpenRouter provider routing OpenRouter can serve one model through several inference providers. `openRouterRouting` keeps diff --git a/docs-site/src/content/docs/reference/configuration/routing.md b/docs-site/src/content/docs/reference/configuration/routing.md index 13def1b9b67..bce76e5594a 100644 --- a/docs-site/src/content/docs/reference/configuration/routing.md +++ b/docs-site/src/content/docs/reference/configuration/routing.md @@ -86,6 +86,11 @@ a caller `Authorization` header addressed to the source route is stripped, as fo A disabled destination, a forward destination, or a key/OAuth destination without usable stored credentials keeps the source route's legacy behavior. Under a routing policy, every redirect target must itself be a declared eligible candidate, even when it keeps the same provider. +The initial evaluation's eligible provider/model set remains fixed through fallback and subagent +recovery. Retries use concrete candidates, so a combo alias cannot replace a candidate's destination. +Different candidates that resolve to a destination already attempted do not send the turn there again. +An eligible virtual model keeps its normal wire-model mapping without adding that wire id to the +profile's allowed candidates. If a local replacement is skipped, fallback retains the last upstream failure. Malformed redirect maps are ignored with a warning on load and rejected on configuration writes. Redirected routes record `blocked-model-redirect`; omitting the setting leaves routing unchanged. @@ -146,6 +151,9 @@ namespace, and cannot use reserved bare native families such as `gpt-*`, `o1-*`, | `alias?` | `string` | — | Optional public model id in place of the canonical picker slug. | | `nativeAlias?` | `boolean` | `false` | Let a currently supported bare native id take precedence only for that unqualified id. Bare `gpt-5.6-*` ids use Codex Pool/Direct credentials. Account-qualified routes remain distinct. Provider-qualified routes such as `openai-apikey/gpt-5.6-*` use their configured API-key route and never fall through to the native alias. | | `displayName?` | `string` | — | Display-only catalog label, required and non-empty for a native alias. | +| `decisionProvider?` | `string` | `"jev"` | `strategy: "jev"` only. `"jev"` (the same as omitting it, and stored as omission) is the TypeSafe decision service, valid without a provider row; any other value must name a configured provider with `adapter: "jev-decision"` whose `baseUrl` ends in `/systemone`. | +| `decisionModel?` | `string` | unset | `strategy: "jev"` only, mutually exclusive with `decisionProvider`. An ordinary opencodex route (for example `ollama/qwen3:4b`) asked to pick one offered option as JSON. It runs with the selected provider's stored credentials, never the caller's, and cannot resolve to this combo, any JEV combo, or a `jev-decision` row. | +| `decisionTimeoutMs?` | `number` | `4000` | `strategy: "jev"` only. Decision deadline before failing open, 1000–120000 ms. | ```json { @@ -167,13 +175,58 @@ namespace, and cannot use reserved bare native families such as `gpt-*`, `o1-*`, For strategy behavior, retryable failures, cooldowns, encrypted v2 task limits, and management commands, see [Combos](/guides/combos/). -The `jev` strategy is optional and requires the canonical `jev` provider credential. That provider +The `jev` strategy is optional. Only the TypeSafe method, used when neither `decisionProvider` nor +`decisionModel` is set, requires the canonical `jev` provider credential. That provider is a decision service, publishes no directly routable model, and cannot be a Combo target. JEV sees only currently eligible members of `targets`; missing, failed, or invalid decisions use the first eligible member, while caller cancellation remains terminal. Adding the provider or Combo never changes `defaultProvider` or hides direct model rows. See -[JEV: decision-guided first pick](/guides/combos/#jev-decision-guided-first-pick) for setup, privacy -bounds, and the one-decision-per-call contract. +[Decision method](/guides/combos/#decision-method) for the three methods, setup, privacy bounds, and +the one-decision-per-call contract. + +### Self-hosted decision model (e.g. Ollama tev1) + +`decisionProvider` can point a JEV Combo at a self-hosted Jev-API-compatible decision model such as +Ollama's `tev1` (Ollama 0.35+, `POST /v1/systemone`, no API key): + +```json +{ + "providers": { + "ollama-tev1": { + "adapter": "jev-decision", + "baseUrl": "http://127.0.0.1:11434/v1/systemone", + "allowPrivateNetwork": true, + "defaultModel": "tev1:4b", + "liveModels": false + } + }, + "combos": { + "jev-local": { + "strategy": "jev", + "decisionProvider": "ollama-tev1", + "decisionTimeoutMs": 60000, + "targets": [ + { "provider": "openai", "model": "gpt-5.6-sol" }, + { "provider": "openai", "model": "gpt-5.6-luna" } + ] + } + } +} +``` + +The row's `baseUrl` is the full decision endpoint and must end in `/systemone`; its model is +`defaultModel`, else `models[0]` (a row with neither fails open without a request). Loopback needs +the row's own explicit `allowPrivateNetwork: true`, and plain `http:` is accepted only for `localhost` +or a loopback/RFC 1918/ULA address literal reached without a proxy. Only the row's own `apiKey` is +sent — none when it is unset — and a key referencing the TypeSafe environment variables or another +provider's keychain entry is refused, so TypeSafe credentials never reach it. Options are sent as +description strings, which Ollama requires. `tev1` was trained on 2–24 options, so keep target × +effort pairs at 24 or fewer (fewer than 2 or more than 26 fail open without a request); +its effective context is about 2k tokens and OpenCodex clips the task text to 500 characters. Keep the +model resident (`OLLAMA_KEEP_ALIVE=-1`) and raise `decisionTimeoutMs` for slow services; see +[System One-compatible server](/guides/combos/#system-one-compatible-server). +In the dashboard, choose **System One-compatible server** in the JEV Combo's **Decision method** +section under **Models → Combos**, then pick the row and set **Decision timeout (ms)**. ## Routing policy profiles (`config.routingProfiles`) diff --git a/docs-site/src/content/docs/reference/configuration/server.md b/docs-site/src/content/docs/reference/configuration/server.md index e97fab4cdca..e9fc0cde0ca 100644 --- a/docs-site/src/content/docs/reference/configuration/server.md +++ b/docs-site/src/content/docs/reference/configuration/server.md @@ -38,6 +38,8 @@ runs helper features around provider requests. | `codexAccountNamespaces?` | `Record` | unset | Public selector to Codex account mapping. The main login maps to the config-only sentinel `@main`; added accounts map to their account id. Normally generated by enabling `codexAccountPickerEnabled` rather than hand-authored. An empty map means no account-qualified rows exist, including the Luna Reserve row. | | `codexClientCompaction?` | `boolean` | `false` | Opt into Codex client-side compaction on an authenticated loopback bind. Uses the dedicated `opencodex` provider identity with `requires_openai_auth = true`, preventing new routed compactions from storing OpenCodeX-owned `ocx1:` state. `codexDesktopAuthless` takes precedence when both are enabled and keeps `requires_openai_auth = false`. V2 sub-agent routing is unchanged. `ocx system settings --client-compaction on`. See [Codex integration](/guides/codex-integration/#client-side-compaction-opt-in). | | `codexProviderDisplayName?` | `string` | `"OpenCodex Proxy"` | Label Codex shows for the injected `opencodex` provider, written as its `name` field in `config.toml` and the reference profile. Presentation only: routing resolves through the provider id `opencodex`, so a rename never moves `model_provider = "opencodex"` or the `[model_providers.opencodex]` header and cannot orphan threads already tagged with that id. Codex refuses to load a provider with no name, so there is no way to omit the field — choose a neutral label instead. A blank, over-128-character, or control-character value is ignored and the default label is written. | +| `codexMainAccountHardLock?` | `boolean` | `true` | Block new identity-matched main-account requests at their per-window threshold. Only explicit `false` opts out. See [main-account protection](/reference/cli/providers-accounts/#main-account-quota-protection). | +| `codexMainAccountHardLockThresholds?` | `{ short?: number; long?: number }` | `{ short: 90, long: 98 }` | Independent 5h and weekly/monthly thresholds. Integers 80–100, with `short <= long`. Invalid persisted fields fall back to defaults; the settings API rejects invalid writes before mutation. | | `resetCreditAutoRedeem?` | `{ enabled?: boolean; leadTimeMinutes?: number }` | off | Opt-in: redeem the main Codex account's soonest-expiring reset credit `leadTimeMinutes` (1–60, default 10) before it expires. Every attempt re-reads the upstream credit list first and skips when the credit is gone (for example, redeemed by hand); the `redeem_request_id` is journaled in `$OPENCODEX_HOME/reset-credit-auto-redeem.json` before the call so a crash replays the same idempotent request instead of spending a second credit. Servers sharing this configuration directory coordinate reservations and settlements so one process does not replace another's request record. Logs carry a hashed account key only. | | `syncResumeHistory?` | `boolean` | `true` | Reversible Codex App history compatibility. Original metadata is backed up and restored by `ocx stop` / `ocx restore`. | | `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | Redirect recognized Codex helper/shadow calls to a chosen model while preserving the request's configured reasoning effort. The default source prefixes are `gpt-6-luna` and `gpt-5.6-luna`; older clients through 0.144.x used `gpt-5.4-mini`, which `sourceModels` can restore. | @@ -578,7 +580,7 @@ modes, the preview, and the per-request trace. | `protocols.unrepresentable?` | `"legacy" \| "reject"` | `"legacy"` | `legacy` sends a request whose path drops a feature and records the loss in the trace. `reject` refuses it with HTTP 400 before any send, naming only the feature keys. | | `protocols.rollout.nativeChatCombos?` | `boolean` | `false` | Send an eligible Chat candidate inside a combo natively from its own copy of the client body. | | `protocols.rollout.managedMessagesNative?` | `boolean` | `false` | Send Messages natively to a direct, key-authenticated Anthropic provider instead of through the internal Responses bridge. | -| `protocols.rollout.managedMessagesNativeOAuth?` | `boolean` | `false` | Native Messages for the unpooled `anthropic` OAuth provider on `api.anthropic.com`. Read as off unless `managedMessagesNative` is on; a pooled account set stays on the bridge. | +| `protocols.rollout.managedMessagesNativeOAuth?` | `boolean` | `false` | Native Messages for the unpooled `anthropic` OAuth provider on `api.anthropic.com`. Read as off unless `managedMessagesNative` is on; a pooled account set stays on the bridge. Recognized JSON-string `metadata.user_id` account UUIDs are aligned with the serving OAuth credential; device and session fields are preserved. | | `protocols.rollout.directEncoders?` | `boolean` | `false` | Encode Chat and Messages answers from a non-Responses upstream directly from adapter events. | | `protocols.rollout.shadowPlan?` | `boolean` | `false` | Compare each Chat or Messages request's path with the plan a preview predicts and mark a disagreement as `planMismatch` on its log row. Sends nothing extra. | diff --git a/docs-site/src/content/docs/reference/management-api.md b/docs-site/src/content/docs/reference/management-api.md index 41e40d28302..2692bae66fb 100644 --- a/docs-site/src/content/docs/reference/management-api.md +++ b/docs-site/src/content/docs/reference/management-api.md @@ -75,6 +75,7 @@ route-specific results rather than repeating this table. | --- | --- | --- | | `GET, PUT /api/v2` | Read or change native multi-agent v2 mode and thread settings | 400 invalid settings; 502 transition or persistence failure | | `GET, PUT /api/injection-model` | Read or set the injected sub-agent model, effort, prompt, and guidance settings | 400 invalid model, effort, or body | +| `POST /api/injection-model/suggest` | Size a described delegated workload and propose a delegation model and effort without writing | 400 invalid work or model; 409 no sizing model | | `GET, PUT /api/effort-caps` | Read or set global and sub-agent reasoning-effort ceilings | 400 invalid ladder value | | `GET, PUT /api/subagent-models` | Read or order the models advertised to sub-agents | 400 invalid list or more than five models | | `GET, PUT /api/subagent-model-fallback` | Read or set the ordered fallback chain and poll interval | 400 invalid list or poll interval | @@ -678,6 +679,7 @@ manager. Its routes are: | `PUT /api/codex-auth/accounts/alias` | Set or clear an account alias | 400 invalid account/alias | | `PUT /api/codex-auth/accounts/pause` | Pause or resume one account | 400 invalid account/state; 404 missing account | | `PUT /api/codex-auth/accounts/pause-exhausted` | Pause accounts whose quota is exhausted | Mutation-lock failures become 503 | +| `PUT /api/codex-auth/accounts/credits` | Allow or stop spending ChatGPT credits after the usage limit. Body `{ id, creditsAfterLimit }` for one account, including `__main__`: true adds the id to `creditCodexAccountIds`, false removes it. Body `{ all }` for the global switch: true lists `__main__` and every pool account, false clears the list. Applies to the next selection. | 400 invalid id or non-boolean value; 404 missing account | | `PUT /api/settings` with `codexQuotaAutoRefresh: { id, window, enabled }` | Enable or disable 5-hour or weekly automatic window activation for one account | 400 invalid id/window/state; 404 missing account; 409 unavailable window | | `POST /api/codex-auth/accounts/clear-cooldown` | Clear runtime cooldown for one account or all accounts | 400 invalid id | | `GET, PUT /api/codex-auth/active` | Read or select the active account | 400 invalid or missing account; 409 paused/legacy-row conflict | diff --git a/docs-site/src/content/docs/ru/getting-started/quickstart.md b/docs-site/src/content/docs/ru/getting-started/quickstart.md index 9552ed50b42..3662e395830 100644 --- a/docs-site/src/content/docs/ru/getting-started/quickstart.md +++ b/docs-site/src/content/docs/ru/getting-started/quickstart.md @@ -18,7 +18,7 @@ ocx init `ocx init` проведёт вас по следующим шагам: -1. **Выбор провайдера** — выберите один из 100 встроенных пресетов реестра или `custom`, чтобы +1. **Выбор провайдера** — выберите один из 102 встроенных пресетов реестра или `custom`, чтобы ввести базовый URL и адаптер вручную. 2. **API-ключ** — вставьте ключ или сошлитесь на переменную окружения вида `${ANTHROPIC_API_KEY}`. 3. **Модель по умолчанию** — для провайдеров с ключом, локальных и `custom` примите значение из diff --git a/docs-site/src/content/docs/ru/guides/providers.md b/docs-site/src/content/docs/ru/guides/providers.md index 4bc043e2fa7..606d4739e0b 100644 --- a/docs-site/src/content/docs/ru/guides/providers.md +++ b/docs-site/src/content/docs/ru/guides/providers.md @@ -206,7 +206,7 @@ Inline JSON и лишние позиционные аргументы откло ## 3. Каталог API-ключей -opencodex поставляется с 100 встроенными пресетами: 83 на основе ключей, 13 OAuth, три локальных и +opencodex поставляется с 102 встроенными пресетами: 84 на основе ключей, 14 OAuth, три локальных и один пресет ChatGPT-форварда по умолчанию. Селектор **Add provider** в дашборде открывает страницу выдачи ключей провайдера, проверяет ключ и сохраняет его; проверка зависит от провайдера. Наиболее заметные записи: @@ -269,10 +269,20 @@ opencodex поставляется с 100 встроенными пресета | Xiaomi MiMo | `https://api.xiaomimimo.com/anthropic` | | Xiaomi MiMo (OpenAI Chat) | `https://api.xiaomimimo.com/v1` | | Kilo | `https://api.kilo.ai/api/gateway` | +| OpenGateway | `https://apis.opengateway.ai/v1` | | GitLab Duo | `https://cloud.gitlab.com/ai/v1/proxy/openai/v1` | | Cloudflare AI Gateway | `https://gateway.ai.cloudflare.com/v1/{account-id}/{gateway}/anthropic` | | …и другие | opencode zen, Vercel AI Gateway, Venice, NanoGPT, Synthetic, Qianfan, Alibaba, Parallel, ZenMux, LiteLLM | +**OpenGateway** — OpenAI-совместимый шлюз компании Sionic AI по адресу +`https://apis.opengateway.ai/v1`. Публичный каталог содержит около 80 активных моделей +(проверено 2026-10-02). Пресет автоматически обновляет список через публичный +`GET /v1/models`, оставляя активные модели Chat Completions (а также `openai/o3-pro`, доступную только через Responses и направляемую в Responses). Модели, обслуживаемые Sionic, +`deepseek/deepseek-v4.1-flash-ultrafast` и `z-ai/glm-5.3-flash-ultrafast`, показаны первыми. +Создайте ключ в [панели OpenGateway](https://opengateway.ai/api-keys), затем выполните +`ocx provider add opengateway` или выберите **OpenGateway** в панели. Chat-запросы используют +настроенный Bearer-ключ; публичный список не подтверждает действительность ключа. + **OpenCode Zen** (`opencode-zen`) и бесключевой пресет **OpenCode Free** используют один `https://opencode.ai/zen/v1`. Бесплатные модели на этом шлюзе часто упираются в короткое окно примерно 15–20 запросов в минуту (оценка сообщества; OpenCode не публикует RPM). diff --git a/docs-site/src/content/docs/ru/reference/configuration/providers.md b/docs-site/src/content/docs/ru/reference/configuration/providers.md index 8c1f6e22c5d..1d36ca37398 100644 --- a/docs-site/src/content/docs/ru/reference/configuration/providers.md +++ b/docs-site/src/content/docs/ru/reference/configuration/providers.md @@ -245,8 +245,10 @@ reauth или эффективного порога исчерпания это ### `anthropicAccountPool` (experimental) Этот opt-in объединяет несколько Anthropic OAuth-аккаунтов, уже сохранённых в `auth.json`. По -умолчанию функция выключена и не считается battle-tested. Аккаунты внутри одной организации могут -делить общую quota, а автоматическая ротация может вызвать ограничения со стороны провайдера. +умолчанию функция выключена и остаётся экспериментальной; Anthropic не одобрял автоматический пул +аккаунтов. Аккаунты внутри одной организации могут делить общую quota, а переключение аккаунтов не +защищает от ограничений со стороны провайдера. Условия, для которых предназначен пул, описаны в +[руководстве по Claude Code](/guides/claude-code/#claude-oauth-account-pool-experimental). | Ключ | Тип | По умолчанию | Описание | | --- | --- | --- | --- | @@ -256,8 +258,7 @@ reauth или эффективного порога исчерпания это | `anthropicAccountPool.quotaWindow?` | `"five-hour" \| "weekly" \| "max-utilization"` | `"five-hour"` | Кешированная полоса использования, сообщённая провайдером и применяемая при выборе по использованию. `five-hour` сохраняет прежнее поведение. `weekly` использует недельный bar и пропускает аккаунты с исчерпанным 5-hour bar, пока остаётся другой доступный аккаунт, но возвращается к ним, если других нет. `max-utilization` использует наибольшее известное значение, поэтому до появления недельных данных может использовать 5-hour usage; если неизвестны оба значения, аккаунт следует порядку unknown usage. Известное использование ранжируется раньше unknown, но если у всех доступных аккаунтов оно неизвестно, выбирается аккаунт в доступном порядке. После описанного сравнения по меньшему значению 5-hour usage полное равенство также сохраняет этот порядок. Здоровая сессия с affinity не перебалансируется заранее. При назначении новой сессии и восстановлении маршрутизации после допустимой замены при 429 `quota` напрямую ранжирует доступных кандидатов по этому окну, `fill-first` идёт в стабильном порядке с учётом порога и правил исчерпания этого окна, а `round-robin` игнорирует настройку. Cooldown, лимиты failover и допустимость повторной аутентификации остаются отдельным локальным состоянием. Недельные bar'ы аккаунтов известны только после опроса на странице Providers в dashboard. | | `anthropicAccountPool.stickyLimit?` | `number` | `1` | Сколько успешных bind'ов новых сессий удерживать на одном выборе round-robin. Диапазон 1–100. | -Если функция включена, 429 записывает ограниченный cooldown из `Retry-After` или из default -backoff и может переключить аккаунт уже внутри текущего запроса. Affinity локальна для процесса и +Только 429 с подтверждённым отказом общей пятичасовой или недельной квоты охлаждает аккаунт и допускает переключение. Временный лимит скорости приостанавливает допуск аккаунта, сохраняя affinity: на запрос разрешены один короткий повтор на том же аккаунте и один переход к другому подходящему аккаунту. Без доказательных заголовков 429 допускает только один короткий повтор на том же аккаунте, без cooldown и без выдуманного Retry-After. Поведение по умолчанию с одним аккаунтом не меняется. Отказ только для Fable не ограничивает Sonnet; ручной выбор и affinity также проверяют общую квоту и квоту запрошенного семейства. Пассивные данные семейства устаревают через тридцать минут или при известном сбросе и перепроверяются одним обслуживающим запросом за раз. Пороги остаются мягкими предпочтениями с прежним fallback при исчерпании всех кандидатов, без жёсткого лимита использования или оплаты. Affinity локальна для процесса и ограничена по размеру. Ошибки обновления токена сохраняют существующие правила повторной аутентификации. Подтверждённый 403 из-за подписки или оплаты аккаунта допускает переключение до вывода и cooldown по `Retry-After` либо на десять минут. Обычный отказ в доступе не переключает аккаунт. Если все eligible-аккаунты в cooldown, клиент получает 429 с `Retry-After`, если он известен, а не authentication error. @@ -396,6 +397,8 @@ semantics Codex. Grok 4.7 поддерживает Fast через OAuth, уровни `low` / `medium` / `high` / `xhigh` и окно в 500,000 токенов. [Стандартная цена xAI](https://docs.x.ai/developers/models/grok-4.7) за миллион токенов составляет $2.00 за ввод, $0.50 за кэшированный ввод и $6.00 за вывод; при контексте от 200,000 токенов действуют цены $4.00 / $1.00 / $12.00. +Если для провайдера явно не настроен `fastWire`, API-ключ opencodex с ограничением `allowedModels` должен разрешать `xai/grok-4.7-build-fast` (или идентификатор без префикса провайдера) для запросов Fast через OAuth. Разрешение только `xai/grok-4.7` не даёт доступа к этому варианту Fast. Ключ с разрешением только модели Fast может использовать этот вариант; обычные запросы или отключённый Fast по-прежнему требуют `xai/grok-4.7`. При явной настройке `fastWire` разрешите фактически отправляемую модель: например, вариант `service-tier` сохраняет `xai/grok-4.7` и требует разрешения для этой модели. Ограничения провайдера также сохраняются. + ## Маршрутизация провайдера OpenRouter OpenRouter может обслуживать одну и ту же модель через нескольких inference-провайдеров. diff --git a/docs-site/src/content/docs/tr/getting-started/quickstart.md b/docs-site/src/content/docs/tr/getting-started/quickstart.md index 21056695bba..955ff8c0f68 100644 --- a/docs-site/src/content/docs/tr/getting-started/quickstart.md +++ b/docs-site/src/content/docs/tr/getting-started/quickstart.md @@ -14,7 +14,7 @@ ocx init `ocx init` adım adım size rehberlik eder: -1. **Bir sağlayıcı seçin** — yerleşik kayıt defterindeki 100 önayardan birini +1. **Bir sağlayıcı seçin** — yerleşik kayıt defterindeki 102 önayardan birini veya bir temel URL ile adaptör yazmak için `custom` seçeneğini belirleyin. 2. **API anahtarı** — bir anahtar yapıştırın veya `${ANTHROPIC_API_KEY}` gibi bir ortam değişkenine başvurun. diff --git a/docs-site/src/content/docs/tr/guides/claude-code.md b/docs-site/src/content/docs/tr/guides/claude-code.md index 54c702c2665..07fa51bcdbb 100644 --- a/docs-site/src/content/docs/tr/guides/claude-code.md +++ b/docs-site/src/content/docs/tr/guides/claude-code.md @@ -23,9 +23,16 @@ seçim yapar: `quota` (varsayılan), `autoSwitchThreshold` üzerinde olduğunda seçer (`five-hour` varsayılandır; `weekly` ve `max-utilization` da kullanılabilir); `round-robin` eşit olarak dağıtır (`stickyLimit`, varsayılan `1`); `fill-first`, bekleme süresi, yeniden kimlik doğrulama veya eşiğe kadar aktif hesabı tüketir, ardından ilerler. **Varsayılan -olarak kapalıdır**, bir GUI uyarısı gösterir ve sahada kapsamlı olarak test -edilmemiştir — Anthropic otomatik rotasyona benzeyen hesapları kısıtlayabilir; -rotasyon sağlayıcı yaptırımlarına karşı koruma sağlamaz. +olarak kapalıdır** ve deneyseldir. + +Pano, havuzun hangi koşullar için tasarlandığını gösterir: size ait olan veya kullanma yetkiniz bulunan +abonelikler, gerçek Claude Code istemcisi ve oturumu gözeten bir kişi. Anthropic otomatik hesap havuzunu +onaylamamıştır; aynı kuruluştaki hesaplar kotayı paylaşabilir (yeni hesap kapasite eklemeyebilir) ve hesap +değiştirmek sağlayıcı yaptırımlarına karşı koruma sağlamaz. OpenCodex hesapları sıcak tutmak için (keep-warm) +istek göndermez ve varsayılan olarak Claude jetonlarını arka planda yenilemez ya da kullanımı arka planda +okumaz: kullanım, pano, menü çubuğu uygulaması veya bir `ocx` komutu istediğinde okunur. Eşikler hesap +seçimi için tercihtir; kullanım veya faturalandırma sınırı değildir. Bu bir ürün açıklamasıdır, hukuki +tavsiye değildir; Anthropic'in güncel koşullarını kontrol edin. Etkinleştirildiğinde operasyonel sözleşme: diff --git a/docs-site/src/content/docs/tr/guides/providers.md b/docs-site/src/content/docs/tr/guides/providers.md index 76201a96233..142c3948a87 100644 --- a/docs-site/src/content/docs/tr/guides/providers.md +++ b/docs-site/src/content/docs/tr/guides/providers.md @@ -327,7 +327,7 @@ olmayan bir makineden oturum açmak bundan etkilenmez. ## 3. API anahtarı kataloğu -opencodex 100 yerleşik önayar ile birlikte gelir: 83 anahtar tabanlı, 13 +opencodex 102 yerleşik önayar ile birlikte gelir: 84 anahtar tabanlı, 14 OAuth, üç yerel ve bir varsayılan ChatGPT iletme önayarı. Kontrol panelinin **Sağlayıcı ekle** seçicisi bir anahtar sağlayıcısının kontrol panelini açar, anahtarı doğrular ve saklar; doğrulama sağlayıcıya özgüdür. Dikkate değer @@ -396,10 +396,20 @@ yalnızca Cline IDE/CLI içinde mevcuttur; `minimax/minimax-m2.5` belgelenmiş A | Xiaomi MiMo | `https://api.xiaomimimo.com/anthropic` | | Xiaomi MiMo (OpenAI Chat) | `https://api.xiaomimimo.com/v1` | | Kilo | `https://api.kilo.ai/api/gateway` | +| OpenGateway | `https://apis.opengateway.ai/v1` | | GitLab Duo | `https://cloud.gitlab.com/ai/v1/proxy/openai/v1` | | Cloudflare AI Gateway | `https://gateway.ai.cloudflare.com/v1/{account-id}/{gateway}/anthropic` | | …ve daha fazlası | opencode zen, Vercel AI Gateway, Venice, NanoGPT, Synthetic, Qianfan, Alibaba, Parallel, ZenMux, LiteLLM | +**OpenGateway**, Sionic AI tarafından işletilen OpenAI uyumlu bir ağ geçididir: +`https://apis.opengateway.ai/v1`. Genel katalogda yaklaşık 80 etkin model bulunur +(2026-10-02 tarihinde doğrulandı). Önayar, genel `GET /v1/models` üzerinden listeyi otomatik +yeniler ve etkin Chat Completions modellerini (ve yalnızca Responses ile sunulup Responses'a yönlendirilen `openai/o3-pro` modelini) tutar. Sionic tarafından sunulan +`deepseek/deepseek-v4.1-flash-ultrafast` ve `z-ai/glm-5.3-flash-ultrafast` ilk sırada listelenir. +[OpenGateway panelinde](https://opengateway.ai/api-keys) bir anahtar oluşturun, ardından +`ocx provider add opengateway` çalıştırın veya panelde **OpenGateway** seçin. Chat istekleri +yapılandırılmış Bearer anahtarını kullanır; genel model listesi anahtarı doğrulamaz. + **OpenCode Zen** (`opencode-zen`) ve anahtarsız **OpenCode Free** önayarı `https://opencode.ai/zen/v1` adresini paylaşır. Bu ağ geçidindeki ücretsiz modeller genellikle yaklaşık 15–20 istek/dakika civarında kısa pencereli bir diff --git a/docs-site/src/content/docs/tr/reference/configuration/providers.md b/docs-site/src/content/docs/tr/reference/configuration/providers.md index 989ec7af701..0d7bb915565 100644 --- a/docs-site/src/content/docs/tr/reference/configuration/providers.md +++ b/docs-site/src/content/docs/tr/reference/configuration/providers.md @@ -275,8 +275,7 @@ ve otomatik rotasyon sağlayıcı kısıtlamalarını tetikleyebilir. | `anthropicAccountPool.quotaWindow?` | `"five-hour" \| "weekly" \| "max-utilization"` | `"five-hour"` | Kullanıma dayalı hesap seçiminde kullanılan, sağlayıcının bildirdiği önbelleğe alınmış kullanım çubuğu. `five-hour` mevcut davranışı korur. `weekly` haftalık çubuğu kullanır ve başka uygun hesap kaldığı sürece 5 saatlik çubuğu tükenmiş hesapları atlar; hiçbiri kalmazsa bu hesaplara geri döner. `max-utilization` bilinen en yüksek değeri kullanır; haftalık değer henüz yokken 5 saatlik değeri kullanabilir, ikisi de bilinmiyorsa hesap unknown kullanım sırasını izler. Bilinen kullanım unknown değerlerden önce gelir; tüm uygun hesaplar unknown olsa bile uygun sıradaki bir hesap seçilir. Belgelenen daha düşük 5 saatlik kullanım eşitlik bozmasından sonra tam eşitlikte de uygun sıra korunur. Sağlıklı affinity oturumları önceden yeniden dengelenmez. Yeni oturum ataması ve uygun bir 429 yedeğine geçildikten sonraki yönlendirme kurtarmasında `quota`, uygun adayları doğrudan bu pencereye göre sıralar; `fill-first`, bu pencerenin eşik ve tükenme kurallarıyla kararlı sırada ilerler; `round-robin` ayarı yok sayar. Cooldown, yük devretme sınırları ve yeniden kimlik doğrulama uygunluğu ayrı yerel durum olarak kalır. Hesap başına haftalık çubuklar ancak dashboard Sağlayıcılar sayfasında sorgulandıktan sonra bilinir. | | `anthropicAccountPool.stickyLimit?` | `number` | `1` | Bir round-robin seçiminde tutulan başarılı yeni oturum bağlamaları. Aralık 1–100. | -Etkinleştirildiğinde 429, `Retry-After`'dan veya varsayılan bir geri çekilmeden -sınırlı soğuma kaydeder ve istek içinde dönebilir. Bağlılık işleme özeldir ve +Yalnızca ortak 5 saatlik veya haftalık kotanın reddini doğrulayan 429 hesabı soğutur ve değiştirir. Geçici hız sınırı bağlılığı koruyarak hesap kabulünü duraklatır; istek başına aynı hesapta bir kısa yeniden deneme ve uygun başka hesaba bir geçiş yapılabilir. Kanıt başlığı olmayan 429 yalnızca aynı hesapta bir kısa yeniden denemeye izin verir, hesabı soğutmaz ve Retry-After üretmez. Varsayılan tek hesap davranışı değişmez. Fable reddi Sonnet erişimini engellemez; elle seçim ve bağlılık da istenen modelin ortak ve aile kotalarını denetler. Pasif aile bilgisi otuz dakika veya bilinen sıfırlama anında eskir ve tek bir hizmet isteğiyle yeniden doğrulanır. Eşikler esnek tercihlerdir; tüm adaylar tükenince mevcut geri dönüş korunur. Bunlar katı kullanım veya faturalama tavanları değildir. Bağlılık işleme özeldir ve boyut sınırlıdır. Token yenileme hataları mevcut yeniden kimlik doğrulama kurallarını korur. Doğrulanmış abonelik veya hesap ödeme 403 hatası çıktı başlamadan hesap değiştirebilir ve `Retry-After` veya varsayılan on dakika soğuma uygular. Genel izin reddi hesap değiştirmez. Uygun tüm hesaplar soğuyorsa istemciler bir kimlik doğrulama hatası değil, bilindiğinde `Retry-After` ile 429 alır. @@ -426,6 +425,8 @@ yürütmeyi kapalı bırakın. Grok 4.7, OAuth üzerinde Fast ile `low` / `medium` / `high` / `xhigh` düzeylerini ve 500.000 tokenlık bağlam penceresini destekler. [xAI standart fiyatı](https://docs.x.ai/developers/models/grok-4.7) milyon token başına giriş için $2,00, önbellekli giriş için $0,50 ve çıkış için $6,00; 200.000 token ve üzeri bağlamda sırasıyla $4,00 / $1,00 / $12,00’dır. +Sağlayıcı için açık bir `fastWire` yapılandırılmadığında, `allowedModels` ile sınırlandırılmış bir opencodex API anahtarının OAuth Fast istekleri için `xai/grok-4.7-build-fast` (veya sağlayıcı öneki olmayan model kimliği) iznine sahip olması gerekir. Yalnızca `xai/grok-4.7` izni bu Fast modeline erişim sağlamaz. Yalnızca Fast modeline izin veren anahtar bu modeli kullanabilir; normal istekler veya Fast kapalıyken yapılan istekler yine `xai/grok-4.7` izni gerektirir. Açık bir `fastWire` yapılandırılmışsa gerçekten gönderilen modele izin verin: örneğin, `service-tier` türü `xai/grok-4.7` modelini korur ve bu modelin iznini gerektirir. Sağlayıcı kısıtlamaları geçerliliğini korur. + ## OpenRouter sağlayıcı yönlendirmesi OpenRouter bir modeli birkaç çıkarım sağlayıcısı aracılığıyla sunabilir. diff --git a/docs-site/src/content/docs/troubleshooting/spend-ledger-synced-folder.md b/docs-site/src/content/docs/troubleshooting/spend-ledger-synced-folder.md new file mode 100644 index 00000000000..d9bc77574fc --- /dev/null +++ b/docs-site/src/content/docs/troubleshooting/spend-ledger-synced-folder.md @@ -0,0 +1,68 @@ +--- +title: Spend Ledger Refused in a Synced Folder +description: Why requests can fail with "Spend-ledger storage could not be opened safely" when the opencodex state directory is inside iCloud Drive or another synced folder, and how to fix it. +--- + +Some macOS users saw requests fail intermittently with HTTP 502 and this message, while +other requests in the same session succeeded: + +```text +Provider unreachable: Spend-ledger storage could not be opened safely. +``` + +Current builds say which file and which check refused it, for example: + +```text +Spend-ledger storage could not be opened safely (journal: extra-hard-link). +``` + +## What the check is + +opencodex keeps a spend ledger in its state directory (`~/.opencodex` by default, or +`OPENCODEX_HOME`): a journal file, `spend-ledger.jsonl`, and a salt file, `spend-ledger.salt`. +Before every write it checks that each file is a regular file owned by you, is not a symbolic +link, and has exactly one directory entry. A second hard link would mean another name elsewhere +on the volume can see or change the same bytes, so opencodex refuses instead of writing through +it. This check stays strict on purpose. + +| Condition in the message | Meaning | +| --- | --- | +| `extra-hard-link` | Another directory entry points at the same file. | +| `symbolic-link` | The ledger file is a symbolic link. | +| `not-regular-file` | Something other than a regular file sits at the ledger path. | +| `foreign-owner` | The file belongs to a different user. | +| `invalid-salt` | The salt file exists but its content is not a valid salt. | + +The role in the message is `journal`, `journal-compaction` (the temporary file written while +the journal is compacted) or `salt`. + +## Why a synced folder triggers it + +macOS sync services, including iCloud Drive with "Desktop & Documents Folders" turned on and +File Provider clients such as OneDrive, Dropbox and Google Drive, can briefly keep a second link +to a file while they stage or upload a change. If the state directory is inside such a folder, +the journal can have two links for a moment after an ordinary write. A request that lands in +that moment is refused, and the next one may succeed. Once the sync settles, the file is back to +one link, so inspecting it afterwards shows nothing wrong. + +At startup opencodex now warns when the state directory resolves inside iCloud Drive +(`~/Library/Mobile Documents`), a File Provider folder (`~/Library/CloudStorage`), or Desktop +or Documents while iCloud Desktop & Documents sync appears to be on. The warning is advisory. +The detection reads the folder layout and can be wrong in either direction. + +## Fix + +Keep the state directory outside synced folders. The default `~/.opencodex` is not synced. + +1. Stop opencodex. +2. Move or copy the state directory to an unsynced location, for example `~/.opencodex-trial`. +3. Set `OPENCODEX_HOME` to that location, or unset it to use the default, and start opencodex + again. + +Do not delete the journal, relax its permissions, or remove the check to make the error go away. +The journal holds your recorded spend, and the check is what keeps it from being written through +an unexpected link. + +If the message names a condition other than `extra-hard-link`, or the state directory is not in +a synced folder, please open an issue with the full refusal message. It contains no path, +account or request content. diff --git a/docs-site/src/content/docs/troubleshooting/update-failed.md b/docs-site/src/content/docs/troubleshooting/update-failed.md index 93b35c7d3dd..939ab75f997 100644 --- a/docs-site/src/content/docs/troubleshooting/update-failed.md +++ b/docs-site/src/content/docs/troubleshooting/update-failed.md @@ -79,10 +79,24 @@ proxy stays in that state, and `ocx restart` may report that no proxy is running still holds the port; use `ocx service restart`, or end the process whose `pid` the `/healthz` response shows and then run `ocx start`. -## What is not known yet - -The report's `ENOTDIR` on `mkdir` means npm tried to create a folder where a path component was a -file. The job log withholds the path, so which component that was is not known, and the leftover -folders above have not been shown to cause it. If it keeps happening, open an issue with the full -terminal output of `ocx update` and the output of `npm config get prefix` and -`npm config get cache` (for example, a cache on a different drive). +## If the update stops at the npm cache check + +`ENOTDIR` on `mkdir` means npm tried to create a folder where a path component was a file, or +a link whose target no longer exists. One confirmed cause +([#6288](https://github.com/lidge-jun/opencodex/issues/6288)) is a `%LOCALAPPDATA%\npm-cache` +junction that pointed to another drive after the target folder had been deleted or the drive had +been removed. + +The updater checks npm's cache folder before it stops the proxy, and the npm staging install uses +the same folder that `npm config get cache` reports. If the cache folder cannot be used, the update +stops with `cache_root_dangling_link` or `cache_root_not_directory` and leaves the proxy running: + +- `cache_root_dangling_link`: the cache folder is a link or junction whose target is missing. + Recreate the target folder, or remove the link so npm can create a normal folder. +- `cache_root_not_directory`: the cache folder, or a folder above it, is a file or sits on a drive + that is not available. Move the file aside, or point npm at another cache with + `npm config set cache `. + +Then run `ocx update` again. If `ENOTDIR` persists after the check passes, open an issue with the +full terminal output of `ocx update` and the output of `npm config get prefix` and +`npm config get cache`. diff --git a/docs-site/src/content/docs/zh-cn/getting-started/quickstart.md b/docs-site/src/content/docs/zh-cn/getting-started/quickstart.md index b8561722c38..892e33377e0 100644 --- a/docs-site/src/content/docs/zh-cn/getting-started/quickstart.md +++ b/docs-site/src/content/docs/zh-cn/getting-started/quickstart.md @@ -18,7 +18,7 @@ ocx init `ocx init` 会引导你完成: -1. **选择 provider** — 从内置 registry 的 100 个预设中选择一个,或选择 `custom` 手动输入 base URL 和 adapter。 +1. **选择 provider** — 从内置 registry 的 102 个预设中选择一个,或选择 `custom` 手动输入 base URL 和 adapter。 2. **API key** — 粘贴一个 key,或引用一个环境变量,例如 `${ANTHROPIC_API_KEY}`。 3. **默认模型** — 对于 key、本地和 custom provider,接受预设值或输入模型 id。 4. **代理端口** — 默认为 `10100`。 diff --git a/docs-site/src/content/docs/zh-cn/guides/providers.md b/docs-site/src/content/docs/zh-cn/guides/providers.md index 0e8df8f42af..351261f6a74 100644 --- a/docs-site/src/content/docs/zh-cn/guides/providers.md +++ b/docs-site/src/content/docs/zh-cn/guides/providers.md @@ -181,7 +181,7 @@ Kiro 登录需要 Kiro CLI:Unix 使用 `curl -fsSL https://cli.kiro.dev/instal ## 3. API 密钥目录 -opencodex 内置 100 个预设:83 个密钥预设、13 个 OAuth 预设、3 个本地预设,以及 1 个默认的 +opencodex 内置 102 个预设:84 个密钥预设、14 个 OAuth 预设、3 个本地预设,以及 1 个默认的 ChatGPT 转发预设。仪表盘的 **Add provider** 选择器会打开密钥提供商的控制台,验证并保存密钥。 验证因提供商而异。主要条目包括: @@ -243,10 +243,20 @@ Cline IDE/CLI 中提供,不能通过 API 使用;`minimax/minimax-m2.5` 是 | Xiaomi MiMo | `https://api.xiaomimimo.com/anthropic` | | Xiaomi MiMo (OpenAI Chat) | `https://api.xiaomimimo.com/v1` | | Kilo | `https://api.kilo.ai/api/gateway` | +| OpenGateway | `https://apis.opengateway.ai/v1` | | GitLab Duo | `https://cloud.gitlab.com/ai/v1/proxy/openai/v1` | | Cloudflare AI Gateway | `https://gateway.ai.cloudflare.com/v1/{account-id}/{gateway}/anthropic` | | ……以及更多 | opencode zen、Vercel AI Gateway、Venice、NanoGPT、Synthetic、Qianfan、Alibaba、Parallel、ZenMux、LiteLLM | +**OpenGateway** 是 Sionic AI 运营的 OpenAI 兼容网关,base URL 为 +`https://apis.opengateway.ai/v1`。公开目录包含约 80 个活跃模型(2026-10-02 核实)。 +预设通过公开 `GET /v1/models` 自动刷新列表,仅保留活跃的 Chat Completions 模型(以及仅支持 Responses、并固定走 Responses 的 `openai/o3-pro`)。 +Sionic 提供的 `deepseek/deepseek-v4.1-flash-ultrafast` 和 +`z-ai/glm-5.3-flash-ultrafast` 排在最前。请在 +[OpenGateway 控制台](https://opengateway.ai/api-keys)创建密钥,再运行 +`ocx provider add opengateway` 或在控制台选择 **OpenGateway**。Chat 请求使用配置的 +Bearer 密钥;公开模型列表不能验证密钥有效性。 + MiniMax 和 MiniMax (CN) 的提供商卡片也会在配置的密钥有有效 Coding Plan 时显示用量。 仪表盘读取 5 小时窗口,以及套餐提供时的每周窗口;这些仅用于展示,不会改变模型路由。 diff --git a/docs-site/src/content/docs/zh-cn/reference/configuration/providers.md b/docs-site/src/content/docs/zh-cn/reference/configuration/providers.md index 92225c1226d..0edbc641cad 100644 --- a/docs-site/src/content/docs/zh-cn/reference/configuration/providers.md +++ b/docs-site/src/content/docs/zh-cn/reference/configuration/providers.md @@ -248,7 +248,7 @@ affinity。这些策略不能规避 provider enforcement。 | `anthropicAccountPool.quotaWindow?` | `"five-hour" \| "weekly" \| "max-utilization"` | `"five-hour"` | 基于用量选择账户时使用的、由提供商报告并缓存的用量条。`five-hour` 保持原有行为。`weekly` 使用每周用量条,并在仍有其他可用账户时跳过 5 小时用量已耗尽的账户;若没有其他账户,则回退使用这些账户。`max-utilization` 使用已知值中的最高值,因此每周用量尚不可用时仍可使用 5 小时用量;两者都未知时,账户遵循 unknown 用量排序。已知用量排在 unknown 之前,但如果所有可用账户都未知,仍会按可用顺序选择一个账户。在前述较低 5 小时用量的同分判定之后,完全相同时也保留可用顺序。不会主动重新平衡健康且已建立亲和性的会话。在新会话分配和符合条件的 429 替代后的路由恢复中,`quota` 直接按此窗口对可用候选账户排序;`fill-first` 按此窗口的阈值和耗尽规则以稳定顺序前进;`round-robin` 忽略此设置。冷却状态、故障转移上限和重新认证资格仍是独立的本地状态。各账户的每周用量只有在控制面板的提供商页面完成查询后才可用。 | | `anthropicAccountPool.stickyLimit?` | `number` | `1` | 在一次轮询选择中保留的成功新会话绑定次数。范围 1–100。 | -启用后,429 会根据 `Retry-After` 记录有界冷却,或者使用默认退避,并且可能在同一请求内轮换。亲和性是进程本地的,并且有大小上限。令牌刷新失败保留原有重新认证规则。明确的订阅或账户计费 403 可在输出前切换账户,并按 `Retry-After` 或默认十分钟冷却;普通权限拒绝不会切换。如果所有合格账户都在冷却,客户端会在已知时收到带 `Retry-After` 的 429,而不是身份验证错误。 +只有共享5小时或每周额度明确拒绝的429才会冷却账户并切换。临时速率限制保留亲和性,只暂停该账户的请求准入;每个请求最多一次短暂的同账户重试和一次合格兄弟账户切换。没有依据响应头的429只允许一次同账户短暂重试,不冷却账户,也不生成Retry-After。默认单账户行为不变。Fable专属拒绝不限制Sonnet;手动选择和亲和性均检查请求模型的共享与家族额度。被动家族信息在30分钟或已知重置时到期,由一个服务请求串行重新验证。使用阈值仍是软偏好,全部候选耗尽时保留原有回退,不是用量或账单硬上限。亲和性是进程本地的,并且有大小上限。令牌刷新失败保留原有重新认证规则。明确的订阅或账户计费 403 可在输出前切换账户,并按 `Retry-After` 或默认十分钟冷却;普通权限拒绝不会切换。如果所有合格账户都在冷却,客户端会在已知时收到带 `Retry-After` 的 429,而不是身份验证错误。 :::caution[Experimental] 除非你理解 Anthropic 账户策略风险,否则请保持关闭。若不确定,优先手动使用 `ocx account use anthropic ` 切换。 @@ -360,6 +360,8 @@ Cursor 由服务端驱动的本地工具默认是禁用的。Codex 继续使用 Grok 4.7 在 OAuth 上支持 Fast,提供 `low` / `medium` / `high` / `xhigh`,上下文窗口为 500,000。按 [xAI 标准价格](https://docs.x.ai/developers/models/grok-4.7),每百万 token 的输入、缓存输入和输出费用分别为 $2.00、$0.50 和 $6.00;上下文达到 200,000 token 时分别为 $4.00 / $1.00 / $12.00。 +未显式配置提供商的 `fastWire` 时,通过 `allowedModels` 限制的 opencodex API 密钥必须允许 `xai/grok-4.7-build-fast`(或不带提供商前缀的模型 ID),才能发送 OAuth Fast 请求。仅允许 `xai/grok-4.7` 不会授予此 Fast 模型的权限。仅允许 Fast 模型的密钥可以使用该模型;普通请求或关闭 Fast 时仍需允许 `xai/grok-4.7`。显式配置 `fastWire` 时,应允许实际发送的模型。例如,`service-tier` 方式保留 `xai/grok-4.7`,因此需要该模型的权限。提供商限制仍然有效。 + ## OpenRouter 提供者路由 OpenRouter 可以通过多个推理提供者来提供同一个模型。`openRouterRouting` 会让请求停留在偏好的提供者上;`modelOpenRouterRouting` 则会对精确模型 id 进行替换。对于提示缓存亲和性来说,这很有用,因为不同推理提供者的缓存支持、保留策略、命中率和定价都不同。 diff --git a/docs-site/src/content/docs/zh-tw/getting-started/quickstart.md b/docs-site/src/content/docs/zh-tw/getting-started/quickstart.md index f0b02c6ab89..42b37e373c7 100644 --- a/docs-site/src/content/docs/zh-tw/getting-started/quickstart.md +++ b/docs-site/src/content/docs/zh-tw/getting-started/quickstart.md @@ -13,7 +13,7 @@ ocx init `ocx init` 會引導你完成: -1. **選擇 provider** —— 從內建 registry 的 100 個預設中選擇一個,或選擇 `custom` 手動輸入 +1. **選擇 provider** —— 從內建 registry 的 102 個預設中選擇一個,或選擇 `custom` 手動輸入 base URL 和 adapter。 2. **API key** —— 貼上一個 key,或引用一個環境變數,例如 `${ANTHROPIC_API_KEY}`。 3. **預設模型** —— 對於 API key、本機和 custom provider,可接受預設值或輸入模型 id。 diff --git a/docs-site/src/content/docs/zh-tw/guides/claude-code.md b/docs-site/src/content/docs/zh-tw/guides/claude-code.md index 47ec6f2e7ca..bc755d3d23e 100644 --- a/docs-site/src/content/docs/zh-tw/guides/claude-code.md +++ b/docs-site/src/content/docs/zh-tw/guides/claude-code.md @@ -18,8 +18,13 @@ sticky session affinity 與依用量的新工作階段選擇。它**不**控制 `anthropicAccountPool.quotaWindow` 所設定的視窗挑選已知用量最低者(`five-hour` 為預設,亦可選 `weekly` 或 `max-utilization`); `round-robin` 平均分散(`stickyLimit`,預設 `1`);`fill-first` 一直使用作用中帳號直到冷卻、重新認證 -或達到閾值,然後前進。它**預設關閉**、會在 GUI 顯示警告,而且尚未經過實戰驗證——Anthropic 可能 -限制看起來像自動輪換的帳號;輪換並不能保護你免受供應商執行機制的處置。 +或達到閾值,然後前進。它**預設關閉**,仍屬實驗性功能。 + +儀表板會列出帳號池的適用條件:你本人擁有或獲授權使用的訂閱、官方 Claude Code 用戶端,以及有人看顧的工作階段。 +Anthropic 未認可自動帳號池;同一組織的帳號可能共用配額(新增帳號不一定能增加容量),切換帳號也無法避免供應商的 +執行處置。OpenCodex 不會傳送保溫(keep-warm)請求,預設也不會在背景更新 Claude 權杖或讀取用量:只有儀表板、選單列 +應用程式或 `ocx` 指令要求時才會讀取用量。門檻是選擇帳號的偏好,而非用量或計費上限。以上為產品說明,並非法律意見; +請查閱 Anthropic 的現行條款。 啟用時的營運契約: diff --git a/docs-site/src/content/docs/zh-tw/guides/providers.md b/docs-site/src/content/docs/zh-tw/guides/providers.md index 3b8c18a509e..c0ea41bc752 100644 --- a/docs-site/src/content/docs/zh-tw/guides/providers.md +++ b/docs-site/src/content/docs/zh-tw/guides/providers.md @@ -249,7 +249,7 @@ database 並移除目前的 WAL、SHM 與 journal sidecar,再發布先前的 s ## 3. API 金鑰目錄 -opencodex 內建 100 個 preset:83 個 key-based、13 個 OAuth、3 個 local,以及 1 個預設 ChatGPT-forward +opencodex 內建 102 個 preset:84 個 key-based、14 個 OAuth、3 個 local,以及 1 個預設 ChatGPT-forward preset。儀表板的 **Add provider** picker 會開啟 key provider 的 dashboard、驗證金鑰並儲存;驗證方式 依 provider 而異。主要條目如下。 @@ -310,10 +310,20 @@ IDE/CLI,不透過 API;`minimax/minimax-m2.5` 是文件列出的 API 免費 | Xiaomi MiMo | `https://api.xiaomimimo.com/anthropic` | | Xiaomi MiMo (OpenAI Chat) | `https://api.xiaomimimo.com/v1` | | Kilo | `https://api.kilo.ai/api/gateway` | +| OpenGateway | `https://apis.opengateway.ai/v1` | | GitLab Duo | `https://cloud.gitlab.com/ai/v1/proxy/openai/v1` | | Cloudflare AI Gateway | `https://gateway.ai.cloudflare.com/v1/{account-id}/{gateway}/anthropic` | | …以及更多 | opencode zen、Vercel AI Gateway、Venice、NanoGPT、Synthetic、Qianfan、Alibaba、Parallel、ZenMux、LiteLLM | +**OpenGateway** 是 Sionic AI 營運的 OpenAI 相容閘道,base URL 為 +`https://apis.opengateway.ai/v1`。公開目錄包含約 80 個活躍模型(2026-10-02 確認)。 +preset 透過公開 `GET /v1/models` 自動更新清單,只保留活躍的 Chat Completions 模型(以及僅支援 Responses、並固定走 Responses 的 `openai/o3-pro`)。 +Sionic 提供的 `deepseek/deepseek-v4.1-flash-ultrafast` 與 +`z-ai/glm-5.3-flash-ultrafast` 排在最前。請在 +[OpenGateway 控制台](https://opengateway.ai/api-keys)建立金鑰,再執行 +`ocx provider add opengateway` 或在控制台選擇 **OpenGateway**。Chat 請求使用設定的 +Bearer 金鑰;公開模型清單無法驗證金鑰有效性。 + **OpenCode Zen**(`opencode-zen`)與無 key 的 **OpenCode Free** preset 共用 `https://opencode.ai/zen/v1`。該 gateway 的免費模型常遇到短時間 burst limit,約 15–20 requests/minute (社群實測;OpenCode 未公布 RPM)。Zen 可能回傳 generic rate-limit 429,而沒有 `Retry-After`/ diff --git a/docs-site/src/content/docs/zh-tw/reference/configuration/providers.md b/docs-site/src/content/docs/zh-tw/reference/configuration/providers.md index d583396414c..e2834a0d8bd 100644 --- a/docs-site/src/content/docs/zh-tw/reference/configuration/providers.md +++ b/docs-site/src/content/docs/zh-tw/reference/configuration/providers.md @@ -182,7 +182,7 @@ API-key 供應商可持有字面值金鑰或環境參考。OAuth 供應商使用 | `anthropicAccountPool.quotaWindow?` | `"five-hour" \| "weekly" \| "max-utilization"` | `"five-hour"` | 使用量型帳號選擇所採用、由供應商回報並快取的用量列。`five-hour` 保留原有行為。`weekly` 使用每週用量列,並在仍有其他可用帳號時略過 5 小時用量已用盡的帳號;若沒有其他帳號,則退回使用這些帳號。`max-utilization` 使用已知值中的最高值,因此每週用量尚未取得時仍可使用 5 小時用量;兩者都未知時,帳號遵循 unknown 用量排序。已知用量排在 unknown 之前,但若所有可用帳號都是 unknown,仍會依可用順序選出一個。完成前述較低 5 小時用量的同分判定後,完全相同時也保留可用順序。不會主動重新平衡健康且已有 affinity 的 session。在分配新 session 與符合條件的 429 替代後進行路由復原時,`quota` 直接依此視窗排序可用候選帳號;`fill-first` 依此視窗的門檻與用盡規則按穩定順序前進;`round-robin` 忽略此設定。冷卻狀態、容錯移轉上限與重新驗證資格仍是獨立的本機狀態。每個帳號的每週用量只有在 dashboard 的供應商頁面完成查詢後才可得知。 | | `anthropicAccountPool.stickyLimit?` | `number` | `1` | 在一次 round-robin 選擇上保留的成功新 session 綁定。範圍 1–100。 | -啟用時,429 記錄來自 `Retry-After` 或預設 backoff 的有界冷卻,並可能在請求內輪換。親和性為行程本地且有界。Token 更新失敗保留原有重新認證規則。已分類的訂閱或帳號計費 403 可在輸出前切換帳號,並按 `Retry-After` 或預設十分鐘冷卻;一般權限拒絕不切換。若所有合格帳號都在冷卻,客戶端收到附帶已知 `Retry-After` 的 429,而非認證錯誤。 +只有共用5小時或每週額度明確拒絕的429才會冷卻帳號並切換。暫時速率限制保留親和性,只暫停該帳號的請求准入;每個請求最多一次短暫的同帳號重試及一次合格兄弟帳號切換。沒有依據回應標頭的429只允許一次同帳號短暫重試,不冷卻帳號,也不產生Retry-After。預設單帳號行為不變。Fable專屬拒絕不限制Sonnet;手動選擇和親和性均檢查請求模型的共用與家族額度。被動家族資訊在30分鐘或已知重設時到期,由一個服務請求依序重新驗證。用量門檻仍是軟偏好,全部候選耗盡時保留原有退回機制,不是用量或帳單硬上限。親和性為行程本地且有界。Token 更新失敗保留原有重新認證規則。已分類的訂閱或帳號計費 403 可在輸出前切換帳號,並按 `Retry-After` 或預設十分鐘冷卻;一般權限拒絕不切換。若所有合格帳號都在冷卻,客戶端收到附帶已知 `Retry-After` 的 429,而非認證錯誤。 :::caution[實驗性] 除非你了解 Anthropic 帳號政策風險,否則保持停用。不確定時偏好手動 `ocx account use anthropic ` 切換。 @@ -293,6 +293,8 @@ Cursor 伺服器驅動的本機工具預設停用。Codex 繼續使用其自身 Grok 4.7 在 OAuth 上支援 Fast,提供 `low` / `medium` / `high` / `xhigh`,context window 為 500,000。依 [xAI 標準價格](https://docs.x.ai/developers/models/grok-4.7),每百萬 token 的輸入、快取輸入及輸出費用分別為 $2.00、$0.50 及 $6.00;context 達 200,000 token 時分別為 $4.00 / $1.00 / $12.00。 +未明確設定供應商的 `fastWire` 時,透過 `allowedModels` 限制的 opencodex API 金鑰必須允許 `xai/grok-4.7-build-fast`(或不含供應商前綴的模型 ID),才能傳送 OAuth Fast 請求。僅允許 `xai/grok-4.7` 不會授予此 Fast 模型的權限。僅允許 Fast 模型的金鑰可以使用該模型;一般請求或關閉 Fast 時仍需允許 `xai/grok-4.7`。明確設定 `fastWire` 時,應允許實際傳送的模型。例如,`service-tier` 方式保留 `xai/grok-4.7`,因此需要該模型的權限。供應商限制仍然有效。 + ## OpenRouter 供應商路由 OpenRouter 可透過多個推論供應商提供一個模型。`openRouterRouting` 將請求保持在偏好的供應商上;`modelOpenRouterRouting` 為精確 model id 取代它。這對 prompt-cache 親和性很有用,因為 cache 支援、保留、命中率與定價因推論供應商而異。 diff --git a/gui/public/provider-icons/README.md b/gui/public/provider-icons/README.md index e53cfda2211..c54807fc8d3 100644 --- a/gui/public/provider-icons/README.md +++ b/gui/public/provider-icons/README.md @@ -391,3 +391,11 @@ committing it. since they do not affect rendering. `viewBox="0 0 48 48"`, one `#151714` ink, so it is **masked** like `packycode.svg`: as an image it vanishes on the dark tile. The pack's reverse symbol (`#F5F5F1`) is the same geometry and is not needed once the mark follows the theme. + +## OpenGateway (2026-10-02) + +- `opengateway.svg` — OpenGateway's header logo, `https://opengateway.ai/logo.svg` (the asset the + site's own header renders), committed byte for byte (860 bytes, MD5 `b510e841406d8bcb24fb4ad584f128fd`). `viewBox="0 0 1140 650"` + (1.75:1, inside the 2.5 lockup limit), pure geometry (`rect`/`path`), drawn in `currentColor`, so it + is **masked**: as an image `currentColor` resolves to black and vanishes on the dark tile. The site + favicon `https://opengateway.ai/icon.svg` was rejected: its "OG" is a `` glyph. diff --git a/gui/public/provider-icons/opengateway.svg b/gui/public/provider-icons/opengateway.svg new file mode 100644 index 00000000000..d13395441ea --- /dev/null +++ b/gui/public/provider-icons/opengateway.svg @@ -0,0 +1,12 @@ + + OpenGateway + + diff --git a/gui/src/App.tsx b/gui/src/App.tsx index ef9f1a2b245..a12548a13a9 100644 --- a/gui/src/App.tsx +++ b/gui/src/App.tsx @@ -7,6 +7,7 @@ import Subagents from "./pages/Subagents"; import Logs from "./pages/Logs"; import Usage from "./pages/Usage"; import Storage from "./pages/Storage"; +import Claude from "./pages/Claude"; import CodexSet from "./pages/CodexSet"; import Integrations from "./pages/Integrations"; import Startup from "./pages/Startup"; @@ -16,7 +17,7 @@ import ErrorBoundary from "./components/ErrorBoundary"; import QuotaSummaryBar from "./components/quota-summary-bar/QuotaSummaryBar"; import { SidebarGithubRow } from "./components/sidebar-github-row"; import { DesktopStarOnboarding } from "./components/desktop-star-onboarding"; -import { IconGrid, IconServer, IconBoxes, IconBot, IconList, IconActivity, IconHardDrive, IconCodex, IconMenu, IconSun, IconMoon, IconMonitor, IconGlobe, IconPower, IconX, IconRefresh} from "./icons"; +import { IconGrid, IconServer, IconBoxes, IconBot, IconList, IconActivity, IconHardDrive, IconCodex, IconClaude, IconMenu, IconSun, IconMoon, IconMonitor, IconGlobe, IconPower, IconX, IconRefresh} from "./icons"; import { useI18n, useT, LOCALES, localeDisplayName, type Locale, type TKey } from "./i18n/shared"; import { Notice, Select, ToastNotice, type NoticeTone } from "./ui"; import { configureApiTargets, hasApiSession, installApiAuthFetch, installApiSessionFromHtml, logoutApiSession, SESSION_UNAVAILABLE_EVENT } from "./api"; @@ -48,6 +49,7 @@ const PAGE_TKEY: Record = { "remote-workspace": "nav.remoteWorkspace", "codex-set": "nav.codexSet", integrations: "nav.integrations", + claude: "nav.claude", }; const API_BASE = import.meta.env.VITE_API_BASE || ""; @@ -56,14 +58,7 @@ configureApiTargets(INITIAL_TARGETS); installApiAuthFetch(); const THEME_KEY = "ocx-theme"; -/** - * Every sidebar row maps one-to-one onto a page again. - * - * The Claude row was the exception: a second entry pointing at a tab of Integrations, - * which needed `subPath`, `activeHashes`, and an `isNavEntryActive` helper whose only - * job was stopping the sidebar from lighting two rows and claiming the user was in two - * places. Removing the duplicate removed all four. - */ +/** Every sidebar row maps to its own page. */ type NavEntry = { id: Page; tkey: TKey; @@ -73,6 +68,7 @@ type NavEntry = { const NAV: NavEntry[] = [ { id: "dashboard", tkey: "nav.dashboard", Icon: IconGrid }, { id: "codex-set", tkey: "nav.codexSet", Icon: IconCodex }, + { id: "claude", tkey: "nav.claude", Icon: IconClaude }, { id: "providers", tkey: "nav.providers", Icon: IconServer }, { id: "models", tkey: "nav.models", Icon: IconBoxes }, { id: "subagents", tkey: "nav.subagents", Icon: IconBot }, @@ -576,6 +572,7 @@ export default function App() { {page === "remote" && !remotePairingRequired && navigateToPage("remote-workspace")} />} {page === "remote-workspace" && navigateToPage("remote")} />} {page === "codex-set" && } + {page === "claude" && } {page === "integrations" && } )} diff --git a/gui/src/app-routing.ts b/gui/src/app-routing.ts index ee5be633727..398873c58b1 100644 --- a/gui/src/app-routing.ts +++ b/gui/src/app-routing.ts @@ -14,7 +14,8 @@ export type Page = | "remote" | "remote-workspace" | "codex-set" - | "integrations"; + | "integrations" + | "claude"; export const VALID_PAGES = new Set([ "dashboard", @@ -29,6 +30,7 @@ export const VALID_PAGES = new Set([ "remote-workspace", "codex-set", "integrations", + "claude", ]); export function readPageFromHash(hash?: string): Page { @@ -53,8 +55,8 @@ export function readPageFromHash(hash?: string): Page { // the destination page here keeps the initial hook state aligned until the // resolver replaces the hash with the exact nested destination. if (pageId === ("api" as Page) - || pageId === ("claude" as Page) || pageId === ("grok" as Page)) return "integrations"; + if (raw === "integrations/claude" || raw === "integrations/claude/desktop") return "claude"; return VALID_PAGES.has(pageId) ? pageId : "dashboard"; } @@ -74,6 +76,22 @@ export const MODELS_TAB_HASHES = ["models/combos", "models/routing", "models/com /** Action deep link that opens the editable JEV Auto template in the Combos tab. */ export const JEV_AUTO_CREATE_HASH = "models/combos/jev-auto"; +/** JEV Auto deep link; a self-hosted `jev-decision` row rides along as `?decisionProvider=`. */ +export function jevAutoCreateHash(decisionProvider?: string | null): string { + const id = decisionProvider?.trim(); + return id && id !== "jev" + ? `${JEV_AUTO_CREATE_HASH}?${new URLSearchParams({ decisionProvider: id })}` + : JEV_AUTO_CREATE_HASH; +} + +/** The decision provider a JEV Auto deep link pre-fills, or undefined when it is not one. */ +export function jevAutoCreateDecisionProvider(hash: string): string | null | undefined { + const { path, query } = splitHashQuery(normalizeHashPath(hash)); + if (path !== JEV_AUTO_CREATE_HASH) return undefined; + const id = new URLSearchParams(query).get("decisionProvider")?.trim(); + return id && id !== "jev" ? id : null; +} + /** * `#dashboard/update` is an action deep link, not a tab: the sidebar update button uses * it to open the maintenance update dialog over the Overview section. It is listed as a @@ -115,13 +133,16 @@ export const INTEGRATION_TAB_HASHES = [ /** * Routes that own a `?query` suffix: provider settings for one provider - * (`#providers?provider=`) and a protocol-pair prefilter on the compatibility matrix - * (`#models/compatibility?inbound=chat&upstream=messages`). Anywhere else the query is dropped. + * (`#providers?provider=`), a protocol-pair prefilter on the compatibility matrix + * (`#models/compatibility?inbound=chat&upstream=messages`), and the decision service a JEV Auto + * deep link pre-fills (`#models/combos/jev-auto?decisionProvider=`). Anywhere else the + * query is dropped. */ -export const QUERY_HASH_PATHS: readonly string[] = ["providers", "models/compatibility"]; +export const QUERY_HASH_PATHS: readonly string[] = ["providers", "models/compatibility", JEV_AUTO_CREATE_HASH]; export function hashBelongsToPage(rawHash: string, page: Page): boolean { return rawHash === page + || (page === "claude" && ["claude/account", "claude/code", "claude/desktop", "claude/settings"].includes(rawHash)) || (page === "logs" && rawHash === "logs/debug") || (page === "usage" && rawHash === "usage/companion") || (page === "codex-set" && rawHash === "codex-set/prompt") @@ -181,7 +202,8 @@ export function resolveAppHashChange(rawHash: string): AppHashChangeAction { /* Legacy top-level integration pages. */ if (rawHash === "api") return { page: "integrations", replaceTo: "integrations/keys" }; - if (rawHash === "claude") return { page: "integrations", replaceTo: "integrations/claude" }; + if (rawHash === "integrations/claude") return { page: "claude", replaceTo: "claude/code" }; + if (rawHash === "integrations/claude/desktop") return { page: "claude", replaceTo: "claude/desktop" }; if (rawHash === "grok") return { page: "integrations", replaceTo: "integrations/grok" }; // Legacy deep link from the removed dual-layout era. diff --git a/gui/src/codex-credit-spend.ts b/gui/src/codex-credit-spend.ts new file mode 100644 index 00000000000..f86517c7fd6 --- /dev/null +++ b/gui/src/codex-credit-spend.ts @@ -0,0 +1,11 @@ +import type { CodexAccountEntry } from "./hooks/useCodexAccountPool"; + +export interface CreditSpendSummary { + enabled: number; + total: number; +} + +/** Spending credits is opt-in, so a row without the field counts as off. */ +export function creditSpendSummary(rows: readonly CodexAccountEntry[]): CreditSpendSummary { + return { enabled: rows.filter(row => row.creditsAfterLimit === true).length, total: rows.length }; +} diff --git a/gui/src/combo-workspace-data.ts b/gui/src/combo-workspace-data.ts index 28750c0ef87..f579a6f2b28 100644 --- a/gui/src/combo-workspace-data.ts +++ b/gui/src/combo-workspace-data.ts @@ -6,6 +6,16 @@ import { SUPPORTED_NATIVE_OPENAI_SLUGS } from "../../src/codex/catalog/native-models"; import { PROVIDER_QUOTA_MAX_AGE_MS } from "../../src/providers/quota-types"; import type { TKey } from "./i18n/shared"; +import { + CANONICAL_JEV_DECISION_PROVIDER, + JEV_DECISION_TIMEOUT_MAX_MS, + JEV_DECISION_TIMEOUT_MIN_MS, + type JevDecisionIssue, + type JevDecisionRow, + jevDecisionRowIssue, + jevDecisionMethod, + jevDecisionModelForbidden, +} from "./jev-decision-service"; export { SUPPORTED_NATIVE_OPENAI_SLUGS }; @@ -52,6 +62,15 @@ export const COMBO_TARGETS_HINT_KEYS: Record = { const COMBO_STRATEGY_SET = new Set(COMBO_STRATEGIES); +export const JEV_DECISION_ISSUE_LABEL_KEYS: Record = { + missing: "cws.jev.decisionIssue.missing", + notDecision: "cws.jev.decisionIssue.notDecision", + disabled: "cws.jev.decisionIssue.disabled", + endpoint: "cws.jev.decisionIssue.endpoint", + model: "cws.jev.decisionIssue.model", +}; + + /** * Intersection of advertised effort ladders for picker availability. * Unknown ladders are wildcards here only; runtime injection remains fail-closed. @@ -146,6 +165,12 @@ export interface ComboItem { * out of the intersection instead of emptying it for the whole group. */ reasoningEffortMode?: "strict" | "adaptive"; + /** `jev` only: self-hosted decision provider id; null/omitted = canonical TypeSafe JEV. */ + decisionProvider?: string | null; + /** `jev` only: independently routed decision model; mutually exclusive with provider. */ + decisionModel?: string | null; + /** `jev` only: decision deadline in ms; null/omitted = the server default. */ + decisionTimeoutMs?: number | null; targets: ComboTarget[]; } @@ -193,6 +218,15 @@ export function updateComboAliasDraft(item: ComboItem, rawAlias: string): ComboI }; } +function normalizeDecisionProvider(raw: unknown): string | null { + const id = typeof raw === "string" ? raw.trim() : ""; + return id && id !== CANONICAL_JEV_DECISION_PROVIDER ? id : null; +} + +function normalizeDecisionTimeoutMs(raw: unknown): number | null { + return typeof raw === "number" && Number.isInteger(raw) ? raw : null; +} + function normalizeAlias(raw: unknown): string | null { return typeof raw === "string" && raw.trim() ? raw.trim() : null; } @@ -263,6 +297,9 @@ export function parseComboList(payload: unknown): ComboItem[] { ...(typeof tr.modelProfile === "string" ? { modelProfile: tr.modelProfile } : {}), })); } + const decisionProvider = normalizeDecisionProvider(r.decisionProvider); + const decisionModel = normalizeAlias(r.decisionModel); + const decisionTimeoutMs = normalizeDecisionTimeoutMs(r.decisionTimeoutMs); out.push({ id, model: typeof r.model === "string" && r.model.trim() @@ -276,6 +313,10 @@ export function parseComboList(payload: unknown): ComboItem[] { defaultEffort: normalizeDefaultEffort(r.defaultEffort), imageInput: normalizeImageInput(r.imageInput), reasoningEffortMode: normalizeReasoningEffortMode(r.reasoningEffortMode), + // Sparse like the wire: only a JEV combo that names a service or deadline carries them. + ...(decisionProvider !== null ? { decisionProvider } : {}), + ...(decisionModel !== null ? { decisionModel } : {}), + ...(decisionTimeoutMs !== null ? { decisionTimeoutMs } : {}), targets, }); } @@ -445,6 +486,12 @@ export function draftEquals(a: ComboItem, b: ComboItem): boolean { || a.defaultEffort !== b.defaultEffort || (a.imageInput ?? "auto") !== (b.imageInput ?? "auto") || (a.reasoningEffortMode ?? "strict") !== (b.reasoningEffortMode ?? "strict") + // Only JEV sends these; another strategy keeps them in the draft for a switch back. + || (a.strategy === "jev" && ( + (a.decisionProvider ?? null) !== (b.decisionProvider ?? null) + || (a.decisionModel ?? null) !== (b.decisionModel ?? null) + || (a.decisionTimeoutMs ?? null) !== (b.decisionTimeoutMs ?? null) + )) ) return false; if (a.targets.length !== b.targets.length) return false; return a.targets.every((t, i) => { @@ -470,6 +517,9 @@ export function toPutBody(item: ComboItem, options: { renameFrom?: string } = {} alias?: string; nativeAlias?: true; displayName?: string; + decisionProvider?: string | null; + decisionModel?: string | null; + decisionTimeoutMs?: number | null; }; } { const weighted = item.strategy === "round-robin" || item.strategy === "random"; @@ -499,6 +549,17 @@ export function toPutBody(item: ComboItem, options: { renameFrom?: string } = {} ...(item.alias && item.alias.trim() ? { alias: item.alias.trim() } : {}), ...(item.nativeAlias ? { nativeAlias: true } : {}), ...(item.displayName && item.displayName.trim() ? { displayName: item.displayName.trim() } : {}), + // JEV only: explicit null selects the default, since the server keeps an omitted field. + // Other strategies omit both; the server drops stored values and rejects sent ones. + // One normalized model value drives both selectors, so a blank model input cannot send + // provider null and model null together and silently select TypeSafe. + ...(item.strategy === "jev" + ? { + decisionProvider: item.decisionModel?.trim() ? null : normalizeDecisionProvider(item.decisionProvider), + decisionModel: item.decisionModel?.trim() || null, + decisionTimeoutMs: item.decisionTimeoutMs ?? null, + } + : {}), }, }; } @@ -524,16 +585,22 @@ export type ComboDraftError = | "invalidWeight" | "invalidReasoningEfforts" | "invalidModelProfile" + | "invalidDecisionTimeout" + | "invalidDecisionProvider" + | "invalidDecisionModel" | "noEnabledTarget"; export function validateComboDraft( item: ComboItem, options: { existingIds: readonly string[]; + combos?: readonly ComboItem[]; + decisionModels?: readonly string[]; /** Aliases already taken by OTHER combos (callers exclude the edited combo). */ existingAliases?: readonly string[]; isCreate: boolean; - providers: Readonly>; + /** Every configured provider; decision fields let a JEV decision service be checked. */ + providers: Readonly>; }, ): ComboDraftError | null { const id = item.id.trim(); @@ -602,6 +669,23 @@ export function validateComboDraft( } } + if (item.strategy === "jev" && item.decisionTimeoutMs != null + && (!Number.isInteger(item.decisionTimeoutMs) + || item.decisionTimeoutMs < JEV_DECISION_TIMEOUT_MIN_MS + || item.decisionTimeoutMs > JEV_DECISION_TIMEOUT_MAX_MS)) { + return "invalidDecisionTimeout"; + } + if (item.strategy === "jev" && jevDecisionMethod(item) === "model") { + const route = item.decisionModel?.trim() ?? ""; + if (!route || route.length > 512 || item.decisionProvider != null + || jevDecisionModelForbidden(route, options.combos ?? [], item) + || options.decisionModels && !options.decisionModels.includes(route)) return "invalidDecisionModel"; + } + if (item.strategy === "jev" && jevDecisionMethod(item) === "systemone" + && (!item.decisionProvider?.trim() || jevDecisionProviderIssue(item.decisionProvider, options.providers) !== null)) { + return "invalidDecisionProvider"; + } + if (!item.targets.some((target) => options.providers[target.provider.trim()]?.disabled !== true)) { return "noEnabledTarget"; } @@ -630,6 +714,7 @@ const JEV_AUTO_MODEL_IDS = ["gpt-6-astra", "gpt-5.6-sol", "gpt-5.6-luna"] as con export function jevAutoDraft( models: readonly { provider: string; id: string }[], eligibleProviders?: ReadonlySet, + decisionProvider?: string | null, ): ComboItem { const targets = JEV_AUTO_MODEL_IDS.flatMap((id) => { const model = models.find((candidate) => candidate.id === id @@ -647,6 +732,69 @@ export function jevAutoDraft( defaultEffort: null, imageInput: "auto", reasoningEffortMode: "adaptive", + decisionProvider: normalizeDecisionProvider(decisionProvider), targets: targets.length > 0 ? targets : [newComboTarget()], }; } + +/** Why a combo's `decisionProvider` is unusable, or null for TypeSafe or a usable row. */ +export function jevDecisionProviderIssue( + decisionProvider: string | null | undefined, + providers: Readonly>, +): JevDecisionIssue | null { + const id = normalizeDecisionProvider(decisionProvider); + if (id === null) return null; + return jevDecisionRowIssue(Object.hasOwn(providers, id) ? providers[id] : undefined); +} + +export interface JevDecisionServiceOption { + /** Provider id written to `decisionProvider`; null selects canonical TypeSafe JEV. */ + id: string | null; + baseUrl?: string; + /** Why the server would reject or the runtime would skip this service; absent when usable. */ + issue?: JevDecisionIssue; +} + +/** + * Decision services a JEV combo may name: canonical TypeSafe first, then every configured + * `adapter: "jev-decision"` row except `jev` itself, each annotated with any usability issue. + * A stored id that is not a decision row stays listed so opening the editor never silently + * rewrites it. + */ +export function jevDecisionServiceOptions( + providers: readonly (JevDecisionRow & { name: string })[], + current?: string | null, +): JevDecisionServiceOption[] { + const rows: JevDecisionServiceOption[] = providers + .filter(provider => provider.adapter === "jev-decision" && provider.name !== CANONICAL_JEV_DECISION_PROVIDER) + .toSorted((a, b) => a.name.localeCompare(b.name)) + .map(provider => { + const issue = jevDecisionRowIssue(provider); + return { + id: provider.name, + ...(provider.baseUrl ? { baseUrl: provider.baseUrl } : {}), + ...(issue ? { issue } : {}), + }; + }); + const selected = normalizeDecisionProvider(current); + const stale: JevDecisionServiceOption[] = selected !== null && !rows.some(row => row.id === selected) + ? [{ id: selected, issue: providers.some(provider => provider.name === selected) ? "notDecision" : "missing" }] + : []; + return [{ id: null }, ...rows, ...stale]; +} + +/** Read-only decision-service facts for a JEV combo, or null for other strategies. */ +export function jevDecisionSummary( + item: Pick, + providers: readonly { name: string; adapter?: string; baseUrl?: string }[], +): { provider: string | null; model: string | null; baseUrl: string | null; timeoutMs: number | null } | null { + if (item.strategy !== "jev") return null; + const provider = normalizeDecisionProvider(item.decisionProvider); + const row = provider === null ? undefined : providers.find(candidate => candidate.name === provider); + return { + provider, + model: item.decisionModel?.trim() || null, + baseUrl: row?.baseUrl?.trim() || null, + timeoutMs: item.decisionTimeoutMs ?? null, + }; +} diff --git a/gui/src/components/ClaudeDesktopPicker.tsx b/gui/src/components/ClaudeDesktopPicker.tsx index f8147b88927..e44f77986d5 100644 --- a/gui/src/components/ClaudeDesktopPicker.tsx +++ b/gui/src/components/ClaudeDesktopPicker.tsx @@ -1,3 +1,5 @@ +import ClaudeInterceptStart from "./ClaudeInterceptStart"; +import { interceptReasonKey } from "../pages/claude-code-first-party"; import { useState } from "react"; import { useI18n, type TKey } from "../i18n/shared"; import { readJsonOrThrow } from "../fetch-json"; @@ -72,8 +74,12 @@ export default function ClaudeDesktopPicker({ apiBase, picker, onUpdated, + pickerReason, + pickerFailurePort, }: { apiBase: string; + pickerReason?: string | null; + pickerFailurePort?: number; picker: DesktopPickerStatus; onUpdated?: (picker: DesktopPickerStatus) => void; }) { @@ -93,7 +99,7 @@ export default function ClaudeDesktopPicker({ headers: { "Content-Type": "application/json" }, body: JSON.stringify({ enabled: !current.desired, persist: true }), }); - let payload: { ok?: boolean; picker?: DesktopPickerStatus } | undefined; + let payload: { ok?: boolean; reason?: string; port?: number; picker?: DesktopPickerStatus } | undefined; if (response.ok) { payload = await readJsonOrThrow<{ ok?: boolean; picker?: DesktopPickerStatus }>( response, @@ -103,11 +109,11 @@ export default function ClaudeDesktopPicker({ // A refused enable is a useful status response (not a transport failure); 503 // carries the controller's reason so the card can explain the missing proxy. try { - payload = await response.json() as { ok?: boolean; picker?: DesktopPickerStatus }; + payload = await response.json() as { ok?: boolean; reason?: string; port?: number; picker?: DesktopPickerStatus }; } catch { throw new Error(t("claudeDesktop.updateFailed")); } - if (!payload.picker) throw new Error(t("claudeDesktop.updateFailed")); + if (!payload.picker) throw new Error(t(interceptReasonKey(payload.reason), { port: payload.port ?? "" })); } if (!payload || payload.ok !== true && !payload.picker || !payload.picker || !isDesktopPickerStatus(payload.picker)) { throw new Error(t("claudeDesktop.updateFailed")); @@ -123,6 +129,8 @@ export default function ClaudeDesktopPicker({ return (
+ {!current.listenerReady && pickerReason && {t(interceptReasonKey(pickerReason === "port_in_use" && !pickerFailurePort ? "failed" : pickerReason), { port: pickerFailurePort ?? "" })}} + {!current.listenerReady && !pickerReason && { setLocalPicker(null); onUpdated?.(current); }} />}

{t("claudeDesktop.picker.title")}

diff --git a/gui/src/components/ClaudeInterceptStart.tsx b/gui/src/components/ClaudeInterceptStart.tsx new file mode 100644 index 00000000000..4513db68916 --- /dev/null +++ b/gui/src/components/ClaudeInterceptStart.tsx @@ -0,0 +1,31 @@ +import { useState } from "react"; +import { useT } from "../i18n/shared"; +import { interceptReasonKey } from "../pages/claude-code-first-party"; + +const FAILED_REASON = "failed"; + +export default function ClaudeInterceptStart({ apiBase, reason, port, onStarted }: { + apiBase: string; reason?: string | null; port?: number; onStarted: () => void; +}) { + const t = useT(); + const [pending, setPending] = useState(false); + const [failure, setFailure] = useState<{ reason: string; port?: number } | null>(null); + const start = async () => { + if (pending) return; + setPending(true); + setFailure(null); + try { + const response = await fetch(`${apiBase}/api/claude-intercept/start`, { method: "POST" }); + const result = await response.json() as { ok?: boolean; reason?: string; port?: number }; + if (!response.ok || !result.ok) setFailure({ reason: result.reason ?? FAILED_REASON, port: result.port }); + else onStarted(); + } catch { setFailure({ reason: FAILED_REASON }); } + finally { setPending(false); } + }; + return + {t(interceptReasonKey(failure?.reason ?? reason), { port: failure?.port ?? port ?? "" })}{" "} + + ; +} diff --git a/gui/src/components/CodexAccountPool.tsx b/gui/src/components/CodexAccountPool.tsx index 541b81c3f39..86a9375bb3c 100644 --- a/gui/src/components/CodexAccountPool.tsx +++ b/gui/src/components/CodexAccountPool.tsx @@ -28,6 +28,8 @@ import { DEFAULT_ACCOUNT_POOL_STRATEGY } from "../account-pool-strategy"; import type { CodexAccountMutationCompletion } from "../codex-account-mutation"; import { createBoundedFetch, type BoundedFetch } from "../bounded-fetch"; import CodexQuotaAutoRefreshSetting from "./CodexQuotaAutoRefreshSetting"; +import { CodexCreditSpendSwitch } from "./CodexCreditSpend"; +import { creditSpendSummary } from "../codex-credit-spend"; import { quotaActivationWindows, readQuotaActivationSettings, type QuotaAutoRefreshSettings } from "../codex-quota-activation"; // Single definition lives with the controller that owns this data (WP3). @@ -274,6 +276,14 @@ export default function CodexAccountPool({ apiBase, accountModeState = null, ban }), result.ok ? "ok" : "err"); }; + const toggleAllCreditsAfterLimit = async (enabled: boolean) => { + const result = await controller.setAllCreditsAfterLimit(enabled); + if (!result.ok && result.reason === "busy") return; + showActionFeedback(t(result.ok + ? enabled ? "codexAuth.creditsAllOnSucceeded" : "codexAuth.creditsAllOffSucceeded" + : "codexAuth.creditsAllUpdateFailed"), result.ok ? "ok" : "err"); + }; + const changePriority = async (account: CodexAccountEntry, priority: number) => { // Same guard as the pool strategy control (CodexPoolStrategySetting.tsx), and here it is // load-bearing rather than just thrift: `Select` calls onChange for the clicked option @@ -305,6 +315,16 @@ export default function CodexAccountPool({ apiBase, accountModeState = null, ban return result.ok; }; + const toggleCreditsAfterLimit = async (account: CodexAccountEntry, enabled: boolean) => { + const result = await controller.setAccountCreditsAfterLimit(account.id, enabled); + if (!result.ok && result.reason === "busy") return; + showActionFeedback(t(result.ok + ? enabled ? "codexAuth.creditsOnSucceeded" : "codexAuth.creditsOffSucceeded" + : "codexAuth.creditsUpdateFailed", { + email: account.alias ?? account.email, + }), result.ok ? "ok" : "err"); + }; + const remove = async (id: string) => { const label = accounts.find(account => account.id === id)?.email ?? t("pws.accountOrdinal", { count: "1" }); if (!(await confirmAction({ message: t("codexAuth.removeConfirm", { id: label }), confirmLabel: t("common.remove"), tone: "danger" }))) return; @@ -448,6 +468,9 @@ export default function CodexAccountPool({ apiBase, accountModeState = null, ban const main = accounts.find(a => a.isMain); const pool = accounts.filter(a => !a.isMain); + // The rows the global credit switch covers: the main login once it has a credential, then the pool. + const creditRows = [...(main?.hasCredential ? [main] : []), ...pool]; + const creditSummary = creditSpendSummary(creditRows); const isMainActive = !main?.paused && (!activeId || activeId === "__main__"); const switchActionLabel = t(accountModeState === "direct" ? "codexAuth.prepareForPool" : "codexAuth.setAsNext"); const pauseBusy = pauseUpdatingId !== null || pausingExhausted; @@ -465,6 +488,13 @@ export default function CodexAccountPool({ apiBase, accountModeState = null, ban creditsVisible={credits.visible} creditsBusy={credits.busy} onToggleCredits={() => { void credits.toggle(); }} + creditSpendControl={loadState === "loading" && accounts.length === 0 ? undefined : ( + { void toggleAllCreditsAfterLimit(enabled); }} + /> + )} refreshingQuota={refreshingQuota} actionFeedback={actionFeedback} actionFeedbackTone={actionFeedbackTone} @@ -520,6 +550,8 @@ export default function CodexAccountPool({ apiBase, accountModeState = null, ban priorityUpdatingId={priorityUpdatingId} onAutoSwitchThresholdChange={changeAccountAutoSwitchThreshold} autoSwitchDisabled={accountAutoSwitchDisabled} + onToggleCreditsAfterLimit={(entry, enabled) => { void toggleCreditsAfterLimit(entry, enabled); }} + creditsAfterLimitUpdatingId={controller.creditsAfterLimitUpdatingId} switchingId={switchingId} pinnedId={activePinnedId} onOpenReset={openResetPopup} @@ -566,6 +598,8 @@ export default function CodexAccountPool({ apiBase, accountModeState = null, ban priorityUpdatingId={priorityUpdatingId} onAutoSwitchThresholdChange={changeAccountAutoSwitchThreshold} autoSwitchDisabled={accountAutoSwitchDisabled} + onToggleCreditsAfterLimit={(entry, enabled) => { void toggleCreditsAfterLimit(entry, enabled); }} + creditsAfterLimitUpdatingId={controller.creditsAfterLimitUpdatingId} switchingId={switchingId} pinnedId={activePinnedId} onReauth={openReauth} diff --git a/gui/src/components/CodexCreditSpend.tsx b/gui/src/components/CodexCreditSpend.tsx new file mode 100644 index 00000000000..54c5e474ebf --- /dev/null +++ b/gui/src/components/CodexCreditSpend.tsx @@ -0,0 +1,86 @@ +import { useId } from "react"; +import { useT } from "../i18n/shared"; +import type { CreditSpendSummary } from "../codex-credit-spend"; +import "../styles/codex-credits.css"; + +/** + * Spending ChatGPT credits after a usage limit is opt-in (#6334). Upstream keeps serving an account + * that holds credits at 100% and draws the balance, so by default every account is switched out at + * 100% and returns after its reset. The header carries one global switch beside the credits + * display; each account's own switch lives in that card's "more" disclosure. The global switch is + * derived from the accounts: off when none may spend, mixed when some may, on when all may. + */ +export function CodexCreditSpendSwitch({ summary, busy, onToggleAll }: { + summary: CreditSpendSummary; + busy: boolean; + /** The requested global state: true allows every account, false clears them all. */ + onToggleAll(enabled: boolean): void; +}) { + const t = useT(); + const hintId = useId(); + const all = summary.total > 0 && summary.enabled === summary.total; + const mixed = summary.enabled > 0 && !all; + return ( + + {t("codexAuth.creditSpend")} + + {t("codexAuth.creditsAfterLimitHint")} + + ); +} + +/** One account's switch, rendered inside its card's "more" disclosure. */ +export function AccountCreditsToggle({ accountLabel, enabled, saving, disabled, hint, onChange }: { + accountLabel: string; + /** Absent on rows from an older server; the default is off. */ + enabled: boolean | undefined; + saving: boolean; + disabled: boolean; + /** Replaces the shared hint; the main login uses it to name its hard lock. */ + hint?: string; + onChange(enabled: boolean): void; +}) { + const t = useT(); + const hintId = useId(); + const on = enabled === true; + const description = hint ?? t("codexAuth.creditsAfterLimitHint"); + return ( +
+ {t("codexAuth.creditsAfterLimit")} + + {description} +
+ ); +} + +/** Marks an account allowed to spend credits, so the exception is visible on its card. */ +export function CreditsOnBadge({ enabled }: { enabled: boolean | undefined }) { + const t = useT(); + if (enabled !== true) return null; + return ( + + {t("codexAuth.creditsOn")} + + ); +} diff --git a/gui/src/components/ComboWorkspace.tsx b/gui/src/components/ComboWorkspace.tsx index c0415cabd16..0d5413dfa15 100644 --- a/gui/src/components/ComboWorkspace.tsx +++ b/gui/src/components/ComboWorkspace.tsx @@ -6,6 +6,7 @@ import { filterCombos, groupCombos, jevAutoDraft, + jevDecisionProviderIssue, } from "../combo-workspace-data"; import { IconChevron, IconPlus, IconSearch, IconShuffle } from "../icons"; import { useT } from "../i18n/shared"; @@ -31,12 +32,19 @@ export default function ComboWorkspace({ onAdd, adding, addIntent, + addDecisionProvider, onCloseAdd, onCreated, }: ComboWorkspaceProps) { const t = useT(); const providerMap = useMemo( - () => Object.fromEntries(providers.map((provider) => [provider.name, { disabled: provider.disabled }])), + () => Object.fromEntries(providers.map((provider) => [provider.name, { + disabled: provider.disabled, + adapter: provider.adapter, + baseUrl: provider.baseUrl, + defaultModel: provider.defaultModel, + models: provider.models, + }])), [providers], ); const [query, setQuery] = useState(""); @@ -57,10 +65,13 @@ export default function ComboWorkspace({ .map(provider => provider.name)), [providers], ); - const addDraft = useMemo( - () => addIntent === "jev-auto" ? jevAutoDraft(models, jevTargetProviders) : undefined, - [addIntent, jevTargetProviders, models], - ); + const addDraft = useMemo(() => { + if (addIntent !== "jev-auto") return undefined; + // A deep-linked row that is not a usable decision service falls back to TypeSafe. + const usable = providers.length === 0 + || jevDecisionProviderIssue(addDecisionProvider, providerMap) === null; + return jevAutoDraft(models, jevTargetProviders, usable ? addDecisionProvider : null); + }, [addDecisionProvider, addIntent, jevTargetProviders, models, providerMap, providers.length]); const filtered = useMemo(() => filterCombos(combos, query), [combos, query]); const sections = useMemo(() => groupCombos(filtered), [filtered]); @@ -206,6 +217,7 @@ export default function ComboWorkspace({ key={baseline.id} apiBase={apiBase} baseline={baseline} + combos={combos.filter((c) => c.id !== baseline.id)} otherIds={otherComboIds} otherAliases={otherComboAliases} providerMap={providerMap} @@ -232,6 +244,7 @@ export default function ComboWorkspace({ ) : creatingFirstCombo ? ( trySelect(id)} onAdd={onAdd} /> @@ -263,7 +277,9 @@ export default function ComboWorkspace({ {adding && ( c.id)} existingAliases={existingComboAliases} providerMap={providerMap} diff --git a/gui/src/components/MainAccountHardLockSetting.tsx b/gui/src/components/MainAccountHardLockSetting.tsx index 7f5e8953313..e46079b9e8e 100644 --- a/gui/src/components/MainAccountHardLockSetting.tsx +++ b/gui/src/components/MainAccountHardLockSetting.tsx @@ -2,7 +2,7 @@ import { useCallback, useEffect, useId, useRef, useState } from "react"; import { createBoundedFetch } from "../bounded-fetch"; import { startVisibilityPoll } from "../visibility-poll"; import { useT } from "../i18n/shared"; -import type { MainAccountHardLockStatus } from "../hooks/useCodexAccountPool"; +import { hardLockThresholds, type MainAccountHardLockStatus } from "../hooks/useCodexAccountPool"; type Props = { apiBase: string; onSaved: () => Promise }; type Snapshot = { codexMainAccountHardLock: boolean; mainAccountHardLock: MainAccountHardLockStatus }; @@ -20,8 +20,9 @@ function readSnapshot(value: unknown): Snapshot { return payload as Snapshot; } -function HardLockConfirmation({ pending, onCancel, onConfirm }: { +function HardLockConfirmation({ pending, thresholds, onCancel, onConfirm }: { pending: boolean; + thresholds: { short: number; long: number }; onCancel: () => void; onConfirm: () => void; }) { @@ -49,7 +50,7 @@ function HardLockConfirmation({ pending, onCancel, onConfirm }: {
+ {snapshot?.mainAccountHardLock.externalUsage &&

{t("codexAuth.mainExternalUsageWarning")}

} {(saveError || loadError) &&

{t(saveError ? "codexAuth.mainHardLockSaveFailed" : "codexAuth.mainHardLockLoadFailed")}{" "}

} - {saved !== null && !refreshError &&

{t(saved ? "codexAuth.mainHardLockEnabled" : "codexAuth.mainHardLockDisabled")}

} + {saved !== null && !refreshError &&

{t(saved ? "codexAuth.mainHardLockEnabled" : "codexAuth.mainHardLockDisabled", thresholds)}

} {refreshError &&

{t("codexAuth.mainHardLockRefreshFailed")}{" "}

}
- {confirming && { void save(true); }} />} + {confirming && { void save(true); }} />}
); } diff --git a/gui/src/components/OAuthTosWarningModal.tsx b/gui/src/components/OAuthTosWarningModal.tsx index 5bac71be28d..31e3766b6dd 100644 --- a/gui/src/components/OAuthTosWarningModal.tsx +++ b/gui/src/components/OAuthTosWarningModal.tsx @@ -5,11 +5,7 @@ import { useCallback, useEffect, useId, useRef, useState } from "react"; import { useT } from "../i18n/shared"; import { IconAlert } from "../icons"; -import { - oauthTosRisk, - oauthTosRiskBodyKey, - oauthTosRiskTitleKey, -} from "../oauth-tos-risk"; +import { oauthTosCopyKeys, oauthTosRisk } from "../oauth-tos-risk"; export default function OAuthTosWarningModal({ providerId, @@ -25,6 +21,7 @@ export default function OAuthTosWarningModal({ const t = useT(); const titleId = useId(); const bodyId = useId(); + const conditionsId = useId(); const dialogRef = useRef(null); const submittedRef = useRef(false); const [acknowledged, setAcknowledged] = useState(false); @@ -46,13 +43,7 @@ export default function OAuthTosWarningModal({ // Unmarked provider: render nothing (callers must gate with oauthTosRisk). if (!level) return null; - const normalizedProviderId = providerId.trim().toLowerCase(); - const bodyKey = normalizedProviderId === "anthropic" - ? "oauthTos.anthropicBody" - : oauthTosRiskBodyKey(level); - const showApiKeySaferPath = - normalizedProviderId === "anthropic" - || normalizedProviderId === "google-antigravity"; + const copy = oauthTosCopyKeys(providerId, level); const handleContinue = () => { if (!acknowledged || submittedRef.current) return; @@ -65,7 +56,7 @@ export default function OAuthTosWarningModal({ @@ -75,7 +66,7 @@ export default function OAuthTosWarningModal({ onClick={e => e.stopPropagation()} style={{ maxWidth: 460 }} > -

{t(oauthTosRiskTitleKey(level), { provider: providerLabel })}

+

{t(copy.title, { provider: providerLabel })}

- {showApiKeySaferPath && ( + {copy.conditions && ( +

+ {t(copy.conditions)} +

+ )} + {copy.saferPath && (

- {t("oauthTos.saferPath")} + {t(copy.saferPath)}

)}
diff --git a/gui/src/components/codex-account-pool-cards.tsx b/gui/src/components/codex-account-pool-cards.tsx index beb5961ee33..4055a63daee 100644 --- a/gui/src/components/codex-account-pool-cards.tsx +++ b/gui/src/components/codex-account-pool-cards.tsx @@ -6,6 +6,7 @@ import { displayAccountId } from "../lib/privacy"; import AccountPriorityControl, { AccountPriorityBadge } from "./AccountPriorityControl"; import { DEFAULT_ACCOUNT_PRIORITY, normalizeAccountPriority } from "../account-priority"; import AccountAutoSwitchControl from "./AccountAutoSwitchControl"; +import { AccountCreditsToggle, CreditsOnBadge } from "./CodexCreditSpend"; import type { CodexAccountEntry } from "./codex-account-pool-types"; import type { CodexAccountModeState } from "../codex-multi-state"; import QuotaBars from "./QuotaBars"; @@ -43,6 +44,8 @@ export function CodexAccountPoolCards({ onReauth, onEditAlias, onRemove, + onToggleCreditsAfterLimit, + creditsAfterLimitUpdatingId = null, onCopyDoctor, doctorCopyOutcomeFor, }: { @@ -74,6 +77,9 @@ export function CodexAccountPoolCards({ onReauth: (id: string) => void; onEditAlias: (account: CodexAccountEntry) => void; onRemove: (id: string) => void; + /** Writes one account's "use credits after limit" switch, shown in its "more" disclosure. */ + onToggleCreditsAfterLimit?: (account: CodexAccountEntry, enabled: boolean) => void; + creditsAfterLimitUpdatingId?: string | null; onCopyDoctor?: (accountId: string) => void; doctorCopyOutcomeFor?: (accountId: string) => "copied" | "unavailable" | null; }) { @@ -114,6 +120,7 @@ export function CodexAccountPoolCards({ )} + {a.id === pinnedId && !a.paused && {t("codexAuth.pinned")}} onOpenReset(a)} /> {healthLabel && ( @@ -178,6 +185,15 @@ export function CodexAccountPoolCards({ + {onToggleCreditsAfterLimit && ( + onToggleCreditsAfterLimit(a, enabled)} + /> + )} )} + {/* Same disclosure as the pool cards' "more" actions; the main login only carries its + credits switch there. */} + {main?.hasCredential && onToggleCreditsAfterLimit && ( +
+ ⋯ +
+ onToggleCreditsAfterLimit(mainSwitchEntry, enabled)} + /> +
+
+ )} {t("codexAuth.appLogin")}
@@ -203,12 +228,13 @@ export function CodexAccountPoolMainCard({ {policy?.enabled && (

{t(hardLocked ? "codexAuth.mainHardLockBlocked" - : policy.state === "ready" ? "codexAuth.mainHardLockMonitoring" : "codexAuth.mainHardLockUnknown")}

+ : policy.state === "ready" ? "codexAuth.mainHardLockMonitoring" : "codexAuth.mainHardLockUnknown", hardLockThresholds(policy.thresholds))}

{onManageMainHardLock ? : }
)} + {policy?.externalUsage &&

{t("codexAuth.mainExternalUsageWarning")}

} {healthSummary && (
{healthSummary}
)} @@ -285,6 +311,7 @@ export function CodexAccountPoolPageHead({ creditsVisible, creditsBusy, onToggleCredits, + creditSpendControl, }: { t: TFn; embedded: boolean; @@ -292,6 +319,8 @@ export function CodexAccountPoolPageHead({ creditsVisible?: boolean; creditsBusy?: boolean; onToggleCredits?: () => void; + /** The global "use credits" switch, rendered beside the credits display switch. */ + creditSpendControl?: ReactNode; refreshingQuota: boolean; pausingExhausted: boolean; pauseBusy?: boolean; @@ -330,6 +359,7 @@ export function CodexAccountPoolPageHead({ )} + {creditSpendControl} {/* The standalone pause/refresh row sits next to the account cards. Embedded surfaces keep those actions beside feedback because there is no page title. */} {embedded && ( diff --git a/gui/src/components/combo-workspace-add-modal.tsx b/gui/src/components/combo-workspace-add-modal.tsx index e6c71388d85..87557f9aa2a 100644 --- a/gui/src/components/combo-workspace-add-modal.tsx +++ b/gui/src/components/combo-workspace-add-modal.tsx @@ -13,10 +13,17 @@ import { useT } from "../i18n/shared"; import { Notice } from "../ui"; import type { ModelOption, ProviderOption } from "./combo-workspace-types"; import { ComboCapabilities, EffortSelect, StrategySeg, TargetEditor } from "./combo-workspace-controls"; +import { ComboJevDecisionSection } from "./combo-workspace-jev-decision"; import { COMBO_STRATEGY_HINT_KEYS, COMBO_TARGETS_HINT_KEYS } from "../combo-workspace-data"; -import { clampedNumberInput } from "./combo-workspace-utils"; +import { clampedNumberInput, comboDraftErrorText } from "./combo-workspace-utils"; +import type { JevDecisionRow } from "../jev-decision-service"; + +/** Stable default so an omitted combo list does not change identity every render. */ +const NO_COMBOS: readonly ComboItem[] = []; export function AddComboModal({ + apiBase, + combos = NO_COMBOS, existingIds, existingAliases, providerMap, @@ -27,9 +34,12 @@ export function AddComboModal({ onClose, onSubmit, }: { + apiBase?: string; + /** Existing combos; a JEV decision model may not name the new combo or any JEV combo. */ + combos?: readonly ComboItem[]; existingIds: string[]; existingAliases: string[]; - providerMap: Readonly>; + providerMap: Readonly>; providerQuotaStates: ProviderQuotaStates; providers: ProviderOption[]; models: ModelOption[]; @@ -80,11 +90,12 @@ export function AddComboModal({ const code = validateComboDraft(draft, { existingIds, existingAliases, + combos, isCreate: true, providers: providerMap, }); if (code) { - setError(t(`cws.err.${code}`)); + setError(comboDraftErrorText(t, code, draft, providerMap)); return; } setBusy(true); @@ -180,6 +191,21 @@ export function AddComboModal({ {t(COMBO_STRATEGY_HINT_KEYS[draft.strategy])}

+ {draft.strategy === "jev" && ( + setDraft((d) => ({ ...d, ...patch }))} + /> + )}
model.provider === row.provider && model.id === row.model, )?.reasoningEfforts; - const selectableReasoningEfforts = advertisedReasoningEfforts === undefined + const advertisedEffortSet = advertisedReasoningEfforts === undefined ? undefined : new Set(advertisedReasoningEfforts); + const selectableReasoningEfforts = advertisedEffortSet === undefined ? undefined - : COMBO_EFFORTS.filter(effort => advertisedReasoningEfforts.includes(effort)); + : COMBO_EFFORTS.filter(effort => advertisedEffortSet.has(effort)); + const selectableEffortSet = new Set(selectableReasoningEfforts ?? []); const selectedReasoningEfforts = selectableReasoningEfforts === undefined ? [] : row.reasoningEfforts === undefined ? selectableReasoningEfforts - : row.reasoningEfforts.filter(effort => selectableReasoningEfforts.includes(effort)); + : row.reasoningEfforts.filter(effort => selectableEffortSet.has(effort)); + const selectedEffortSet = new Set(selectedReasoningEfforts); return (
{t("cws.jev.allowedEfforts")} {selectableReasoningEfforts.map((effort) => { - const checked = selectedReasoningEfforts.includes(effort); + const checked = selectedEffortSet.has(effort); return (
+ {draft.strategy === "jev" && ( + updateDraft((d) => ({ ...d, ...patch }))} + /> + )}
= { + typesafe: "cws.jev.method.typesafe", + systemone: "cws.jev.method.systemone", + model: "cws.jev.method.model", +}; + +/** + * "Decision method" for a JEV combo: TypeSafe, a System One-compatible server, or any model + * opencodex routes. Callers render it only for `strategy: "jev"`. The Test button probes the + * unsaved selection through POST /api/combos/decision-test; nothing is written. + */ +export function ComboJevDecisionSection({ + idPrefix, + apiBase, + combo, + combos, + providers, + models, + decisionProvider, + decisionModel, + decisionTimeoutMs, + disabled, + onChange, +}: { + idPrefix: string; + apiBase?: string; + /** The combo being edited; excluded from the model list together with every JEV combo. */ + combo: { id: string; alias?: string | null; model?: string }; + combos: readonly { id: string; alias?: string | null; model: string; strategy: string }[]; + providers: ProviderOption[]; + models: ModelOption[]; + decisionProvider: string | null; + decisionModel: string | null; + decisionTimeoutMs: number | null; + disabled?: boolean; + onChange: (patch: JevDecisionPatch) => void; +}) { + const t = useT(); + const method = jevDecisionMethod({ decisionProvider, decisionModel }); + const servers = useMemo( + () => jevDecisionServiceOptions(providers, decisionProvider).filter(option => option.id !== null), + [providers, decisionProvider], + ); + const selectedServer = servers.find(option => option.id === decisionProvider?.trim()); + // The parent passes the whole draft, which changes identity on every keystroke; only its + // identity fields affect the route list. + const { id: comboId, alias: comboAlias, model: comboModel } = combo; + const modelRoutes = useMemo( + () => jevDecisionModelOptions(models, providers, combos, { id: comboId, alias: comboAlias, model: comboModel }), + [models, providers, combos, comboId, comboAlias, comboModel], + ); + // Each result remembers the selection it describes; a different selection shows no result. + const [probe, setProbe] = useState<{ key: string; result: DecisionTestResult } | null>(null); + const testAbort = useRef(null); + const selectionKey = `${method}\0${decisionProvider ?? ""}\0${decisionModel ?? ""}\0${decisionTimeoutMs ?? ""}`; + const test = probe?.key === selectionKey ? probe.result : null; + // Concatenated, not a template: the i18n lint reads template text outside JSX id props as copy. + const modelListId = idPrefix + "-decision-model-options"; + + // Changing the selection (or unmounting) cancels an in-flight probe for the old one. + useEffect(() => () => { + testAbort.current?.abort(); + testAbort.current = null; + }, [selectionKey]); + + const selectMethod = (next: JevDecisionMethod) => { + if (next === method) return; + if (next === "typesafe") onChange({ decisionProvider: null, decisionModel: null }); + else if (next === "systemone") { + const first = servers.find(option => !option.issue)?.id ?? servers[0]?.id ?? ""; + onChange({ decisionProvider: first, decisionModel: null }); + } else onChange({ decisionProvider: null, decisionModel: modelRoutes[0] ?? "" }); + }; + + const canTest = method === "typesafe" + || (method === "systemone" && !!selectedServer && !selectedServer.issue) + || (method === "model" && !!decisionModel?.trim()); + + const runTest = async () => { + testAbort.current?.abort(); + const controller = new AbortController(); + testAbort.current = controller; + const key = selectionKey; + const setTest = (result: DecisionTestResult) => setProbe({ key, result }); + setTest({ state: "running" }); + try { + const response = await fetch(`${apiBase ?? ""}/api/combos/decision-test`, { + method: "POST", + headers: { "content-type": "application/json" }, + signal: controller.signal, + body: JSON.stringify({ + ...(combo.id.trim() ? { comboId: combo.id.trim() } : {}), + ...(method === "systemone" && decisionProvider ? { decisionProvider: decisionProvider.trim() } : {}), + ...(method === "model" && decisionModel ? { decisionModel: decisionModel.trim() } : {}), + ...(decisionTimeoutMs !== null ? { decisionTimeoutMs } : {}), + }), + }); + if (!response.ok) { + // The management API reports refusals as { error }; anything else falls back to the status. + const failure = await response.json().catch(() => null) as Record | null; + if (controller.signal.aborted) return; + setTest({ state: "error", message: failure && typeof failure.error === "string" ? failure.error : String(response.status) }); + return; + } + const data = await response.json().catch(() => null) as Record | null; + if (controller.signal.aborted) return; + if (!data || typeof data.gate !== "string") { + setTest({ state: "error", message: String(response.status) }); + return; + } + setTest({ + state: "done", + ok: data.ok === true, + backend: typeof data.backend === "string" ? data.backend : "unknown", + gate: data.gate, + latencyMs: typeof data.latencyMs === "number" ? Math.round(data.latencyMs) : 0, + }); + } catch (error) { + if (controller.signal.aborted) return; + setTest({ state: "error", message: error instanceof Error ? error.message : String(error) }); + } finally { + if (testAbort.current === controller) testAbort.current = null; + } + }; + + return ( +
+ {t("cws.jev.decisionMethod")} +
+ {METHODS.map(option => ( + + ))} +
+

+ {method === "typesafe" && t("cws.jev.decisionServiceDefaultHint")} + {method === "systemone" && servers.length === 0 && t("cws.jev.method.systemoneEmpty")} +

+ + {method === "systemone" && servers.length > 0 && ( +
+ + +

+ {t("cws.jev.decisionServiceHint")} + {selectedServer?.baseUrl && <> {selectedServer.baseUrl}} +

+ {selectedServer?.issue && ( +

+ {t("cws.err.invalidDecisionProvider", { + name: selectedServer.id ?? "", + reason: t(JEV_DECISION_ISSUE_LABEL_KEYS[selectedServer.issue]), + })} +

+ )} +
+ )} + + {method === "model" && ( +
+ + onChange({ decisionModel: e.target.value, decisionProvider: null })} + /> + + {modelRoutes.map(route => +

+ {t("cws.jev.decisionModelHint")} +

+
+ )} + +
+ + { + if (e.target.value === "") { + onChange({ decisionTimeoutMs: null }); + return; + } + const value = Number(e.target.value); + if (Number.isFinite(value)) onChange({ decisionTimeoutMs: value }); + }} + /> +

+ {t("cws.jev.decisionTimeoutHint", { + default: JEV_DECISION_TIMEOUT_DEFAULT_MS, + min: JEV_DECISION_TIMEOUT_MIN_MS, + max: JEV_DECISION_TIMEOUT_MAX_MS, + })} +

+
+ +
+ + + {test === null && t("cws.jev.testHint")} + {test?.state === "done" && (test.ok + ? t("cws.jev.testOk", { ms: test.latencyMs, backend: t(JEV_BACKEND_LABEL_KEYS[test.backend] ?? "cws.jev.backend.unknown") }) + : t("cws.jev.testFailOpen", { gate: test.gate, ms: test.latencyMs }))} + {test?.state === "error" && t("cws.jev.testError", { error: test.message })} + +
+
+ ); +} diff --git a/gui/src/components/combo-workspace-overview-panel.tsx b/gui/src/components/combo-workspace-overview-panel.tsx index e997efe1db3..235818e8ea9 100644 --- a/gui/src/components/combo-workspace-overview-panel.tsx +++ b/gui/src/components/combo-workspace-overview-panel.tsx @@ -1,7 +1,8 @@ import type { ComboItem, ProviderQuotaStates } from "../combo-workspace-data"; -import { buildComboAttention, groupCombos } from "../combo-workspace-data"; +import { buildComboAttention, groupCombos, jevDecisionSummary } from "../combo-workspace-data"; import { IconAlert, IconChevron, IconPlus } from "../icons"; import { useT, type TFn } from "../i18n/shared"; +import type { ProviderOption } from "./combo-workspace-types"; function attentionCopy( reason: "empty-targets" | "few-targets" | "catalog-omitted" | "all-targets-exhausted", @@ -18,6 +19,7 @@ export function OverviewPanel({ cataloguedComboIds, providerMap, providerQuotaStates, + providers, onSelect, onAdd, }: { @@ -25,6 +27,8 @@ export function OverviewPanel({ cataloguedComboIds?: ReadonlySet; providerMap: Readonly>; providerQuotaStates: ProviderQuotaStates; + /** Configured providers; resolves a JEV combo's decision service to its endpoint. */ + providers: readonly ProviderOption[]; onSelect: (id: string) => void; onAdd: () => void; }) { @@ -35,6 +39,10 @@ export function OverviewPanel({ providers: providerMap, providerQuotaStates, }); + const jevCombos = combos.flatMap((item) => { + const decision = jevDecisionSummary(item, providers); + return decision ? [{ item, decision }] : []; + }); return (
@@ -57,6 +65,37 @@ export function OverviewPanel({

{t("cws.howBody")}

+ {jevCombos.length > 0 && ( +
+

{t("cws.jev.decisionServicesTitle")}

+
+ {jevCombos.map(({ item, decision }) => ( + + ))} +
+
+ )} + {attention.length > 0 && (

{t("cws.attentionTitle")}

diff --git a/gui/src/components/combo-workspace-types.ts b/gui/src/components/combo-workspace-types.ts index aec36283f47..587aae79555 100644 --- a/gui/src/components/combo-workspace-types.ts +++ b/gui/src/components/combo-workspace-types.ts @@ -7,6 +7,9 @@ export type ProviderOption = { authMode?: string; adapter?: string; baseUrl?: string; + /** Decision-service rows need a model; see jevDecisionRowIssue. */ + defaultModel?: string; + models?: string[]; }; export type ModelOption = { provider: string; @@ -34,6 +37,8 @@ export interface ComboWorkspaceProps { onAdd: (intent?: ComboAddIntent) => void; adding: boolean; addIntent?: ComboAddIntent; + /** Decision service a `jev-auto` add pre-fills; null keeps canonical TypeSafe JEV. */ + addDecisionProvider?: string | null; onCloseAdd: () => void; onCreated: (id: string) => void; } diff --git a/gui/src/components/combo-workspace-utils.ts b/gui/src/components/combo-workspace-utils.ts index 5a5c6ee1c03..6428fe01108 100644 --- a/gui/src/components/combo-workspace-utils.ts +++ b/gui/src/components/combo-workspace-utils.ts @@ -1,5 +1,37 @@ +import { + type ComboDraftError, + type ComboItem, + JEV_DECISION_ISSUE_LABEL_KEYS, + jevDecisionProviderIssue, +} from "../combo-workspace-data"; +import type { TFn } from "../i18n/shared"; +import { + type JevDecisionRow, + JEV_DECISION_TIMEOUT_MAX_MS, + JEV_DECISION_TIMEOUT_MIN_MS, +} from "../jev-decision-service"; import type { ModelOption, ProviderOption } from "./combo-workspace-types"; +/** Localized validation message, with the bounds or the decision-service reason filled in. */ +export function comboDraftErrorText( + t: TFn, + code: ComboDraftError, + draft: ComboItem, + providers: Readonly>, +): string { + if (code === "invalidDecisionTimeout") { + return t("cws.err.invalidDecisionTimeout", { min: JEV_DECISION_TIMEOUT_MIN_MS, max: JEV_DECISION_TIMEOUT_MAX_MS }); + } + if (code === "invalidDecisionProvider") { + const issue = jevDecisionProviderIssue(draft.decisionProvider, providers) ?? "missing"; + return t("cws.err.invalidDecisionProvider", { + name: draft.decisionProvider ?? "", + reason: t(JEV_DECISION_ISSUE_LABEL_KEYS[issue]), + }); + } + return t(`cws.err.${code}`); +} + export function enabledProviders(providers: ProviderOption[]): ProviderOption[] { return providers .filter((p) => !p.disabled && !p.hiddenFromPicker) diff --git a/gui/src/components/jev-stats-panel.tsx b/gui/src/components/jev-stats-panel.tsx index d5b1baa8d80..52ba7166c2c 100644 --- a/gui/src/components/jev-stats-panel.tsx +++ b/gui/src/components/jev-stats-panel.tsx @@ -5,6 +5,7 @@ import { useI18n } from "../i18n/shared"; import { formatProviderDisplayName } from "../provider-icons"; import type { UsageReadMetadata } from "../usage-summary-resource"; import { Notice } from "../ui"; +import { JEV_BACKEND_LABEL_KEYS } from "../jev-decision-service"; import { DataSurfaceSkeleton } from "./data-surface"; import { UsageIncompleteNotice } from "./usage-incomplete-notice"; @@ -37,6 +38,8 @@ interface JevStatsResponse extends UsageReadMetadata { averageChosenProbability: number | null; }; gates: Array<{ gate: string; decisions: number }>; + /** Absent on servers that predate decision backends. */ + backends?: Array<{ backend: string; decisions: number; applied: number; averageLatencyMs: number | null }>; models: Array<{ provider: string; model: string; @@ -196,6 +199,31 @@ export function JevStatsPanel({ ))}
+ {data.backends && data.backends.length > 0 && ( +
+ + + + + + + + + + + {data.backends.map(row => ( + + + + + + + ))} + +
{t("cws.jev.stats.backend")}{t("cws.jev.stats.decisions")}{t("cws.jev.stats.applied")}{t("cws.jev.stats.latency")}
{t(JEV_BACKEND_LABEL_KEYS[row.backend] ?? "cws.jev.backend.unknown")}{row.decisions.toLocaleString(locale)}{row.applied.toLocaleString(locale)}{formatLatency(row.averageLatencyMs, locale)}
+
+ )} +
diff --git a/gui/src/components/native-main-profiles-view.tsx b/gui/src/components/native-main-profiles-view.tsx index 1ab3efc1441..0e773ee60a2 100644 --- a/gui/src/components/native-main-profiles-view.tsx +++ b/gui/src/components/native-main-profiles-view.tsx @@ -119,7 +119,7 @@ export function NativeMainProfilesView({ } {action && s &&
{ if (event.key === "Escape" && !busy) { event.preventDefault(); onSelect(null); } }}> @@ -131,9 +131,9 @@ export function NativeMainProfilesView({ onStopped(event.target.checked)} /> {t("nativeMain.stopped")}

-
- - +
} diff --git a/gui/src/components/provider-workspace/AnthropicAccountPoolSettings.tsx b/gui/src/components/provider-workspace/AnthropicAccountPoolSettings.tsx index 15b5e20b9e1..4d9c1ad1422 100644 --- a/gui/src/components/provider-workspace/AnthropicAccountPoolSettings.tsx +++ b/gui/src/components/provider-workspace/AnthropicAccountPoolSettings.tsx @@ -1,6 +1,8 @@ /** * Opt-in Anthropic OAuth account pool controls (#294). - * Experimental — shows a strong warning because the feature is not battle-tested. + * Experimental. The conditions it is meant for are static helper text next to the toggle, + * with the selection details behind a disclosure: the notice describes how to use the pool, + * so it is not announced as a live alert. Load and save failures keep their own messages. */ import { useCallback, useEffect, useLayoutEffect, useRef, useState } from "react"; import { useT } from "../../i18n/shared"; @@ -20,6 +22,9 @@ import { import AccountPoolStrategyControls from "../AccountPoolStrategyControls"; import { Select } from "../../ui"; +/** The public guide section that explains pool selection, failover and its limits. */ +const ANTHROPIC_POOL_GUIDE_URL = "https://opencodex.me/guides/claude-code/#claude-oauth-account-pool-experimental"; + const QUOTA_WINDOW_LABEL_KEYS = { "five-hour": "accountPool.quotaWindowFiveHour", weekly: "accountPool.quotaWindowWeekly", @@ -34,6 +39,32 @@ type PoolState = { quotaWindow: AccountPoolQuotaWindow; }; +/** + * The enabled status line names only what the selected strategy actually reads + * (src/oauth/anthropic-routing.ts). Round-robin rotates new sessions and refusal recovery + * through the ring and reads no usage, threshold or window. Fill-first drains the active + * account to its threshold in the window, then advances in stable order; at threshold 0 it + * stays until cooldown or sign-in. Quota keeps a healthy active account under the threshold + * and otherwise, and during recovery, picks the lowest usage in the window. + */ +function enabledStatus( + t: ReturnType, + strategy: AccountPoolStrategy, + threshold: number, + quotaWindow: AccountPoolQuotaWindow, +): string { + const window = t(QUOTA_WINDOW_LABEL_KEYS[quotaWindow]); + if (strategy === "round-robin") return t("anthropicPool.enabledRoundRobinDesc"); + if (strategy === "fill-first") { + return threshold === 0 + ? t("anthropicPool.enabledFillFirstNoThresholdDesc") + : t("anthropicPool.enabledFillFirstDesc", { threshold, window }); + } + return threshold === 0 + ? t("anthropicPool.enabledNoProactiveDesc", { window }) + : t("anthropicPool.enabledDesc", { threshold, window }); +} + export default function AnthropicAccountPoolSettings({ apiBase, accountCount, @@ -204,14 +235,7 @@ export default function AnthropicAccountPoolSettings({ : loading ? t("common.loading") : enabled - ? threshold === 0 - ? t("anthropicPool.enabledNoProactiveDesc", { - window: t(QUOTA_WINDOW_LABEL_KEYS[quotaWindow]), - }) - : t("anthropicPool.enabledDesc", { - threshold, - window: t(QUOTA_WINDOW_LABEL_KEYS[quotaWindow]), - }) + ? enabledStatus(t, strategy, threshold, quotaWindow) : t("anthropicPool.disabledDesc")} @@ -236,14 +260,24 @@ export default function AnthropicAccountPoolSettings({ -
+

{t("anthropicPool.experimentalWarning")} -

+

{accountCount < 2 && (
{t("anthropicPool.needTwoAccounts")}
)} +
+ {t("anthropicPool.detailsSummary")} +

{t("anthropicPool.detailsEnabling")}

+

{t("anthropicPool.detailsFailover")}

+

{t("anthropicPool.detailsActivity")}

+

+ {t("anthropicPool.detailsGuide")} +

+
+ {enabled && state && ( <>