From 54819b0d6ac0beff37ea445b74b61e777be5ab06 Mon Sep 17 00:00:00 2001 From: pinvou3-dev Date: Wed, 9 Sep 2026 11:24:23 +0800 Subject: [PATCH 1/7] fix(ci): close exact-head release gates Make command-migration validation use the dispatched commit's parent when pull-request and push event baselines are unavailable. Keep downstream compatibility modules out of generated API docs and replace links to implementation-private items so rustdoc remains warning-free. Signed-off-by: pinvou3-dev --- .github/workflows/ci.yml | 6 ++++++ crates/app-server/src/daemon_socket.rs | 2 +- crates/tui/src/compaction.rs | 2 +- crates/tui/src/config.rs | 18 +++++++++--------- crates/tui/src/core/engine.rs | 6 +++--- crates/tui/src/core/engine/preview.rs | 4 ++-- crates/tui/src/core/events.rs | 6 +++--- crates/tui/src/core/turn.rs | 2 +- crates/tui/src/error_taxonomy.rs | 2 +- crates/tui/src/fleet/role.rs | 4 ++-- crates/tui/src/hooks/config.rs | 2 +- crates/tui/src/lib.rs | 20 ++++++++++++++++++++ crates/tui/src/mcp.rs | 2 +- crates/tui/src/prompts.rs | 2 +- crates/tui/src/session_manager.rs | 6 +++--- crates/tui/src/skills/install.rs | 4 ++-- crates/tui/src/skills/mod.rs | 2 +- crates/tui/src/tools/execution_envelope.rs | 6 +++--- crates/tui/src/tools/github/mod.rs | 4 ++-- crates/tui/src/tools/goal.rs | 2 +- crates/tui/src/tools/js_execution.rs | 2 +- crates/tui/src/tools/lsp.rs | 2 +- crates/tui/src/tools/pandoc.rs | 4 ++-- crates/tui/src/tools/read_media.rs | 2 +- crates/tui/src/tools/registry.rs | 2 +- crates/tui/src/tools/shell/output.rs | 2 +- crates/tui/src/tools/subagent/mod.rs | 4 ++-- crates/tui/src/tools/verify.rs | 10 +++++----- scripts/test_ci_migration_wiring.py | 6 ++++++ 29 files changed, 84 insertions(+), 52 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a4d2ef7561..c534cba443 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -377,11 +377,17 @@ jobs: - name: Check command migration manifest if: needs.changes.outputs.heavy == 'true' env: + EVENT_NAME: ${{ github.event_name }} PR_BASE_SHA: ${{ github.event.pull_request.base.sha }} PUSH_BEFORE_SHA: ${{ github.event.before }} run: | python3 scripts/test_check_command_migration_manifest.py baseline="${PR_BASE_SHA:-${PUSH_BEFORE_SHA:-}}" + # workflow_dispatch has neither a PR base nor a push-before SHA. + # Exact-head full CI still needs a real previous manifest revision. + if [[ "${EVENT_NAME}" == "workflow_dispatch" ]]; then + baseline="$(git rev-parse HEAD^)" + fi if [[ -n "${baseline}" && ! "${baseline}" =~ ^0+$ ]]; then git fetch --no-tags origin "${baseline}" python3 scripts/check-command-migration-manifest.py --baseline-ref "${baseline}" diff --git a/crates/app-server/src/daemon_socket.rs b/crates/app-server/src/daemon_socket.rs index 6ce22ad5c2..e3148762c2 100644 --- a/crates/app-server/src/daemon_socket.rs +++ b/crates/app-server/src/daemon_socket.rs @@ -4,7 +4,7 @@ //! daemon over a local socket instead of a TCP port: local multi-client, //! peer-credential auth, nothing to firewall (CORE-PROTOCOL spec §5). The //! wire is *identical* to the `--stdio` transport — newline-delimited -//! JSON-RPC 2.0 driven by the same [`crate::run_stdio_loop`] — with exactly +//! JSON-RPC 2.0 driven by the same `crate::run_stdio_loop` — with exactly //! one addition in front of it: a `daemon/attach` handshake that establishes //! who this client is and whether it owns the daemon. //! diff --git a/crates/tui/src/compaction.rs b/crates/tui/src/compaction.rs index c53f0b826b..480cc79516 100644 --- a/crates/tui/src/compaction.rs +++ b/crates/tui/src/compaction.rs @@ -161,7 +161,7 @@ duplicating work. Here is the summary produced by the other language model, use in this summary to assist with your own analysis:"; /// Detection marker for committed compaction-summary text: the stable first -/// sentence of [`SUMMARY_HEADER`]. `engine/context.rs` restores summaries by +/// sentence of `SUMMARY_HEADER`. `engine/context.rs` restores summaries by /// the same marker on session load. pub const COMPACTION_SUMMARY_MARKER: &str = "Another language model started to solve this problem"; /// Marker written by pre-v0.9.6 compaction; sessions saved under the old diff --git a/crates/tui/src/config.rs b/crates/tui/src/config.rs index f5c5c3c82b..11b3dbec7e 100644 --- a/crates/tui/src/config.rs +++ b/crates/tui/src/config.rs @@ -605,7 +605,7 @@ pub struct ProviderCapability { /// behind" — for example the Kimi Code membership ids, whose limits live in /// the membership catalog rather than the static model catalogue. Unknown /// must stay unknown: callers may **not** substitute a placeholder ceiling, - /// and in particular [`crate::route_budget`] does not clamp a requested + /// and in particular `crate::route_budget` does not clamp a requested /// `max_tokens` against an unknown compatibility cap. /// /// When `Some`, the value is a documented exact-route maximum or a @@ -1536,7 +1536,7 @@ pub(crate) fn legacy_deepseek_alias_effort_for_route( /// fallback** (#4188). /// /// Preferred sources are the live Models.dev catalog and the offline bundled -/// snapshot via [`crate::provider_lake`]. Call this directly only for +/// snapshot via `crate::provider_lake`. Call this directly only for /// Codewhale-only / local providers Models.dev does not represent, or when /// probing the fallback table in tests. Picker, inventory, and subagent /// surfaces must go through the provider lake. @@ -2314,7 +2314,7 @@ pub struct GoalConfig { /// Goals are unlimited by default; token/time budgets are telemetry only. /// /// `None` uses the built-in default - /// ([`crate::goal_loop::DEFAULT_MAX_GOAL_CONTINUATIONS`], currently `0`); + /// (`crate::goal_loop::DEFAULT_MAX_GOAL_CONTINUATIONS`, currently `0`); /// `0` disables the backstop entirely so only terminal status or user /// control ends the run. #[serde(default)] @@ -2807,8 +2807,8 @@ pub struct AutoRouterConfig { #[serde(default)] pub thinking: Option, /// Classifier call timeout in seconds. Defaults to - /// [`DEFAULT_AUTO_ROUTER_TIMEOUT_SECS`] (4); `0` means "use the default". - /// Values above [`MAX_AUTO_ROUTER_TIMEOUT_SECS`] (300) are clamped so a + /// `DEFAULT_AUTO_ROUTER_TIMEOUT_SECS` (4); `0` means "use the default". + /// Values above `MAX_AUTO_ROUTER_TIMEOUT_SECS` (300) are clamped so a /// hung local router cannot stall a turn indefinitely. #[serde(default)] pub timeout_secs: Option, @@ -3728,7 +3728,7 @@ impl NetworkPolicyToml { } } -/// `[lsp]` table — mirrors [`crate::lsp::LspConfig`]. Documented in +/// `[lsp]` table — mirrors `crate::lsp::LspConfig`. Documented in /// `config.example.toml`. When omitted, defaults from `LspConfig::default()` /// apply (enabled, 5 s poll, 20 diagnostics/file, errors only, no overrides). #[derive(Debug, Clone, Deserialize, Default)] @@ -3757,7 +3757,7 @@ pub struct LspConfigToml { } impl LspConfigToml { - /// Build a runtime [`crate::lsp::LspConfig`] from the on-disk schema, + /// Build a runtime `crate::lsp::LspConfig` from the on-disk schema, /// falling back to defaults for any unset fields. #[must_use] pub fn into_runtime(self) -> crate::lsp::LspConfig { @@ -4757,8 +4757,8 @@ impl Config { } /// Classifier call timeout for `[auto.router]` in seconds. Defaults to - /// [`DEFAULT_AUTO_ROUTER_TIMEOUT_SECS`] (4); `0` means "use the default". - /// Values above [`MAX_AUTO_ROUTER_TIMEOUT_SECS`] (300) are clamped so a + /// `DEFAULT_AUTO_ROUTER_TIMEOUT_SECS` (4); `0` means "use the default". + /// Values above `MAX_AUTO_ROUTER_TIMEOUT_SECS` (300) are clamped so a /// hung local router cannot stall a turn indefinitely. #[must_use] pub fn auto_router_timeout_secs(&self) -> u64 { diff --git a/crates/tui/src/core/engine.rs b/crates/tui/src/core/engine.rs index c61e278dd8..c8a92b26e8 100644 --- a/crates/tui/src/core/engine.rs +++ b/crates/tui/src/core/engine.rs @@ -356,8 +356,8 @@ pub struct EngineConfig { pub translation_enabled: bool, pub verbosity: Option, /// Maximum number of assistant steps before stopping. Ordinary interactive - /// hosts use [`UNBOUNDED_MODEL_STEPS`]; explicit test/embed callers may - /// still install a finite boundary. + /// hosts use the finite `DEFAULT_MODEL_STEPS`; configuration and explicit + /// test/embed callers may install a different bounded value. pub max_steps: u32, /// Maximum number of concurrently active subagents. pub max_subagents: usize, @@ -493,7 +493,7 @@ pub struct EngineConfig { /// Cumulative wall-clock budget for one turn (R1). Counted across every /// model step of the turn, excluding time blocked on a human approval /// decision. Resolved from `[tui].turn_wall_clock_secs`; always finite — - /// see [`turn_budget::resolve_turn_wall_clock`]. + /// see `turn_budget::resolve_turn_wall_clock`. pub turn_wall_clock: Duration, /// Per-step cap on accumulated streamed content, in bytes (R1). Resolved /// from `[tui].stream_max_content_mb`. Pre-R1 this was the hard-coded diff --git a/crates/tui/src/core/engine/preview.rs b/crates/tui/src/core/engine/preview.rs index 8fb3585043..a3d87eac9f 100644 --- a/crates/tui/src/core/engine/preview.rs +++ b/crates/tui/src/core/engine/preview.rs @@ -12,7 +12,7 @@ //! - **Never `session.last_tool_catalog`.** That value is one turn stale and //! stores the pre-activation catalog, so it cannot describe what the *next* //! request would send. The catalog is rebuilt through -//! [`Engine::build_turn_tool_registry_and_catalog`], which returns the same +//! `Engine::build_turn_tool_registry_and_catalog`, which returns the same //! typed policy a real turn consumes. //! - **Never invent a route.** For fixed routes, the host resolves the next //! turn through the same shared planner production dispatch uses. Auto would @@ -21,7 +21,7 @@ //! model, billing, tool budget, or body hash is recycled from the installed //! route. //! - **Never resolve by side effect.** The catalog build runs with -//! [`SubAgentWiring::Inert`] and [`McpAccess::PassiveSnapshot`]: no fork +//! `SubAgentWiring::Inert` and `McpAccess::PassiveSnapshot`: no fork //! snapshot, no spawned drainer, no MCP pool creation, no `connect_all`, no //! status events. When the connected MCP state is not already exactly what //! a turn would use, the tool section is reported unavailable rather than diff --git a/crates/tui/src/core/events.rs b/crates/tui/src/core/events.rs index a1d435a07f..0fb8f4e74e 100644 --- a/crates/tui/src/core/events.rs +++ b/crates/tui/src/core/events.rs @@ -58,7 +58,7 @@ pub struct TurnRoute { pub billing: Option, /// Endpoint this turn's client was frozen against, verbatim. /// - /// [`crate::route_receipt::TurnRouteReceipt`] deliberately keeps only a + /// `crate::route_receipt::TurnRouteReceipt` deliberately keeps only a /// redacted endpoint identity, which billing cannot classify from, so the /// non-secret URL travels here. Captured from the resolved route candidate /// at the client-freeze boundary, before any ambient selection state can @@ -69,7 +69,7 @@ pub struct TurnRoute { /// at the same instant. /// /// Together with `provider_identity` and `base_url` this is a complete - /// [`crate::route_billing::DispatchedReceipt`]: every fact billing needs, + /// `crate::route_billing::DispatchedReceipt`: every fact billing needs, /// frozen at the client-freeze boundary. Consumers must classify from /// these fields and must never re-read an ambient `Config` after the turn /// starts — by `TurnComplete` a provider switch, an auto-router hop, or a @@ -89,7 +89,7 @@ pub struct TurnRoute { /// /// - `base_url` + `billing_product` + `provider_identity` are frozen at the /// **client-freeze** boundary and answer *which route is this and how does -/// it bill* — a [`crate::route_billing::DispatchedReceipt`]. They must be +/// it bill* — a `crate::route_billing::DispatchedReceipt`. They must be /// readable from `TurnStarted` onward so a child turn arriving mid-flight /// can be billed against the parent's frozen route. /// - This envelope is stamped at the **wire** boundary and answers *what was diff --git a/crates/tui/src/core/turn.rs b/crates/tui/src/core/turn.rs index 14e5aff243..90b4c086dd 100644 --- a/crates/tui/src/core/turn.rs +++ b/crates/tui/src/core/turn.rs @@ -372,7 +372,7 @@ pub(crate) fn parse_snapshot_label(label: &str) -> ParsedSnapshotLabel { /// Take a `pre-turn:` workspace snapshot. /// /// `cap_bytes` is the workspace-size ceiling that gates first-init -/// (passed through to [`SnapshotRepo::open_or_init_with_cap`]); pass +/// (passed through to `SnapshotRepo::open_or_init_with_cap`); pass /// `0` to disable the cap. /// `user_prompt` is an optional snippet of the user's message for this /// turn, embedded in the snapshot label so `/restore` listings are diff --git a/crates/tui/src/error_taxonomy.rs b/crates/tui/src/error_taxonomy.rs index 9eb7100d7d..ee655c41a8 100644 --- a/crates/tui/src/error_taxonomy.rs +++ b/crates/tui/src/error_taxonomy.rs @@ -206,7 +206,7 @@ impl ErrorEnvelope { } } -/// Classify a boundary error from the typed [`LlmError`] when one is +/// Classify a boundary error from the typed `LlmError` when one is /// available, keeping the caller's display message. /// /// Boundaries that only have an `anyhow::Error` historically stringified it diff --git a/crates/tui/src/fleet/role.rs b/crates/tui/src/fleet/role.rs index 88aca8a799..040d5074aa 100644 --- a/crates/tui/src/fleet/role.rs +++ b/crates/tui/src/fleet/role.rs @@ -28,7 +28,7 @@ use crate::worker_profile::{ShellPolicy, ToolScope, WorkerRuntimeProfile}; /// Canonical model-facing Fleet role values, in schema order. This is the /// closed `enum` advertised on the Agent tool's `type` property. Legacy /// aliases are accepted only at replay/deserialization boundaries -/// ([`migrate_legacy_role_token`]) and are never advertised to models. +/// (`migrate_legacy_role_token`) and are never advertised to models. pub(crate) const FLEET_ROLE_SCHEMA_VALUES: [&str; 8] = [ "general", "explore", @@ -54,7 +54,7 @@ pub(crate) const VALID_ROLE_ALIASES: &str = "general; explore; planner; reviewer /// vocabulary one-to-one. Serialization, prompts, receipts, and UI always /// use [`Self::as_str`]. Legacy wire spellings (`worker`, `scout`, `plan`, /// `review`, `implementer`, …) are accepted only through -/// [`migrate_legacy_role_token`] at deserialization / parse boundaries. +/// `migrate_legacy_role_token` at deserialization / parse boundaries. /// /// This is the closed runtime role set. It is distinct from /// `codewhale_config::FleetRole`, which is the open config-side role diff --git a/crates/tui/src/hooks/config.rs b/crates/tui/src/hooks/config.rs index 83debb745d..63f514238d 100644 --- a/crates/tui/src/hooks/config.rs +++ b/crates/tui/src/hooks/config.rs @@ -483,7 +483,7 @@ impl HooksConfig { /// A condition that references context the event never carries can never /// match, so a hook wearing one is inert — the dangerous version of that /// is a `deny` gate the operator believes is armed. Those are reported as - /// `rejected` and dropped by [`Self::apply_validation`] rather than left + /// `rejected` and dropped by `Self::apply_validation` rather than left /// to fail silently at dispatch time. Problems that only affect how a /// hook is scheduled are reported as warnings and the hook still runs. #[must_use] diff --git a/crates/tui/src/lib.rs b/crates/tui/src/lib.rs index 65d16b1f95..bea28e80b1 100644 --- a/crates/tui/src/lib.rs +++ b/crates/tui/src/lib.rs @@ -20,11 +20,15 @@ use crate::dependencies::ExternalTool; use rust_i18n::i18n; i18n!("locales", fallback = ["en"]); +// Modules marked `doc(hidden)` below are public only as a downstream host- +// compatibility bridge. Keep the unstable facade out of generated API docs. mod acp_server; mod approval_log; +#[doc(hidden)] pub mod artifacts; mod audit; mod auto_reasoning; +#[doc(hidden)] pub mod automation_manager; mod child_env; mod client; @@ -32,15 +36,18 @@ pub mod cloud_dispatch; mod codex_model_cache; mod command_safety; mod commands; +#[doc(hidden)] pub mod compaction; mod composer_history; mod composer_stash; pub mod computer_meter; +#[doc(hidden)] pub mod config; mod config_persistence; mod context_budget; mod context_report; mod continual_harness; +#[doc(hidden)] pub mod core; mod cost_status; mod credentials; @@ -50,17 +57,20 @@ mod doctor; mod doctor_fix; mod dsh_credentials; mod elapsed; +#[doc(hidden)] pub mod error_taxonomy; mod eval; mod execpolicy; mod external_credentials; mod fast_hash; +#[doc(hidden)] pub mod features; mod fleet; pub use fleet::profile::WORKSPACE_AGENT_PROFILE_DIR; pub use fleet::roster::FleetRoster; mod goal_loop; mod hashing; +#[doc(hidden)] pub mod hooks; mod image_attach; mod import_claude; @@ -71,6 +81,7 @@ mod llm_response_cache; mod localization; mod logging; mod lsp; +#[doc(hidden)] pub mod mcp; mod mcp_server; mod media_originals; @@ -80,9 +91,11 @@ mod model_inventory; mod model_profile; mod model_registry; mod model_routing; +#[doc(hidden)] pub mod models; mod models_dev_live; mod native_memory; +#[doc(hidden)] pub mod network_policy; mod oauth; mod operate; @@ -93,6 +106,7 @@ mod pricing; mod project_context; mod project_context_cache; mod prompt_zones; +#[doc(hidden)] pub mod prompts; mod provider_lake; mod provider_readiness; @@ -110,6 +124,7 @@ pub mod rlm; mod route_billing; mod route_budget; mod route_receipt; +#[doc(hidden)] pub mod route_runtime; mod runtime_api; mod runtime_chat_relay; @@ -132,6 +147,7 @@ mod doctor_loader_tests; #[cfg(test)] mod session_control_acceptance; #[allow(dead_code)] +#[doc(hidden)] pub mod session_manager; mod session_peek; mod session_projection; @@ -140,9 +156,11 @@ pub mod session_tree; mod settings; mod shell_dispatcher; mod skill_state; +#[doc(hidden)] pub mod skills; mod snapshot; mod startup_trace; +#[doc(hidden)] pub mod task_manager; mod telemetry_notice; #[cfg(test)] @@ -154,11 +172,13 @@ mod todo_snapshot; mod tool_history_repair; mod tool_inspection; mod tool_output_receipts; +#[doc(hidden)] pub mod tools; mod tui; pub use tui::app::AppMode; pub use tui::approval::ApprovalMode; mod turn_route_plan; +#[doc(hidden)] pub mod utils; mod vision; mod work_graph; diff --git a/crates/tui/src/mcp.rs b/crates/tui/src/mcp.rs index 38ff684fc5..38f509afdb 100644 --- a/crates/tui/src/mcp.rs +++ b/crates/tui/src/mcp.rs @@ -3169,7 +3169,7 @@ impl McpPool { /// Connect to all enabled servers, returning errors for failed connections. /// - /// Servers connect **concurrently** (bounded by [`Self::CONNECT_CONCURRENCY`]). + /// Servers connect **concurrently** (bounded by `Self::CONNECT_CONCURRENCY`). /// This used to be a sequential loop over `get_or_connect`, so every /// server paid the slowest server's spawn+handshake from its own budget: /// with the default 10s connect timeout, N servers meant a worst case of diff --git a/crates/tui/src/prompts.rs b/crates/tui/src/prompts.rs index d7223ddf6e..83984717f2 100644 --- a/crates/tui/src/prompts.rs +++ b/crates/tui/src/prompts.rs @@ -2,7 +2,7 @@ //! System prompt composition. //! //! Prompts are assembled from composable layers loaded at compile time from -//! the single [`text`] module: +//! the single `text` module: //! constitution + personality overlay → `message[0]` (byte-stable). //! approval policy → request-time runtime metadata. //! Tool availability comes only from the per-turn model catalog. diff --git a/crates/tui/src/session_manager.rs b/crates/tui/src/session_manager.rs index 6f7b07ddf1..1f9e7e7a62 100644 --- a/crates/tui/src/session_manager.rs +++ b/crates/tui/src/session_manager.rs @@ -181,7 +181,7 @@ pub struct SessionMetadata { /// and stay loadable; they are hidden from the default browse surfaces /// and are never chosen by auto-resume. /// - /// This mirrors `ThreadRecord::archived` in [`crate::runtime_threads`] so + /// This mirrors `ThreadRecord::archived` in `crate::runtime_threads` so /// the TUI session surfaces and the Runtime API/web dashboard project the /// same lifecycle field instead of two divergent notions of "put away". /// Additive and `skip_serializing_if`-guarded: sessions written before @@ -377,7 +377,7 @@ pub fn current_session_boot_id() -> &'static str { /// Which archive states a session listing includes. /// /// Deliberately the same three-way shape as -/// [`crate::runtime_threads::ThreadListFilter`] so `/v1/sessions` and +/// `crate::runtime_threads::ThreadListFilter` so `/v1/sessions` and /// `/v1/threads` answer the same `include_archived` / `archived_only` query /// pair with the same semantics. #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] @@ -508,7 +508,7 @@ impl SessionCostSnapshot { /// Session + subagent spend as **one** dual-currency accumulator. /// /// The persisted USD and CNY columns are projections of per-turn - /// [`crate::pricing::CostEstimate`]s that were accumulated jointly; every + /// `crate::pricing::CostEstimate`s that were accumulated jointly; every /// display total is derived from this single fold so the two currencies /// cannot be re-summed by separate code paths that then drift (#4939). /// CNY is *not* an FX multiple of USD: a turn carries CNY only when its diff --git a/crates/tui/src/skills/install.rs b/crates/tui/src/skills/install.rs index b67274642f..6bd5afe4f9 100644 --- a/crates/tui/src/skills/install.rs +++ b/crates/tui/src/skills/install.rs @@ -610,11 +610,11 @@ struct CacheMeta { /// For every skill listed in `index.json` this function: /// /// 1. Resolves the download URL (same logic as `install`). -/// 2. Checks the cached [`CacheMeta`] (etag + sha256) for freshness; skips +/// 2. Checks the cached `CacheMeta` (etag + sha256) for freshness; skips /// the download if unchanged. /// 3. Downloads SKILL.md (and any companion files if the source is a tarball) /// into `//`. -/// 4. Writes updated [`CacheMeta`] so the next sync is fast. +/// 4. Writes updated `CacheMeta` so the next sync is fast. /// /// Failures per-skill are non-fatal: [`SkillSyncOutcome::Failed`] is recorded /// and the sync continues. The caller decides how to surface per-skill errors. diff --git a/crates/tui/src/skills/mod.rs b/crates/tui/src/skills/mod.rs index 093e59e910..afba32e72b 100644 --- a/crates/tui/src/skills/mod.rs +++ b/crates/tui/src/skills/mod.rs @@ -405,7 +405,7 @@ impl SkillRegistry { /// `.git/`. The provided `dir` itself is always honored, even if /// hidden — that's what the user explicitly configured. /// Symlinked directories are followed when they resolve to directories, - /// with canonical path tracking plus [`Self::MAX_DISCOVERY_DEPTH`] keeping + /// with canonical path tracking plus `Self::MAX_DISCOVERY_DEPTH` keeping /// the walk finite when a skills layout contains cycles. #[must_use] pub fn discover(dir: &Path) -> Self { diff --git a/crates/tui/src/tools/execution_envelope.rs b/crates/tui/src/tools/execution_envelope.rs index 455715b11f..aab8ef0853 100644 --- a/crates/tui/src/tools/execution_envelope.rs +++ b/crates/tui/src/tools/execution_envelope.rs @@ -2,7 +2,7 @@ //! **executes**, **mutates**, or **reaches the network**. //! //! Before this module the answer was spread across three hand-maintained name -//! lists ([`crate::fleet::role::RAW_SHELL_DENYLIST`] and its siblings) plus a +//! lists (`crate::fleet::role::RAW_SHELL_DENYLIST` and its siblings) plus a //! role posture that keyed on `ShellPolicy::Full`. That shape had a structural //! hole: a name list can only deny the execution primitives someone remembered //! to write down, and `shell = "full"` was being read as "may run arbitrary @@ -24,7 +24,7 @@ //! //! The classification is derived, never listed: it comes from the tool's own //! [`ToolCapability`] set and from `is_read_only_for` applied to the **actual -//! input**, after [`canonical_action_alias`] has resolved the family/action +//! input**, after `canonical_action_alias` has resolved the family/action //! pair. That is what makes it cover tools this file has never heard of — //! plugins, runtime MCP servers, and anything registered later. //! @@ -49,7 +49,7 @@ //! whole purpose of a read-only verifier, and the shipped `verifier` role is //! exactly `write = false, shell = "full"`. Classifying it by tool name would //! either take the role's job away or hand it a program launcher, so the -//! bound is read off the concrete call by [`classify_verification`]: +//! bound is read off the concrete call by `classify_verification`: //! argument-free and pure test *selection* both cost shell authority (each //! forks a process, which `analyst`/`scout` were never granted), and //! anything that can name a program is held to the raw-shell bar. Every diff --git a/crates/tui/src/tools/github/mod.rs b/crates/tui/src/tools/github/mod.rs index b83c02750e..05540aed2f 100644 --- a/crates/tui/src/tools/github/mod.rs +++ b/crates/tui/src/tools/github/mod.rs @@ -6,8 +6,8 @@ //! //! This file is the surface and its guards — which action a call names, and //! whether the input is allowed to run it. The work itself is split by -//! responsibility: [`schema`] declares the input contracts, [`actions`] runs -//! the actions, [`cli`] builds every `gh`/`git` invocation, and [`shape`] +//! responsibility: `schema` declares the input contracts, `actions` runs +//! the actions, `cli` builds every `gh`/`git` invocation, and `shape` //! turns payloads into tool results. use async_trait::async_trait; diff --git a/crates/tui/src/tools/goal.rs b/crates/tui/src/tools/goal.rs index a0fae6f905..249704d8c1 100644 --- a/crates/tui/src/tools/goal.rs +++ b/crates/tui/src/tools/goal.rs @@ -179,7 +179,7 @@ fn normalize_explicit_goal_objective(raw: &str) -> Option { /// this. /// /// The whole "is this real work?" rule lives here: the prompt is work when it -/// has at least [`OPERATE_GOAL_MIN_WORDS`] words, or opens (after "please") +/// has at least `OPERATE_GOAL_MIN_WORDS` words, or opens (after "please") /// with an imperative work verb and has at least three words. Greetings, /// acknowledgements, and short questions stay chat. The objective is the whole /// prompt, whitespace-collapsed and bounded so continuation prompts stay small; diff --git a/crates/tui/src/tools/js_execution.rs b/crates/tui/src/tools/js_execution.rs index 9929fa68ba..23de4d4525 100644 --- a/crates/tui/src/tools/js_execution.rs +++ b/crates/tui/src/tools/js_execution.rs @@ -8,7 +8,7 @@ //! `execute_code_execution_tool`) keeps the dependency-probe and //! tempfile-spawn logic isolated for the test pin. //! -//! Registration is gated by [`crate::dependencies::resolve_node`]: +//! Registration is gated by `crate::dependencies::resolve_node`: //! when Node is missing the tool is simply not advertised, so the //! model never sees a runtime it can't actually use. See //! `core::engine::tool_catalog::ensure_advanced_tooling` for the diff --git a/crates/tui/src/tools/lsp.rs b/crates/tui/src/tools/lsp.rs index 9855b99db5..339d74585f 100644 --- a/crates/tui/src/tools/lsp.rs +++ b/crates/tui/src/tools/lsp.rs @@ -1,6 +1,6 @@ //! Model-facing LSP code-intelligence tool. //! -//! Extends the existing [`crate::lsp::LspManager`] lifecycle — never spawns a +//! Extends the existing `crate::lsp::LspManager` lifecycle — never spawns a //! competing server pool. Operations: diagnostics, read_lints, symbols, //! definition, references. diff --git a/crates/tui/src/tools/pandoc.rs b/crates/tui/src/tools/pandoc.rs index 9ac2f4c982..7a534a40d0 100644 --- a/crates/tui/src/tools/pandoc.rs +++ b/crates/tui/src/tools/pandoc.rs @@ -9,7 +9,7 @@ //! changelog as ..." workflows that previously required the user //! to drop into a terminal between turns. //! -//! Registration is gated by [`crate::dependencies::resolve_pandoc`] +//! Registration is gated by `crate::dependencies::resolve_pandoc` //! (see [`crate::tools::registry::ToolRegistryBuilder::with_pandoc_tools`]). //! When pandoc isn't installed the tool simply doesn't appear in the //! catalog, so the model never sees a binary it can't actually use. @@ -25,7 +25,7 @@ //! system dependencies (LaTeX engines, ImageMagick) beyond pandoc //! itself. //! -//! Adding a format: append to [`SUPPORTED_TARGET_FORMATS`] and the +//! Adding a format: append to `SUPPORTED_TARGET_FORMATS` and the //! schema description; the dispatch logic is whitelist-driven so //! anything in the list goes through unchanged. diff --git a/crates/tui/src/tools/read_media.rs b/crates/tui/src/tools/read_media.rs index c2db20f620..1b6280ffed 100644 --- a/crates/tui/src/tools/read_media.rs +++ b/crates/tui/src/tools/read_media.rs @@ -10,7 +10,7 @@ //! longest-edge halving) until the payload fits the budget. Results carry a //! delivery note stating exactly how the image was delivered (untouched / //! downsampled / crop / full) with zoom guidance, and the pre-compression -//! original is persisted in the content-addressed [`crate::media_originals`] +//! original is persisted in the content-addressed `crate::media_originals` //! store so a later crop read can pull the full-resolution source. When no //! ladder result fits the budget the tool fails closed — nothing is sent — //! with the exact conversion command to retry with. diff --git a/crates/tui/src/tools/registry.rs b/crates/tui/src/tools/registry.rs index 83c5e0d826..2bc7c5db3f 100644 --- a/crates/tui/src/tools/registry.rs +++ b/crates/tui/src/tools/registry.rs @@ -1137,7 +1137,7 @@ impl ToolRegistryBuilder { } /// Include the model-facing LSP intelligence tools. They reuse the - /// session [`crate::lsp::LspManager`] attached to `ToolContext` and never + /// session `crate::lsp::LspManager` attached to `ToolContext` and never /// spawn a second server lifecycle. #[must_use] pub fn with_lsp_tool(self) -> Self { diff --git a/crates/tui/src/tools/shell/output.rs b/crates/tui/src/tools/shell/output.rs index 205a01d41b..bd02a08854 100644 --- a/crates/tui/src/tools/shell/output.rs +++ b/crates/tui/src/tools/shell/output.rs @@ -35,7 +35,7 @@ pub(super) struct BoundedOutputAccumulator { full_output_path: Option, /// Why the on-disk spill file could not be created (disk full, descriptor /// exhaustion, unwritable temp dir). The stream still runs and the bounded - /// tail is still delivered; only "Full output: " is unavailable. + /// tail is still delivered; only `Full output: ` is unavailable. spill_unavailable: Option, } diff --git a/crates/tui/src/tools/subagent/mod.rs b/crates/tui/src/tools/subagent/mod.rs index 3e880b4dbb..c558a029d7 100644 --- a/crates/tui/src/tools/subagent/mod.rs +++ b/crates/tui/src/tools/subagent/mod.rs @@ -42,7 +42,7 @@ use crate::core::events::{AgentProgressEventMeta, Event}; use crate::core::session::ToolActivationCache; use crate::dependencies::{ExternalTool, Git}; /// Compatibility re-export: the closed role set lives in -/// [`crate::fleet::role`], the lightweight role surface every spawn path +/// `crate::fleet::role`, the lightweight role surface every spawn path /// consumes. Existing `tools::subagent::FleetRole` paths keep resolving. pub use crate::fleet::role::FleetRole; use crate::fleet::role::{ @@ -410,7 +410,7 @@ impl SubAgentAssignment { } /// Role presentation: system prompts and config key lookup. The role itself -/// ([`FleetRole`], parsing, posture) lives in [`crate::fleet::role`]; this +/// ([`FleetRole`], parsing, posture) lives in `crate::fleet::role`; this /// impl stays on the agent tool because it renders prompt text the tool owns. impl FleetRole { /// Pre-Fleet model-override key (`explorer_model` / `ni_model` tables). diff --git a/crates/tui/src/tools/verify.rs b/crates/tui/src/tools/verify.rs index 09f6ae5a16..cbe5d1be72 100644 --- a/crates/tui/src/tools/verify.rs +++ b/crates/tui/src/tools/verify.rs @@ -10,9 +10,9 @@ //! # Why elevated reasoning is the mechanism //! //! The critic request explicitly sets `reasoning_effort` to a high tier -//! ([`VerifyTool::critic_effort`], default [`ReasoningEffort::Max`]). Elevated +//! (`VerifyTool::critic_effort`, default `ReasoningEffort::Max`). Elevated //! reasoning IS the test-time-compute lever, so the critic never inherits a low -//! session tier — [`build_critic_request`] threads the effort onto the outgoing +//! session tier — `build_critic_request` threads the effort onto the outgoing //! [`MessageRequest`], which the client forwards to the provider. //! //! # Bounded / no runaway (hard requirement) @@ -21,12 +21,12 @@ //! guards enforce this: //! //! 1. **Structural (primary):** the critic is a single model call with -//! `tools: None` (see [`build_critic_request`]). With no tools of any kind, +//! `tools: None` (see `build_critic_request`). With no tools of any kind, //! the critic literally cannot invoke `verify` — recursion is impossible by //! construction, not by a denylist that could be forgotten. //! 2. **Re-entry guard (defense in depth):** [`VerifyTool::execute`] refuses if //! it is entered while a critique is already in progress on the same task -//! (tracked via the [`static@VERIFY_ACTIVE`] task-local). This protects any +//! (tracked via the `VERIFY_ACTIVE` task-local). This protects any //! future path that might run the critic inside a tool loop. //! //! # Relationship to neighbouring tools @@ -282,7 +282,7 @@ pub struct VerifyTool { } impl VerifyTool { - /// Construct with the default critic effort ([`ReasoningEffort::Max`]). + /// Construct with the default critic effort (`ReasoningEffort::Max`). #[must_use] pub fn new(client: Option, model: String) -> Self { Self { diff --git a/scripts/test_ci_migration_wiring.py b/scripts/test_ci_migration_wiring.py index 4fd97ccc10..e88dea169a 100644 --- a/scripts/test_ci_migration_wiring.py +++ b/scripts/test_ci_migration_wiring.py @@ -68,6 +68,12 @@ def test_migration_live_scan_receives_fetched_baseline(self) -> None: ) self.assertIn('--baseline-ref "${baseline}"', block) + def test_workflow_dispatch_uses_the_exact_heads_parent_as_baseline(self) -> None: + block = migration_step_block(load_ci()) + self.assertIn("EVENT_NAME", block) + self.assertIn('"${EVENT_NAME}" == "workflow_dispatch"', block) + self.assertIn('baseline="$(git rev-parse HEAD^)"', block) + def test_migration_commands_are_ordered_self_test_first(self) -> None: ci = load_ci() block = migration_step_block(ci) From ff9959bfc6b72e370756a2ab768bd905d072c43e Mon Sep 17 00:00:00 2001 From: pinvou3-dev Date: Wed, 9 Sep 2026 11:38:14 +0800 Subject: [PATCH 2/7] test(engine): remove obsolete full-access harness The active fork regression deliberately verifies that non-bypassable tools are blocked in Full Access. Remove the unused upstream comparison helper instead of consuming an additional dead-code budget exception. Signed-off-by: pinvou3-dev --- crates/tui/src/core/engine/tests.rs | 221 ---------------------------- 1 file changed, 221 deletions(-) diff --git a/crates/tui/src/core/engine/tests.rs b/crates/tui/src/core/engine/tests.rs index acef2de4fc..91825fb364 100644 --- a/crates/tui/src/core/engine/tests.rs +++ b/crates/tui/src/core/engine/tests.rs @@ -12353,227 +12353,6 @@ async fn assert_full_access_model_tool_batch_is_blocked( assert!(saw_turn_complete); } -// Retained as an upstream comparison harness: Pinvou deliberately blocks this -// path for non-bypassable tools, so the active regression uses the deny helper. -#[allow(dead_code)] -async fn assert_full_access_model_tool_batch_runs( - engine_config: EngineConfig, - tool_calls: Vec<(&'static str, serde_json::Value)>, - expected_names: &[&str], -) { - use wiremock::matchers::{body_string_contains, method, path}; - use wiremock::{Mock, MockServer, ResponseTemplate}; - - let server = MockServer::start().await; - let model_tool_calls = tool_calls - .iter() - .enumerate() - .map(|(index, (name, arguments))| { - json!({ - "index": index, - "id": format!("call_full_access_{index}"), - "type": "function", - "function": { - "name": name, - "arguments": arguments.to_string(), - }, - }) - }) - .collect::>(); - let tool_delta = json!({ - "id": "chatcmpl-full-access-blocked", - "choices": [{ - "index": 0, - "delta": {"tool_calls": model_tool_calls}, - "finish_reason": serde_json::Value::Null, - }], - }); - let tool_finish = json!({ - "id": "chatcmpl-full-access-blocked", - "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}], - }); - let tool_call_sse = format!("data: {tool_delta}\n\ndata: {tool_finish}\n\ndata: [DONE]\n\n"); - // Request 1's "model" response: discover the deferred specialized tools - // through tool_search, exactly as the lowercase contract expects before - // the first direct call. - let search_tool_calls = vec![ - json!({ - "index": 0, - "id": "call_search_mcp", - "type": "function", - "function": { - "name": "tool_search", - "arguments": r#"{"query":"mcp server"}"#, - }, - }), - json!({ - "index": 1, - "id": "call_search_rlm", - "type": "function", - "function": { - "name": "tool_search", - "arguments": r#"{"query":"rlm"}"#, - }, - }), - ]; - let search_delta = json!({ - "id": "chatcmpl-full-access-search", - "choices": [{ - "index": 0, - "delta": {"tool_calls": search_tool_calls}, - "finish_reason": serde_json::Value::Null, - }], - }); - let search_finish = json!({ - "id": "chatcmpl-full-access-search", - "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}], - }); - let search_sse = format!("data: {search_delta}\n\ndata: {search_finish}\n\ndata: [DONE]\n\n"); - let done_sse = concat!( - "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,", - "\"delta\":{\"content\":\"done\"},\"finish_reason\":null}]}\n\n", - "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,\"delta\":{},", - "\"finish_reason\":\"stop\"}]}\n\n", - "data: [DONE]\n\n", - ); - - // wiremock keeps every mounted mock matching even after its expected - // count is reached, and request history accumulates across the three - // steps, so substring matchers on the *calls* would re-fire forever. - // Anchor each mock on the tool-result ids that exist in exactly one - // request body: request 3 carries the executed batch's results, request 2 - // carries the search results, and request 1 carries neither. - - // Request 3 (exec batch results) terminates with the done turn. - Mock::given(method("POST")) - .and(path("/v1/chat/completions")) - .and(body_string_contains( - "\"tool_call_id\":\"call_full_access_0\"", - )) - .respond_with( - ResponseTemplate::new(200) - .insert_header("content-type", "text/event-stream") - .set_body_string(done_sse), - ) - .expect(1) - .with_priority(1) - .mount(&server) - .await; - // Request 2 (search results) executes the deferred specialized tools that - // request 1 discovered through tool_search. The search call activates - // them in the session cache, so this request runs them under Full Access - // without an approval modal. - Mock::given(method("POST")) - .and(path("/v1/chat/completions")) - .and(body_string_contains("\"tool_call_id\":\"call_search_mcp\"")) - .respond_with( - ResponseTemplate::new(200) - .insert_header("content-type", "text/event-stream") - .set_body_string(tool_call_sse), - ) - .expect(1) - .with_priority(2) - .mount(&server) - .await; - // Request 1: the model discovers the deferred specialized tools it needs. - Mock::given(method("POST")) - .and(path("/v1/chat/completions")) - .respond_with( - ResponseTemplate::new(200) - .insert_header("content-type", "text/event-stream") - .set_body_string(search_sse), - ) - .expect(1) - .with_priority(3) - .mount(&server) - .await; - - let api_config = Config { - api_key: Some("test-key".to_string()), - base_url: Some(server.uri()), - ..Config::default() - }; - let (engine, handle) = Engine::new(engine_config, &api_config); - let run_task = tokio::spawn(engine.run()); - - handle - .send(Op::SendMessage { - content: "exercise the Full Access auto-approval boundary".to_string(), - mode: AppMode::Agent, - route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL), - compaction: Box::new(CompactionConfig::default()), - goal_objective: None, - goal_token_budget: None, - goal_status: crate::tools::goal::GoalStatus::Active, - reasoning_effort: None, - reasoning_effort_auto: false, - auto_model: false, - allow_shell: true, - trust_mode: true, - auto_approve: true, - approval_mode: crate::tui::approval::ApprovalMode::Bypass, - translation_enabled: false, - allowed_tools: None, - dynamic_tools: Vec::new(), - hook_executor: None, - verbosity: None, - provenance: UserInputProvenance::ExternalUser, - turn_tool_security: None, - }) - .await - .expect("send Full Access model turn"); - - let expected = expected_names - .iter() - .copied() - .collect::>(); - let mut seen = HashSet::new(); - let mut saw_turn_complete = false; - let mut rx = handle.rx_event.write().await; - while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv()) - .await - .expect("timed out waiting for Full Access auto-approval event") - { - match event { - Event::ApprovalRequired { tool_name, .. } => { - panic!( - "Full Access must not open an approval modal for auto-approved tool {tool_name}" - ) - } - Event::ToolCallComplete { name, result, .. } if expected.contains(name.as_str()) => { - if let Err(error) = &result { - let message = error.to_string(); - assert!( - !message.contains("blocked in Full Access"), - "Full Access auto-approves non-bypassable tools: {message}" - ); - } - seen.insert(name); - } - Event::TurnComplete { status, error, .. } => { - assert_eq!( - status, - TurnOutcomeStatus::Completed, - "Full Access turn must complete: {error:?}" - ); - saw_turn_complete = true; - break; - } - _ => {} - } - } - drop(rx); - - handle.send(Op::Shutdown).await.expect("shutdown engine"); - run_task.await.expect("engine task"); - assert_eq!( - seen.len(), - expected.len(), - "every tool must reach execution, seen: {seen:?}" - ); - assert!(saw_turn_complete); -} - #[tokio::test] #[allow(clippy::await_holding_lock)] async fn full_access_blocks_non_bypassable_registered_tools_without_prompting() { From 6615af7cac0dcf13cc68afe2a5e385bbfaeb2f9b Mon Sep 17 00:00:00 2001 From: pinvou3-dev Date: Wed, 9 Sep 2026 11:47:36 +0800 Subject: [PATCH 3/7] docs(fork): remove stale policy section references Describe the registered product reversals directly so test comments do not point maintainers to a nonexistent fork-policy subsection. Signed-off-by: pinvou3-dev --- crates/tui/src/core/engine/tests.rs | 4 ++-- crates/tui/src/prompts.rs | 6 +++--- crates/tui/src/skills/tests.rs | 2 +- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/crates/tui/src/core/engine/tests.rs b/crates/tui/src/core/engine/tests.rs index 91825fb364..2a6788cd4a 100644 --- a/crates/tui/src/core/engine/tests.rs +++ b/crates/tui/src/core/engine/tests.rs @@ -8740,7 +8740,7 @@ fn non_bypassable_registered_tools_auto_approve_in_full_access() { // Upstream comparison: the shared resolver still returns "do not prompt" // for Bypass. Pinvou's execution boundary deliberately turns that result // into a denial via `registered_tool_blocked_in_full_access`; the - // fork-policy §3.4 integration test below records that reversal. + // result-level integration test below records that registered reversal. // Upstream #3866's resolver contract maps these holds to unprompted in // Full Access while Ask still prompts. The fork keeps that shared helper // unchanged and adds its stricter denial only at final engine dispatch. @@ -12356,7 +12356,7 @@ async fn assert_full_access_model_tool_batch_is_blocked( #[tokio::test] #[allow(clippy::await_holding_lock)] async fn full_access_blocks_non_bypassable_registered_tools_without_prompting() { - // Fork-policy §3.4: this intentionally reverses upstream v0.9.12's + // This intentionally reverses upstream v0.9.12's // `full_access_auto_approves_non_bypassable_registered_tools`. Pinvou // treats a tool that requires an explicit human decision as unavailable // in Full Access, whose posture cannot open an approval prompt. diff --git a/crates/tui/src/prompts.rs b/crates/tui/src/prompts.rs index 83984717f2..a0c83e51b8 100644 --- a/crates/tui/src/prompts.rs +++ b/crates/tui/src/prompts.rs @@ -2043,9 +2043,9 @@ mod tests { #[test] fn forkguard_system_prompt_uses_only_explicit_configured_skills_dir() { - // Fork-policy §3.4: the upstream default merge contract above still - // holds. Pinvou selects this separate, explicit-host mode so ambient - // workspace Skills cannot join the host-reviewed Skill authority. + // The registered Pinvou product boundary keeps the upstream default + // merge contract above intact while selecting a separate explicit-host + // mode so ambient workspace Skills cannot join the reviewed authority. let _env_guard = crate::test_support::lock_test_env(); let tmp = tempdir().expect("tempdir"); let _home = ScopedHome::set(tmp.path().join("home")); diff --git a/crates/tui/src/skills/tests.rs b/crates/tui/src/skills/tests.rs index 22f90bf72a..17954028e8 100644 --- a/crates/tui/src/skills/tests.rs +++ b/crates/tui/src/skills/tests.rs @@ -1131,7 +1131,7 @@ fn discover_for_workspace_and_dir_merges_workspace_and_configured_sources() { #[test] fn forkguard_explicit_skills_dir_excludes_ambient_workspace_sources() { - // Fork-policy §3.4: this is an additional explicit-host path, not a + // This is an additional registered explicit-host path, not a // reversal of the upstream default merge test above. Pinvou uses it to // keep ambient workspace roots outside the reviewed Skill authority. let tmpdir = TempDir::new().unwrap(); From 409138dbe41cf8a05991da1cd7fc7c7dca3f7e84 Mon Sep 17 00:00:00 2001 From: pinvou3-dev Date: Wed, 9 Sep 2026 12:12:38 +0800 Subject: [PATCH 4/7] chore(contract): record v0.9.12 runtime surface Signed-off-by: pinvou3-dev --- scripts/runtime-contract-budget.json | 26 +++++++++++++------------- 1 file changed, 13 insertions(+), 13 deletions(-) diff --git a/scripts/runtime-contract-budget.json b/scripts/runtime-contract-budget.json index 25cbca8092..0afc278f97 100644 --- a/scripts/runtime-contract-budget.json +++ b/scripts/runtime-contract-budget.json @@ -1,5 +1,5 @@ { - "_comment": "One-way numeric ceilings and exact structural identities for the provider-free runtime contract. Decreases pass; increases or identity changes fail. Lock in decreases with: python3 scripts/check-runtime-contract-budget.py --update The v0.9.8 child-receipt restore grew every production tool surface by 1496 schema bytes / 374 estimated tokens (agent tool). The v0.9.8 workshop read/tool-result byte fields then grew every production tool surface by 371 schema bytes / 93 estimated tokens. Both raises are explicit maintainer decisions; identities stay on the pre-raise digests only if the name set is unchanged \u2014 re-measure on Linux CI if Lint reports identity drift. The v0.9.8 pinned session prefix added the sentence to the base prompt (5848 -> 6084 bytes, every representative stage re-hashed), and the host-side Workflow/Goal verbs plus honest child posture grew the tool catalog (active 16531 -> 16602 bytes, full 71473 -> 72371); both are explicit v0.9.8 maintainer decisions measured from the release train. The v0.9.9 configured-skills change hides only custom configured-root paths, preserves discoverable default-root paths, normalizes Windows prompt separators, and trims 50 redundant skills-prompt bytes. The skill/memory/goal/handoff identities were re-measured without raising any ceiling. Explicit maintainer decision for #5473/#5492. The v0.9.10 full surfaces intentionally add the safe read_media tool; their measured schemas remain below the prior byte/token ceilings. Representative prompt byte metrics now use the same host-independent normalized text as their identities; the normalized base is 6089 bytes. The v0.9.11 model-visible sub-agent surface intentionally retires six legacy agents/* tools in favor of the canonical agent tool; all affected schema and prompt metrics decrease. The v0.9.12 plugin prompt-match slice intentionally adds the request_plugin_install tool to the full tool surfaces (plan full: +518 schema bytes / +130 estimated tokens / 29 -> 30 tools) so a strong prompt match can surface the human review CTA; explicit maintainer decision for #5663/#5579.", + "_comment": "One-way numeric ceilings and exact structural identities for the provider-free runtime contract. Decreases pass; increases or identity changes fail. Lock in decreases with: python3 scripts/check-runtime-contract-budget.py --update The v0.9.8 child-receipt restore grew every production tool surface by 1496 schema bytes / 374 estimated tokens (agent tool). The v0.9.8 workshop read/tool-result byte fields then grew every production tool surface by 371 schema bytes / 93 estimated tokens. Both raises are explicit maintainer decisions; identities stay on the pre-raise digests only if the name set is unchanged \u2014 re-measure on Linux CI if Lint reports identity drift. The v0.9.8 pinned session prefix added the sentence to the base prompt (5848 -> 6084 bytes, every representative stage re-hashed), and the host-side Workflow/Goal verbs plus honest child posture grew the tool catalog (active 16531 -> 16602 bytes, full 71473 -> 72371); both are explicit v0.9.8 maintainer decisions measured from the release train. The v0.9.9 configured-skills change hides only custom configured-root paths, preserves discoverable default-root paths, normalizes Windows prompt separators, and trims 50 redundant skills-prompt bytes. The skill/memory/goal/handoff identities were re-measured without raising any ceiling. Explicit maintainer decision for #5473/#5492. The v0.9.10 full surfaces intentionally add the safe read_media tool; their measured schemas remain below the prior byte/token ceilings. Representative prompt byte metrics now use the same host-independent normalized text as their identities; the normalized base is 6089 bytes. The v0.9.11 model-visible sub-agent surface intentionally retires six legacy agents/* tools in favor of the canonical agent tool; all affected schema and prompt metrics decrease. The v0.9.12 plugin prompt-match slice intentionally adds the request_plugin_install tool to the full tool surfaces (plan full: +518 schema bytes / +130 estimated tokens / 29 -> 30 tools) so a strong prompt match can surface the human review CTA; explicit maintainer decision for #5663/#5579. The final v0.9.12 release then added provider-route fields to the canonical automation and tasks full schemas (+548 bytes) while its Agent role-only copy shrank (-361), an official net +187 bytes in Act/Operate full. Pinvou r1 adds bounded write-size guidance and exact host-presented prompt-only profiles (+181 bytes versus the official release). Raising Act/Operate full to 65887 bytes / 16472 estimated tokens while tightening every active surface to 12704 / 3176 and Plan full to 38303 / 9576 is an explicit maintainer decision; tool identities are unchanged.", "document_kind": "codewhale.runtime_contract_budget", "representative_context": { "fixture_id": "representative-v1", @@ -85,9 +85,9 @@ "modes": { "act": { "active": { - "bytes": 12884, + "bytes": 12704, "identity_sha256": "8411bddfc0d7fce72ec53dfeef341f28e6a3cfe37b98a721dc05b17d53b0a13e", - "tokens_est": 3221, + "tokens_est": 3176, "tool_names": [ "agent", "bash", @@ -100,9 +100,9 @@ "tools": 7 }, "full": { - "bytes": 65519, + "bytes": 65887, "identity_sha256": "b5ceb72d89dc1e0e4161bd360ed2a05c433ce7b60b2300987165406c33d155af", - "tokens_est": 16380, + "tokens_est": 16472, "tool_names": [ "Git", "Run", @@ -161,9 +161,9 @@ }, "operate": { "active": { - "bytes": 12884, + "bytes": 12704, "identity_sha256": "8411bddfc0d7fce72ec53dfeef341f28e6a3cfe37b98a721dc05b17d53b0a13e", - "tokens_est": 3221, + "tokens_est": 3176, "tool_names": [ "agent", "bash", @@ -176,9 +176,9 @@ "tools": 7 }, "full": { - "bytes": 65519, + "bytes": 65887, "identity_sha256": "b5ceb72d89dc1e0e4161bd360ed2a05c433ce7b60b2300987165406c33d155af", - "tokens_est": 16380, + "tokens_est": 16472, "tool_names": [ "Git", "Run", @@ -237,9 +237,9 @@ }, "plan": { "active": { - "bytes": 12884, + "bytes": 12704, "identity_sha256": "8411bddfc0d7fce72ec53dfeef341f28e6a3cfe37b98a721dc05b17d53b0a13e", - "tokens_est": 3221, + "tokens_est": 3176, "tool_names": [ "agent", "bash", @@ -252,9 +252,9 @@ "tools": 7 }, "full": { - "bytes": 38483, + "bytes": 38303, "identity_sha256": "ac8af1f4988199825be7b00b054c258724a44074b1d4e6de6c92ade7c1cffe63", - "tokens_est": 9621, + "tokens_est": 9576, "tool_names": [ "Git", "Web", From 881cf4444403fce1978c0ff04db5d1a4898e3dab Mon Sep 17 00:00:00 2001 From: pinvou3-dev Date: Wed, 9 Sep 2026 12:50:30 +0800 Subject: [PATCH 5/7] fix(ci): let macOS wrapper smoke finish Signed-off-by: pinvou3-dev --- .github/workflows/ci.yml | 6 +++++- scripts/test_ci_migration_wiring.py | 10 ++++++++++ 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c534cba443..36b7dbab16 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -654,7 +654,11 @@ jobs: # ubuntu-only, so the required "npm wrapper smoke (ubuntu-latest)" # context is unaffected. Heavy pull requests execute the Ubuntu smoke # here; their branches may not be mirrored to CNB. - timeout-minutes: 30 + # A cold macOS release build can exceed 30 minutes when the optional + # sccache service becomes unavailable (the official v0.9.12 release and + # Pinvou exact-head verification both hit that path). Keep the smoke + # authoritative instead of letting infrastructure cancel it mid-build. + timeout-minutes: 60 runs-on: ${{ needs.changes.outputs.heavy == 'true' && matrix.os || 'ubuntu-latest' }} strategy: matrix: diff --git a/scripts/test_ci_migration_wiring.py b/scripts/test_ci_migration_wiring.py index e88dea169a..f9a86521b9 100644 --- a/scripts/test_ci_migration_wiring.py +++ b/scripts/test_ci_migration_wiring.py @@ -38,6 +38,12 @@ def migration_step_block(ci: str) -> str: return ci[step_start:next_step] +def npm_wrapper_job_block(ci: str) -> str: + start = ci.index("\n npm-wrapper-smoke:") + end = ci.index("\n mobile-smoke:", start) + return ci[start:end] + + class CiWiringTests(unittest.TestCase): def test_boundary_step_still_present(self) -> None: ci = load_ci() @@ -129,6 +135,10 @@ def test_safety_gate_is_hermetic_for_config_home(self) -> None: self.assertIn("unset CODEWHALE_CONFIG_PATH DEEPSEEK_CONFIG_PATH DEEPSEEK_HOME", block) self.assertIn("command_safety auto_review authority sandbox", block) + def test_npm_wrapper_smoke_allows_a_cold_macos_release_build(self) -> None: + block = npm_wrapper_job_block(load_ci()) + self.assertIn("timeout-minutes: 60", block) + def test_valid_wiring_passes_all_assertions(self) -> None: # The live workflow must satisfy every structural invariant above. ci = load_ci() From baa87f4de77d75ad69043817da1e541cb5e80345 Mon Sep 17 00:00:00 2001 From: pinvou3-dev Date: Wed, 9 Sep 2026 13:14:45 +0800 Subject: [PATCH 6/7] fix(automation): preserve overdue one-shot runs Signed-off-by: pinvou3-dev --- crates/tui/src/automation_manager.rs | 80 +++++++++++++++++----------- 1 file changed, 48 insertions(+), 32 deletions(-) diff --git a/crates/tui/src/automation_manager.rs b/crates/tui/src/automation_manager.rs index afb2647845..c48233634b 100644 --- a/crates/tui/src/automation_manager.rs +++ b/crates/tui/src/automation_manager.rs @@ -1410,12 +1410,15 @@ impl AutomationManager { AutomationRunStatus::Queued | AutomationRunStatus::Running ) }); - let missed_while_offline = now.signed_duration_since(due_at) - > Duration::seconds(AUTOMATION_MISFIRE_GRACE_SECS); - - // Recurring Pinvou tasks neither backfill offline slots nor overlap - // an active attempt. Jump directly to the first future slot. - if existing_for_slot || has_active_run || missed_while_offline { + let missed_recurring_slot = !matches!(&schedule, AutomationSchedule::Once { .. }) + && now.signed_duration_since(due_at) + > Duration::seconds(AUTOMATION_MISFIRE_GRACE_SECS); + + // Every schedule avoids duplicate and overlapping attempts. Only + // recurring tasks skip slots missed while the scheduler was + // offline: a one-shot has no future slot to advance to, so it must + // remain deliverable until its single run is durably enqueued. + if existing_for_slot || has_active_run || missed_recurring_slot { self.advance_automation_after_slot(&mut automation, &schedule, now, now)?; continue; } @@ -3190,14 +3193,20 @@ model = "private-model" ); } - #[test] - fn once_schedule_fires_once_and_auto_completes() { + #[tokio::test] + async fn forkguard_once_schedule_missed_while_offline_enqueues_exactly_one_run() -> Result<()> { let tempdir = tempfile::tempdir().expect("tempdir"); - let manager = AutomationManager::open(tempdir.path().to_path_buf()).expect("manager"); - // Exercise a due one-shot inside the scheduler's offline-misfire - // grace. Older one-shots are intentionally skipped by the Pinvou - // no-backfill contract. - let due_at = Utc::now() - Duration::seconds(10); + let task_manager = TaskManager::start_with_executor( + automation_task_config(tempdir.path().join("tasks")), + std::sync::Arc::new(AutomationNoopExecutor), + ) + .await?; + let manager = AutomationManager::open(tempdir.path().join("automations"))?; + // A one-shot missed while the process was offline has no future slot. + // It must still produce its one durable task/run instead of being + // silently advanced to `None` and paused. + let due_at = + Utc::now() - Duration::seconds(AUTOMATION_MISFIRE_GRACE_SECS.saturating_add(30)); let automation = AutomationRecord { rrule: due_at .format("FREQ=ONCE;AT=%Y-%m-%dT%H:%M:%S+00:00") @@ -3210,28 +3219,35 @@ model = "private-model" manager .save_automation(&automation) .expect("save automation"); + let shared: SharedAutomationManager = Arc::new(Mutex::new(manager)); + + scheduler_tick_shared(&shared, &task_manager).await?; - let due = manager - .collect_due_runs(Utc::now()) - .expect("collect due runs"); - assert_eq!(due.len(), 1); - let (_automation, run) = &due[0]; - assert_eq!(run.scheduled_for, due_at); + let task_id = { + let manager = shared.lock().await; + let runs = manager.list_runs(&automation.id, None)?; + assert_eq!(runs.len(), 1, "the expired one-shot must form one run"); + assert_eq!(runs[0].scheduled_for, due_at); + let task_id = runs[0] + .task_id + .clone() + .expect("the run must reference its enqueued task"); + let updated = manager.get_automation(&automation.id)?; + assert_eq!(updated.status, AutomationStatus::Paused); + assert_eq!(updated.next_run_at, None); + task_id + }; + assert_eq!(task_manager.get_task(&task_id).await?.id, task_id); - manager - .finish_scheduled_run(run, Utc::now()) - .expect("finish one-shot run"); - let updated = manager - .get_automation(&automation.id) - .expect("updated automation"); - assert_eq!(updated.status, AutomationStatus::Paused); - assert_eq!(updated.next_run_at, None); - assert!( - manager - .collect_due_runs(Utc::now() + Duration::hours(1)) - .expect("later tick") - .is_empty() + scheduler_tick_shared(&shared, &task_manager).await?; + assert_eq!( + shared.lock().await.list_runs(&automation.id, None)?.len(), + 1, + "a completed one-shot slot must not enqueue twice" ); + + task_manager.shutdown(); + Ok(()) } #[test] From 1fafee7e26b60a59457a43bce50c63aa2ad9dbaf Mon Sep 17 00:00:00 2001 From: pinvou3-dev Date: Wed, 9 Sep 2026 13:14:45 +0800 Subject: [PATCH 7/7] fix(search): restore actionable backend guidance Signed-off-by: pinvou3-dev --- crates/tui/src/tools/web/backend.rs | 30 +++++++++++++++++++++++++++-- 1 file changed, 28 insertions(+), 2 deletions(-) diff --git a/crates/tui/src/tools/web/backend.rs b/crates/tui/src/tools/web/backend.rs index 148ae8b562..95012f1c5f 100644 --- a/crates/tui/src/tools/web/backend.rs +++ b/crates/tui/src/tools/web/backend.rs @@ -10,6 +10,15 @@ use crate::client::ProviderNativeSearchRequest; use crate::config::SearchProvider; use crate::tools::spec::{ToolContext, ToolError}; +const SEARCH_BACKEND_CONFIGURATION_HINT: &str = concat!( + "Check network access, or configure `[search] provider` and `[search] api_key` in ", + "config.toml. Keyed providers include tavily, bocha, metaso, baidu, volcengine, and ", + "sofya; metaso also accepts METASO_API_KEY, baidu accepts BAIDU_SEARCH_API_KEY, ", + "volcengine accepts VOLCENGINE_API_KEY / VOLCENGINE_ARK_API_KEY / ARK_API_KEY, and ", + "sofya accepts SOFYA_API_KEY. For a keyless route, set ", + "`[search] provider = \"firecrawl\"` or `[search] provider = \"bing\"`." +); + #[async_trait] pub(crate) trait SearchBackend: Send + Sync { fn id(&self) -> BackendId; @@ -263,7 +272,7 @@ async fn run_backend_chain( .collect::>() .join(", "); Err(ToolError::not_available(format!( - "web search backends unavailable: {backend_ids}" + "web search backends unavailable: {backend_ids}. {SEARCH_BACKEND_CONFIGURATION_HINT}" ))) } @@ -837,7 +846,7 @@ mod tests { } #[tokio::test] - async fn all_unavailable_returns_typed_error_with_backend_ids_only() { + async fn all_unavailable_returns_actionable_error_without_private_details() { let private_error = "secret provider response"; let api = FakeBackend { id: BackendId::Bocha, @@ -860,6 +869,23 @@ mod tests { assert!(matches!(error, ToolError::NotAvailable { .. })); assert!(message.contains("bocha, duckduckgo")); + for provider in ["tavily", "bocha", "metaso", "baidu", "volcengine"] { + assert!( + message.contains(provider), + "configuration hint must name {provider}: `{message}`" + ); + } + assert!(message.contains("[search] provider")); + assert!(message.contains("[search] api_key")); + assert!(message.contains("config.toml")); + assert!(message.contains("METASO_API_KEY")); + assert!(message.contains("BAIDU_SEARCH_API_KEY")); + assert!(message.contains("VOLCENGINE_API_KEY")); + assert!(message.contains("VOLCENGINE_ARK_API_KEY")); + assert!(message.contains("ARK_API_KEY")); + assert!(message.contains("SOFYA_API_KEY")); + assert!(message.contains("provider = \"firecrawl\"")); + assert!(message.contains("provider = \"bing\"")); assert!(!message.contains(private_error)); assert!(!message.contains("different private response")); }