diff --git a/crates/app-server/src/chat_completions.rs b/crates/app-server/src/chat_completions.rs index f767714b38..d268aa6326 100644 --- a/crates/app-server/src/chat_completions.rs +++ b/crates/app-server/src/chat_completions.rs @@ -19,7 +19,7 @@ use codewhale_agent::ModelRegistry; use codewhale_config::{ ConfigApiKeyValueKind, ConfigToml, ProviderKind, auth_mode_disables_api_key, classify_config_api_key_value, is_upstream_auth_header, - provider::WireFormat, + provider::{WireFormat, wire_dialect_override}, provider_base_url_is_official, provider_preserves_custom_base_url_model, route::{LogicalModelRef, RouteError, RouteRequest, RouteResolver}, }; @@ -118,6 +118,16 @@ fn resolve_endpoint( saved_provider_model: None, base_url_override: Some(base_url.clone()), limit_overrides: Vec::new(), + // The wire dialect rides the same literal `[providers.custom]` table + // that supplied the endpoint and key above (this ingress reads the + // legacy field, not the named-table map the tui route layer + // resolves), so a `wire = "responses"` table cannot be silently + // served as chat here: the resolver mints a Responses candidate and + // the handler's ChatCompletions-only guard rejects it (fail closed) + // instead of forwarding to `{base}/chat/completions`. + wire_override: (provider_kind == ProviderKind::Custom) + .then(|| wire_dialect_override("custom", provider_cfg.wire.as_deref())) + .flatten(), })?; let model = route.wire_model_id().as_str().to_string(); @@ -1044,6 +1054,39 @@ api_key = {provider_api_key:?} )); } + /// A `wire = "responses"` custom table must reach this pass-through's + /// resolution: the endpoint then carries the Responses wire and the + /// handler's ChatCompletions-only guard rejects the request (fail closed) + /// instead of silently forwarding a chat body to `{base}/chat/completions` + /// — the same mis-route the runtime-route fix removed from per-turn + /// clients. + #[test] + fn custom_table_wire_override_reaches_the_app_route() { + let mut config = ConfigToml { + provider: ProviderKind::Custom, + ..ConfigToml::default() + }; + config.providers.custom.base_url = Some("https://relay.example.test/v1".to_string()); + config.providers.custom.model = Some("gpt-6-sol".to_string()); + config.providers.custom.wire = Some("responses".to_string()); + + let endpoint = resolve_endpoint(&config, &ModelRegistry::default(), Some("gpt-6-sol")) + .expect("custom responses table resolves"); + assert_eq!(endpoint.provider, ProviderKind::Custom); + assert_eq!(endpoint.model, "gpt-6-sol"); + assert_eq!( + endpoint.wire_format, + WireFormat::Responses, + "the table's wire dialect rides the same provider_cfg as its endpoint" + ); + + // An ordinary chat table keeps the forwardable Chat wire. + config.providers.custom.wire = None; + let endpoint = resolve_endpoint(&config, &ModelRegistry::default(), Some("gpt-6-sol")) + .expect("custom chat table resolves"); + assert_eq!(endpoint.wire_format, WireFormat::ChatCompletions); + } + #[test] fn foreign_model_is_rejected_before_credentials_or_headers_can_cross() { let mut config = ConfigToml { @@ -1482,25 +1525,119 @@ api_key = {provider_api_key:?} assert_eq!(response.status(), StatusCode::UNAUTHORIZED); } + /// The headline fail-closed path for a `wire = "responses"` custom + /// table: the pass-through must reject with + /// `provider_wire_format_unsupported` instead of silently forwarding a + /// chat body to `{base}/chat/completions`. Posted through the real + /// router so the resolver→guard composite is pinned end to end. #[tokio::test] - async fn non_chat_completions_provider_rejected() { - // Use the test to verify WireFormat checks work for non-ChatCompletions providers. - // Anthropic's wire format is AnthropicMessages; OpenaiCodex is Responses. - let endpoint = ResolvedModelEndpoint { - provider: ProviderKind::Anthropic, - base_url: "https://api.anthropic.com".to_string(), - model: "claude-sonnet-4-20250514".to_string(), - api_key: Some("sk-ant-test".to_string()), - auth_disabled: false, - http_headers: BTreeMap::new(), - path_suffix: None, - insecure_skip_tls_verify: false, - wire_format: WireFormat::AnthropicMessages, - }; + async fn custom_responses_wire_table_is_rejected_fail_closed() { + install_crypto_provider(); + let tmp = tempfile::tempdir().expect("tempdir"); + let config_path = tmp.path().join("config.toml"); + // The base URL is never contacted: the guard fires before any + // upstream I/O. + fs::write( + &config_path, + r#" +provider = "custom" + +[providers.custom] +wire = "responses" +base_url = "https://relay.example/v1" +api_key = "custom-responses-key" +model = "gpt-6-sol" +"#, + ) + .expect("write config"); + let state = build_state(Some(config_path), None).expect("state"); + let app = app_router(state, &[]); + + let body = serde_json::json!({ + "messages": [{"role": "user", "content": "hello"}] + }); + let response = app + .oneshot( + Request::builder() + .method(Method::POST) + .uri("/v1/chat/completions") + .header("content-type", "application/json") + .body(Body::from(serde_json::to_vec(&body).unwrap())) + .unwrap(), + ) + .await + .unwrap(); - assert_ne!(endpoint.wire_format, WireFormat::ChatCompletions); - // The handler would reject this; we verify the wire format here. - assert_eq!(endpoint.wire_format, WireFormat::AnthropicMessages); + assert_eq!(response.status(), StatusCode::BAD_REQUEST); + let bytes = axum::body::to_bytes(response.into_body(), 64 * 1024) + .await + .expect("error body"); + let payload: serde_json::Value = serde_json::from_slice(&bytes).expect("json error body"); + assert_eq!(payload["error"]["code"], "provider_wire_format_unsupported"); + assert_eq!(payload["error"]["type"], "unsupported_provider"); + assert!( + payload["error"]["message"] + .as_str() + .is_some_and(|message| message.contains("Responses")), + "the error names the offending dialect: {payload}" + ); + } + + /// Same fail-closed contract for the Anthropic dialect: a + /// `wire = "anthropic"` custom table resolves to a Messages-protocol + /// route and must be rejected by the Chat-Completions-only guard, not + /// forwarded a chat body. + #[tokio::test] + async fn custom_anthropic_wire_table_is_rejected_fail_closed() { + install_crypto_provider(); + let tmp = tempfile::tempdir().expect("tempdir"); + let config_path = tmp.path().join("config.toml"); + // The base URL is never contacted: the guard fires before any + // upstream I/O. + fs::write( + &config_path, + r#" +provider = "custom" + +[providers.custom] +wire = "anthropic" +base_url = "https://relay.example/v1" +api_key = "custom-anthropic-key" +model = "custom-claude" +"#, + ) + .expect("write config"); + let state = build_state(Some(config_path), None).expect("state"); + let app = app_router(state, &[]); + + let body = serde_json::json!({ + "messages": [{"role": "user", "content": "hello"}] + }); + let response = app + .oneshot( + Request::builder() + .method(Method::POST) + .uri("/v1/chat/completions") + .header("content-type", "application/json") + .body(Body::from(serde_json::to_vec(&body).unwrap())) + .unwrap(), + ) + .await + .unwrap(); + + assert_eq!(response.status(), StatusCode::BAD_REQUEST); + let bytes = axum::body::to_bytes(response.into_body(), 64 * 1024) + .await + .expect("error body"); + let payload: serde_json::Value = serde_json::from_slice(&bytes).expect("json error body"); + assert_eq!(payload["error"]["code"], "provider_wire_format_unsupported"); + assert_eq!(payload["error"]["type"], "unsupported_provider"); + assert!( + payload["error"]["message"] + .as_str() + .is_some_and(|message| message.contains("AnthropicMessages")), + "the error names the offending dialect: {payload}" + ); } #[test] diff --git a/crates/config/src/catalog.rs b/crates/config/src/catalog.rs index 125bf2e7db..0f6ea076a5 100644 --- a/crates/config/src/catalog.rs +++ b/crates/config/src/catalog.rs @@ -770,9 +770,15 @@ impl CatalogCompiler { /// Normalization folds case in the scheme/host, trims trailing slashes, and /// drops a default-port suffix, so cosmetically different spellings of the same /// endpoint share a cache scope while genuinely different endpoints do not. The -/// fingerprint is a SHA-256 digest. Secret-bearing URLs are mapped to one -/// constant redacted input before hashing, so userinfo, query credentials, and -/// fragments never enter the digest function at all. +/// fingerprint is a SHA-256 digest. Userinfo, query credentials, and fragments +/// never enter the digest function at all: a query or fragment glued onto the +/// authority fingerprints the credential-free host (distinct hosts stay +/// distinct), while non-http(s) schemes and empty or missing authorities map +/// to one constant redacted input. Scheme-less inputs take the fall-through +/// branch instead and fingerprint their credential-free authority and path. +/// Unlike the constant's inputs, the glued-query form does construct a client +/// and can carry captured state — which is why it fingerprints its host +/// rather than collapsing onto the shared constant. #[must_use] pub fn base_url_fingerprint(base_url: &str) -> String { use sha2::Digest as _; @@ -798,7 +804,21 @@ fn secret_free_fingerprint_input(base_url: &str) -> String { let authority_end = rest.find('/').unwrap_or(rest.len()); let authority_with_userinfo = &rest[..authority_end]; if authority_with_userinfo.contains(['?', '#']) { - return REDACTED.to_string(); + // A query or fragment glued onto the authority is a malformed but + // constructible base URL (`https://gw.example?tenant=1`). The host + // still identifies the endpoint, so fingerprint the + // credential-free host: distinct hosts must not collapse into one + // shared digest, or captured opaque state could replay across + // them. The query/fragment never enters the digest. + let head = authority_with_userinfo + .split(['?', '#']) + .next() + .unwrap_or_default(); + let authority = head.rsplit_once('@').map_or(head, |(_, host)| host); + if authority.is_empty() { + return REDACTED.to_string(); + } + return normalize_base_url(&format!("{scheme}://{authority}")); } let authority = authority_with_userinfo .rsplit_once('@') diff --git a/crates/config/src/catalog/tests.rs b/crates/config/src/catalog/tests.rs index 420249fe33..ff556e2164 100644 --- a/crates/config/src/catalog/tests.rs +++ b/crates/config/src/catalog/tests.rs @@ -357,6 +357,43 @@ fn fingerprint_never_hashes_secret_bearing_url_text() { } } +#[test] +fn fingerprint_of_an_authority_glued_query_is_the_host_not_one_constant() { + // `https://gw.example?tenant=1` glues the query onto the authority. That + // branch used to return one shared redacted constant, so two distinct + // hosts shared a single fingerprint — one replay scope for captured + // opaque reasoning state across genuinely different endpoints. Distinct + // hosts must stay distinct. + let host_a = base_url_fingerprint("https://gw-a.example?tenant=1"); + assert_ne!( + host_a, + base_url_fingerprint("https://gw-b.example?tenant=2"), + "distinct hosts must not share a fingerprint" + ); + // Same host, different glued query: one endpoint scope. The query is + // credential-bearing text and never enters the digest, so it must not + // split the scope either. + assert_eq!( + host_a, + base_url_fingerprint("https://gw-a.example?other=2"), + "a glued query must not change the host's fingerprint" + ); + // The host fingerprint agrees with the ordinary spelling of the same + // endpoint, and userinfo strips exactly like the path branch. + assert_eq!(host_a, base_url_fingerprint("https://gw-a.example")); + assert_eq!( + host_a, + base_url_fingerprint("https://user:secret@gw-a.example?tenant=1"), + "userinfo on a glued-query URL must not influence the digest" + ); + // A fragment glued onto the authority behaves the same way. + assert_eq!( + host_a, + base_url_fingerprint("https://gw-a.example#frag"), + "a glued fragment must not change the host's fingerprint" + ); +} + #[test] fn fingerprint_strips_userinfo_from_a_scheme_less_base_url() { // A base_url typed without a scheme took the fall-through branch, which diff --git a/crates/config/src/lib.rs b/crates/config/src/lib.rs index 68bd01ed87..62ec7fdcff 100644 --- a/crates/config/src/lib.rs +++ b/crates/config/src/lib.rs @@ -18,6 +18,8 @@ pub mod resolve; pub mod route; pub mod settings_schema; pub mod setup_state; +#[cfg(test)] +mod test_tracing; pub mod user_constitution; mod xai_credentials; pub use config_document::{ @@ -159,9 +161,14 @@ pub struct ProviderConfigToml { pub context_window: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub mode: Option, - /// Wire dialect preference for dual-protocol vendors (DeepSeek, MiniMax, - /// Model Studio): `openai` (Chat Completions, default) or `anthropic` - /// (Messages). Not a separate catalog provider — a power-user toggle. + /// Wire dialect override. Named custom-provider tables accept + /// `responses`, `anthropic` (or `messages`/`claude`), and `chat` or + /// `openai` (the Chat Completions default); dual-protocol built-in + /// vendors (DeepSeek, MiniMax, Model Studio) accept `openai` (default) + /// or `anthropic` (Messages). An unrecognized value falls back to the + /// default policy — custom tables log a warning as they degrade, while + /// a built-in vendor's dialect space is silently `openai`/`anthropic`. + /// Not a separate catalog provider — a power-user toggle. #[serde( default, skip_serializing_if = "Option::is_none", @@ -3378,6 +3385,21 @@ impl ConfigToml { // protocol, and endpoint come from a ReadyRouteCandidate. Auth/key // resolution above is unchanged. A resolver error keeps the existing // model string so this method stays total. + // The label must name the table `provider_cfg` above actually read: + // the named table only on the same Config-source condition that + // selected it, else the literal legacy table a CLI/env-forced Custom + // selection resolves — a typo warning that points at `relay_a` while + // the degraded dialect came from `[providers.custom]` would send the + // user to the wrong table. + let custom_table_label = if provider == ProviderKind::Custom + && matches!(provider_source, ProviderSource::Config) + { + self.named_custom_provider_id() + .unwrap_or("custom") + .to_string() + } else { + "custom".to_string() + }; let route = crate::route::RouteResolver::new() .resolve(&crate::route::RouteRequest { explicit_provider: Some(provider), @@ -3385,6 +3407,17 @@ impl ConfigToml { saved_provider_model: None, base_url_override: Some(base_url.clone()), limit_overrides: Vec::new(), + // provider_cfg above is the active custom table (named or + // legacy); its dialect is the same fact the tui route layer + // threads, so this receipt cannot disagree with the turn. + wire_override: (provider == ProviderKind::Custom) + .then(|| { + provider::wire_dialect_override( + &custom_table_label, + provider_cfg.wire.as_deref(), + ) + }) + .flatten(), }) .ok(); @@ -4579,6 +4612,10 @@ fn moonshot_base_url_uses_kimi_code(base_url: &str) -> bool { } /// Dual-wire vendors: dialect is config (`wire`), not a separate ProviderKind. +/// The `anthropic` alias tail is the canonical parse in +/// [`provider::wire_dialect_prefers_anthropic`] — the built-in dialect space +/// only ever branches openai/anthropic, so it shares that alias list rather +/// than keeping a second one that can drift. fn wire_prefers_anthropic(kind: ProviderKind, wire: Option<&str>) -> bool { if matches!( kind, @@ -4589,19 +4626,7 @@ fn wire_prefers_anthropic(kind: ProviderKind, wire: Option<&str>) -> bool { ) { return true; } - let Some(raw) = wire.map(str::trim).filter(|value| !value.is_empty()) else { - return false; - }; - let normalized = raw.to_ascii_lowercase().replace(['_', ' '], "-"); - matches!( - normalized.as_str(), - "anthropic" - | "anthropic-messages" - | "messages" - | "claude" - | "anthropic-compatible" - | "anthropic-compat" - ) + provider::wire_dialect_prefers_anthropic(wire) } fn modelstudio_mode_is_coding_plan(kind: ProviderKind, mode: Option<&str>) -> bool { diff --git a/crates/config/src/provider.rs b/crates/config/src/provider.rs index 686150f36b..85e77bdc5f 100644 --- a/crates/config/src/provider.rs +++ b/crates/config/src/provider.rs @@ -1718,15 +1718,92 @@ impl Provider for Custom { fn wire_policy(&self) -> WirePolicy { // Static default remains Chat Completions for backward compatibility. // Per-config `wire = "responses" | "anthropic" | "chat"` overrides are - // honored in `crates/tui/src/client.rs::provider_wire_format_for_config` - // and `crates/tui/src/config.rs::provider_capability`, which read - // `ProviderConfig::wire` for the `Custom` catalog identity. This keeps + // parsed by [`wire_dialect_override`] here in the config crate and + // honored by every consumer that reads `ProviderConfig::wire` for the + // `Custom` catalog identity (the tui wire-format/capability readers + // and the route resolver's `RouteRequest::wire_override`). This keeps // the `Provider` trait `Fixed` while giving custom endpoints the same // three-way switch (`responses` / `anthropic` / `chat`) as built-ins. WirePolicy::Fixed(WireFormat::ChatCompletions) } } +/// Whether a per-config `wire` dialect string names the Anthropic Messages +/// endpoint. Canonical parse shared by the tui wire-format/capability readers, +/// the route resolver, the app-server pass-through, and the built-in +/// dual-wire base-URL resolvers, so one alias list cannot drift from another. +#[must_use] +pub fn wire_dialect_prefers_anthropic(wire: Option<&str>) -> bool { + let Some(raw) = wire.map(str::trim).filter(|value| !value.is_empty()) else { + return false; + }; + let normalized = raw.to_ascii_lowercase().replace(['_', ' '], "-"); + matches!( + normalized.as_str(), + "anthropic" + | "anthropic-messages" + | "messages" + | "claude" + | "anthropic-compatible" + | "anthropic-compat" + ) +} + +/// Whether a per-config `wire` dialect string names the OpenAI Responses +/// endpoint. See [`wire_dialect_prefers_anthropic`] for the sharing contract. +#[must_use] +pub fn wire_dialect_prefers_responses(wire: Option<&str>) -> bool { + let Some(raw) = wire.map(str::trim).filter(|value| !value.is_empty()) else { + return false; + }; + let normalized = raw.to_ascii_lowercase().replace(['_', ' '], "-"); + matches!( + normalized.as_str(), + "responses" + | "responses-api" + | "openai-responses" + | "openai-responses-api" + | "response" + | "response-api" + | "openai-responses-compat" + | "responses-compat" + ) +} + +/// The wire override a per-config `wire` dialect asks for: `Some(Responses)` / +/// `Some(AnthropicMessages)` when the string names that endpoint, `None` for +/// `chat` / absent / unrecognized values (the static descriptor policy +/// applies). The resolver honors the override only for `ProviderKind::Custom`, +/// and every other consumer gates the same way, so the warning below can only +/// fire for a custom table — `table` names it in the log, since several +/// tables can coexist and the typo must be locatable. A non-empty +/// unrecognized value is most likely a typo of the one string that switches +/// the endpoint's protocol, so it is logged before degrading to the default +/// policy — the parse stays total and forward-compatible. +#[must_use] +pub fn wire_dialect_override(table: &str, wire: Option<&str>) -> Option { + if wire_dialect_prefers_responses(wire) { + Some(WireFormat::Responses) + } else if wire_dialect_prefers_anthropic(wire) { + Some(WireFormat::AnthropicMessages) + } else { + if let Some(raw) = wire.map(str::trim).filter(|value| !value.is_empty()) { + let normalized = raw.to_ascii_lowercase().replace(['_', ' '], "-"); + if !matches!( + normalized.as_str(), + "chat" | "chat-completions" | "openai" | "openai-chat" | "openai-chat-completions" + ) { + tracing::warn!( + table = %table, + dialect = %raw, + "unrecognized custom-provider wire dialect; using the default Chat Completions policy" + ); + } + } + None + } +} + static DEEPSEEK: Deepseek = Deepseek; static DEEPSEEK_ANTHROPIC: DeepseekAnthropic = DeepseekAnthropic; static NVIDIA_NIM: NvidiaNim = NvidiaNim; @@ -1899,6 +1976,136 @@ pub fn provider_for_kind(kind: ProviderKind) -> &'static dyn Provider { #[cfg(test)] mod tests { use super::*; + use crate::test_tracing::CapturedEvents; + + const TEST_TABLE: &str = "pinvou_responses"; + + #[test] + fn wire_dialect_override_parses_the_canonical_alias_sets() { + // The exact alias sets are the cross-crate contract: the tui route + // layer, the app-server pass-through, the ambient client, and the + // route receipt all resolve dialects through these two lists. + for dialect in [ + "responses", + "responses-api", + "openai-responses", + "openai-responses-api", + "response", + "response-api", + "openai-responses-compat", + "responses-compat", + ] { + assert_eq!( + wire_dialect_override(TEST_TABLE, Some(dialect)), + Some(WireFormat::Responses), + "{dialect} must parse as the Responses dialect" + ); + } + for dialect in [ + "anthropic", + "anthropic-messages", + "messages", + "claude", + "anthropic-compatible", + "anthropic-compat", + ] { + assert_eq!( + wire_dialect_override(TEST_TABLE, Some(dialect)), + Some(WireFormat::AnthropicMessages), + "{dialect} must parse as the Anthropic Messages dialect" + ); + } + } + + #[test] + fn wire_dialect_override_normalizes_case_whitespace_and_separators() { + assert_eq!( + wire_dialect_override(TEST_TABLE, Some(" Responses ")), + Some(WireFormat::Responses) + ); + assert_eq!( + wire_dialect_override(TEST_TABLE, Some("OPENAI_RESPONSES")), + Some(WireFormat::Responses) + ); + assert_eq!( + wire_dialect_override(TEST_TABLE, Some("Anthropic Messages")), + Some(WireFormat::AnthropicMessages) + ); + assert_eq!( + wire_dialect_override(TEST_TABLE, Some("anthropic-messages")), + Some(WireFormat::AnthropicMessages) + ); + assert_eq!( + wire_dialect_override(TEST_TABLE, Some("CLAUDE")), + Some(WireFormat::AnthropicMessages) + ); + } + + #[test] + fn wire_dialect_override_defaults_chat_and_degrades_unknowns() { + assert_eq!(wire_dialect_override(TEST_TABLE, None), None); + assert_eq!(wire_dialect_override(TEST_TABLE, Some("")), None); + assert_eq!(wire_dialect_override(TEST_TABLE, Some(" ")), None); + + // Recognized explicit-chat spellings degrade silently to the static + // policy — they are deliberate, not typos. + for dialect in [ + "chat", + "chat-completions", + "openai", + "openai-chat", + "openai-chat-completions", + ] { + assert_eq!( + wire_dialect_override(TEST_TABLE, Some(dialect)), + None, + "{dialect} must stay a silent no-preference value" + ); + } + + // A typo of the one string that switches the endpoint's protocol + // degrades to the default Chat policy (the tui route layer pins the + // same contract at the runtime candidate). + assert_eq!(wire_dialect_override(TEST_TABLE, Some("respones")), None); + assert!(!wire_dialect_prefers_responses(Some("respones"))); + assert!(!wire_dialect_prefers_anthropic(Some("respones"))); + } + + #[test] + fn unrecognized_dialect_warns_once_and_names_the_table() { + // The degrade contract above pins the return value; this pins the + // diagnostic itself — exactly the unrecognized non-empty dialect + // warns, and the warning names the table so the typo is locatable + // among several named tables. + let captured = CapturedEvents::default(); + tracing::subscriber::with_default(captured.clone(), || { + assert_eq!( + wire_dialect_override(TEST_TABLE, Some("respones")), + None, + "typo still degrades to the default policy" + ); + assert_eq!( + wire_dialect_override(TEST_TABLE, Some("responses")), + Some(WireFormat::Responses), + "the recognized dialect still parses (silently)" + ); + assert_eq!(wire_dialect_override(TEST_TABLE, None), None); + assert_eq!(wire_dialect_override(TEST_TABLE, Some(" ")), None); + }); + let events = captured.events(); + assert_eq!( + events.len(), + 1, + "only the unrecognized non-empty dialect warns: {events:?}" + ); + assert_eq!(events[0].field("table"), Some(TEST_TABLE)); + assert_eq!(events[0].field("dialect"), Some("respones")); + assert!( + events[0].message.contains("unrecognized"), + "warning must describe the degrade: {:?}", + events[0].message + ); + } #[test] fn credential_help_covers_every_provider_without_guessing_non_key_urls() { diff --git a/crates/config/src/route/authority.rs b/crates/config/src/route/authority.rs index b20c379d94..f23171188e 100644 --- a/crates/config/src/route/authority.rs +++ b/crates/config/src/route/authority.rs @@ -212,6 +212,7 @@ mod tests { saved_provider_model: None, base_url_override: None, limit_overrides: Vec::new(), + wire_override: None, } } diff --git a/crates/config/src/route/conformance_tests.rs b/crates/config/src/route/conformance_tests.rs index e2caeb887f..f3c9182da9 100644 --- a/crates/config/src/route/conformance_tests.rs +++ b/crates/config/src/route/conformance_tests.rs @@ -19,6 +19,7 @@ fn none_request(kind: ProviderKind) -> RouteRequest { saved_provider_model: None, base_url_override: None, limit_overrides: Vec::new(), + wire_override: None, } } @@ -125,6 +126,7 @@ fn every_provider_kind_resolves_the_auto_selector() { saved_provider_model: None, base_url_override: None, limit_overrides: Vec::new(), + wire_override: None, }; let candidate = resolver .resolve(&request) diff --git a/crates/config/src/route/resolver.rs b/crates/config/src/route/resolver.rs index 82574373b6..e54c26a1c5 100644 --- a/crates/config/src/route/resolver.rs +++ b/crates/config/src/route/resolver.rs @@ -27,6 +27,7 @@ //! There is deliberately no prompt-text / freeform field on [`RouteRequest`], //! which structurally bars prompt-content routing. +use super::RequestProtocol; use super::candidate::{ LimitField, PricingSku, ReadyRouteCandidate, ResolvedAuthSource, ResolvedEndpoint, SourcedLimitOverride, ValidationReport, @@ -62,6 +63,15 @@ pub struct RouteRequest { /// for adjusting a route's effective limits: the candidate itself is /// immutable once minted. pub limit_overrides: Vec, + /// Explicit wire-format override for `ProviderKind::Custom` routes. + /// + /// The Custom descriptor's static policy is Chat Completions for backward + /// compatibility; a per-config `[providers.] wire = "responses" | + /// "anthropic" | "chat"` table must mint a wire-true candidate so the + /// billing receipt, preflight, and the per-turn client binding all read + /// the same protocol. Ignored for every other kind: built-ins keep their + /// descriptor policy. + pub wire_override: Option, } /// Resolves [`RouteRequest`]s into [`ReadyRouteCandidate`]s. @@ -251,6 +261,27 @@ impl RouteResolver { selected.capabilities = RouteCapabilities::default(); selected.pricing = PricingSku::UnknownOrStale; } + // The Custom-only wire channel, bound once: the endpoint-key remap + // here and the protocol selection further down must read the same + // override, or a future edit to one `Custom` gate silently desyncs + // the key from the protocol. Built-ins keep their descriptor policy + // even if a stray override is set. + let wire_override = (provider_kind == ProviderKind::Custom) + .then_some(req.wire_override) + .flatten(); + if let Some(protocol_override) = wire_override { + // A per-config `wire` override names the endpoint the custom + // table actually serves; the static descriptor policy stays Chat + // Completions for backward compatibility, so the candidate's + // endpoint key and protocol must come from the override (the + // tui route layer reads `[providers.] wire` into the + // request; see `route_runtime::custom_wire_override_for`). + selected.endpoint_key = match protocol_override { + RequestProtocol::Responses => "responses".to_string(), + RequestProtocol::AnthropicMessages => "messages".to_string(), + RequestProtocol::ChatCompletions => selected.endpoint_key, + }; + } if provider_kind == ProviderKind::Zai { let effective_base_url = req .base_url_override @@ -274,8 +305,8 @@ impl RouteResolver { ); } - let protocol = descriptor - .protocol_for_endpoint(&selected.endpoint_key) + let protocol = wire_override + .or_else(|| descriptor.protocol_for_endpoint(&selected.endpoint_key)) .ok_or_else(|| RouteError::UnsupportedModelProtocol { provider: provider_id.clone(), model: selected.wire_model_id.as_str().to_string(), diff --git a/crates/config/src/route/tests.rs b/crates/config/src/route/tests.rs index 5247707593..ad302b50a6 100644 --- a/crates/config/src/route/tests.rs +++ b/crates/config/src/route/tests.rs @@ -19,6 +19,7 @@ fn req(provider: Option, model: Option<&str>) -> RouteRequest { saved_provider_model: None, base_url_override: None, limit_overrides: Vec::new(), + wire_override: None, } } @@ -431,6 +432,7 @@ fn resolver_routes_only_official_deepseek_flash_over_responses() { saved_provider_model: None, base_url_override: Some("https://compatible.example/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("custom compatible Flash route remains pass-through"); assert_eq!(custom.protocol(), RequestProtocol::ChatCompletions); @@ -455,6 +457,7 @@ fn resolver_routes_deepseek_vision_exp_over_chat_with_image_input() { saved_provider_model: None, base_url_override: base_url_override.map(str::to_string), limit_overrides: Vec::new(), + wire_override: None, }) .expect("experimental vision route resolves"); @@ -493,6 +496,7 @@ fn resolver_keeps_custom_deepseek_same_name_capabilities_unverified() { saved_provider_model: None, base_url_override: Some("https://deepseek-proxy.example.test/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("same-name custom proxy route resolves"); @@ -775,6 +779,7 @@ fn resolver_direct_owned_row_match_survives_casing_mismatch() { saved_provider_model: None, base_url_override: Some("https://compatible.example.test/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }; let out = r .resolve(&custom) @@ -841,6 +846,7 @@ fn resolver_custom_endpoint_allows_namespaced_selector_for_strict_provider() { saved_provider_model: None, base_url_override: Some("https://example.local/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }; let out = r .resolve(&request) @@ -864,6 +870,7 @@ fn resolver_treats_every_official_deepseek_endpoint_as_strict_direct() { saved_provider_model: None, base_url_override: Some(base_url.to_string()), limit_overrides: Vec::new(), + wire_override: None, }; assert!( matches!( @@ -884,6 +891,7 @@ fn resolver_does_not_trust_deepseek_hostname_substrings() { saved_provider_model: None, base_url_override: Some("https://api.deepseek.com.evil.example/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }; let route = resolver .resolve(&request) @@ -903,6 +911,7 @@ fn resolver_explicit_custom_with_base_url_override_passes_model_through_verbatim saved_provider_model: None, base_url_override: Some("https://api.example.com/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }; let out = r .resolve(&request) @@ -1054,6 +1063,7 @@ fn together_custom_endpoint_preserves_its_explicit_model_id() { saved_provider_model: None, base_url_override: Some("http://127.0.0.1:8000/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("custom Together-compatible endpoint should resolve"); @@ -1089,6 +1099,7 @@ fn openrouter_custom_endpoint_preserves_qwen37_alias() { saved_provider_model: None, base_url_override: Some("https://gateway.example.test/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("custom OpenRouter-compatible endpoint should resolve"); @@ -1144,6 +1155,7 @@ fn opencode_go_resolver_rejects_messages_models_even_on_custom_base_urls() { saved_provider_model: None, base_url_override, limit_overrides: Vec::new(), + wire_override: None, }; assert!( matches!( @@ -1228,6 +1240,7 @@ fn opencode_zen_resolver_fails_closed_for_unproven_protocols() { saved_provider_model: None, base_url_override, limit_overrides: Vec::new(), + wire_override: None, }; assert!( matches!( @@ -1284,6 +1297,7 @@ fn resolver_empty_saved_provider_model_is_empty_model_error() { saved_provider_model: Some(WireModelId::from("")), base_url_override: None, limit_overrides: Vec::new(), + wire_override: None, }; assert!(matches!(r.resolve(&request), Err(RouteError::EmptyModel))); } @@ -1479,6 +1493,7 @@ fn provider_native_web_search_requires_exact_direct_endpoint_offering() { saved_provider_model: None, base_url_override: Some("https://gateway.example.test/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("custom compatible endpoint resolves"); assert_eq!( @@ -1520,6 +1535,7 @@ fn mimo_native_search_is_exact_to_documented_chat_models() { saved_provider_model: None, base_url_override: Some("https://compatible.example.test/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("custom MiMo-compatible route resolves"); assert_eq!( @@ -1544,6 +1560,7 @@ fn zai_native_search_requires_exact_general_api_product() { saved_provider_model: None, base_url_override: Some(base_url.to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("general API route resolves"); assert_eq!( @@ -1565,6 +1582,7 @@ fn zai_native_search_requires_exact_general_api_product() { saved_provider_model: None, base_url_override: Some(base_url.to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("adjacent route resolves"); assert_eq!( @@ -1614,6 +1632,7 @@ fn qwen_native_search_is_exact_to_token_plan_responses_routes() { saved_provider_model: None, base_url_override: Some(base_url.to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("alternate product route resolves"); assert_eq!( @@ -1643,6 +1662,7 @@ fn moonshot_native_search_requires_exact_product_model_pair() { saved_provider_model: None, base_url_override: Some(base_url.to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("documented Moonshot/Kimi route resolves"); assert_eq!( @@ -1665,6 +1685,7 @@ fn moonshot_native_search_requires_exact_product_model_pair() { saved_provider_model: None, base_url_override: Some(base_url.to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("adjacent Moonshot/Kimi route resolves"); assert_eq!( @@ -1707,6 +1728,7 @@ fn custom_endpoint_does_not_inherit_first_party_pricing() { saved_provider_model: None, base_url_override: Some("https://deepseek-proxy.example.test/v1".to_string()), limit_overrides: Vec::new(), + wire_override: None, }) .expect("same-name custom proxy route resolves"); @@ -1756,6 +1778,7 @@ fn req_with_base(provider: ProviderKind, model: &str, base_url: &str) -> RouteRe saved_provider_model: None, base_url_override: Some(base_url.to_string()), limit_overrides: Vec::new(), + wire_override: None, } } @@ -1823,3 +1846,137 @@ fn https_endpoint_has_no_warning() { out.validation().messages ); } + +/// A `Custom` route's per-config `wire` override must mint a wire-true +/// candidate: the protocol and endpoint key come from the override, not the +/// descriptor's backward-compatible Chat Completions static policy. +#[test] +fn custom_wire_override_mints_a_wire_true_candidate() { + let resolver = RouteResolver::new(); + let base = |wire: Option| RouteRequest { + explicit_provider: Some(ProviderKind::Custom), + model_selector: Some(LogicalModelRef::from("gpt-6-sol".to_string())), + saved_provider_model: None, + base_url_override: Some("https://api.openai.com/v1".to_string()), + limit_overrides: Vec::new(), + wire_override: wire, + }; + + let responses = resolver + .resolve(&base(Some(RequestProtocol::Responses))) + .expect("responses override resolves"); + assert_eq!(responses.protocol(), RequestProtocol::Responses); + assert_eq!(responses.endpoint().endpoint_key, "responses"); + assert_eq!(responses.endpoint().base_url, "https://api.openai.com/v1"); + assert_eq!(responses.wire_model_id().as_str(), "gpt-6-sol"); + + let anthropic = resolver + .resolve(&base(Some(RequestProtocol::AnthropicMessages))) + .expect("anthropic override resolves"); + assert_eq!(anthropic.protocol(), RequestProtocol::AnthropicMessages); + assert_eq!(anthropic.endpoint().endpoint_key, "messages"); + + // An explicit `chat` override is the third arm of the dialect: it must + // resolve exactly like the absent case (the static Chat policy), not fall + // out of the override block with a stale endpoint key. + let explicit_chat = resolver + .resolve(&base(Some(RequestProtocol::ChatCompletions))) + .expect("explicit chat override resolves"); + assert_eq!(explicit_chat.protocol(), RequestProtocol::ChatCompletions); + assert_eq!(explicit_chat.endpoint().endpoint_key, "chat"); + + // No override keeps the documented backward-compatible default. + let chat = resolver.resolve(&base(None)).expect("default resolves"); + assert_eq!(chat.protocol(), RequestProtocol::ChatCompletions); + assert_eq!(chat.endpoint().endpoint_key, "chat"); +} + +/// The override channel is Custom-only: built-ins keep their descriptor +/// policy even when a request carries a wire override, so a stray override +/// cannot silently rewire a first-party route. +#[test] +fn wire_override_is_ignored_for_builtin_kinds() { + let resolver = RouteResolver::new(); + let out = resolver + .resolve(&RouteRequest { + explicit_provider: Some(ProviderKind::Openai), + model_selector: Some(LogicalModelRef::from("gpt-6-sol".to_string())), + saved_provider_model: None, + base_url_override: None, + limit_overrides: Vec::new(), + wire_override: Some(RequestProtocol::Responses), + }) + .expect("builtin route resolves"); + assert_eq!( + out.protocol(), + RequestProtocol::ChatCompletions, + "the builtin openai policy stays Chat Completions" + ); +} + +/// The override's edges: it mints the wire even with the descriptor's +/// placeholder base URL, it never resurrects unowned offering facts on a +/// custom endpoint, and a Responses-descriptor built-in ignores a Chat +/// override exactly like the Chat-policy kind ignores a Responses override. +#[test] +fn custom_wire_override_edges_stay_fail_closed() { + let resolver = RouteResolver::new(); + + // No base_url_override: the wire still comes from the override while the + // endpoint stays the Custom descriptor's loopback placeholder — failing + // closed locally instead of guessing a public host. + let placeholder = resolver + .resolve(&RouteRequest { + explicit_provider: Some(ProviderKind::Custom), + model_selector: Some(LogicalModelRef::from("gpt-6-sol".to_string())), + saved_provider_model: None, + base_url_override: None, + limit_overrides: Vec::new(), + wire_override: Some(RequestProtocol::Responses), + }) + .expect("override resolves without a base URL override"); + assert_eq!(placeholder.protocol(), RequestProtocol::Responses); + assert_eq!(placeholder.endpoint().endpoint_key, "responses"); + assert_eq!(placeholder.endpoint().base_url, "http://localhost/v1"); + + // A wire-true custom candidate must not restore capability or pricing + // facts the custom endpoint never proved, even when the override makes + // the route look first-party Responses. + let wire_true = resolver + .resolve(&RouteRequest { + explicit_provider: Some(ProviderKind::Custom), + model_selector: Some(LogicalModelRef::from("gpt-6-sol".to_string())), + saved_provider_model: None, + base_url_override: Some("https://api.openai.com/v1".to_string()), + limit_overrides: Vec::new(), + wire_override: Some(RequestProtocol::Responses), + }) + .expect("wire-true custom candidate resolves"); + assert_eq!( + wire_true.capabilities(), + RouteCapabilities::default(), + "the override must not resurrect capability facts" + ); + assert!(matches!( + wire_true.pricing(), + Some(super::candidate::PricingSku::UnknownOrStale) + )); + + // Demotion is structurally impossible: a built-in whose descriptor + // already serves Responses keeps its protocol under a Chat override. + let zen = resolver + .resolve(&RouteRequest { + explicit_provider: Some(ProviderKind::OpencodeZen), + model_selector: Some(LogicalModelRef::from("gpt-5.6-sol".to_string())), + saved_provider_model: None, + base_url_override: None, + limit_overrides: Vec::new(), + wire_override: Some(RequestProtocol::ChatCompletions), + }) + .expect("zen route resolves"); + assert_eq!( + zen.protocol(), + RequestProtocol::Responses, + "a Chat override cannot demote a Responses-descriptor builtin" + ); +} diff --git a/crates/config/src/test_tracing.rs b/crates/config/src/test_tracing.rs new file mode 100644 index 0000000000..2e168512bd --- /dev/null +++ b/crates/config/src/test_tracing.rs @@ -0,0 +1,74 @@ +//! Test-only `tracing` capture shared by config-crate unit tests. +//! +//! A minimal in-memory [`tracing::Subscriber`] so tests can assert on their +//! own diagnostics without a subscriber dependency. Clones share the same +//! buffer; install with +//! `tracing::subscriber::with_default(captured.clone(), || …)` around the +//! synchronous code whose events should be captured. + +/// One captured `tracing` event: the message plus every structured field the +/// event carried (a repeated field keeps its last value). +#[derive(Debug, Clone, Default)] +pub(crate) struct CapturedEvent { + pub(crate) message: String, + fields: std::collections::BTreeMap, +} + +impl CapturedEvent { + pub(crate) fn field(&self, name: &str) -> Option<&str> { + self.fields.get(name).map(String::as_str) + } +} + +/// A capturing subscriber. Clone it to install and to read events back. +#[derive(Clone, Default)] +pub(crate) struct CapturedEvents(std::sync::Arc>>); + +impl CapturedEvents { + pub(crate) fn events(&self) -> Vec { + self.0.lock().expect("event capture lock").clone() + } +} + +impl tracing::Subscriber for CapturedEvents { + fn enabled(&self, metadata: &tracing::Metadata<'_>) -> bool { + metadata.level() <= &tracing::Level::WARN + } + + fn new_span(&self, _attributes: &tracing::span::Attributes<'_>) -> tracing::Id { + tracing::Id::from_u64(1) + } + + fn record(&self, _span: &tracing::Id, _values: &tracing::span::Record<'_>) {} + + fn event(&self, event: &tracing::Event<'_>) { + #[derive(Default)] + struct Visitor { + captured: CapturedEvent, + } + impl tracing::field::Visit for Visitor { + fn record_debug(&mut self, field: &tracing::field::Field, value: &dyn std::fmt::Debug) { + if field.name() == "message" { + self.captured.message = format!("{value:?}"); + } else { + self.captured + .fields + .insert(field.name().to_string(), format!("{value:?}")); + } + } + } + + let mut visitor = Visitor::default(); + event.record(&mut visitor); + self.0 + .lock() + .expect("event capture lock") + .push(visitor.captured); + } + + fn enter(&self, _span: &tracing::Id) {} + + fn exit(&self, _span: &tracing::Id) {} + + fn record_follows_from(&self, _span: &tracing::Id, _follows_from: &tracing::Id) {} +} diff --git a/crates/config/src/tests.rs b/crates/config/src/tests.rs index 63466aea5d..076293665b 100644 --- a/crates/config/src/tests.rs +++ b/crates/config/src/tests.rs @@ -5198,6 +5198,7 @@ model = "gpt-5.5" saved_provider_model: None, base_url_override: None, limit_overrides: Vec::new(), + wire_override: None, }) .expect("documented Zen model must resolve"); assert_eq!(route.protocol(), crate::route::RequestProtocol::Responses); @@ -8772,6 +8773,117 @@ fn resolved_runtime_options_mints_a_route_candidate() { assert_eq!(route.endpoint().base_url, resolved.base_url); } +/// The runtime receipt for a custom table resolves that table's own `wire` +/// dialect, so `codewhale model resolve` receipts and readiness surfaces read +/// the same protocol the per-turn client binds. Deleting the receipt's +/// `wire_override` must fail this pin. +#[test] +fn resolved_runtime_options_threads_the_custom_tables_wire() { + let resolved = |wire: &str| -> crate::route::ReadyRouteCandidate { + let config: ConfigToml = toml::from_str(&format!( + r#" +provider = "custom" + +[providers.custom] +wire = "{wire}" +base_url = "https://relay.example/v1" +api_key = "receipt-wire-test-key" +model = "gpt-6-sol" +"# + )) + .expect("custom table config parses"); + config + .resolve_runtime_options(&CliRuntimeOverrides::default()) + .route + .expect("RouteResolver is the runtime path") + }; + + let responses = resolved("responses"); + assert_eq!(responses.protocol(), crate::provider::WireFormat::Responses); + assert_eq!(responses.endpoint().endpoint_key, "responses"); + assert_eq!(responses.endpoint().base_url, "https://relay.example/v1"); + + let anthropic = resolved("anthropic"); + assert_eq!( + anthropic.protocol(), + crate::provider::WireFormat::AnthropicMessages + ); + assert_eq!(anthropic.endpoint().endpoint_key, "messages"); + + // Explicit chat (and unset) stay on the static Chat policy. + let chat = resolved("chat"); + assert_eq!( + chat.protocol(), + crate::provider::WireFormat::ChatCompletions + ); + assert_eq!(chat.endpoint().endpoint_key, "chat"); +} + +/// The receipt's unrecognized-dialect warning must name the table the +/// receipt actually read. A CLI/env-forced Custom selection resolves the +/// literal legacy `[providers.custom]` table even while the config file +/// selects a named table, so the label must not follow the named selection — +/// a warning pointing at `relay_a` while the degraded dialect came from +/// `[providers.custom]` would send the user to the wrong table. Deleting the +/// Config-source condition on the label fails exactly this pin. +#[test] +fn cli_forced_custom_receipt_warns_against_the_table_it_reads() { + let _guard = env_lock(); + let captured = crate::test_tracing::CapturedEvents::default(); + let resolved = tracing::subscriber::with_default(captured.clone(), || { + let mut config: ConfigToml = toml::from_str( + r#" +[providers.custom] +wire = "respones" +base_url = "https://legacy.example/v1" +api_key = "legacy-key" +model = "legacy-model" + +[providers.relay_a] +kind = "openai-compatible" +base_url = "https://relay-a.example/v1" +api_key = "relay-key" +model = "relay-model" +"#, + ) + .expect("two-table config parses"); + config + .set_value("provider", "relay_a") + .expect("named selection resolves"); + config.resolve_runtime_options(&CliRuntimeOverrides { + provider: Some(ProviderKind::Custom), + ..CliRuntimeOverrides::default() + }) + }); + + // The receipt itself stays coherent: the forced selection reads the + // legacy table and degrades its typo to the static Chat policy. + let route = resolved.route.expect("legacy custom table resolves"); + assert_eq!( + route.protocol(), + crate::provider::WireFormat::ChatCompletions + ); + + // Filter to the dialect warning itself: the resolution's ambient-env + // readers may legitimately warn about junk exported into this shell, and + // the pin is about the dialect event, not about a clean environment. + let dialect_warnings: Vec<_> = captured + .events() + .into_iter() + .filter(|event| event.message.contains("wire dialect")) + .collect(); + assert_eq!( + dialect_warnings.len(), + 1, + "exactly the unrecognized dialect warns: {dialect_warnings:?}" + ); + assert_eq!( + dialect_warnings[0].field("table"), + Some("custom"), + "the warning must name the legacy table the receipt read, not the file's named selection: {dialect_warnings:?}" + ); +} + /// #5441: the runtime receipt carries the same source the surfaces print. #[test] fn resolved_runtime_options_reports_telemetry_source() { diff --git a/crates/core/src/request.rs b/crates/core/src/request.rs index 92e0650270..0c75f75a0f 100644 --- a/crates/core/src/request.rs +++ b/crates/core/src/request.rs @@ -130,6 +130,24 @@ pub struct OpaqueReasoningState { #[serde(skip_serializing_if = "Option::is_none")] pub id: Option, pub encrypted_content: String, + /// Fingerprint of the endpoint URL the state was captured from. A present + /// fingerprint must equal the requesting client's, so editing a named + /// Custom table's `base_url` stops replaying the previous endpoint's + /// opaque blobs — the provider tag alone pins the table name, not the URL + /// behind it. States minted before this field existed carry `None`: + /// those fail closed on Custom tags (no proof of origin, and the endpoint + /// can move under a stable tag), and built-in tags keep replaying only + /// while the requesting client still points at the provider's official + /// endpoint — a re-pointed client has no proof of where an old state was + /// captured, so it fails closed too. A state carrying this field read by + /// a build older than the field drops it as unknown JSON, which restores + /// that build's pre-fingerprint gate for the session; `custom/` + /// tags still drop there (the old gate never minted them). The one + /// mixed-version corner: a bare-`custom`-tagged state minted by this + /// build replays on a pre-fingerprint build without an endpoint check, + /// because the old gate's equality matches the root tag. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub endpoint: Option, } /// A single content block inside a message. diff --git a/crates/tui/src/client.rs b/crates/tui/src/client.rs index 3578514409..2cdbc7e8da 100644 --- a/crates/tui/src/client.rs +++ b/crates/tui/src/client.rs @@ -23,7 +23,9 @@ use codewhale_config::catalog::{ CatalogOffering, CatalogRefreshError, CatalogSnapshot, CatalogSource, CatalogStatus, ProviderCatalogCache, ProviderCatalogDelta, base_url_fingerprint, now_unix, }; -use codewhale_config::provider::WireFormat; +use codewhale_config::provider::{ + WireFormat, wire_dialect_prefers_anthropic, wire_dialect_prefers_responses, +}; use codewhale_config::route::{ LogicalModelRef, ReadyRouteCandidate, RouteLimits, RouteRequest, RouteResolver, }; @@ -1187,6 +1189,12 @@ impl DeepSeekClient { /// `Config`: `ReadyRouteCandidate` is secret-free by design (it carries only /// an auth-source *class*), so the API key and provider are still read from /// `config`. + /// + /// `config` must be the same identity-scoped config the candidate was + /// resolved from: the reasoning-provider tag freezes `config`'s table + /// identity, so pairing a pinned candidate with an ambient config would + /// mint captured state under the wrong table tag (replay then fails + /// closed). pub fn from_candidate(config: &Config, candidate: &ReadyRouteCandidate) -> Result { Self::from_parts( candidate.endpoint().base_url.clone(), @@ -1421,6 +1429,41 @@ impl DeepSeekClient { redact_model_bound_text(text, &self.model_bound_secret_values) } + /// The wire override that pins this transport's own endpoint on internal + /// re-resolutions: Custom clients are endpoint-scoped, so a fresh + /// config-aware resolution must reproduce — never downgrade — the wire + /// they were built with. Non-Custom transports resolve by descriptor + /// policy (`None`). + fn pinned_wire_override(&self) -> Option { + (self.api_provider == ApiProvider::Custom).then_some(self.wire_format) + } + + /// Provider tag minted into captured [`OpaqueReasoningState`] and required + /// by the replay gate. Built-in backends replay under their provider slug + /// (the gate pairs it with the official-endpoint rule for fingerprint-less + /// states); every named Custom table shares the `custom` slug, so the tag + /// carries the frozen table identity — encrypted reasoning captured for + /// one table must never replay onto another. + pub(super) fn reasoning_provider_tag(&self) -> String { + if self.api_provider != ApiProvider::Custom { + return self.api_provider.as_str().to_string(); + } + match self.provider_identity.as_str() { + "custom" | "" => ApiProvider::Custom.as_str().to_string(), + table => format!("custom/{table}"), + } + } + + /// Endpoint fingerprint minted into captured [`OpaqueReasoningState`] and + /// required by the replay gate: the tag pins the table name, not the URL + /// behind it, so a table whose `base_url` is edited between sessions must + /// stop replaying the previous endpoint's opaque blobs. Both capture and + /// replay derive this from the same frozen `base_url`, so an unchanged + /// endpoint always matches itself. + pub(super) fn reasoning_endpoint_fingerprint(&self) -> String { + base_url_fingerprint(&self.base_url) + } + /// Resolve `model` through the central route resolver and rebuild this /// client whenever its exact wire identity, limits, or protocol differs /// from the route bound at construction (#5042). `Ok(None)` means the @@ -1441,6 +1484,11 @@ impl DeepSeekClient { saved_provider_model: None, base_url_override: Some(self.base_url.clone()), limit_overrides: Vec::new(), + // Engine-internal rebinds keep this client's own wire: the + // transport is endpoint-scoped, and a fresh config-aware + // resolution would only reproduce it (both read the same + // named-table `wire`). + wire_override: self.pinned_wire_override(), }) .map_err(anyhow::Error::msg)?; let candidate_limits = crate::route_budget::known_route_limits(candidate.limits()); @@ -1461,6 +1509,10 @@ impl DeepSeekClient { rebound.route_limits = candidate_limits; return Ok(Some(rebound)); } + // A Custom client never reaches this rebuild/error tail: its pinned + // wire override makes `candidate.protocol() == self.wire_format` by + // construction, so only built-in model-aware kinds can demand a + // protocol the bound transport cannot speak. let config = config.ok_or_else(|| { anyhow::anyhow!( "{} model {:?} uses {:?}, but this client is bound to {:?} and no configuration is available to rebuild it", @@ -1493,6 +1545,9 @@ impl DeepSeekClient { saved_provider_model: None, base_url_override: Some(self.base_url.clone()), limit_overrides: Vec::new(), + // Same-client request routing: keep this transport's own wire + // (endpoint-scoped, mirrors `rebound_for_model_protocol`). + wire_override: self.pinned_wire_override(), }) { Ok(candidate) => candidate, Err(error) if model_aware => return Err(anyhow::Error::msg(error)), @@ -1763,7 +1818,7 @@ fn provider_wire_format_for_config( | ApiProvider::MinimaxAnthropic | ApiProvider::ModelstudioTokenPlanAnthropic | ApiProvider::ModelstudioCodingPlanAnthropic - ) || wire_config_prefers_anthropic(wire); + ) || wire_dialect_prefers_anthropic(wire); if prefers_anthropic && matches!( @@ -1789,10 +1844,10 @@ fn provider_wire_format_for_config( // anthropic: "anthropic" | "messages" | "claude" | "anthropic-messages" | ... // responses: "responses" | "responses-api" | "openai-responses" | "openai_responses" | ... if api_provider == ApiProvider::Custom { - if wire_config_prefers_anthropic(wire) { + if wire_dialect_prefers_anthropic(wire) { return WireFormat::AnthropicMessages; } - if wire_config_prefers_responses(wire) { + if wire_dialect_prefers_responses(wire) { return WireFormat::Responses; } } @@ -1813,40 +1868,6 @@ fn provider_wire_format_for_config( }) } -fn wire_config_prefers_anthropic(wire: Option<&str>) -> bool { - let Some(raw) = wire.map(str::trim).filter(|value| !value.is_empty()) else { - return false; - }; - let normalized = raw.to_ascii_lowercase().replace(['_', ' '], "-"); - matches!( - normalized.as_str(), - "anthropic" - | "anthropic-messages" - | "messages" - | "claude" - | "anthropic-compatible" - | "anthropic-compat" - ) -} - -fn wire_config_prefers_responses(wire: Option<&str>) -> bool { - let Some(raw) = wire.map(str::trim).filter(|value| !value.is_empty()) else { - return false; - }; - let normalized = raw.to_ascii_lowercase().replace(['_', ' '], "-"); - matches!( - normalized.as_str(), - "responses" - | "responses-api" - | "openai-responses" - | "openai-responses-api" - | "response" - | "response-api" - | "openai-responses-compat" - | "responses-compat" - ) -} - fn api_provider_skips_models_probe(api_provider: ApiProvider) -> bool { // Concentrate's `GET /v1/models` is explicitly unauthenticated // (docs/PROVIDERS.md): a 2xx proves nothing about the key, so guided @@ -2147,8 +2168,13 @@ impl DeepSeekClient { )) } WireFormat::Responses => { - let body = - responses::build_responses_body_for_provider(&request, self.api_provider); + let body = responses::build_responses_body_for_provider( + &request, + self.api_provider, + &self.reasoning_provider_tag(), + &self.reasoning_endpoint_fingerprint(), + &self.base_url, + ); let is_codex = self.api_provider == ApiProvider::OpenaiCodex; let url = if is_codex { format!("{}{}", self.base_url, responses::CODEX_RESPONSES_PATH) @@ -2224,6 +2250,7 @@ impl DeepSeekClient { saved_provider_model: None, base_url_override: Some(self.base_url.clone()), limit_overrides: Vec::new(), + wire_override: self.pinned_wire_override(), }) .ok() .and_then(|candidate| crate::route_budget::known_route_limits(candidate.limits())) @@ -6798,6 +6825,59 @@ mod tests { assert!(body.get("messages").is_none(), "Responses body: {body}"); } + /// Zen's Responses roster sends `include: ["reasoning.encrypted_content"]` + /// and `store: false`, so multi-turn tool continuations replay only if the + /// stream captured the encrypted reasoning items — the same discipline as + /// the Codex backend, tagged with the Zen provider slug. + #[tokio::test] + async fn opencode_zen_responses_stream_captures_encrypted_reasoning() { + let server = MockServer::start().await; + let sse_body = concat!( + "data: {\"type\":\"response.output_item.added\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_zen\"}}\n\n", + "data: {\"type\":\"response.output_item.done\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_zen\",\"summary\":[],\"encrypted_content\":\"enc_zen_state\"}}\n\n", + "data: [DONE]\n\n", + ); + Mock::given(method("POST")) + .and(path("/v1/responses")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("Content-Type", "text/event-stream") + .set_body_string(sse_body), + ) + .expect(1) + .mount(&server) + .await; + + let client = opencode_zen_client(&server, "gpt-5.5"); + assert_eq!(client.wire_format, WireFormat::Responses); + let mut stream = client + .create_message_stream(minimal_zen_request("gpt-5.5")) + .await + .expect("Zen Responses request should start"); + let mut captured = None; + while let Some(event) = stream.next().await { + if let StreamEvent::ContentBlockDelta { + delta: Delta::ReasoningStateDelta { state }, + .. + } = event.expect("Zen Responses stream event") + { + captured = Some(state); + } + } + + let state = captured.expect("encrypted reasoning state delta on the Zen route"); + assert_eq!(state.provider, ApiProvider::OpencodeZen.as_str()); + assert_eq!(state.api, "openai-responses"); + assert_eq!(state.id.as_deref(), Some("rs_zen")); + assert_eq!(state.encrypted_content, "enc_zen_state"); + assert_eq!(state.model, "gpt-5.5"); + assert_eq!( + state.endpoint, + Some(base_url_fingerprint(&client.base_url)), + "the captured state is bound to the Zen endpoint" + ); + } + #[tokio::test] async fn opencode_zen_messages_request_shape_uses_api_key_anthropic_route() { let server = MockServer::start().await; @@ -11826,4 +11906,232 @@ mod tests { ); assert_eq!(route.candidate.wire_model_id().as_str(), "custom-model-v1"); } + + /// The per-turn client for a `wire = "responses"` Custom table must POST + /// the generic `/responses` endpoint: the turn path is + /// resolve_runtime_route → from_candidate, so the candidate's protocol — + /// not the ambient spawn-time client — decides the wire (Pinvou PR #625). + #[tokio::test] + async fn forkguard_custom_responses_route_turn_client_posts_to_the_responses_endpoint() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/v1/responses")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("Content-Type", "text/event-stream") + .set_body_string(concentrate_sse_fixture("gpt-6-sol")), + ) + .expect(1) + .mount(&server) + .await; + + let _env_lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + &format!("{}/v1", server.uri()), + "custom-responses-key", + "gpt-6-sol", + ); + let route = crate::route_runtime::resolve_runtime_route( + &config, + ApiProvider::Custom, + Some("gpt-6-sol"), + ) + .expect("named table resolves"); + // The production turn path builds from the resolved route's + // identity-scoped config, not the ambient config — mirror it here so + // the pin would catch an ambient/identity divergence (in a two-table + // setup the ambient table would freeze the wrong identity). + let client = DeepSeekClient::from_candidate(&route.config, &route.candidate) + .expect("per-turn client builds"); + assert_eq!( + client.wire_format, + WireFormat::Responses, + "from_candidate binds the wire-true candidate protocol" + ); + assert_eq!(client.api_provider, ApiProvider::Custom); + + let mut stream = client + .create_message_stream(minimal_zen_request("gpt-6-sol")) + .await + .expect("Custom Responses request should start"); + let mut text = String::new(); + while let Some(event) = stream.next().await { + if let StreamEvent::ContentBlockDelta { + delta: Delta::TextDelta { text: piece }, + .. + } = event.expect("Custom stream event") + { + text.push_str(&piece); + } + } + assert_eq!(text, "ok from stub"); + + let requests = server.received_requests().await.expect("recorded request"); + assert_eq!(requests.len(), 1); + let request = &requests[0]; + assert_eq!( + request.url.path(), + "/v1/responses", + "the turn must hit the Responses endpoint, not /chat/completions" + ); + assert_eq!( + request + .headers + .get(AUTHORIZATION) + .and_then(|value| value.to_str().ok()), + Some("Bearer custom-responses-key") + ); + let body: Value = serde_json::from_slice(&request.body).expect("Responses JSON body"); + assert_eq!(body["model"], "gpt-6-sol", "model id verbatim: {body}"); + assert_eq!(body["stream"], true); + assert_eq!(body["store"], false, "stateless Custom route: {body}"); + assert_eq!( + body["include"], + json!(["reasoning.encrypted_content"]), + "encrypted reasoning replay stays available on this route: {body}" + ); + assert!( + body.get("messages").is_none(), + "Responses body, not Chat: {body}" + ); + } + + /// A wire-bound Custom client's engine-internal re-resolutions must pin + /// the transport's own wire: `rebound_for_model_protocol` and + /// `bind_request_to_protocol` feed `pinned_wire_override` back into the + /// resolver, so a model switch rebinds on the same Responses transport + /// instead of letting the fresh resolution fall back to Custom's static + /// Chat policy and downgrade the client mid-session. Deleting either + /// plumb fails exactly this pin (the rebind rebuilds as Chat, the + /// per-request route bails on the protocol mismatch). + #[test] + fn forkguard_wire_bound_client_rebinds_without_downgrading_the_wire() { + let _env_lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + "https://wire-rebind.example/v1", + "custom-responses-key", + "gpt-6-sol", + ); + let route = crate::route_runtime::resolve_runtime_route( + &config, + ApiProvider::Custom, + Some("gpt-6-sol"), + ) + .expect("named table resolves"); + let client = DeepSeekClient::from_candidate(&route.config, &route.candidate) + .expect("per-turn client builds"); + assert_eq!(client.wire_format, WireFormat::Responses); + + // Engine-internal rebind on a different model: same Responses + // transport, same endpoint, new wire model. + let rebound = client + .rebound_for_model_protocol(Some(&route.config), "gpt-6-mini") + .expect("alternate model resolves") + .expect("model change requires a rebound"); + assert_eq!( + rebound.wire_format, + WireFormat::Responses, + "the rebind must not downgrade the endpoint-scoped wire" + ); + assert_eq!(rebound.base_url, "https://wire-rebind.example/v1"); + assert_eq!(rebound.default_model, "gpt-6-mini"); + assert!( + client + .rebound_for_model_protocol(Some(&route.config), "gpt-6-sol") + .expect("same-model rebind resolves") + .is_none(), + "a matching binding must not rebuild the client" + ); + + // Per-request routing on the same client: the bound wire survives a + // model switch instead of bailing on a protocol mismatch. + let (request, _) = client + .bind_request_to_protocol(minimal_zen_request("gpt-6-mini")) + .expect("per-request routing keeps the bound wire"); + assert_eq!(request.model, "gpt-6-mini"); + } + + /// A `wire = "anthropic"` Custom table must reach a real + /// Messages-protocol transport: the per-turn client + /// (resolve_runtime_route → from_candidate) POSTs `{base}/v1/messages` + /// with the Anthropic credential and version headers. The Responses + /// wire has the full transport pin above; this closes the same gap for + /// the Messages wire. + #[tokio::test] + async fn forkguard_custom_anthropic_route_turn_client_posts_to_the_messages_endpoint() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/v1/messages")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "id": "msg_custom", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "ok from stub"}], + "model": "custom-claude", + "stop_reason": "end_turn", + "stop_sequence": null, + "usage": {"input_tokens": 1, "output_tokens": 1} + }))) + .expect(1) + .mount(&server) + .await; + + let _env_lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_messages", + Some("anthropic"), + &format!("{}/v1", server.uri()), + "custom-anthropic-key", + "custom-claude", + ); + let route = crate::route_runtime::resolve_runtime_route( + &config, + ApiProvider::Custom, + Some("custom-claude"), + ) + .expect("named table resolves"); + assert_eq!( + route.candidate.protocol(), + WireFormat::AnthropicMessages, + "the minted candidate speaks Messages" + ); + assert_eq!(route.candidate.endpoint().endpoint_key, "messages"); + let client = DeepSeekClient::from_candidate(&route.config, &route.candidate) + .expect("per-turn client builds"); + assert_eq!(client.wire_format, WireFormat::AnthropicMessages); + + client + .create_message(minimal_zen_request("custom-claude")) + .await + .expect("Custom Messages request should succeed"); + + let requests = server.received_requests().await.expect("recorded request"); + assert_eq!(requests.len(), 1); + assert_eq!( + requests[0].url.path(), + "/v1/messages", + "the turn must hit the Messages endpoint, not /chat/completions" + ); + assert_eq!( + requests[0] + .headers + .get("x-api-key") + .and_then(|value| value.to_str().ok()), + Some("custom-anthropic-key"), + "the Messages transport authenticates via x-api-key" + ); + assert_eq!( + requests[0] + .headers + .get("anthropic-version") + .and_then(|value| value.to_str().ok()), + Some("2023-06-01") + ); + let body: Value = serde_json::from_slice(&requests[0].body).expect("Messages JSON body"); + assert_eq!(body["model"], "custom-claude", "model id verbatim: {body}"); + } } diff --git a/crates/tui/src/client/responses.rs b/crates/tui/src/client/responses.rs index 0d4d07a46b..3672dc8ad7 100644 --- a/crates/tui/src/client/responses.rs +++ b/crates/tui/src/client/responses.rs @@ -8,6 +8,8 @@ //! (`client/chat.rs`) to avoid protocol hacks. use anyhow::{Context, Result}; +use codewhale_config::provider::WireFormat; +use codewhale_config::provider_base_url_is_official; use serde_json::{Value, json}; use crate::config::ApiProvider; @@ -33,7 +35,41 @@ pub(super) const CODEX_RESPONSES_PATH: &str = "/codex/responses"; /// Build the Responses API request body from a `MessageRequest`. #[cfg(test)] pub(super) fn build_responses_body(request: &MessageRequest) -> Value { - build_responses_body_for_provider(request, ApiProvider::OpenaiCodex) + build_responses_body_for_provider( + request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + "fp-codex-endpoint", + // The catalog Codex endpoint — the only URL fingerprint-less legacy + // states may replay onto (see the gate in + // `convert_messages_to_responses_input`). + "https://chatgpt.com/backend-api", + ) +} + +/// Whether a Responses route's request body carries +/// `include: ["reasoning.encrypted_content"]`: every Responses-wire provider +/// except DeepSeek (stateless, plain `reasoning_text`, no `include`) and +/// Concentrate (documented fields only). The capture gate in +/// `handle_responses_stream` derives from this same predicate, so include and +/// capture cover the same route set by construction. +fn responses_route_sends_encrypted_reasoning_include(provider: ApiProvider) -> bool { + !matches!( + provider, + ApiProvider::Deepseek | ApiProvider::DeepseekCN | ApiProvider::Concentrate + ) +} + +/// Whether a captured provider tag names endpoint-scoped Custom identity: the +/// legacy root table (`custom`) or a named table (`custom/`), exactly +/// the tags `DeepSeekClient::reasoning_provider_tag` mints for Custom. These +/// are the tags whose endpoint can change while the tag stays put, so they +/// are the fingerprint-less (pre-fingerprint) states that always fail closed. +/// Built-in tags take the narrower rule at the replay gate: a fingerprint-less +/// state may only ride the provider's official endpoint — without a +/// fingerprint it is the one origin this gate still vouches for. +fn is_custom_reasoning_tag(tag: &str) -> bool { + tag == "custom" || tag.starts_with("custom/") } /// Build a provider-aware Responses API request body. @@ -42,9 +78,30 @@ pub(super) fn build_responses_body(request: &MessageRequest) -> Value { /// and exposes plain reasoning text rather than OpenAI encrypted summaries. /// Keep those exact-route differences here instead of leaking them into the /// provider-neutral message model. +/// +/// `reasoning_provider_tag` is the tag the caller's capture side mints into +/// [`OpaqueReasoningState`] (see `DeepSeekClient::reasoning_provider_tag`); +/// the replay gate below only reattaches reasoning items whose state carries +/// that exact tag, so encrypted reasoning minted by one endpoint — a named +/// Custom table — is never replayed to another. `reasoning_endpoint_fingerprint` +/// is the same client's endpoint fingerprint (see +/// `DeepSeekClient::reasoning_endpoint_fingerprint`): the tag pins the table +/// name, not the URL behind it, so a state captured before the table's +/// `base_url` was edited stops replaying. States minted before fingerprints +/// existed carry no proof of origin: Custom endpoints can move under a +/// stable tag, so those fail closed, and built-in tags keep replaying only +/// while `current_base_url` is still the provider's official endpoint — a +/// client re-pointed by config has no way to prove where an old state was +/// captured, so it fails closed too. +/// +/// `current_base_url` is the endpoint this request is about to be POSTed to +/// (the client's frozen base URL); it decides that legacy arm. pub(super) fn build_responses_body_for_provider( request: &MessageRequest, provider: ApiProvider, + reasoning_provider_tag: &str, + reasoning_endpoint_fingerprint: &str, + current_base_url: &str, ) -> Value { let is_deepseek = matches!(provider, ApiProvider::Deepseek | ApiProvider::DeepseekCN); // Concentrate documents `model`, `input`, `stream`, `max_output_tokens`, @@ -91,7 +148,13 @@ pub(super) fn build_responses_body_for_provider( .unwrap_or_else(|| "You are a helpful assistant.".to_string()); // Convert messages to Responses input items. - let mut input = convert_messages_to_responses_input(request, provider); + let mut input = convert_messages_to_responses_input( + request, + provider, + reasoning_provider_tag, + reasoning_endpoint_fingerprint, + current_base_url, + ); if is_concentrate { input.insert( 0, @@ -136,9 +199,12 @@ pub(super) fn build_responses_body_for_provider( }; } - // OpenAI Codex can replay encrypted reasoning. DeepSeek exposes plain - // `reasoning_text` and does not support `include`. - if !is_deepseek && !is_concentrate { + // Every Responses route that receives this builder can replay encrypted + // reasoning. The include predicate is the same one the capture gate in + // `handle_responses_stream` derives from, so include and capture cover + // the same route set by construction — a future Responses-wire provider + // cannot silently start sending `include` without capturing. + if responses_route_sends_encrypted_reasoning_include(provider) { body["include"] = json!(["reasoning.encrypted_content"]); } @@ -162,8 +228,22 @@ impl DeepSeekClient { // remapping — rather than borrowing the request that no longer exists // at this layer. let wire_model = prepared.wire_model.clone(); - let reasoning_origin = (self.api_provider == ApiProvider::OpenaiCodex) - .then(|| (self.api_provider.as_str().to_string(), wire_model.clone())); + // Encrypted-reasoning capture applies to every Responses route whose + // request carries `include: ["reasoning.encrypted_content"]` — the + // same predicate the body builder uses to send the include — and + // replays by the endpoint-scoped provider tag from + // `reasoning_provider_tag` plus the endpoint fingerprint from + // `reasoning_endpoint_fingerprint`. The transport-wire half keeps + // Chat-wire Custom tables (and any other dialect) excluded. + let reasoning_origin = (self.wire_format == WireFormat::Responses + && responses_route_sends_encrypted_reasoning_include(self.api_provider)) + .then(|| { + ( + self.reasoning_provider_tag(), + self.reasoning_endpoint_fingerprint(), + wire_model.clone(), + ) + }); // The bearer Authorization header is already installed as a default // header on both the dual and the HTTP/1.1 twin client (resolved from @@ -485,7 +565,7 @@ impl DeepSeekClient { } "response.output_item.done" => { if let Some(idx) = current_block_index { - if let (Some((provider, model)), Some(item)) = + if let (Some((provider, endpoint, model)), Some(item)) = (reasoning_origin.as_ref(), event.get("item")) && item.get("type").and_then(Value::as_str) == Some("reasoning") @@ -506,6 +586,7 @@ impl DeepSeekClient { .and_then(Value::as_str) .map(str::to_string), encrypted_content: encrypted_content.to_string(), + endpoint: Some(endpoint.clone()), }, }, }); @@ -711,11 +792,33 @@ pub(super) fn responses_tool_output(content: &str, content_blocks: Option<&[Valu } /// Convert Codewhale messages to Responses API input items. +/// +/// `reasoning_provider_tag` scopes opaque-reasoning replay to the endpoint +/// that minted the state; see [`build_responses_body_for_provider`]. pub(super) fn convert_messages_to_responses_input( request: &MessageRequest, provider: ApiProvider, + reasoning_provider_tag: &str, + reasoning_endpoint_fingerprint: &str, + current_base_url: &str, ) -> Vec { let is_deepseek = matches!(provider, ApiProvider::Deepseek | ApiProvider::DeepseekCN); + // Fingerprint-less (pre-fingerprint) states carry no proof of which + // endpoint minted them, so they may only replay where the endpoint cannot + // have moved: a built-in on its official endpoint family. Custom is + // excluded by its tag at the gate below (and + // `provider_base_url_is_official` rejects it outright), and a built-in + // re-pointed by config or env loses the credit — it has no way to prove + // an old state came from the new URL. + let fingerprintless_replay_official = provider + .kind() + .is_some_and(|kind| provider_base_url_is_official(kind, current_base_url)); + // Replay rides the same predicate that sends the `include`: a provider + // whose body omits `include: ["reasoning.encrypted_content"]` (DeepSeek's + // plain `reasoning_text`, Concentrate's documented-fields contract) must + // never receive an encrypted replay item either, so the replay gate + // cannot drift from the include gate. + let replays_encrypted_reasoning = responses_route_sends_encrypted_reasoning_include(provider); let mut items = Vec::new(); for msg in &request.messages { @@ -808,9 +911,33 @@ pub(super) fn convert_messages_to_responses_input( thinking, state, .. } => { if let Some(state) = state { - if state.provider == provider.as_str() + // Endpoint-scoped replay: the tag must match + // the endpoint this request is bound to + // (provider slug, plus the table identity for + // Custom), alongside api shape, exact model, + // and the endpoint fingerprint — a table whose + // base_url was edited stops replaying the + // previous endpoint's blobs. Legacy states + // minted before fingerprints existed carry no + // proof of which endpoint produced them: + // Custom endpoints can move under a stable + // tag, so those fail closed, and a built-in + // keeps replaying only while the client still + // points at the provider's official endpoint + // (`fingerprintless_replay_official` above) — + // a re-pointed client gets no such credit. + let endpoint_matches = match &state.endpoint { + None => { + !is_custom_reasoning_tag(&state.provider) + && fingerprintless_replay_official + } + Some(captured) => captured == reasoning_endpoint_fingerprint, + }; + if replays_encrypted_reasoning + && state.provider == reasoning_provider_tag && state.api == "openai-responses" && state.model == request.model + && endpoint_matches { let mut item = json!({ "type": "reasoning", diff --git a/crates/tui/src/client/responses/tests.rs b/crates/tui/src/client/responses/tests.rs index 652619abcb..548851cc89 100644 --- a/crates/tui/src/client/responses/tests.rs +++ b/crates/tui/src/client/responses/tests.rs @@ -12,6 +12,15 @@ use crate::models::SystemPrompt; use wiremock::matchers::{method, path}; use wiremock::{Mock, MockServer, Request, Respond, ResponseTemplate}; +/// The catalog Codex endpoint: `provider_base_url_is_official` matches the +/// exact URL, so this literal pins the official side of the fingerprint-less +/// replay arm (the config crate keeps the constant crate-private). +const OFFICIAL_CODEX_BASE_URL: &str = "https://chatgpt.com/backend-api"; + +/// Placeholder endpoint for body-builder tests that do not exercise the +/// fingerprint-less replay arm — its officialness only matters there. +const BODY_TEST_BASE_URL: &str = "https://relay.example.test/v1"; + #[derive(Clone)] struct RetryThenSuccess { attempts: Arc, @@ -568,7 +577,13 @@ fn concentrate_responses_body_sends_only_documented_fields() { cache_control: None, }]); - let body = build_responses_body_for_provider(&request, ApiProvider::Concentrate); + let body = build_responses_body_for_provider( + &request, + ApiProvider::Concentrate, + ApiProvider::Concentrate.as_str(), + "fp-concentrate-endpoint", + BODY_TEST_BASE_URL, + ); let documented = [ "model", "input", @@ -617,7 +632,13 @@ fn concentrate_responses_body_sends_only_documented_fields() { // The same request on the generic Responses path still carries the // OpenAI-only fields, so the Concentrate branch is a deliberate subset. - let generic = build_responses_body_for_provider(&request, ApiProvider::Openai); + let generic = build_responses_body_for_provider( + &request, + ApiProvider::Openai, + ApiProvider::Openai.as_str(), + "fp-generic-endpoint", + BODY_TEST_BASE_URL, + ); assert!( generic.get("store").is_some() && generic.get("include").is_some() @@ -644,7 +665,13 @@ fn deepseek_flash_responses_body_uses_stateless_0731_contract() { }, ); - let body = build_responses_body_for_provider(&request, ApiProvider::Deepseek); + let body = build_responses_body_for_provider( + &request, + ApiProvider::Deepseek, + ApiProvider::Deepseek.as_str(), + "fp-deepseek-endpoint", + BODY_TEST_BASE_URL, + ); assert_eq!(body["model"], "deepseek-v4-flash"); assert_eq!(body["max_output_tokens"], 128); @@ -677,7 +704,13 @@ fn codex_responses_body_omits_the_output_cap_the_backend_rejects() { let mut request = minimal_responses_request(); request.max_tokens = 4_096; - let codex = build_responses_body_for_provider(&request, ApiProvider::OpenaiCodex); + let codex = build_responses_body_for_provider( + &request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + "fp-codex-endpoint", + OFFICIAL_CODEX_BASE_URL, + ); assert!( codex.get("max_output_tokens").is_none(), "Codex Responses body names a parameter its backend rejects: {codex}" @@ -687,19 +720,27 @@ fn codex_responses_body_omits_the_output_cap_the_backend_rejects() { "no alternate output-cap spelling may sneak onto the Codex wire: {codex}" ); - let deepseek = build_responses_body_for_provider(&request, ApiProvider::Deepseek); + let deepseek = build_responses_body_for_provider( + &request, + ApiProvider::Deepseek, + ApiProvider::Deepseek.as_str(), + "fp-deepseek-endpoint", + BODY_TEST_BASE_URL, + ); assert_eq!(deepseek["max_output_tokens"], json!(4_096)); } #[test] fn codex_replays_only_exact_model_opaque_reasoning_state() { const SENTINEL: &str = "readable private reasoning must not be replayed"; + const ENDPOINT_FP: &str = "fp-codex-endpoint"; let state = OpaqueReasoningState { provider: ApiProvider::OpenaiCodex.as_str().to_string(), api: "openai-responses".to_string(), model: "gpt-5.5".to_string(), id: Some("rs_opaque".to_string()), encrypted_content: "enc_opaque_payload".to_string(), + endpoint: Some(ENDPOINT_FP.to_string()), }; let mut request = minimal_responses_request(); request.messages.insert( @@ -714,7 +755,13 @@ fn codex_replays_only_exact_model_opaque_reasoning_state() { }, ); - let exact = build_responses_body_for_provider(&request, ApiProvider::OpenaiCodex); + let exact = build_responses_body_for_provider( + &request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + ENDPOINT_FP, + OFFICIAL_CODEX_BASE_URL, + ); let exact_wire = exact.to_string(); assert!(!exact_wire.contains(SENTINEL), "{exact}"); assert_eq!(exact.pointer("/input/0/type"), Some(&json!("reasoning"))); @@ -726,7 +773,13 @@ fn codex_replays_only_exact_model_opaque_reasoning_state() { ); request.model = "gpt-5.6".to_string(); - let switched_model = build_responses_body_for_provider(&request, ApiProvider::OpenaiCodex); + let switched_model = build_responses_body_for_provider( + &request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + ENDPOINT_FP, + OFFICIAL_CODEX_BASE_URL, + ); assert!(!switched_model.to_string().contains(SENTINEL)); assert!( switched_model @@ -736,7 +789,13 @@ fn codex_replays_only_exact_model_opaque_reasoning_state() { "{switched_model}" ); - let switched_provider = build_responses_body_for_provider(&request, ApiProvider::Deepseek); + let switched_provider = build_responses_body_for_provider( + &request, + ApiProvider::Deepseek, + ApiProvider::Deepseek.as_str(), + ENDPOINT_FP, + BODY_TEST_BASE_URL, + ); let switched_wire = switched_provider.to_string(); assert!(!switched_wire.contains(SENTINEL), "{switched_provider}"); assert!( @@ -796,6 +855,13 @@ async fn codex_stream_captures_encrypted_reasoning_as_opaque_state() { assert_eq!(state.model, "gpt-5.5"); assert_eq!(state.id.as_deref(), Some("rs_1")); assert_eq!(state.encrypted_content, "enc_state"); + assert_eq!( + state.endpoint, + Some(codewhale_config::catalog::base_url_fingerprint( + &client.base_url + )), + "the captured state is bound to the capturing endpoint" + ); } #[test] @@ -1090,7 +1156,13 @@ fn responses_input_includes_user_role_tool_results() { top_p: None, }; - let input = convert_messages_to_responses_input(&request, ApiProvider::OpenaiCodex); + let input = convert_messages_to_responses_input( + &request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + "fp-codex-endpoint", + OFFICIAL_CODEX_BASE_URL, + ); assert_eq!(input[0]["type"], "function_call"); assert_eq!(input[0]["call_id"], "call_abc"); @@ -1126,7 +1198,13 @@ fn responses_input_encodes_tool_call_names() { top_p: None, }; - let input = convert_messages_to_responses_input(&request, ApiProvider::OpenaiCodex); + let input = convert_messages_to_responses_input( + &request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + "fp-codex-endpoint", + OFFICIAL_CODEX_BASE_URL, + ); assert_eq!(input[0]["type"], "function_call"); assert_eq!(input[0]["name"], to_api_tool_name("web.run")); @@ -1253,7 +1331,13 @@ fn user_image_becomes_an_input_image_item() { }, }); - let items = convert_messages_to_responses_input(&request, ApiProvider::OpenaiCodex); + let items = convert_messages_to_responses_input( + &request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + "fp-codex-endpoint", + OFFICIAL_CODEX_BASE_URL, + ); let user = items .iter() @@ -1305,7 +1389,13 @@ fn tool_result_image_becomes_native_function_output_content() { }, ]; - let items = convert_messages_to_responses_input(&request, ApiProvider::OpenaiCodex); + let items = convert_messages_to_responses_input( + &request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + "fp-codex-endpoint", + OFFICIAL_CODEX_BASE_URL, + ); let output = items .iter() .find(|item| item["type"] == "function_call_output") @@ -1343,7 +1433,13 @@ fn responses_input_keeps_system_role_history_messages() { }, ); - let items = convert_messages_to_responses_input(&request, ApiProvider::OpenaiCodex); + let items = convert_messages_to_responses_input( + &request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + "fp-codex-endpoint", + OFFICIAL_CODEX_BASE_URL, + ); let system = items .iter() @@ -1358,3 +1454,792 @@ fn responses_input_keeps_system_role_history_messages() { }) ); } + +/// A `wire = "responses"` Custom table captures encrypted reasoning exactly +/// like the Codex backend: the stream yields an opaque reasoning-state delta +/// tagged with the client's provider tag, so the replay gate +/// (`state.provider == reasoning_provider_tag()`) matches on the next turn +/// (Pinvou PR #625). +#[tokio::test] +async fn forkguard_custom_responses_stream_captures_encrypted_reasoning_as_opaque_state() { + let server = MockServer::start().await; + let sse_body = concat!( + "data: {\"type\":\"response.output_item.added\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_custom\"}}\n\n", + "data: {\"type\":\"response.output_item.done\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_custom\",\"summary\":[],\"encrypted_content\":\"enc_custom_state\"}}\n\n", + "data: [DONE]\n\n", + ); + Mock::given(method("POST")) + .and(path("/v1/responses")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("Content-Type", "text/event-stream") + .set_body_string(sse_body), + ) + .mount(&server) + .await; + + let client = { + let _env_lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + &format!("{}/v1", server.uri()), + "custom-responses-key", + "gpt-6-sol", + ); + DeepSeekClient::new(&config).expect("Custom responses client should resolve") + // `DeepSeekClient::new` reads the table's `wire` dialect + // (`provider_wire_format_for_config`), so this ambient client speaks + // Responses; after the runtime-route fix the per-turn + // `from_candidate` client carries the same wire. + }; + assert_eq!(client.wire_format, WireFormat::Responses); + let mut stream = client + .handle_responses_stream( + &client + .prepare_outbound_request(minimal_responses_request(), true) + .expect("responses request prepares"), + ) + .await + .unwrap(); + let mut captured = None; + while let Some(event) = stream.next().await { + if let StreamEvent::ContentBlockDelta { + delta: Delta::ReasoningStateDelta { state }, + .. + } = event.unwrap() + { + captured = Some(state); + } + } + + let state = captured.expect("encrypted reasoning state delta on the Custom route"); + assert_eq!( + state.provider, "custom/pinvou_responses", + "the tag must carry the minting table, not the shared `custom` slug" + ); + assert_eq!(state.api, "openai-responses"); + assert_eq!(state.id.as_deref(), Some("rs_custom")); + assert_eq!(state.encrypted_content, "enc_custom_state"); + assert_eq!( + state.model, "gpt-5.5", + "the captured wire model is the replay gate's other key" + ); + assert_eq!( + state.endpoint, + Some(codewhale_config::catalog::base_url_fingerprint( + &client.base_url + )), + "the captured state is bound to the minting table's endpoint" + ); +} + +/// A Chat-wire Custom table must not capture encrypted reasoning even if a +/// Responses-shaped stream reaches `handle_responses_stream`: the capture +/// gate keys on the transport's wire, not just on the stream's event shape. +#[tokio::test] +async fn forkguard_custom_chat_stream_does_not_capture_encrypted_reasoning() { + let server = MockServer::start().await; + let sse_body = concat!( + "data: {\"type\":\"response.output_item.added\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_chat\"}}\n\n", + "data: {\"type\":\"response.output_item.done\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_chat\",\"summary\":[],\"encrypted_content\":\"enc_chat_state\"}}\n\n", + "data: [DONE]\n\n", + ); + Mock::given(method("POST")) + .and(path("/v1/chat/completions")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("Content-Type", "text/event-stream") + .set_body_string(sse_body), + ) + .mount(&server) + .await; + + let client = { + let _env_lock = crate::test_support::lock_test_env(); + // No `wire`: the legacy table default is the Chat transport. + let config = crate::test_support::custom_named_table_config( + "pinvou_chat_table", + None, + &format!("{}/v1", server.uri()), + "custom-chat-key", + "vendor-model", + ); + DeepSeekClient::new(&config).expect("Custom chat client should resolve") + }; + assert_eq!(client.wire_format, WireFormat::ChatCompletions); + let mut stream = client + .handle_responses_stream( + &client + .prepare_outbound_request(minimal_responses_request(), true) + .expect("request prepares"), + ) + .await + .unwrap(); + let mut captured = None; + let mut blocks_closed = 0; + while let Some(event) = stream.next().await { + match event.unwrap() { + StreamEvent::ContentBlockDelta { + delta: Delta::ReasoningStateDelta { state }, + .. + } => captured = Some(state), + StreamEvent::ContentBlockStop { .. } => blocks_closed += 1, + _ => {} + } + } + assert!( + captured.is_none(), + "a Chat-wire Custom table must not mint opaque reasoning state: {captured:?}" + ); + // The fixture carries exactly one reasoning item whose done event closes + // its block. Asserting the close proves the stream actually flowed + // through the parser, so the empty capture above is the transport-wire + // gate's doing — not a dead stream passing the test vacuously. + assert_eq!(blocks_closed, 1, "the reasoning block must still close"); +} + +/// A reasoning item without (or with an empty) `encrypted_content` must not +/// be captured — replaying an empty blob would poison every later turn — +/// but the stream itself keeps flowing and closes the block. +#[tokio::test] +async fn forkguard_custom_responses_capture_tolerates_missing_or_empty_encrypted_content() { + let server = MockServer::start().await; + let sse_body = concat!( + "data: {\"type\":\"response.output_item.added\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_missing\"}}\n\n", + "data: {\"type\":\"response.output_item.done\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_missing\",\"summary\":[]}}\n\n", + "data: {\"type\":\"response.output_item.added\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_empty\"}}\n\n", + "data: {\"type\":\"response.output_item.done\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_empty\",\"summary\":[],\"encrypted_content\":\"\"}}\n\n", + "data: [DONE]\n\n", + ); + Mock::given(method("POST")) + .and(path("/v1/responses")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("Content-Type", "text/event-stream") + .set_body_string(sse_body), + ) + .mount(&server) + .await; + + let client = { + let _env_lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + &format!("{}/v1", server.uri()), + "custom-responses-key", + "gpt-6-sol", + ); + DeepSeekClient::new(&config).expect("Custom responses client should resolve") + }; + let mut stream = client + .handle_responses_stream( + &client + .prepare_outbound_request(minimal_responses_request(), true) + .expect("responses request prepares"), + ) + .await + .unwrap(); + let mut captured = None; + let mut blocks_closed = 0; + while let Some(event) = stream.next().await { + match event.unwrap() { + StreamEvent::ContentBlockDelta { + delta: Delta::ReasoningStateDelta { state }, + .. + } => captured = Some(state), + StreamEvent::ContentBlockStop { .. } => blocks_closed += 1, + _ => {} + } + } + assert!( + captured.is_none(), + "missing or empty encrypted_content must not be captured: {captured:?}" + ); + assert_eq!(blocks_closed, 2, "both reasoning blocks still close"); +} + +/// The replay gate matches Custom-tagged reasoning state by endpoint-scoped +/// provider tag, endpoint fingerprint, and exact model: an exact +/// table+endpoint+model match replays the encrypted item, while a model +/// switch, a different table, a different provider, or an edited `base_url` +/// must not — table A's encrypted reasoning never rides to table B or to +/// whatever endpoint table A later points at. A pre-fingerprint state (no +/// endpoint field) carries no proof of origin, so on a Custom tag it fails +/// closed too: it must not ride a table's wire even at the table's current +/// URL, because the URL may not be the one that minted it. (Built-in tags +/// take the narrower official-endpoint rule, pinned by the two +/// legacy-state tests further down.) +#[test] +fn forkguard_custom_responses_replays_only_exact_model_opaque_reasoning_state() { + const SENTINEL: &str = "readable private reasoning must not be replayed"; + const MINTING_TABLE: &str = "custom/pinvou_responses"; + const MINTING_ENDPOINT_FP: &str = "fp-pinvou-responses-endpoint"; + let state = OpaqueReasoningState { + provider: MINTING_TABLE.to_string(), + api: "openai-responses".to_string(), + model: "gpt-6-sol".to_string(), + id: Some("rs_custom".to_string()), + encrypted_content: "enc_custom_payload".to_string(), + endpoint: Some(MINTING_ENDPOINT_FP.to_string()), + }; + let mut request = minimal_responses_request(); + request.model = "gpt-6-sol".to_string(); + request.messages.insert( + 0, + Message { + role: Role::Assistant, + content: vec![ContentBlock::Thinking { + thinking: SENTINEL.to_string(), + signature: None, + state: Some(state), + }], + }, + ); + + let exact = build_responses_body_for_provider( + &request, + ApiProvider::Custom, + MINTING_TABLE, + MINTING_ENDPOINT_FP, + BODY_TEST_BASE_URL, + ); + let exact_wire = exact.to_string(); + assert!(!exact_wire.contains(SENTINEL), "{exact}"); + assert_eq!(exact.pointer("/input/0/type"), Some(&json!("reasoning"))); + assert_eq!(exact.pointer("/input/0/id"), Some(&json!("rs_custom"))); + assert_eq!(exact.pointer("/input/0/summary"), Some(&json!([]))); + assert_eq!( + exact.pointer("/input/0/encrypted_content"), + Some(&json!("enc_custom_payload")) + ); + + // Same table and model, but the table's base_url was edited: the tag + // alone pins the table NAME, so the endpoint fingerprint must stop the + // old endpoint's opaque blobs from riding to the new one. + let switched_endpoint = build_responses_body_for_provider( + &request, + ApiProvider::Custom, + MINTING_TABLE, + "fp-new-endpoint-after-base-url-edit", + BODY_TEST_BASE_URL, + ); + assert!( + !switched_endpoint.to_string().contains("enc_custom_payload"), + "state must not replay onto a re-pointed endpoint: {switched_endpoint}" + ); + assert!( + switched_endpoint + .get("input") + .and_then(Value::as_array) + .is_some_and(|items| items.iter().all(|item| item["type"] != "reasoning")), + "{switched_endpoint}" + ); + + // States minted before endpoint fingerprints existed carry none and fail + // closed on Custom: with no proof of which endpoint minted the blob, a + // stable table tag proves nothing about the URL behind it, so the blob + // must not ride the wire even at the table's current URL. The state + // stays in history and is dropped on every turn until compaction gives + // the session a clean slate; fresh captures carry fingerprints and + // replay resumes from the next captured turn. + request.messages[0].content = vec![ContentBlock::Thinking { + thinking: SENTINEL.to_string(), + signature: None, + state: Some(OpaqueReasoningState { + provider: MINTING_TABLE.to_string(), + api: "openai-responses".to_string(), + model: "gpt-6-sol".to_string(), + id: Some("rs_legacy".to_string()), + encrypted_content: "enc_legacy_payload".to_string(), + endpoint: None, + }), + }]; + let legacy_state = build_responses_body_for_provider( + &request, + ApiProvider::Custom, + MINTING_TABLE, + MINTING_ENDPOINT_FP, + BODY_TEST_BASE_URL, + ); + let legacy_wire = legacy_state.to_string(); + assert!(!legacy_wire.contains(SENTINEL), "{legacy_state}"); + assert!( + !legacy_wire.contains("enc_legacy_payload"), + "fingerprint-less Custom state must fail closed, not replay: {legacy_state}" + ); + assert!( + legacy_state + .get("input") + .and_then(Value::as_array) + .is_some_and(|items| items.iter().all(|item| item["type"] != "reasoning")), + "{legacy_state}" + ); + + // The legacy-root custom table (identity "custom", tag "custom") is + // endpoint-scoped Custom identity too and fails closed the same way. + request.messages[0].content = vec![ContentBlock::Thinking { + thinking: SENTINEL.to_string(), + signature: None, + state: Some(OpaqueReasoningState { + provider: "custom".to_string(), + api: "openai-responses".to_string(), + model: "gpt-6-sol".to_string(), + id: Some("rs_legacy_root".to_string()), + encrypted_content: "enc_legacy_root_payload".to_string(), + endpoint: None, + }), + }]; + let legacy_root = build_responses_body_for_provider( + &request, + ApiProvider::Custom, + "custom", + MINTING_ENDPOINT_FP, + BODY_TEST_BASE_URL, + ); + assert!( + !legacy_root.to_string().contains("enc_legacy_root_payload"), + "fingerprint-less legacy-root Custom state must fail closed: {legacy_root}" + ); + + // Same model on a DIFFERENT named table: the shared `custom` slug must + // not match, so table A's opaque state never rides table B's wire. + request.messages[0].content = vec![ContentBlock::Thinking { + thinking: SENTINEL.to_string(), + signature: None, + state: Some(OpaqueReasoningState { + provider: MINTING_TABLE.to_string(), + api: "openai-responses".to_string(), + model: "gpt-6-sol".to_string(), + id: Some("rs_custom".to_string()), + encrypted_content: "enc_custom_payload".to_string(), + endpoint: Some(MINTING_ENDPOINT_FP.to_string()), + }), + }]; + let switched_table = build_responses_body_for_provider( + &request, + ApiProvider::Custom, + "custom/other_relay", + MINTING_ENDPOINT_FP, + BODY_TEST_BASE_URL, + ); + let switched_table_wire = switched_table.to_string(); + assert!(!switched_table_wire.contains(SENTINEL)); + assert!( + !switched_table_wire.contains("enc_custom_payload") + && !switched_table_wire.contains("enc_legacy_payload"), + "cross-table replay must drop the foreign encrypted item: {switched_table}" + ); + assert!( + switched_table + .get("input") + .and_then(Value::as_array) + .is_some_and(|items| items.iter().all(|item| item["type"] != "reasoning")), + "{switched_table}" + ); + + // State minted by a DIFFERENT provider (Codex) must not replay on a + // Custom route even with the same model and api shape: the tag is exact. + request.messages[0].content = vec![ContentBlock::Thinking { + thinking: SENTINEL.to_string(), + signature: None, + state: Some(OpaqueReasoningState { + provider: ApiProvider::OpenaiCodex.as_str().to_string(), + api: "openai-responses".to_string(), + model: "gpt-6-sol".to_string(), + id: Some("rs_codex".to_string()), + encrypted_content: "enc_codex_payload".to_string(), + endpoint: Some(MINTING_ENDPOINT_FP.to_string()), + }), + }]; + let codex_state_on_custom = build_responses_body_for_provider( + &request, + ApiProvider::Custom, + MINTING_TABLE, + MINTING_ENDPOINT_FP, + BODY_TEST_BASE_URL, + ); + let codex_state_wire = codex_state_on_custom.to_string(); + assert!(!codex_state_wire.contains(SENTINEL)); + assert!( + !codex_state_wire.contains("enc_codex_payload"), + "foreign-provider state must not replay onto a Custom route: {codex_state_on_custom}" + ); + + request.model = "gpt-6-luna".to_string(); + let switched_model = build_responses_body_for_provider( + &request, + ApiProvider::Custom, + MINTING_TABLE, + MINTING_ENDPOINT_FP, + BODY_TEST_BASE_URL, + ); + assert!(!switched_model.to_string().contains(SENTINEL)); + assert!( + switched_model + .get("input") + .and_then(Value::as_array) + .is_some_and(|items| items.iter().all(|item| item["type"] != "reasoning")), + "{switched_model}" + ); +} + +/// The legacy migration is fail-closed wherever the endpoint can move. A +/// fingerprint-less state from a built-in provider keeps replaying only on +/// that provider's official endpoint — the one origin the gate vouches for +/// without a fingerprint — so pre-upgrade sessions on the official route +/// keep their reasoning continuity. +#[test] +fn forkguard_fixed_endpoint_legacy_state_without_fingerprint_keeps_replaying() { + const SENTINEL: &str = "readable private reasoning must not be replayed"; + const ENDPOINT_FP: &str = "fp-codex-endpoint"; + let mut request = minimal_responses_request(); + request.messages.insert( + 0, + Message { + role: Role::Assistant, + content: vec![ContentBlock::Thinking { + thinking: SENTINEL.to_string(), + signature: None, + state: Some(OpaqueReasoningState { + provider: ApiProvider::OpenaiCodex.as_str().to_string(), + api: "openai-responses".to_string(), + model: request.model.clone(), + id: Some("rs_legacy_codex".to_string()), + encrypted_content: "enc_legacy_codex_payload".to_string(), + endpoint: None, + }), + }], + }, + ); + let body = build_responses_body_for_provider( + &request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + ENDPOINT_FP, + OFFICIAL_CODEX_BASE_URL, + ); + assert_eq!( + body.pointer("/input/0/type"), + Some(&json!("reasoning")), + "fingerprint-less fixed-endpoint state must keep replaying: {body}" + ); + assert_eq!( + body.pointer("/input/0/encrypted_content"), + Some(&json!("enc_legacy_codex_payload")) + ); +} + +/// The carve-out above must not extend to a re-pointed built-in: config and +/// env can move a built-in provider's base URL while its tag stays put +/// (`OPENAI_CODEX_BASE_URL`, a root/base_url override, a CLI flag), so a +/// fingerprint-less state carries no proof it came from the new endpoint. +/// The same tag that keeps replaying on the official route must fail closed +/// the moment the client points elsewhere — same rule, same direction as the +/// Custom fail-closed arm above. +#[test] +fn forkguard_repointed_builtin_legacy_state_without_fingerprint_fails_closed() { + const SENTINEL: &str = "readable private reasoning must not be replayed"; + let mut request = minimal_responses_request(); + request.messages.insert( + 0, + Message { + role: Role::Assistant, + content: vec![ContentBlock::Thinking { + thinking: SENTINEL.to_string(), + signature: None, + state: Some(OpaqueReasoningState { + provider: ApiProvider::OpenaiCodex.as_str().to_string(), + api: "openai-responses".to_string(), + model: request.model.clone(), + id: Some("rs_legacy_codex".to_string()), + encrypted_content: "enc_legacy_codex_payload".to_string(), + endpoint: None, + }), + }], + }, + ); + let body = build_responses_body_for_provider( + &request, + ApiProvider::OpenaiCodex, + ApiProvider::OpenaiCodex.as_str(), + "fp-codex-endpoint", + "https://proxy.example.test/backend-api", + ); + assert!( + !body.to_string().contains("enc_legacy_codex_payload"), + "a fingerprint-less built-in state must not replay onto a re-pointed endpoint: {body}" + ); + assert!( + body.get("input") + .and_then(Value::as_array) + .is_some_and(|items| items.iter().all(|item| item["type"] != "reasoning")), + "{body}" + ); +} + +/// Replay rides the same predicate that sends the `include`: Concentrate's +/// body deliberately omits `include: ["reasoning.encrypted_content"]` +/// (documented fields only), so even a Concentrate-tagged encrypted state +/// that matches tag, api, model, and endpoint must not be attached — the +/// replay gate cannot drift from the include gate. Deleting the predicate +/// half of the gate fails exactly this pin. +#[test] +fn forkguard_replay_never_rides_a_route_that_omits_the_include() { + const SENTINEL: &str = "readable private reasoning must not be replayed"; + let state = OpaqueReasoningState { + provider: ApiProvider::Concentrate.as_str().to_string(), + api: "openai-responses".to_string(), + model: "gpt-5.5".to_string(), + id: Some("rs_concentrate".to_string()), + encrypted_content: "enc_concentrate_payload".to_string(), + endpoint: Some("fp-concentrate-endpoint".to_string()), + }; + let mut request = minimal_responses_request(); + request.messages.insert( + 0, + Message { + role: Role::Assistant, + content: vec![ContentBlock::Thinking { + thinking: SENTINEL.to_string(), + signature: None, + state: Some(state), + }], + }, + ); + let body = build_responses_body_for_provider( + &request, + ApiProvider::Concentrate, + ApiProvider::Concentrate.as_str(), + "fp-concentrate-endpoint", + BODY_TEST_BASE_URL, + ); + assert!( + !body.to_string().contains("enc_concentrate_payload"), + "a route that omits the include must not receive an encrypted replay item: {body}" + ); + assert!( + body.get("input") + .and_then(Value::as_array) + .is_some_and(|items| items.iter().all(|item| item["type"] != "reasoning")), + "{body}" + ); +} + +/// Capture and replay must agree through the real per-turn client: the state +/// captured off turn 1's stream, placed back into history the way the turn +/// loop commits it (a Thinking block ahead of any tool call), reaches turn +/// 2's prepared request body as the paired reasoning item. The pure replay +/// test above pins the gate in isolation; this pins the seam where the +/// capture side's tag and endpoint fingerprint must equal the replay side's. +#[tokio::test] +async fn forkguard_custom_responses_captured_state_replays_on_the_next_turn() { + let server = MockServer::start().await; + let sse_body = concat!( + "data: {\"type\":\"response.output_item.added\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_custom\"}}\n\n", + "data: {\"type\":\"response.output_item.done\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_custom\",\"summary\":[],\"encrypted_content\":\"enc_custom_state\"}}\n\n", + "data: [DONE]\n\n", + ); + Mock::given(method("POST")) + .and(path("/v1/responses")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("Content-Type", "text/event-stream") + .set_body_string(sse_body), + ) + .expect(1) + .mount(&server) + .await; + + let _env_lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + &format!("{}/v1", server.uri()), + "custom-responses-key", + "gpt-6-sol", + ); + let route = crate::route_runtime::resolve_runtime_route( + &config, + ApiProvider::Custom, + Some("gpt-6-sol"), + ) + .expect("named table resolves"); + let client = DeepSeekClient::from_candidate(&route.config, &route.candidate) + .expect("per-turn client builds"); + + let mut first = minimal_responses_request(); + first.model = "gpt-6-sol".to_string(); + let mut stream = client + .handle_responses_stream( + &client + .prepare_outbound_request(first, true) + .expect("first request prepares"), + ) + .await + .unwrap(); + let mut captured = None; + while let Some(event) = stream.next().await { + if let StreamEvent::ContentBlockDelta { + delta: Delta::ReasoningStateDelta { state }, + .. + } = event.expect("first-turn stream event") + { + captured = Some(state); + } + } + let state = captured.expect("state captured on turn 1"); + assert_eq!(state.provider, "custom/pinvou_responses"); + assert_eq!(state.encrypted_content, "enc_custom_state"); + + // Turn loop placement: the Thinking block (state included) precedes any + // tool call in the committed history. + let mut follow_up = minimal_responses_request(); + follow_up.model = "gpt-6-sol".to_string(); + follow_up.messages.insert( + 0, + Message { + role: Role::Assistant, + content: vec![ContentBlock::Thinking { + thinking: String::new(), + signature: None, + state: Some(state), + }], + }, + ); + let prepared = client + .prepare_outbound_request(follow_up, true) + .expect("second request prepares"); + let second = &prepared.body; + assert_eq!( + second.pointer("/input/0/type"), + Some(&serde_json::json!("reasoning")), + "the reasoning item must lead the replayed input: {second}" + ); + assert_eq!( + second.pointer("/input/0/encrypted_content"), + Some(&serde_json::json!("enc_custom_state")), + "turn 2's wire carries the state captured on turn 1: {second}" + ); + assert_eq!( + second.pointer("/input/0/id"), + Some(&serde_json::json!("rs_custom")) + ); +} + +/// The headline replay bound must compose through real clients, not just the +/// body builder: the state captured off a live stream at endpoint X has to be +/// dropped by a client rebuilt — through the production +/// `resolve_runtime_route` → `from_candidate` path — for the same table after +/// its `base_url` was edited. The pure gate test above pins the comparison +/// with hand-written fingerprints; this pins that the capture side and the +/// replay side derive the same fingerprint from the same URL spelling rules, +/// so someone normalizing the URL differently per side fails here. +#[tokio::test] +async fn forkguard_capture_replay_drops_across_a_base_url_edit() { + let server = MockServer::start().await; + let sse_body = concat!( + "data: {\"type\":\"response.output_item.added\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_custom\"}}\n\n", + "data: {\"type\":\"response.output_item.done\",\"item\":{\"type\":\"reasoning\",\"id\":\"rs_custom\",\"summary\":[],\"encrypted_content\":\"enc_custom_state\"}}\n\n", + "data: [DONE]\n\n", + ); + Mock::given(method("POST")) + .and(path("/v1/responses")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("Content-Type", "text/event-stream") + .set_body_string(sse_body), + ) + .expect(1) + .mount(&server) + .await; + + let captured = { + let _env_lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + &format!("{}/v1", server.uri()), + "custom-responses-key", + "gpt-6-sol", + ); + let route = crate::route_runtime::resolve_runtime_route( + &config, + ApiProvider::Custom, + Some("gpt-6-sol"), + ) + .expect("named table resolves"); + let client = DeepSeekClient::from_candidate(&route.config, &route.candidate) + .expect("per-turn client builds"); + + let mut first = minimal_responses_request(); + first.model = "gpt-6-sol".to_string(); + let mut stream = client + .handle_responses_stream( + &client + .prepare_outbound_request(first, true) + .expect("first request prepares"), + ) + .await + .unwrap(); + let mut captured = None; + while let Some(event) = stream.next().await { + if let StreamEvent::ContentBlockDelta { + delta: Delta::ReasoningStateDelta { state }, + .. + } = event.expect("capture-turn stream event") + { + captured = Some(state); + } + } + captured.expect("state captured at the original endpoint") + }; + + // Re-point the SAME table at a different endpoint and rebuild through the + // production path; the edited URL is never contacted (prepare only). + let _env_lock = crate::test_support::lock_test_env(); + let repointed = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + "https://repointed.example/v1", + "custom-responses-key", + "gpt-6-sol", + ); + let route = crate::route_runtime::resolve_runtime_route( + &repointed, + ApiProvider::Custom, + Some("gpt-6-sol"), + ) + .expect("repointed table resolves"); + let client = DeepSeekClient::from_candidate(&route.config, &route.candidate) + .expect("per-turn client builds"); + + let mut follow_up = minimal_responses_request(); + follow_up.model = "gpt-6-sol".to_string(); + follow_up.messages.insert( + 0, + Message { + role: Role::Assistant, + content: vec![ContentBlock::Thinking { + thinking: String::new(), + signature: None, + state: Some(captured), + }], + }, + ); + let prepared = client + .prepare_outbound_request(follow_up, true) + .expect("second request prepares"); + let body = &prepared.body; + assert!( + !body.to_string().contains("enc_custom_state"), + "state captured at the old endpoint must not ride the re-pointed table: {body}" + ); + assert!( + body.get("input") + .and_then(Value::as_array) + .is_some_and(|items| items.iter().all(|item| item["type"] != "reasoning")), + "{body}" + ); +} diff --git a/crates/tui/src/client/role_placement.rs b/crates/tui/src/client/role_placement.rs index 54257f5df9..490c12147c 100644 --- a/crates/tui/src/client/role_placement.rs +++ b/crates/tui/src/client/role_placement.rs @@ -377,6 +377,11 @@ mod adapter_agreement_tests { let items = responses::convert_messages_to_responses_input( &request(transcript()), ApiProvider::Openai, + ApiProvider::Openai.as_str(), + "fp-openai-endpoint", + // Inert here: this test carries no fingerprint-less reasoning + // state, so the endpoint's officialness never enters the gate. + "https://relay.example.test/v1", ); assert_eq!( roles(&items), diff --git a/crates/tui/src/commands/groups/core/core.rs b/crates/tui/src/commands/groups/core/core.rs index caf371354f..cb77e7cb93 100644 --- a/crates/tui/src/commands/groups/core/core.rs +++ b/crates/tui/src/commands/groups/core/core.rs @@ -335,6 +335,9 @@ pub fn model(app: &mut App, model_name: Option<&str>) -> CommandResult { route_base_url, app.active_context_window_override, None, + // This receipt feeds only base_url/limits below; the static + // policy is fine because protocol is never consumed here. + None, ) { Ok(resolution) => Some(resolution), Err(reason) => return CommandResult::error(reason), diff --git a/crates/tui/src/config.rs b/crates/tui/src/config.rs index 2d2e095ea7..3c6ea26876 100644 --- a/crates/tui/src/config.rs +++ b/crates/tui/src/config.rs @@ -43,6 +43,10 @@ pub use models::*; #[cfg(test)] pub(crate) use codewhale_config::API_KEYRING_SENTINEL; pub(crate) use codewhale_config::{ConfigApiKeyValueKind, classify_config_api_key_value}; +// The per-config `wire` dialect parse is owned by the config crate next to the +// `ProviderConfigToml::wire` field it reads; local copies would let an alias +// added there silently miss the ambient wire/capability readers. +use codewhale_config::provider::{wire_dialect_prefers_anthropic, wire_dialect_prefers_responses}; pub const DEFAULT_ZAI_PROVIDER_MAX_CONCURRENCY: usize = 3; pub const MAX_PROVIDER_REQUEST_CONCURRENCY: usize = 64; @@ -676,7 +680,7 @@ pub fn provider_capability_with_wire( // Custom wire overrides must be checked before the generic fallback so // `[providers.] wire = "responses"` / `"anthropic"` is honored. if provider == ApiProvider::Custom { - if wire_config_prefers_anthropic(wire) { + if wire_dialect_prefers_anthropic(wire) { return ProviderCapability { provider, resolved_model: resolved_model.to_string(), @@ -689,7 +693,7 @@ pub fn provider_capability_with_wire( alias_deprecation: None, }; } - if wire_config_prefers_responses(wire) { + if wire_dialect_prefers_responses(wire) { return ProviderCapability { provider, resolved_model: resolved_model.to_string(), @@ -9702,40 +9706,6 @@ fn xiaomi_mimo_env_api_key_for_runtime( xiaomi_mimo_env_var(TOKEN_PLAN_ENV_VARS).or_else(|| xiaomi_mimo_env_var(STANDARD_ENV_VARS)) } -fn wire_config_prefers_anthropic(wire: Option<&str>) -> bool { - let Some(raw) = wire.map(str::trim).filter(|value| !value.is_empty()) else { - return false; - }; - let normalized = raw.to_ascii_lowercase().replace(['_', ' '], "-"); - matches!( - normalized.as_str(), - "anthropic" - | "anthropic-messages" - | "messages" - | "claude" - | "anthropic-compatible" - | "anthropic-compat" - ) -} - -fn wire_config_prefers_responses(wire: Option<&str>) -> bool { - let Some(raw) = wire.map(str::trim).filter(|value| !value.is_empty()) else { - return false; - }; - let normalized = raw.to_ascii_lowercase().replace(['_', ' '], "-"); - matches!( - normalized.as_str(), - "responses" - | "responses-api" - | "openai-responses" - | "openai-responses-api" - | "response" - | "response-api" - | "openai-responses-compat" - | "responses-compat" - ) -} - fn modelstudio_mode_is_coding_plan(provider: ApiProvider, mode: Option<&str>) -> bool { if matches!( provider, @@ -9766,7 +9736,7 @@ fn resolve_modelstudio_base_url_for_tui( let anthropic = matches!( provider, ApiProvider::ModelstudioTokenPlanAnthropic | ApiProvider::ModelstudioCodingPlanAnthropic - ) || wire_config_prefers_anthropic(wire); + ) || wire_dialect_prefers_anthropic(wire); match (coding, anthropic) { (true, true) => MODELSTUDIO_CODING_PLAN_ANTHROPIC_BASE_URL.to_string(), (true, false) => DEFAULT_MODELSTUDIO_CODING_PLAN_BASE_URL.to_string(), @@ -9783,7 +9753,7 @@ fn resolve_minimax_base_url_for_tui( if let Some(url) = configured.filter(|value| !value.trim().is_empty()) { return url; } - if matches!(provider, ApiProvider::MinimaxAnthropic) || wire_config_prefers_anthropic(wire) { + if matches!(provider, ApiProvider::MinimaxAnthropic) || wire_dialect_prefers_anthropic(wire) { DEFAULT_MINIMAX_ANTHROPIC_BASE_URL.to_string() } else { DEFAULT_MINIMAX_BASE_URL.to_string() @@ -9798,7 +9768,7 @@ fn resolve_deepseek_base_url_for_tui( if let Some(url) = configured.filter(|value| !value.trim().is_empty()) { return url; } - if matches!(provider, ApiProvider::DeepseekAnthropic) || wire_config_prefers_anthropic(wire) { + if matches!(provider, ApiProvider::DeepseekAnthropic) || wire_dialect_prefers_anthropic(wire) { DEFAULT_DEEPSEEK_ANTHROPIC_BASE_URL.to_string() } else { DEFAULT_DEEPSEEK_BASE_URL.to_string() diff --git a/crates/tui/src/provider_readiness.rs b/crates/tui/src/provider_readiness.rs index 7a9fba4f13..49a3b639c6 100644 --- a/crates/tui/src/provider_readiness.rs +++ b/crates/tui/src/provider_readiness.rs @@ -348,19 +348,17 @@ fn explicit_provider_credential_present( })) } -/// Validate the configured provider/model/endpoint route without making a -/// network request. This is shared by model inventory, `/model`, and Fleet so -/// none of them can mark a route selectable when `/provider` would reject it. -pub(crate) fn route_is_valid_for_model( +/// The route request `route_is_valid_for_model` validates. Split out from the +/// bool probe so the test suite can observe the wire threading directly: +/// Custom validation is protocol-independent today, so deleting the +/// `wire_override` plumb cannot change the resolution outcome — an +/// observable request can. +fn readiness_route_request( config: &crate::config::Config, provider: ApiProvider, + kind: codewhale_config::ProviderKind, model: Option<&str>, -) -> bool { - let compatibility_kind = - (provider == ApiProvider::DeepseekCN).then_some(codewhale_config::ProviderKind::Deepseek); - let Some(kind) = provider.kind().or(compatibility_kind) else { - return true; - }; +) -> RouteRequest { let configured = config.provider_config_for(provider); let configured_model = model .map(str::trim) @@ -376,7 +374,7 @@ pub(crate) fn route_is_valid_for_model( let active_model = (provider == config.api_provider()) .then(|| config.default_model()) .filter(|model| !model.trim().is_empty() && !model.eq_ignore_ascii_case("auto")); - let request = RouteRequest { + RouteRequest { explicit_provider: Some(kind), model_selector: configured_model.or(active_model).map(LogicalModelRef::from), saved_provider_model: None, @@ -397,7 +395,43 @@ pub(crate) fn route_is_valid_for_model( .map(str::to_string) }, limit_overrides: Vec::new(), + + // Validate the same wire the per-turn route would mint: read the + // dialect of the same table `provider_config_for` resolves above (the + // ambient selection, which also supplies the base URL in every arm), + // so preflight cannot disagree with the client this readiness + // describes. Identity-pinned tables are validated by the route + // layer's identity-scoped resolution instead. Outcome-identical + // today (Custom validation is protocol-independent), pinned against + // future drift. The ambient selection's name labels the dialect + // warning when the table's `wire` is a typo. + wire_override: (kind == codewhale_config::ProviderKind::Custom) + .then(|| { + configured.and_then(|entry| { + codewhale_config::provider::wire_dialect_override( + config.provider.as_deref().unwrap_or("custom"), + entry.wire.as_deref(), + ) + }) + }) + .flatten(), + } +} + +/// Validate the configured provider/model/endpoint route without making a +/// network request. This is shared by model inventory, `/model`, and Fleet so +/// none of them can mark a route selectable when `/provider` would reject it. +pub(crate) fn route_is_valid_for_model( + config: &crate::config::Config, + provider: ApiProvider, + model: Option<&str>, +) -> bool { + let compatibility_kind = + (provider == ApiProvider::DeepseekCN).then_some(codewhale_config::ProviderKind::Deepseek); + let Some(kind) = provider.kind().or(compatibility_kind) else { + return true; }; + let request = readiness_route_request(config, provider, kind, model); RouteResolver::new() .resolve(&request) .is_ok_and(|candidate| candidate.validation().ok) @@ -745,6 +779,91 @@ mod tests { ); } + /// A `wire = "responses"` custom table must keep validating through + /// `route_is_valid_for_model`: the wire threading must never turn into a + /// resolution failure. Outcome-identical with the override today + /// (validation is protocol-independent), pinned against future drift. + #[test] + fn custom_wire_tables_validate_through_the_route_resolver() { + let _lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + "https://relay.example/v1", + "readiness-wire-test-key", + "gpt-6-sol", + ); + assert!(route_is_valid_for_model( + &config, + ApiProvider::Custom, + Some("gpt-6-sol") + )); + } + + /// The readiness plumb must be observable, not just outcome-preserving: + /// Custom validation is protocol-independent, so the bool probe above + /// stays green even if the `wire_override` threading is deleted. This + /// pins the request readiness actually resolves — it must carry the + /// table's dialect and mint the same protocol the per-turn authority + /// binds for the same config. Deleting the threading in + /// `readiness_route_request` fails exactly this pin. + #[test] + fn readiness_threads_the_tables_wire_like_the_turn_path() { + let _lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + "https://relay.example/v1", + "readiness-wire-test-key", + "gpt-6-sol", + ); + let request = readiness_route_request( + &config, + ApiProvider::Custom, + codewhale_config::ProviderKind::Custom, + Some("gpt-6-sol"), + ); + assert_eq!( + request.wire_override, + Some(codewhale_config::provider::WireFormat::Responses), + "readiness must read the table's dialect, not the static policy" + ); + let candidate = RouteResolver::new() + .resolve(&request) + .expect("readiness request resolves"); + let turn = crate::route_runtime::resolve_runtime_route( + &config, + ApiProvider::Custom, + Some("gpt-6-sol"), + ) + .expect("turn route resolves"); + assert!( + candidate.validation().ok, + "the wire threading must keep the table valid" + ); + assert_eq!( + candidate.protocol(), + turn.candidate.protocol(), + "preflight cannot disagree with the per-turn client binding" + ); + + // A table without the dialect keeps the static-policy default. + let chat = crate::test_support::custom_named_table_config( + "pinvou_chat", + None, + "https://relay.example/v1", + "readiness-chat-test-key", + "vendor-model", + ); + let request = readiness_route_request( + &chat, + ApiProvider::Custom, + codewhale_config::ProviderKind::Custom, + Some("vendor-model"), + ); + assert_eq!(request.wire_override, None); + } + #[test] fn deepseek_cn_compatibility_alias_uses_real_key_readiness() { let _lock = crate::test_support::lock_test_env(); diff --git a/crates/tui/src/route_runtime.rs b/crates/tui/src/route_runtime.rs index f3be5e85eb..67a0d42379 100644 --- a/crates/tui/src/route_runtime.rs +++ b/crates/tui/src/route_runtime.rs @@ -1,4 +1,5 @@ use chrono::{DateTime, Duration, Utc}; +use codewhale_config::provider::{WireFormat, wire_dialect_override}; use codewhale_config::route::{ LimitField, LogicalModelRef, OverrideSource, ReadyRouteCandidate, RouteLimits, RouteRequest, RouteResolver, SourcedLimitOverride, WireModelId, @@ -378,6 +379,13 @@ pub(crate) fn resolve_route_candidate( base_url_override, context_window_override, None, + // Config-free by contract: this wrapper's callers (unpinned child + // admission, the fleet hermetic fallback) build no client from the + // candidate. The fallback renders `candidate.protocol()` as a display + // label only, and its `Custom` arm returns `None` above, so the + // static policy is display-correct there too — no wire-override table + // can reach a protocol-consuming consumer through this wrapper. + None, ) .map(|resolution| resolution.candidate) } @@ -427,6 +435,13 @@ pub(crate) fn resolve_unpinned_model_candidate( /// Code bare-K3 endpoint, only at the documented 1M entitlement, and only /// while fresh; this prevents generic Moonshot or stale metadata from being /// inherited by a membership-plan route. +/// +/// `custom_wire_override` must carry the dialect of the same table that +/// supplied `base_url_override` when `provider` is +/// [`ApiProvider::Custom`] — pass [`custom_wire_override_for`] on the config +/// the base URL came from. A `None` here resolves a Custom route under the +/// static Chat policy, which is only correct for callers that never consume +/// `candidate.protocol()`. pub(crate) fn resolve_route_candidate_with_context_metadata( provider: ApiProvider, model_selector: Option<&str>, @@ -434,6 +449,7 @@ pub(crate) fn resolve_route_candidate_with_context_metadata( base_url_override: Option, context_window_override: Option, provider_reported_context: Option, + custom_wire_override: Option, ) -> Result { resolve_route_candidate_with_context_metadata_and_host_limits( provider, @@ -443,9 +459,32 @@ pub(crate) fn resolve_route_candidate_with_context_metadata( context_window_override, provider_reported_context, None, + custom_wire_override, ) } +/// Wire-format override a `Custom` route's named table asks for via its +/// per-config `wire = "responses" | "anthropic" | "chat"` dialect. +/// +/// The Custom descriptor's static policy stays Chat Completions for backward +/// compatibility; the override must flow into the resolver so the minted +/// candidate is wire-true (receipts, preflight, and the per-turn +/// `from_candidate` binding all read `candidate.protocol()`). `None` means +/// "no preference" — the static policy applies, matching +/// `provider_wire_format_for_config` and `provider_capability_with_wire`. +/// +/// Callers must pass the config that actually scopes the route's endpoint: +/// for identity-pinned resolution that is the identity-scoped clone (whose +/// `provider` names the pinned table), never the ambient selection — the two +/// can name different tables on every thread/pin/restore path. +pub(crate) fn custom_wire_override_for(config: &Config) -> Option { + // The scoped `provider` names the table this config resolves (the + // identity-scoped clone's table on pinned paths), so the unrecognized- + // dialect warning points at the right table. + let table = config.provider.as_deref().unwrap_or("custom"); + wire_dialect_override(table, config.provider_wire_dialect(ApiProvider::Custom)) +} + #[allow(clippy::too_many_arguments)] fn resolve_route_candidate_with_context_metadata_and_host_limits( provider: ApiProvider, @@ -455,6 +494,7 @@ fn resolve_route_candidate_with_context_metadata_and_host_limits( context_window_override: Option, provider_reported_context: Option, host_limits: Option, + custom_wire_override: Option, ) -> Result { let effective_base_url = base_url_override .as_deref() @@ -470,9 +510,12 @@ fn resolve_route_candidate_with_context_metadata_and_host_limits( .map(|model| WireModelId::from(model.to_string())), base_url_override, limit_overrides: Vec::new(), + wire_override: custom_wire_override, }; - // First pass: resolve the route without overrides to learn the effective - // endpoint, wire model id, and catalog limits. Candidates are immutable, so + // First pass: resolve the route without limit overrides to learn the + // effective endpoint, wire model id, and catalog limits (the wire + // override rides both passes, so both mint the same protocol and + // endpoint key). Candidates are immutable, so // limit adjustments are planned from this read-only resolution and then // requested through `RouteRequest::limit_overrides` on a second pass; the // resolver applies them BEFORE minting the final candidate and records @@ -775,6 +818,14 @@ fn resolve_runtime_route_for_identity_with_limits( .then(|| model_roster().preferred_model_id().map(str::to_string)) .flatten(); let model_selector = model_selector.or(roster_preferred.as_deref()); + // The override must come from the identity-scoped clone: its `provider` + // names the pinned table that also supplies the endpoint below, while the + // ambient `config` may select a different table (per-thread routing, + // fleet pins, session restore). Reading the dialect from the ambient + // config wired one table's protocol onto another table's endpoint. + let custom_wire_override = (provider == ApiProvider::Custom) + .then(|| custom_wire_override_for(&route_config)) + .flatten(); let resolution = resolve_route_candidate_with_context_metadata_and_host_limits( provider, model_selector, @@ -783,6 +834,7 @@ fn resolve_runtime_route_for_identity_with_limits( route_config.context_window_for_provider_config(provider), None, host_limits, + custom_wire_override, )?; let candidate = resolution.candidate; let model = candidate.wire_model_id().as_str().to_string(); @@ -1270,6 +1322,7 @@ mod tests { base.clone(), None, None, + None, ) .expect("Kimi Code route"); assert_eq!(static_floor.context_window.tokens, 262_144); @@ -1288,6 +1341,7 @@ mod tests { context_tokens: 1_048_576, observed_at: Utc::now(), }), + None, ) .expect("configured route"); assert_eq!(configured.context_window.tokens, 1_048_576); @@ -1306,6 +1360,7 @@ mod tests { context_tokens: 1_048_576, observed_at: Utc::now(), }), + None, ) .expect("fresh documented provider metadata"); assert_eq!(reported.context_window.tokens, 1_048_576); @@ -1324,6 +1379,7 @@ mod tests { context_tokens: 1_048_576, observed_at: Utc::now() - Duration::hours(25), }), + None, ) .expect("stale metadata falls back safely"); assert_eq!( @@ -1341,6 +1397,7 @@ mod tests { context_tokens: 1_048_576, observed_at: Utc::now(), }), + None, ) .expect_err("bare k3 is rejected on the direct Moonshot endpoint (#4687)"); assert!( @@ -1777,3 +1834,190 @@ mod tests { assert_eq!(host_overrides, 3); } } + +/// The named-custom table's per-config `wire` dialect must reach the runtime +/// route candidate: `resolve_runtime_route` is the per-turn authority, and +/// the client it binds reads `candidate.protocol()`. A `wire = "responses"` +/// table therefore resolves a Responses candidate (Pinvou PR #625), an +/// `anthropic` table a Messages candidate, and absent/`chat` keep the +/// backward-compatible Chat Completions default. +#[cfg(test)] +mod custom_wire_override_tests { + use super::*; + use crate::config::{ProviderConfig, ProvidersConfig}; + + #[test] + fn forkguard_named_table_wire_responses_reaches_the_runtime_candidate() { + let _env_lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + "https://api.openai.com/v1", + "test-key", + "gpt-6-sol", + ); + let route = resolve_runtime_route(&config, ApiProvider::Custom, Some("gpt-6-sol")) + .expect("named table resolves"); + assert_eq!( + route.candidate.protocol(), + WireFormat::Responses, + "the per-turn candidate must carry the table's Responses wire" + ); + assert_eq!( + route.candidate.endpoint().base_url, + "https://api.openai.com/v1" + ); + assert_eq!(route.candidate.wire_model_id().as_str(), "gpt-6-sol"); + assert_eq!(route.candidate.endpoint().endpoint_key, "responses"); + } + + #[test] + fn forkguard_named_table_wire_anthropic_reaches_the_runtime_candidate() { + let _env_lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("anthropic"), + "https://relay.example.test/v1", + "test-key", + "claude-sonnet", + ); + let route = resolve_runtime_route(&config, ApiProvider::Custom, Some("claude-sonnet")) + .expect("named table resolves"); + assert_eq!(route.candidate.protocol(), WireFormat::AnthropicMessages); + assert_eq!(route.candidate.endpoint().endpoint_key, "messages"); + } + + #[test] + fn forkguard_named_table_without_wire_keeps_the_chat_default() { + let _env_lock = crate::test_support::lock_test_env(); + for wire in [None, Some("chat")] { + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + wire, + "https://relay.example.test/v1", + "test-key", + "vendor-model", + ); + let route = resolve_runtime_route(&config, ApiProvider::Custom, Some("vendor-model")) + .expect("named table resolves"); + assert_eq!( + route.candidate.protocol(), + WireFormat::ChatCompletions, + "wire {wire:?} keeps the static Chat default" + ); + } + } + + /// A typo'd dialect (`wire = "respones"`) must degrade to the legacy + /// Chat default, not fail the config and not half-resolve to another + /// wire. Pinned so the silent-degrade contract in the shared dialect + /// parser stays deliberate. + #[test] + fn forkguard_named_table_unrecognized_wire_keeps_the_chat_default() { + let _env_lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("respones"), + "https://relay.example.test/v1", + "test-key", + "vendor-model", + ); + assert_eq!(custom_wire_override_for(&config), None); + let route = resolve_runtime_route(&config, ApiProvider::Custom, Some("vendor-model")) + .expect("named table resolves"); + assert_eq!(route.candidate.protocol(), WireFormat::ChatCompletions); + } + + /// Build one config carrying two named custom tables: the ambient + /// selection (`config.provider`) and a second table a persisted identity + /// can pin. Each table has a distinct base URL so the test can prove the + /// wire came from the same table as the endpoint. + fn two_table_config(ambient: (&str, Option<&str>), pinned: (&str, Option<&str>)) -> Config { + let table = |name: &str, wire: Option<&str>| { + ( + name.to_string(), + ProviderConfig { + kind: Some("openai-compatible".to_string()), + base_url: Some(format!("https://{name}.example.test/v1")), + model: Some("shared-model".to_string()), + api_key: Some("test-key".to_string()), + wire: wire.map(str::to_string), + ..ProviderConfig::default() + }, + ) + }; + let (ambient_name, ambient_wire) = ambient; + let (pinned_name, pinned_wire) = pinned; + let custom = [ + table(ambient_name, ambient_wire), + table(pinned_name, pinned_wire), + ] + .into_iter() + .collect(); + Config { + provider: Some(ambient_name.to_string()), + providers: Some(ProvidersConfig { + custom, + ..ProvidersConfig::default() + }), + ..Config::default() + } + } + + /// Direction (i): a thread pinned to a responses-wire table must keep the + /// override even while the ambient selection is an ordinary chat table. + /// The override previously read the ambient config, so pinned turns rode + /// Chat Completions — the exact mis-route this feature fixes. + #[test] + fn forkguard_identity_pinned_route_reads_the_pinned_tables_wire() { + let _env_lock = crate::test_support::lock_test_env(); + let config = two_table_config( + ("ambient_chat", None), + ("pinned_responses", Some("responses")), + ); + let identity = config + .resolve_persisted_provider_identity(Some("custom"), Some("pinned_responses")) + .expect("pinned identity resolves"); + let route = resolve_runtime_route_for_identity(&config, &identity, Some("shared-model")) + .expect("pinned table resolves"); + assert_eq!( + route.candidate.protocol(), + WireFormat::Responses, + "the pinned table's wire, not the ambient table's" + ); + assert_eq!(route.candidate.endpoint().endpoint_key, "responses"); + assert_eq!( + route.candidate.endpoint().base_url, + "https://pinned_responses.example.test/v1", + "wire and endpoint must come from the same table" + ); + } + + /// Direction (ii): an ambient responses-wire selection must not wire its + /// protocol onto a pinned chat-only relay. The override previously read + /// the ambient config, so reload/restore installed a Responses candidate + /// for a `{base}/chat/completions` endpoint — a regression against the + /// static-policy base for every multi-table setup. + #[test] + fn forkguard_ambient_wire_override_does_not_leak_onto_pinned_chat_tables() { + let _env_lock = crate::test_support::lock_test_env(); + let config = two_table_config( + ("ambient_responses", Some("responses")), + ("pinned_chat", None), + ); + let identity = config + .resolve_persisted_provider_identity(Some("custom"), Some("pinned_chat")) + .expect("pinned identity resolves"); + let route = resolve_runtime_route_for_identity(&config, &identity, Some("shared-model")) + .expect("pinned table resolves"); + assert_eq!( + route.candidate.protocol(), + WireFormat::ChatCompletions, + "a chat-only relay keeps its static wire under a responses ambient selection" + ); + assert_eq!( + route.candidate.endpoint().base_url, + "https://pinned_chat.example.test/v1" + ); + } +} diff --git a/crates/tui/src/test_support.rs b/crates/tui/src/test_support.rs index 138bb0cab1..4620e34291 100644 --- a/crates/tui/src/test_support.rs +++ b/crates/tui/src/test_support.rs @@ -319,6 +319,43 @@ pub(crate) fn test_tui_options(workspace: impl AsRef) -> crate::tui::app:: } } +/// A one-table Custom test config: `provider` selects the named +/// `[providers.]` table with the given wire dialect, endpoint, +/// credential, and model. `wire = None` keeps the legacy Chat default. +/// +/// Consolidates the per-test `[providers.]` literals (wire-override +/// route tests, transport pins, readiness, prompt suggestion); tests that +/// need a second table or extra fields mutate the returned value at the call +/// site, per the same convention as [`test_tui_options`]. +pub(crate) fn custom_named_table_config( + name: &str, + wire: Option<&str>, + base_url: &str, + api_key: &str, + model: &str, +) -> crate::config::Config { + crate::config::Config { + provider: Some(name.to_string()), + providers: Some(crate::config::ProvidersConfig { + custom: [( + name.to_string(), + crate::config::ProviderConfig { + kind: Some("openai-compatible".to_string()), + wire: wire.map(str::to_string), + base_url: Some(base_url.to_string()), + api_key: Some(api_key.to_string()), + model: Some(model.to_string()), + ..crate::config::ProviderConfig::default() + }, + )] + .into_iter() + .collect(), + ..crate::config::ProvidersConfig::default() + }), + ..crate::config::Config::default() + } +} + /// Build an `App` whose observable state does not depend on the developer's /// machine. /// diff --git a/crates/tui/src/tui/app/init.rs b/crates/tui/src/tui/app/init.rs index 41699f5ede..2d55dd7236 100644 --- a/crates/tui/src/tui/app/init.rs +++ b/crates/tui/src/tui/app/init.rs @@ -402,6 +402,16 @@ impl App { Some(configured_route_base_url.clone()), active_context_window_override, None, + // The launch identity's dialect — `effective_auth_config` + // is scoped to the same identity that resolved + // `configured_route_base_url` above, which is not always + // the ambient selection: `settings.default_provider` can + // re-point the launch at a different custom table. + (provider == ApiProvider::Custom) + .then(|| { + crate::route_runtime::custom_wire_override_for(&effective_auth_config) + }) + .flatten(), ) .map(|resolution| { ( diff --git a/crates/tui/src/tui/prompt_suggestion.rs b/crates/tui/src/tui/prompt_suggestion.rs index 0dfe5343b0..b0f1078a05 100644 --- a/crates/tui/src/tui/prompt_suggestion.rs +++ b/crates/tui/src/tui/prompt_suggestion.rs @@ -14,6 +14,7 @@ use tracing::debug; use crate::config::{ApiProvider, Config}; use crate::core::events::TurnRoute; use crate::route_receipt::{TurnRouteReceipt, endpoint_identity}; +use codewhale_config::provider::WireFormat; /// The exact route authority a turn was launched against. /// @@ -149,11 +150,13 @@ impl fmt::Debug for SuggestionLaunch { } } -/// Whether a provider speaks the ordinary OpenAI-compatible +/// Whether a provider can ever carry the ordinary OpenAI-compatible /// `/chat/completions` shape [`generate_suggestion`] hardcodes. /// -/// Gate on wire protocol, not a vendor enum: Anthropic Messages and the -/// OpenAI Responses API are different request shapes and stay out. +/// Static pre-filter only: it reads the provider's default wire and cannot +/// see a named Custom table's `wire` dialect, so a Custom provider passes it +/// here and [`resolve_credentials_for_identity`] applies the authoritative +/// gate on the resolved route candidate's protocol. #[must_use] pub fn route_is_supported_suggestion_provider(provider: ApiProvider) -> bool { crate::client::provider_speaks_chat_completions(provider) @@ -164,8 +167,10 @@ pub fn route_is_supported_suggestion_provider(provider: ApiProvider) -> bool { /// The identity is revalidated against live config and then scoped with /// `resolve_runtime_route_for_identity`, so the key and endpoint come from that /// route's own configuration rather than from whichever provider happens to be -/// selected now. An identity that no longer resolves, or that now resolves to a -/// different provider kind, yields `None`. +/// selected now. An identity that no longer resolves, that now resolves to a +/// different provider kind, or whose resolved candidate no longer speaks Chat +/// Completions (a named table's `wire` dialect — invisible to the static +/// provider gate) yields `None`. /// /// The returned `base_url` is the **resolved route candidate's** endpoint, not /// `Config::deepseek_base_url()`. Those two are not the same string: the config @@ -195,6 +200,21 @@ fn resolve_credentials_for_identity( if resolved.identity.provider != provider { return None; } + // The route candidate is the authority on this identity's wire. The + // static provider gate above cannot see a named table's `wire` dialect, + // so a `wire = "responses"` / `"anthropic"` Custom table passes it and + // must be stopped here: this helper only speaks the ordinary Chat + // Completions shape, and a hard-coded Chat body must never reach a route + // the turn itself runs on another protocol. + if resolved.candidate.protocol() != WireFormat::ChatCompletions { + debug!( + provider = ?provider, + identity = %provider_identity, + protocol = ?resolved.candidate.protocol(), + "prompt suggestion skipped: the resolved route does not speak Chat Completions" + ); + return None; + } // This helper intentionally sends the ordinary Chat Completions shape. // A configured path override may describe a provider-specific transport // contract that this bounded feature does not implement, so fail closed @@ -226,7 +246,11 @@ fn resolve_credentials_for_identity( /// /// Unsupported providers — Anthropic Messages, Responses, and any other /// non-Chat-Completions wire — return `None`, and no credential material -/// of any provider is inspected on this path at all. +/// of any provider is inspected on this path at all. The gate is the static +/// default wire only: a named Custom table's `wire` dialect is applied later, +/// in `resolve_credentials_for_identity`, so an authority may be captured +/// here for a table whose resolved route ends up on another protocol (and is +/// then never dispatched). #[must_use] pub fn capture_route_authority(route: &TurnRoute) -> Option { if !route_is_supported_suggestion_provider(route.provider) { @@ -1742,4 +1766,51 @@ mod tests { "an absent turn endpoint must fail closed" ); } + + /// A named table's `wire` dialect is invisible to the static provider + /// gate (Custom's default wire is Chat), so the authoritative protocol + /// gate must live on the resolved route candidate: a `wire = "responses"` + /// table serves its turns on `/responses` and must never receive this + /// helper's hard-coded Chat body, while the same table without the + /// override keeps resolving suggestion credentials. + #[test] + fn forkguard_wire_responses_table_gets_no_suggestion_chat_body() { + let _env_lock = crate::test_support::lock_test_env(); + let responses_config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + "https://relay.example/v1", + "custom-responses-key", + "gpt-6-sol", + ); + assert!( + resolve_credentials_for_identity( + &responses_config, + ApiProvider::Custom, + "pinvou_responses", + "gpt-6-sol", + ) + .is_none(), + "a wire=responses table must not receive the Chat Completions suggestion body" + ); + + // Control: the same table on its legacy Chat wire still resolves, so + // the gate above (not the identity resolution) is what failed closed. + let chat_config = crate::test_support::custom_named_table_config( + "pinvou_chat", + None, + "https://relay.example/v1", + "custom-chat-key", + "vendor-model", + ); + let credentials = resolve_credentials_for_identity( + &chat_config, + ApiProvider::Custom, + "pinvou_chat", + "vendor-model", + ) + .expect("a plain Chat table still resolves suggestion credentials"); + assert_eq!(credentials.base_url, "https://relay.example/v1"); + assert_eq!(credentials.api_key, "custom-chat-key"); + } } diff --git a/crates/tui/src/tui/provider_picker.rs b/crates/tui/src/tui/provider_picker.rs index 29a36ea9cf..c5f5c5cf55 100644 --- a/crates/tui/src/tui/provider_picker.rs +++ b/crates/tui/src/tui/provider_picker.rs @@ -697,6 +697,12 @@ impl ProviderDashboardRow { .flatten(), config.context_window_for_provider_config(provider), None, + // The dialect of the same table that supplied the base URL above + // (this row's scoped config), so the row's supported-protocol + // display matches what a turn on this row would bind. + (provider == ApiProvider::Custom) + .then(|| crate::route_runtime::custom_wire_override_for(config)) + .flatten(), ); let ( base_url, @@ -7010,6 +7016,35 @@ mod tests { assert!(row.available_model_count >= 3); } + #[test] + fn provider_dashboard_row_surfaces_the_wire_tables_protocol() { + // Round-3 of #79 shipped exactly this class of bug: the picker row + // resolved through the config-free wrapper and showed `chat` for a + // `wire = "responses"` table. Pin the named-table row to the same + // wire-true protocol the turn path mints. + let _lock = crate::test_support::lock_test_env(); + let config = crate::test_support::custom_named_table_config( + "pinvou_responses", + Some("responses"), + "https://picker.example/v1", + "picker-key", + "gpt-6-sol", + ); + let row = ProviderDashboardRow::from_custom_config_with_runtime_status( + "pinvou_responses", + ApiProvider::Custom, + &config, + None, + ); + + assert_eq!(row.provider_id, "pinvou_responses"); + assert_eq!( + row.supported_protocols, + vec!["responses".to_string()], + "the row must show the table's wire, not the static Chat default" + ); + } + #[test] fn provider_dashboard_row_surfaces_openmodel_messages_route() { let _lock = crate::test_support::lock_test_env(); diff --git a/crates/tui/src/tui/ui/apply.rs b/crates/tui/src/tui/ui/apply.rs index 4f5b8fb87e..4a4ec1c022 100644 --- a/crates/tui/src/tui/ui/apply.rs +++ b/crates/tui/src/tui/ui/apply.rs @@ -810,6 +810,11 @@ pub(crate) async fn apply_model_picker_choice( Some(config.deepseek_base_url()), config.context_window_for_provider_config(app.api_provider), None, + // The ambient table's dialect — the same table `deepseek_base_url` + // above resolves — so this receipt cannot disagree with the turn. + (app.api_provider == ApiProvider::Custom) + .then(|| crate::route_runtime::custom_wire_override_for(config)) + .flatten(), ) { Ok(resolution) => { resolved_model = resolution.candidate.wire_model_id().as_str().to_string(); diff --git a/crates/tui/src/tui/ui/session_state.rs b/crates/tui/src/tui/ui/session_state.rs index 68cab94ab0..7c2ad636e5 100644 --- a/crates/tui/src/tui/ui/session_state.rs +++ b/crates/tui/src/tui/ui/session_state.rs @@ -948,6 +948,11 @@ pub(crate) fn resolve_loaded_session_route(app: &mut App, config: &Config) { Some(config.deepseek_base_url()), context_override, None, + // The ambient table's dialect — the same table `deepseek_base_url` + // above resolves — so this receipt cannot disagree with the turn. + (app.api_provider == ApiProvider::Custom) + .then(|| crate::route_runtime::custom_wire_override_for(config)) + .flatten(), ) { Ok(resolution) => { app.set_active_route_resolution( diff --git a/docs/CONFIGURATION.md b/docs/CONFIGURATION.md index 81116d2605..41e98d7eca 100644 --- a/docs/CONFIGURATION.md +++ b/docs/CONFIGURATION.md @@ -1784,6 +1784,7 @@ reasoning contract, and all four membership ids omit generic sampling fields. - `context_window` (integer, optional provider-table key): override the total context window for the active `[providers.]` route when an OpenAI-compatible gateway, hosted model alias, or self-hosted runtime has a different limit than Codewhale's static model table. For example, `[providers.openai] context_window = 1000000` lets an OpenAI-compatible DashScope/Qwen route budget against a 1M-token window instead of the conservative fallback. For Kimi Code K3, keep `model = "k3"` and set `[providers.moonshot] context_window = 1048576` only when the membership plan includes 1M access; otherwise omit it to retain the 262,144-token safe baseline. The value must be greater than 0 and affects prompt context notes, compaction thresholds, context-pressure checks, and request output caps. Full resolution order, and how to see which rung produced the current window: [Context length (context window)](#context-length-context-window). - `path_suffix` (string, optional provider-table key): override the chat-completions path for OpenAI-compatible gateways that do not serve `/v1/chat/completions`. For example, `[providers.openai] path_suffix = "/chat/completions"` sends chat requests to the unversioned base URL plus `/chat/completions`; `models` and `beta/*` requests keep their normal routing. - `reasoning_stream_style` (string, optional provider-table key): override how streaming reasoning is separated from answer text for the active provider route. Use `separate_field` for `reasoning_content` / `reasoning` deltas, `inline_tags` for gateways that stream `...` inside `delta.content`, or `none` to render incoming content exactly as answer text. +- `wire` (string, optional provider-table key): select the wire protocol the active route speaks. On a named custom-provider table (`[providers.] kind = "openai-compatible"` selected with `provider = ""`), `wire = "responses"` (aliases `responses-api`, `openai-responses`, `openai-responses-api`, `response-api`, `responses-compat`, ...) switches the route to the OpenAI Responses API at `{base}/responses`, and `wire = "anthropic"` (aliases `messages`, `claude`, `anthropic-messages`, `anthropic-compatible`, `anthropic-compat`) switches it to the Anthropic Messages API at `{base}/messages`; `wire = "chat"` / `wire = "openai"` or omitting the key keeps the default OpenAI Chat Completions contract. The dialect is the route's wire everywhere — the per-turn client binding, receipts, preflight, the `/provider` row, encrypted-reasoning capture, and prompt-suggestion eligibility (suggestions only speak Chat Completions and stay off on a non-chat custom route) — and the built-in OpenAI-compatible `/v1/chat/completions` pass-through rejects a non-chat custom table with `provider_wire_format_unsupported` instead of forwarding it a chat body. An unrecognized value (a typo such as `respones`) degrades to the default Chat policy with a warning naming the table: fix the spelling rather than assuming the endpoint is served. Dual-protocol built-in vendors (DeepSeek, MiniMax, Model Studio) accept the same key for their narrower `openai` / `anthropic` dialect space and degrade silently on anything else. - `[providers..auth]` (table, optional): provider-scoped auth source metadata. `source = "command"` stores a command argv plus optional `timeout_ms`; `source = "secret"` stores a `secret_id`. This slice lets provider readiness, `/provider`, and doctor JSON report the auth source class without exposing command argv output or secret values; executing commands and resolving external secret material is handled by the follow-up resolver work. - `insecure_skip_tls_verify` (bool, optional provider-table key): legacy compatibility key, disabled by default. When true on the active provider table, provider clients reject the configuration instead of skipping TLS certificate verification. Use `SSL_CERT_FILE` for corporate or private CA bundles; `codewhale doctor` reports stale uses of this setting. - `default_text_model` (string, optional): defaults to `deepseek-v4-pro` for DeepSeek and `deepseek-anthropic`, `gpt-5.6` for OpenAI, `grok-4.6` for xAI, `deepseek-ai/deepseek-v4-pro` for NVIDIA NIM, `deepseek-ai/deepseek-v4-flash` for AtlasCloud, `deepseek-reasoner` for Wanjie Ark, `DeepSeek-V4-Pro` for Volcengine Ark, `deepseek/deepseek-v4-pro` for OpenRouter and Novita, `mimo-v2.5-pro` for Xiaomi MiMo, `accounts/fireworks/models/deepseek-v4-pro` for Fireworks, `deepseek-ai/DeepSeek-V4-Pro` for SiliconFlow and DeepInfra, `trinity-large-thinking` for Arcee AI, `kimi-k2.7-code` for Moonshot, `MiniMax-M3` for MiniMax, `GLM-5.3` for Z.ai, `step-3.7-flash` for StepFun, `ernie-4.0-turbo-8k` for Qianfan, `fugu` for Sakana AI, `deepseek-ai/DeepSeek-V4-Pro` for SGLang/vLLM, `deepseek-v4-flash` for local Ollama, and `gpt-oss:120b` for Ollama Cloud. Hugging Face and Together AI both default to `deepseek-ai/DeepSeek-V4-Pro`; `openai-codex` defaults to `gpt-5.6`; `anthropic` defaults to `claude-sonnet-4-6`; `openmodel` defaults to `deepseek-v4-flash`. Current public DeepSeek IDs are `deepseek-v4-pro` and `deepseek-v4-flash`, both with 1M context windows, 384K max output, and thinking mode enabled by default. DeepSeek's live pricing/model page now labels the Pro backend `DeepSeek-V4-Pro-0813`; the callable API ID remains `deepseek-v4-pro`, so Codewhale does not send the backend label or the Claude Code-specific `deepseek-v4-pro[1m]` selector. DeepSeek retires `deepseek-chat` and `deepseek-reasoner` on July 24, 2026; direct first-party routes migrate both to `deepseek-v4-flash`, with omitted reasoning settings preserving their former non-thinking (`off`) and thinking (`high`) intent. Explicit `reasoning_effort` wins, and provider-owned ids on Wanjie Ark, aggregators, self-hosted runtimes, and custom endpoints are not globally rewritten. SiliconFlow retains its own mapping: `deepseek-reasoner` and `deepseek-r1` select its Pro model while `deepseek-chat` and `deepseek-v3` select Flash. Provider-specific mappings translate `deepseek-v4-pro` / `deepseek-v4-flash` to each provider's model ID where supported. OpenRouter also recognizes recent large IDs such as `arcee-ai/trinity-large-thinking`, `minimax/minimax-m3`, `minimax/minimax-m2.7`, `xiaomi/mimo-v2.5-pro`, `qwen/qwen3.6-flash`, `qwen/qwen3.6-35b-a3b`, `qwen/qwen3.6-max-preview`, `qwen/qwen3.6-27b`, `qwen/qwen3.6-plus`, `qwen/qwen3.7-max`, `google/gemma-4-31b-it`, `moonshotai/kimi-k2.7-code`, `moonshotai/kimi-k2.6`, `nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free`, and `nvidia/nemotron-3-ultra-550b-a55b`; direct Arcee uses bare IDs such as `trinity-large-thinking` and `trinity-large-preview`; direct Moonshot recognizes `kimi-k3`, `kimi-k2.7-code`, and `kimi-k2.6`. The exact Kimi Code endpoint recognizes bare `k3` for K3 and `kimi-for-coding` for K2.7; those membership IDs are distinct from the direct Moonshot IDs and are never rewritten across routes. Direct MiniMax recognizes `MiniMax-M3` and the documented M2.x chat model IDs; direct Z.ai recognizes `GLM-5.3` (the default), `GLM-5.2`, `GLM-5.1`, and `GLM-5-Turbo`, and OpenRouter recognizes the matching `z-ai/glm-5.1`, `z-ai/glm-5.2`, `z-ai/glm-5.3`, and `z-ai/glm-5-turbo` IDs — `GLM-5.3` has been live on the Z.ai Coding Plan since 2026-08-13; it inherits its catalog metadata from `GLM-5.2` until Z.ai publishes distinct 5.3 numbers and carries no price, and an explicit `GLM-5.2` selection keeps its own id; direct Sakana recognizes `fugu` and `fugu-ultra-20260615`; direct Xiaomi MiMo recognizes chat IDs `mimo-v2.5-pro`, `mimo-v2.5-pro-ultraspeed`, and `mimo-v2.5`, while TTS IDs are selected through `codewhale speech` / `tts`. Generic `openai`, `atlascloud`, `wanjie-ark`, `xiaomi-mimo`, `arcee`, `moonshot`, `minimax`, `openmodel`, `zai`, `stepfun`, `qianfan`, `sakana`, local Ollama, and Ollama Cloud model IDs are passed through unchanged after known aliases are normalized. OpenRouter and SiliconFlow provider configs with a custom `base_url` also preserve explicit model values, which lets OpenAI-compatible gateways accept bare model IDs. Use `/models` or `codewhale models` to discover live IDs from your configured endpoint. `CODEWHALE_MODEL` overrides this for a single process; `DEEPSEEK_MODEL` is the legacy alias. diff --git a/docs/PROVIDERS.md b/docs/PROVIDERS.md index d7d5dca916..7e752383e4 100644 --- a/docs/PROVIDERS.md +++ b/docs/PROVIDERS.md @@ -255,6 +255,18 @@ Instead, choose the closest shipped route and override its endpoint/model: from an AgentProfile, can use a custom table such as `[providers.lm-studio] kind = "openai-compatible"` and select it with `provider = "lm-studio"` or a profile `provider = "lm-studio"`. +- A custom table defaults to the OpenAI Chat Completions contract. Set `wire` + in the table to serve a different protocol from the same base URL: + `wire = "responses"` selects the OpenAI Responses API (`{base}/responses`) + and `wire = "anthropic"` (or `messages`) the Anthropic Messages API + (`{base}/messages`). The dialect is the route's wire everywhere — turns, + receipts, preflight, the `/provider` row — and an unrecognized value falls + back to Chat Completions with a warning naming the table. The Anthropic + dialect authenticates with `x-api-key` carrying the table's `api_key` + (never `Authorization: Bearer`), and a conflicting `Authorization` header + in `http_headers` is dropped — a Messages-compatible relay that only + accepts Bearer auth is not servable through this dialect yet. See the + `wire` key reference in [CONFIGURATION.md](CONFIGURATION.md). - Local OpenAI-compatible runtimes: use `provider = "vllm"`, `"sglang"`, or `"ollama"` with the matching provider-specific base URL/model values.