diff --git a/docs-site/src/content/docs/guides/combos.md b/docs-site/src/content/docs/guides/combos.md index 2049192ce2..585a81e65a 100644 --- a/docs-site/src/content/docs/guides/combos.md +++ b/docs-site/src/content/docs/guides/combos.md @@ -484,6 +484,24 @@ even when every target supports images — the catalog drops `image` from `input image-bearing requests are rejected with HTTP 400 before any target is called. `"auto"` (or omitting the field) keeps the automatic intersection. +In the dashboard, the **Image / multimodal** switch can be enabled whenever every target is a +known catalog row. Members that advertise image input stay unchanged; members that do not are +automatically declared text-only on save (`modelCapabilities[model].inputModalities = ["text"]`) +so the [Vision Sidecar](/guides/sidecars/) describes their images. The switch's hint names the +members that will be enrolled. A member whose modalities are unknown or have no text input (for +example an audio-only model) cannot be covered by the sidecar; it keeps the switch unavailable and +is named in the hint. If the [Vision Sidecar](/guides/sidecars/) is disabled globally, enabling the +switch shows a warning linking to the dashboard: enrollment still saves, but image requests are +rejected until the sidecar is re-enabled. Turning the switch off disables image input for the combo +but keeps those provider declarations, and `PUT /api/combos` accepts the same enrollment as an +optional top-level `visionSidecarTargets` array of exact `{ provider, model }` targets +(request-only; it is never stored on the combo and is rejected while `imageInput` is +`"disabled"` when the enrollment list is non-empty). Entries that already accept image input or +are declared without text are rejected with HTTP 400 rather than overwritten, entries the sidecar +already covers are skipped, and a failed save restores the combo and every declaration together. +Removing a member from the combo does not remove its text-only declaration; clear it in the +provider editor if you no longer want it. + ## Encrypted v2 sub-agent tasks There is one important limitation for Codex v2 sub-agents ([issue #92](https://github.com/lidge-jun/opencodex/issues/92)). diff --git a/docs-site/src/content/docs/guides/sidecars.md b/docs-site/src/content/docs/guides/sidecars.md index f3a7b8ec84..1663182a17 100644 --- a/docs-site/src/content/docs/guides/sidecars.md +++ b/docs-site/src/content/docs/guides/sidecars.md @@ -140,6 +140,12 @@ allow attachments instead of blocking them before the sidecar runs. When use the `gpt-5.6-luna` fallback. Startup still migrates an explicitly persisted legacy `gpt-5.4-mini` value to `gpt-5.6-luna`; that migration applies to a stored value, not to an absent model field. +Saving a multimodal combo from the dashboard enrolls its non-image members automatically: the +Combos page sends them as `visionSidecarTargets` on `PUT /api/combos`, which writes the exact +text-only declaration on the member's provider, so no manual config edit and reload is needed. +Enrollment never overwrites an existing image-capable declaration, and members whose modalities are +unknown or have no text input are not enrolled — the dashboard names them and keeps the combo's +image switch unavailable. The first-party DeepSeek `deepseek-flash` model is native multimodal (`text` and `image`) and does not use this sidecar by default. Explicit `noVisionModels` or text-only declarations remain authoritative. First-party `deepseek-chat`, `deepseek-reasoner`, and `deepseek-v4-flash` remain diff --git a/gui/src/combo-capabilities.ts b/gui/src/combo-capabilities.ts index 32f91f66b9..af6a5724d5 100644 --- a/gui/src/combo-capabilities.ts +++ b/gui/src/combo-capabilities.ts @@ -1,14 +1,68 @@ import type { ComboTarget } from "./combo-workspace-data"; import type { ModelOption } from "./components/combo-workspace-types"; -/** Whether every selected target advertises image input (incomplete rows fail closed). */ +type ComboImageMemberKind = "vision" | "sidecar" | "blocked" | "missing"; + +/** + * One combo member's image story. Classification must survive a reload, so it + * reads the DECLARED modalities (`inputModalitiesDeclared`) when present: an + * enrolled member advertises image in the catalog, and only the declaration + * still says "declared text-only, covered by the sidecar". Rows with no known + * modalities, or modalities without text, cannot be covered and block the combo. + */ +function imageMemberKind(target: ComboTarget, models: ModelOption[]): ComboImageMemberKind { + const provider = target.provider.trim(); + const modelId = target.model.trim(); + if (!provider || !modelId) return "missing"; + const model = models.find((row) => row.provider === provider && row.id === modelId); + if (!model) return "missing"; + const declared = model.inputModalitiesDeclared ?? model.inputModalities; + if (!declared || declared.length === 0) return "blocked"; + if (declared.includes("image")) return "vision"; + return declared.includes("text") ? "sidecar" : "blocked"; +} + +/** Whether images can be enabled: every target is known and either images natively or can be declared text-only. */ export function comboImagesSupported(targets: ComboTarget[], models: ModelOption[]): boolean { if (targets.length === 0) return false; return targets.every((target) => { - const provider = target.provider.trim(); - const modelId = target.model.trim(); - if (!provider || !modelId) return false; - const model = models.find((row) => row.provider === provider && row.id === modelId); - return !!model?.inputModalities?.includes("image"); + const kind = imageMemberKind(target, models); + return kind === "vision" || kind === "sidecar"; }); } + +/** + * Exact targets that need a text-only declaration so the Vision Sidecar covers + * them when the combo accepts images. Deduplicated in submission order. + */ +export function comboVisionSidecarTargets( + targets: ComboTarget[], + models: ModelOption[], +): Array<{ provider: string; model: string }> { + const seen = new Set(); + const out: Array<{ provider: string; model: string }> = []; + for (const target of targets) { + if (imageMemberKind(target, models) !== "sidecar") continue; + const provider = target.provider.trim(); + const model = target.model.trim(); + const key = `${provider}/${model}`; + if (seen.has(key)) continue; + seen.add(key); + out.push({ provider, model }); + } + return out; +} + +/** + * Exact targets that cannot be covered by the Vision Sidecar: their row is + * known but its modalities are unknown or have no text input. Named in the + * hint so the operator knows which member blocks the image switch. + */ +export function comboImageBlockedTargets( + targets: ComboTarget[], + models: ModelOption[], +): Array<{ provider: string; model: string }> { + return targets + .filter((target) => imageMemberKind(target, models) === "blocked") + .map((target) => ({ provider: target.provider.trim(), model: target.model.trim() })); +} diff --git a/gui/src/combo-workspace-data.ts b/gui/src/combo-workspace-data.ts index 28750c0ef8..e37c267704 100644 --- a/gui/src/combo-workspace-data.ts +++ b/gui/src/combo-workspace-data.ts @@ -457,9 +457,13 @@ export function draftEquals(a: ComboItem, b: ComboItem): boolean { }); } -export function toPutBody(item: ComboItem, options: { renameFrom?: string } = {}): { +export function toPutBody( + item: ComboItem, + options: { renameFrom?: string; visionSidecarTargets?: Array<{ provider: string; model: string }> } = {}, +): { id: string; renameFrom?: string; + visionSidecarTargets?: Array<{ provider: string; model: string }>; combo: { targets: ComboTarget[]; strategy: ComboStrategy; @@ -476,6 +480,7 @@ export function toPutBody(item: ComboItem, options: { renameFrom?: string } = {} return { id: item.id.trim(), ...(options.renameFrom ? { renameFrom: options.renameFrom } : {}), + ...(options.visionSidecarTargets?.length ? { visionSidecarTargets: options.visionSidecarTargets } : {}), combo: { targets: item.targets.map((target) => ({ provider: target.provider.trim(), diff --git a/gui/src/components/ComboWorkspace.tsx b/gui/src/components/ComboWorkspace.tsx index c0415cabd1..9e0db39797 100644 --- a/gui/src/components/ComboWorkspace.tsx +++ b/gui/src/components/ComboWorkspace.tsx @@ -24,6 +24,7 @@ export default function ComboWorkspace({ providers, models, cataloguedComboIds, + visionEnabled, loading, onRefresh, onSave, @@ -212,6 +213,7 @@ export default function ComboWorkspace({ providerQuotaStates={providerQuotaStates} providers={providers} models={models} + visionEnabled={visionEnabled} onBack={() => trySelect(null)} onSaved={(item) => { setDetailDirty(false); @@ -240,6 +242,7 @@ export default function ComboWorkspace({ providerQuotaStates={providerQuotaStates} providers={providers} models={models} + visionEnabled={visionEnabled} onSaved={(item) => { setDetailDirty(false); setSelectedId(item.id); diff --git a/gui/src/components/combo-workspace-controls.tsx b/gui/src/components/combo-workspace-controls.tsx index 8f4f3a9d81..ae15044da3 100644 --- a/gui/src/components/combo-workspace-controls.tsx +++ b/gui/src/components/combo-workspace-controls.tsx @@ -1,6 +1,6 @@ import { useState } from "react"; import type { ComboEffort, ComboStrategy, ComboTarget, ProviderQuotaStates } from "../combo-workspace-data"; -import { comboImagesSupported } from "../combo-capabilities"; +import { comboImageBlockedTargets, comboImagesSupported, comboVisionSidecarTargets } from "../combo-capabilities"; import { COMBO_EFFORTS, COMBO_STRATEGIES, COMBO_STRATEGY_LABEL_KEYS, newComboTarget } from "../combo-workspace-data"; import { IconArrowDown, IconArrowUp, IconGrip, IconPlus, IconTrash } from "../icons"; import { useT } from "../i18n/shared"; @@ -88,6 +88,7 @@ export function ComboCapabilities({ models, imageInput, reasoningEffortMode, + visionEnabled, disabled, onChange, }: { @@ -95,13 +96,30 @@ export function ComboCapabilities({ models: ModelOption[]; imageInput: "auto" | "disabled"; reasoningEffortMode: "strict" | "adaptive"; + /** Vision Sidecar enabled state; undefined = unknown, renders no warning. */ + visionEnabled?: boolean; disabled?: boolean; onChange: (patch: { imageInput?: "auto" | "disabled"; reasoningEffortMode?: "strict" | "adaptive" }) => void; }) { const t = useT(); const imagesSupported = comboImagesSupported(targets, models); + const blockedTargets = comboImageBlockedTargets(targets, models); // Default: checked (auto) when supported; force off when any target lacks image. const effectiveOn = imagesSupported && imageInput !== "disabled"; + const sidecarTargets = comboVisionSidecarTargets(targets, models); + // Enrollment wording only while image input is enabled: a disabled combo is not + // enrolling anything, so it gets the plain hint instead of save-time promises. + const imageHint = !imagesSupported + ? blockedTargets.length > 0 + ? t("cws.capability.imageInputBlockedHint", { + models: blockedTargets.map(({ provider, model }) => `${provider}/${model}`).join(", "), + }) + : t("cws.capability.imageInputUnavailable") + : sidecarTargets.length > 0 && imageInput !== "disabled" + ? t("cws.capability.imageInputSidecarHint", { + models: sidecarTargets.map(({ provider, model }) => `${provider}/${model}`).join(", "), + }) + : t("cws.capability.imageInputHint"); return (
@@ -110,8 +128,14 @@ export function ComboCapabilities({
{t("cws.capability.imageInput")}

- {imagesSupported ? t("cws.capability.imageInputHint") : t("cws.capability.imageInputUnavailable")} + {imageHint}

+ {sidecarTargets.length > 0 && effectiveOn && visionEnabled === false ? ( +

+ {t("cws.capability.imageInputSidecarDisabled")}{" "} + {t("nav.dashboard")} +

+ ) : null}
void; onSaved: (item: ComboItem) => void; onRequestRemove?: () => void; @@ -383,6 +386,7 @@ export function DetailPanel({ models={models} imageInput={draft.imageInput ?? "auto"} reasoningEffortMode={draft.reasoningEffortMode ?? "strict"} + visionEnabled={visionEnabled} disabled={busy} onChange={(patch) => updateDraft((d) => ({ ...d, ...patch }))} /> diff --git a/gui/src/components/combo-workspace-types.ts b/gui/src/components/combo-workspace-types.ts index aec36283f4..1e201c9b0e 100644 --- a/gui/src/components/combo-workspace-types.ts +++ b/gui/src/components/combo-workspace-types.ts @@ -14,6 +14,8 @@ export type ModelOption = { namespaced?: string; reasoningEfforts?: string[]; inputModalities?: string[]; + /** Operator-declared modalities; beats the (sidecar-widened) catalog view on reload. */ + inputModalitiesDeclared?: string[]; }; export type ComboAddIntent = "blank" | "jev-auto"; @@ -27,6 +29,8 @@ export interface ComboWorkspaceProps { models: ModelOption[]; /** Combo ids currently present in the live catalog (`provider === "combo"`). */ cataloguedComboIds?: ReadonlySet; + /** Vision Sidecar enabled state from /api/sidecar-settings; undefined = unknown, no warning. */ + visionEnabled?: boolean; loading?: boolean; onRefresh: () => void; onSave: (item: ComboItem, isCreate: boolean, renameFrom?: string) => Promise<{ ok: boolean; error?: string }>; diff --git a/gui/src/i18n/de.ts b/gui/src/i18n/de.ts index 794bc46324..037fe4de50 100644 --- a/gui/src/i18n/de.ts +++ b/gui/src/i18n/de.ts @@ -2903,8 +2903,11 @@ export const de: Record = { "cws.field.defaultEffort": "Standard-Reasoning", "cws.field.defaultEffortNone": "Keine (Ziel-Standard)", "cws.field.defaultEffortHint": "Nur verwendet, wenn der Client keinen Reasoning-Aufwand sendet. Optionen sind die Schnittmenge der beworbenen Aufwände der gewählten Ziele.", - "cws.capability.imageInputUnavailable": "Erst verfügbar, wenn jedes gewählte Ziel Bildeingabe unterstützt.", + "cws.capability.imageInputUnavailable": "Wähle zuerst alle Ziele aus dem Katalog — unbekannte Modelle kann der Vision Sidecar nicht abdecken.", "cws.capability.imageInputHint": "Standardmäßig aktiv, wenn jedes Ziel Bilder unterstützt. Ausschalten für nur Text.", + "cws.capability.imageInputSidecarHint": "Standardmäßig aktiv. {models} wird beim Speichern als text-only deklariert und nutzt den Vision Sidecar für Bilder.", + "cws.capability.imageInputSidecarDisabled": "Der Vision Sidecar ist ausgeschaltet — {models} lehnt Bilder ab, bis er in den Dashboard-Einstellungen aktiviert wird.", + "cws.capability.imageInputBlockedHint": "{models} kann vom Vision Sidecar nicht abgedeckt werden: Eingabemodalitäten sind unbekannt oder ohne Text.", "cws.capability.imageInput": "Bild / multimodal", "cws.capability.adaptiveEffort": "Adaptive Denkstufen", "cws.capability.adaptiveEffortHint": "Aus: Ziele ohne Denkstufen-Regelung blenden die Auswahl für die gesamte Kombination aus. An: Solche Ziele bleiben nutzbar, und die Auswahl zeigt weiterhin die Stufen der übrigen Ziele.", diff --git a/gui/src/i18n/en.ts b/gui/src/i18n/en.ts index c61d8d17cc..2fdc6fb3f8 100644 --- a/gui/src/i18n/en.ts +++ b/gui/src/i18n/en.ts @@ -3004,8 +3004,11 @@ export const en = { "cws.field.defaultEffort": "Default reasoning", "cws.field.defaultEffortNone": "None (target default)", "cws.field.defaultEffortHint": "Used only when the client omits reasoning effort. Options are the intersection of the selected targets' advertised efforts; targets without catalog effort metadata offer none.", - "cws.capability.imageInputUnavailable": "Unavailable until every selected target supports image input.", + "cws.capability.imageInputUnavailable": "Pick every target from the catalog first — unknown models can't be covered by the Vision Sidecar.", "cws.capability.imageInputHint": "On by default when every target supports images. Turn off to accept text only.", + "cws.capability.imageInputSidecarHint": "On by default. {models} will be declared text-only on save and use the Vision Sidecar for images.", + "cws.capability.imageInputSidecarDisabled": "The Vision Sidecar is turned off — {models} will reject images until it is enabled in dashboard settings.", + "cws.capability.imageInputBlockedHint": "{models} cannot be covered by the Vision Sidecar: its input modalities are unknown or have no text.", "cws.capability.imageInput": "Image / multimodal", "cws.capability.adaptiveEffort": "Adaptive reasoning ladder", "cws.capability.adaptiveEffortHint": "Off: a target with no reasoning control hides the effort picker for the whole combo. On: those targets stay usable and the picker keeps the levels the remaining targets share.", diff --git a/gui/src/i18n/fr.ts b/gui/src/i18n/fr.ts index d306535db9..bd15b922a3 100644 --- a/gui/src/i18n/fr.ts +++ b/gui/src/i18n/fr.ts @@ -2887,8 +2887,11 @@ export const fr: Record = { "cws.emptyTitle": "Créer votre première combinaison", "cws.empty.createDesc": "Nommez un modèle virtuel et enchaînez au moins deux services principaux.", "cws.backToAll": "Retour à toutes les combinaisons", - "cws.capability.imageInputUnavailable": "Indisponible tant que toutes les cibles sélectionnées ne prennent pas en charge les images.", - "cws.capability.imageInputHint": "Activé par défaut lorsque toutes les cibles prennent en charge les images. Désactivez cette option pour n’accepter que du texte.", + "cws.capability.imageInputUnavailable": "Sélectionnez d'abord chaque cible dans le catalogue — les modèles inconnus ne peuvent pas être couverts par le Vision Sidecar.", + "cws.capability.imageInputHint": "Activé par défaut lorsque toutes les cibles prennent en charge les images. Désactivez cette option pour n'accepter que du texte.", + "cws.capability.imageInputSidecarHint": "Activé par défaut. Lors de l’enregistrement, les cibles {models} seront déclarées en mode texte uniquement et utiliseront le Vision Sidecar pour les images.", + "cws.capability.imageInputSidecarDisabled": "Le Vision Sidecar est désactivé — {models} refusera les images tant qu’il ne sera pas activé dans les paramètres du tableau de bord.", + "cws.capability.imageInputBlockedHint": "{models} ne peut pas être couvert par le Vision Sidecar : ses modalités d’entrée sont inconnues ou sans texte.", "cws.capability.imageInput": "Images / multimodal", "cws.capability.adaptiveEffort": "Échelle de raisonnement adaptative", "cws.capability.adaptiveEffortHint": "Désactivé : une cible sans réglage de raisonnement masque le sélecteur pour toute la combinaison. Activé : ces cibles restent utilisables et le sélecteur conserve les niveaux communs aux autres cibles.", diff --git a/gui/src/i18n/ja.ts b/gui/src/i18n/ja.ts index 2efc98321b..b2b0148451 100644 --- a/gui/src/i18n/ja.ts +++ b/gui/src/i18n/ja.ts @@ -2979,8 +2979,11 @@ export const ja: Record = { "cws.field.defaultEffort": "デフォルトの推論", "cws.field.defaultEffortNone": "なし(ターゲットのデフォルト)", "cws.field.defaultEffortHint": "クライアントが推論負荷を省略した場合のみ使用されます。選択肢は選択ターゲットが広告する負荷の交差です。", - "cws.capability.imageInputUnavailable": "選択した全ターゲットが画像入力に対応すると有効になります。", + "cws.capability.imageInputUnavailable": "まずすべてのターゲットをカタログから選択してください。不明なモデルはVision Sidecarでカバーできません。", "cws.capability.imageInputHint": "全ターゲットが画像対応なら既定でオン。オフにするとテキストのみ。", + "cws.capability.imageInputSidecarHint": "既定でオン。{models} は保存時にテキスト専用として宣言され、画像にはVision Sidecarを使用します。", + "cws.capability.imageInputSidecarDisabled": "Vision Sidecarがオフのため、{models} はダッシュボード設定で有効化されるまで画像を拒否します。", + "cws.capability.imageInputBlockedHint": "{models} はVision Sidecarでカバーできません:入力モダリティが不明か、テキストを含みません。", "cws.capability.imageInput": "画像 / マルチモーダル", "cws.capability.adaptiveEffort": "適応的な推論レベル", "cws.capability.adaptiveEffortHint": "オフ: 推論レベルを持たない対象があると、コンボ全体のセレクターが消えます。オン: その対象はそのまま使え、セレクターには残りの対象で共通するレベルが表示されます。", diff --git a/gui/src/i18n/ko.ts b/gui/src/i18n/ko.ts index 8aa5285293..d2fb7ec2a5 100644 --- a/gui/src/i18n/ko.ts +++ b/gui/src/i18n/ko.ts @@ -2942,8 +2942,11 @@ export const ko: Record = { "cws.field.defaultEffort": "기본 추론 수준", "cws.field.defaultEffortNone": "없음 (대상 기본값)", "cws.field.defaultEffortHint": "클라이언트가 추론 수준을 생략한 경우에만 사용합니다. 옵션은 선택한 대상이 광고하는 수준의 교집합입니다.", - "cws.capability.imageInputUnavailable": "선택한 모든 대상이 이미지 입력을 지원해야 사용할 수 있습니다.", + "cws.capability.imageInputUnavailable": "먼저 모든 대상을 카탈로그에서 선택하세요. 알 수 없는 모델은 Vision Sidecar로 처리할 수 없습니다.", "cws.capability.imageInputHint": "모든 대상이 이미지를 지원하면 기본으로 켜집니다. 끄면 텍스트만 허용합니다.", + "cws.capability.imageInputSidecarHint": "기본으로 켜짐. {models}은(는) 저장 시 텍스트 전용으로 선언되어 이미지에 Vision Sidecar를 사용합니다.", + "cws.capability.imageInputSidecarDisabled": "Vision Sidecar가 꺼져 있어 {models}은(는) 대시보드 설정에서 활성화할 때까지 이미지를 거부합니다.", + "cws.capability.imageInputBlockedHint": "{models}은(는) Vision Sidecar로 처리할 수 없습니다: 입력 모달리티를 알 수 없거나 텍스트가 없습니다.", "cws.capability.imageInput": "이미지 / 멀티모달", "cws.capability.adaptiveEffort": "적응형 추론 단계", "cws.capability.adaptiveEffortHint": "끔: 추론 단계를 조절할 수 없는 대상이 하나라도 있으면 콤보 전체의 선택기가 사라집니다. 켬: 그런 대상도 그대로 쓰면서, 선택기에는 나머지 대상이 공통으로 지원하는 단계가 남습니다.", diff --git a/gui/src/i18n/ru.ts b/gui/src/i18n/ru.ts index 2a9f58cc03..c2e7c0004a 100644 --- a/gui/src/i18n/ru.ts +++ b/gui/src/i18n/ru.ts @@ -3069,8 +3069,11 @@ export const ru: Record = { "cws.field.defaultEffort": "Рассуждения по умолчанию", "cws.field.defaultEffortNone": "Нет (по умолчанию для цели)", "cws.field.defaultEffortHint": "Используется, только если клиент не указал уровень рассуждений. Варианты — пересечение заявленных уровней выбранных целей.", - "cws.capability.imageInputUnavailable": "Доступно, когда все выбранные цели поддерживают ввод изображений.", + "cws.capability.imageInputUnavailable": "Сначала выберите все цели из каталога — неизвестные модели нельзя обработать через Vision Sidecar.", "cws.capability.imageInputHint": "Включено по умолчанию, если все цели поддерживают изображения. Выключите, чтобы принимать только текст.", + "cws.capability.imageInputSidecarHint": "Включено по умолчанию. При сохранении для {models} будет указана только текстовая модальность; изображения будут обрабатываться через Vision Sidecar.", + "cws.capability.imageInputSidecarDisabled": "Vision Sidecar выключен — {models} будет отклонять изображения, пока он не включён в настройках панели управления.", + "cws.capability.imageInputBlockedHint": "{models} не может быть покрыт Vision Sidecar: его входные модальности неизвестны или не содержат текст.", "cws.capability.imageInput": "Изображения / мультимодальность", "cws.capability.adaptiveEffort": "Адаптивная шкала рассуждений", "cws.capability.adaptiveEffortHint": "Выкл.: цель без настройки рассуждений скрывает выбор уровня для всей комбинации. Вкл.: такие цели остаются доступными, а в выборе сохраняются уровни, общие для остальных целей.", diff --git a/gui/src/i18n/tr.ts b/gui/src/i18n/tr.ts index a40d6f537e..2aede01b66 100644 --- a/gui/src/i18n/tr.ts +++ b/gui/src/i18n/tr.ts @@ -2945,8 +2945,11 @@ export const tr: Record = { "cws.field.defaultEffort": "Varsayılan akıl yürütme", "cws.field.defaultEffortNone": "Yok (hedef varsayılanı)", "cws.field.defaultEffortHint": "Yalnızca istemci akıl yürütme çabasını belirtmediğinde (atladığında) kullanılır. Seçenekler, seçilen hedeflerin duyurulan çabalarının kesişimidir; katalog çaba meta verisi olmayan hedefler hiçbir seçenek sunmaz.", - "cws.capability.imageInputUnavailable": "Seçilen tüm hedefler görsel girişini destekleyene kadar kullanılamaz.", + "cws.capability.imageInputUnavailable": "Önce tüm hedefleri katalogdan seçin — bilinmeyen modeller Vision Sidecar ile kapsanamaz.", "cws.capability.imageInputHint": "Tüm hedefler görselleri desteklediğinde varsayılan olarak açıktır. Yalnızca metin kabul etmek için kapatın.", + "cws.capability.imageInputSidecarHint": "Varsayılan olarak açık. {models} kaydedildiğinde yalnızca metin olarak bildirilir ve görseller için Vision Sidecar kullanır.", + "cws.capability.imageInputSidecarDisabled": "Vision Sidecar kapalı — {models}, gösterge paneli ayarlarından etkinleştirilene kadar görselleri reddeder.", + "cws.capability.imageInputBlockedHint": "{models} Vision Sidecar tarafından kapsanamaz: giriş modları bilinmiyor veya metin içermiyor.", "cws.capability.imageInput": "Görsel / çok modlu", "cws.capability.adaptiveEffort": "Uyarlanabilir akıl yürütme düzeyi", "cws.capability.adaptiveEffortHint": "Kapalı: akıl yürütme denetimi olmayan bir hedef, tüm kombinasyonun seçicisini gizler. Açık: bu hedefler kullanılabilir kalır ve seçici, kalan hedeflerin ortak düzeylerini gösterir.", diff --git a/gui/src/i18n/vi.ts b/gui/src/i18n/vi.ts index 766c7c39bc..1f9cecac6c 100644 --- a/gui/src/i18n/vi.ts +++ b/gui/src/i18n/vi.ts @@ -2937,6 +2937,9 @@ export const vi: Record = { "cws.field.defaultEffortHint": "Chỉ sử dụng khi client không chỉ định mức độ suy luận (reasoning effort). Các tùy chọn là giao điểm của các mức độ được công bố từ các mục tiêu đã chọn; những mục tiêu không có metadata về mức độ suy luận trong danh mục sẽ không cung cấp tùy chọn này.", "cws.capability.imageInputUnavailable": "Không khả dụng cho đến khi mọi mục tiêu được chọn đều hỗ trợ đầu vào hình ảnh (image input).", "cws.capability.imageInputHint": "Được bật theo mặc định khi mọi mục tiêu đều hỗ trợ hình ảnh. Tắt để chỉ chấp nhận văn bản.", + "cws.capability.imageInputSidecarHint": "Bật theo mặc định. {models} sẽ được khai báo chỉ-text khi lưu và dùng Vision Sidecar cho hình ảnh.", + "cws.capability.imageInputSidecarDisabled": "Vision Sidecar đang tắt — {models} sẽ từ chối hình ảnh cho đến khi được bật trong cài đặt bảng điều khiển.", + "cws.capability.imageInputBlockedHint": "{models} không thể được Vision Sidecar hỗ trợ: modalities đầu vào không rõ hoặc không có văn bản.", "cws.capability.imageInput": "Hình ảnh / đa phương thức (multimodal)", "cws.capability.adaptiveEffort": "Thang suy luận thích ứng (Adaptive reasoning ladder)", "cws.capability.adaptiveEffortHint": "Tắt: một mục tiêu không có kiểm soát suy luận sẽ ẩn đi bộ chọn mức độ (effort picker) cho toàn bộ combo. Bật: những mục tiêu đó vẫn có thể sử dụng và bộ chọn sẽ giữ lại các cấp độ mà các mục tiêu còn lại dùng chung.", diff --git a/gui/src/i18n/zh-TW.ts b/gui/src/i18n/zh-TW.ts index d01b4effeb..232f9bcd0b 100644 --- a/gui/src/i18n/zh-TW.ts +++ b/gui/src/i18n/zh-TW.ts @@ -2207,8 +2207,11 @@ export const zhTW: Record = { "cws.field.defaultEffort": "預設推理級別", "cws.field.defaultEffortNone": "無(使用目標預設)", "cws.field.defaultEffortHint": "僅在客戶端未指定推理級別時使用。客戶端值優先,每個目標會按自身能力進行處理。", - "cws.capability.imageInputUnavailable": "所有已選目標都支援圖片輸入後才可使用。", + "cws.capability.imageInputUnavailable": "請先從目錄選擇所有目標——未知模型無法由 Vision Sidecar 處理。", "cws.capability.imageInputHint": "所有目標都支援圖片時預設開啟;關閉後僅接受文字。", + "cws.capability.imageInputSidecarHint": "預設開啟。{models} 將在儲存時宣告為純文字,並使用 Vision Sidecar 處理圖片。", + "cws.capability.imageInputSidecarDisabled": "Vision Sidecar 已關閉——{models} 在儀表板設定中啟用它之前將拒絕圖片。", + "cws.capability.imageInputBlockedHint": "{models} 無法由 Vision Sidecar 覆蓋:其輸入模態未知或不含文字。", "cws.capability.imageInput": "圖片 / 多模態", "cws.capability.adaptiveEffort": "自適應推理層級", "cws.capability.adaptiveEffortHint": "關閉:只要有一個目標不支援推理層級,整個組合的選擇器都會消失。開啟:這些目標仍可使用,選擇器保留其餘目標共有的層級。", diff --git a/gui/src/i18n/zh.ts b/gui/src/i18n/zh.ts index 16c2d322da..7ab938f9e1 100644 --- a/gui/src/i18n/zh.ts +++ b/gui/src/i18n/zh.ts @@ -2923,8 +2923,11 @@ export const zh: Record = { "cws.field.defaultEffort": "默认推理级别", "cws.field.defaultEffortNone": "无(使用目标默认)", "cws.field.defaultEffortHint": "仅在客户端未指定推理级别时使用。选项为所选目标已公布努力级别的交集。", - "cws.capability.imageInputUnavailable": "所有已选目标均支持图片输入后才可用。", + "cws.capability.imageInputUnavailable": "请先从目录中选择所有目标——未知模型无法由 Vision Sidecar 处理。", "cws.capability.imageInputHint": "所有目标均支持图片时默认开启;关闭后仅接受文本。", + "cws.capability.imageInputSidecarHint": "默认开启。{models} 将在保存时声明为纯文本,并使用 Vision Sidecar 处理图片。", + "cws.capability.imageInputSidecarDisabled": "Vision Sidecar 已关闭——{models} 在仪表板设置中启用它之前将拒绝图片。", + "cws.capability.imageInputBlockedHint": "{models} 无法由 Vision Sidecar 覆盖:其输入模态未知或不含文本。", "cws.capability.imageInput": "图片 / 多模态", "cws.capability.adaptiveEffort": "自适应推理档位", "cws.capability.adaptiveEffortHint": "关闭:只要有一个目标不支持推理档位,整个组合的选择器都会消失。开启:这些目标仍可使用,选择器保留其余目标共有的档位。", diff --git a/gui/src/pages/Combos.tsx b/gui/src/pages/Combos.tsx index 33bba0f04e..f75c9f56f1 100644 --- a/gui/src/pages/Combos.tsx +++ b/gui/src/pages/Combos.tsx @@ -1,5 +1,6 @@ import { useCallback, useEffect, useMemo, useState } from "react"; import ComboWorkspace from "../components/ComboWorkspace"; +import { comboVisionSidecarTargets } from "../combo-capabilities"; import { type ComboItem, comboModelId, @@ -27,7 +28,14 @@ type ProviderOption = { adapter?: string; baseUrl?: string; }; -type ModelOption = { provider: string; id: string; namespaced?: string; reasoningEfforts?: string[]; inputModalities?: string[] }; +type ModelOption = { + provider: string; + id: string; + namespaced?: string; + reasoningEfforts?: string[]; + inputModalities?: string[]; + inputModalitiesDeclared?: string[]; +}; type ProviderDto = { adapter: string; baseUrl: string; @@ -160,6 +168,13 @@ export default function Combos({ const models: ModelOption[] = []; const catalogued = new Set(); + const parseModalities = (raw: unknown): string[] | undefined => + Array.isArray(raw) + ? raw + .filter((modality): modality is string => typeof modality === "string") + .map((modality) => modality.trim()) + .filter(Boolean) + : undefined; for (const row of modelRows) { if (!row || typeof row !== "object") continue; const model = row as { @@ -169,6 +184,7 @@ export default function Combos({ disabled?: unknown; reasoningEfforts?: unknown; inputModalities?: unknown; + inputModalitiesDeclared?: unknown; }; if (typeof model.provider !== "string" || typeof model.id !== "string") continue; const provider = model.provider.trim(); @@ -182,18 +198,15 @@ export default function Combos({ const reasoningEfforts = Array.isArray(model.reasoningEfforts) ? model.reasoningEfforts.filter((effort): effort is string => typeof effort === "string") : undefined; - const inputModalities = Array.isArray(model.inputModalities) - ? model.inputModalities - .filter((modality): modality is string => typeof modality === "string") - .map((modality) => modality.trim()) - .filter(Boolean) - : undefined; + const inputModalities = parseModalities(model.inputModalities); + const inputModalitiesDeclared = parseModalities(model.inputModalitiesDeclared); models.push({ provider, id, namespaced: typeof model.namespaced === "string" ? model.namespaced : undefined, ...(reasoningEfforts ? { reasoningEfforts } : {}), ...(inputModalities && inputModalities.length > 0 ? { inputModalities } : {}), + ...(inputModalitiesDeclared && inputModalitiesDeclared.length > 0 ? { inputModalitiesDeclared } : {}), }); } @@ -234,6 +247,28 @@ export default function Combos({ ); const { state } = resource; + /* + * Vision Sidecar enabled state for the enrollment warning. Deliberately + * separate from the workspace payload: a failure here must never block or + * fail the combo workspace, so errors are swallowed and `undefined` + * (unknown) renders no warning. + */ + const [visionEnabled, setVisionEnabled] = useState(undefined); + useEffect(() => { + if (!active) return; + let cancelled = false; + fetch(`${apiBase}/api/sidecar-settings`) + .then((res) => (res.ok ? res.json() : null)) + .then((data: unknown) => { + if (cancelled || !data || typeof data !== "object") return; + const vision = (data as { vision?: { enabled?: unknown } }).vision; + if (!vision || typeof vision !== "object") return; + setVisionEnabled(vision.enabled !== false); + }) + .catch(() => { /* unknown stays warning-free */ }); + return () => { cancelled = true; }; + }, [active, apiBase]); + const [quotaNow, setQuotaClock] = useState(() => Date.now()); const loadProviderQuotas = useCallback(async (signal?: AbortSignal): Promise => { const response = await fetch(`${apiBase}/api/provider-quotas`, { signal }); @@ -293,7 +328,14 @@ export default function Combos({ const res = await fetch(`${apiBase}/api/combos`, { method: "PUT", headers: { "content-type": "application/json" }, - body: JSON.stringify(toPutBody(item, renameFrom ? { renameFrom } : {})), + body: JSON.stringify(toPutBody(item, { + ...(renameFrom ? { renameFrom } : {}), + // Enabling images declares text-only members for the Vision Sidecar so the + // operator never has to hand-edit provider config for a multimodal combo. + ...(item.imageInput !== "disabled" + ? { visionSidecarTargets: comboVisionSidecarTargets(item.targets, models) } + : {}), + })), }); const data = res.ok ? await res.json() as unknown @@ -388,6 +430,7 @@ export default function Combos({ providers={providers} models={models} cataloguedComboIds={cataloguedComboIds} + visionEnabled={visionEnabled} loading={false} onRefresh={() => { resource.refresh(); quotaResource.refresh(); }} onSave={saveCombo} diff --git a/src/server/management/combo-routes.ts b/src/server/management/combo-routes.ts index 3c802aba67..8dea28f617 100644 --- a/src/server/management/combo-routes.ts +++ b/src/server/management/combo-routes.ts @@ -52,6 +52,14 @@ import { setDebugSettings, type DebugFlag, } from "../../lib/debug-settings"; +import { mergeModelCapabilities, modelCapabilitiesConfigError } from "../../config/provider-validation"; +import { commitProviderPatch } from "./provider-patch-transaction"; +import { ConfigWritePublishedError } from "../../config/persist-unlocked"; +import { + customRowInputModalities, + isVisionSidecarConsumer, + modelAcceptsImageInput, +} from "../../vision/eligibility"; import type { OcxClaudeCodeConfig, OcxComboConfig, OcxConfig, OcxCustomModel, OcxProviderConfig, OcxComboCooldownWaitPolicy } from "../../types"; import { drainAndShutdown } from "../lifecycle"; import { reconcileLiveStateStores } from "../../lib/state-store-registrations"; @@ -114,6 +122,43 @@ function sparseComboConfig, +): { targets?: VisionSidecarTarget[]; error?: string } { + if (raw === undefined) return {}; + if (!Array.isArray(raw)) return { error: "visionSidecarTargets must be an array" }; + const seen = new Set(); + const targets: VisionSidecarTarget[] = []; + for (const entry of raw) { + if (!isPlainRecord(entry)) return { error: "visionSidecarTargets entries must be objects" }; + const provider = typeof entry.provider === "string" ? entry.provider.trim() : ""; + const model = typeof entry.model === "string" ? entry.model.trim() : ""; + if (!provider || !model) { + return { error: "visionSidecarTargets entries must have nonblank provider and model" }; + } + const key = `${provider}/${model}`; + if (!comboTargetKeys.has(key)) { + return { error: `visionSidecarTargets entry "${key}" is not a target of this combo` }; + } + if (seen.has(key)) continue; + seen.add(key); + targets.push({ provider, model }); + } + return { targets }; +} + export async function handleComboRoutes(ctx: ManagementContext): Promise { const { req, url, config, deps, convergeCodexCatalog, syncClaudeAgentDefsBestEffort } = ctx; @@ -229,6 +274,46 @@ export async function handleComboRoutes(ctx: ManagementContext): Promise `${target.provider}/${target.model}`)), + ); + if (sidecarParsed.error) return jsonResponse({ error: sidecarParsed.error }, 400); + const sidecarPatches = new Map>(); + if (sidecarParsed.targets?.length) { + if (normalized.imageInput === "disabled") { + return jsonResponse({ error: "visionSidecarTargets requires imageInput not disabled" }, 400); + } + for (const { provider, model } of sidecarParsed.targets) { + const providerRow = config.providers?.[provider]; + if (!providerRow) { + return jsonResponse({ error: `visionSidecarTargets entry "${provider}/${model}" has no configured provider` }, 400); + } + // A text-only declaration would HIDE a real capability; never overwrite one. + if (modelAcceptsImageInput(config, { provider, id: model }) === true) { + return jsonResponse({ error: `visionSidecarTargets entry "${provider}/${model}" already accepts image input` }, 400); + } + // Read the declaration the way the runtime does: exact capability axis, then the + // operator's custom row. A row without text (audio-only) cannot be widened by the sidecar. + const declared = Object.hasOwn(providerRow.modelCapabilities ?? {}, model) + ? providerRow.modelCapabilities?.[model]?.inputModalities + : customRowInputModalities(config, provider, model); + if (declared !== undefined && !declared.includes("text")) { + return jsonResponse({ error: `visionSidecarTargets entry "${provider}/${model}" is declared without text input; the Vision Sidecar cannot cover it` }, 400); + } + // Already a consumer through any runtime source (exact declaration, custom row, + // noVisionModels, legacy record, registry enrichment): the catalog already + // advertises image for it. Writing the declaration again would be a no-op write. + if (isVisionSidecarConsumer(config, provider, model)) continue; + const patch = sidecarPatches.get(provider) ?? {}; + patch[model] = { inputModalities: ["text"] }; + sidecarPatches.set(provider, patch); + } + for (const patch of sidecarPatches.values()) { + const error = modelCapabilitiesConfigError(patch); + if (error) return jsonResponse({ error }, 400); + } + } // Persist only non-default identity/capability fields so config stays sparse. // Capability defaults (`imageInput`, `reasoningEffortMode`) go through the same // helper the GET/PUT responses use, so the wire shape and the stored shape cannot drift. @@ -291,58 +376,76 @@ export async function handleComboRoutes(ctx: ManagementContext): Promise 0) { - const migrateReference = (model: string): string => migratedModels.get(model) ?? model; - const migrateAgentReference = (model: string): string => { - const migrated = migrateReference(model); - if (migrated !== model) shouldSyncClaudeAgentDefs = true; - return migrated; - }; - if (config.subagentModels) { - config.subagentModels = [...new Set(config.subagentModels.map(migrateAgentReference))]; - } - if (config.injectionModel && migratedModels.has(config.injectionModel)) { - config.injectionModel = migrateReference(config.injectionModel); - } - if (config.shadowCallIntercept?.model && migratedModels.has(config.shadowCallIntercept.model)) { - config.shadowCallIntercept = { - ...config.shadowCallIntercept, - model: migrateReference(config.shadowCallIntercept.model), - }; - } - if (config.claudeCode) { - const claudeCode = { ...config.claudeCode }; - for (const field of ["model", "smallFastModel"] as const) { - if (claudeCode[field]) claudeCode[field] = migrateAgentReference(claudeCode[field]); - } - if (claudeCode.tierModels) { - claudeCode.tierModels = Object.fromEntries( - Object.entries(claudeCode.tierModels).map(([tier, model]) => [tier, migrateAgentReference(model)]), - ); + // One atomic transaction: a failed save must restore the combo, the capability + // declarations, and every migrated reference together, never partially. + let publicationError = false; + try { + commitProviderPatch(config, () => { + config.combos = nextCombos; + if (migratedModels.size > 0) { + const migrateReference = (model: string): string => migratedModels.get(model) ?? model; + const migrateAgentReference = (model: string): string => { + const migrated = migrateReference(model); + if (migrated !== model) shouldSyncClaudeAgentDefs = true; + return migrated; + }; + if (config.subagentModels) { + config.subagentModels = [...new Set(config.subagentModels.map(migrateAgentReference))]; + } + if (config.injectionModel && migratedModels.has(config.injectionModel)) { + config.injectionModel = migrateReference(config.injectionModel); + } + if (config.shadowCallIntercept?.model && migratedModels.has(config.shadowCallIntercept.model)) { + config.shadowCallIntercept = { + ...config.shadowCallIntercept, + model: migrateReference(config.shadowCallIntercept.model), + }; + } + if (config.claudeCode) { + const claudeCode = { ...config.claudeCode }; + for (const field of ["model", "smallFastModel"] as const) { + if (claudeCode[field]) claudeCode[field] = migrateAgentReference(claudeCode[field]); + } + if (claudeCode.tierModels) { + claudeCode.tierModels = Object.fromEntries( + Object.entries(claudeCode.tierModels).map(([tier, model]) => [tier, migrateAgentReference(model)]), + ); + } + if (claudeCode.modelMap) { + claudeCode.modelMap = Object.fromEntries( + Object.entries(claudeCode.modelMap).map(([source, model]) => [source, migrateAgentReference(model)]), + ); + } + if (claudeCode.intercept?.modelMap) { + claudeCode.intercept = { + ...claudeCode.intercept, + modelMap: Object.fromEntries( + Object.entries(claudeCode.intercept.modelMap).map(([pickerId, route]) => [pickerId, migrateAgentReference(route)]), + ), + }; + } + config.claudeCode = claudeCode; + } } - if (claudeCode.modelMap) { - claudeCode.modelMap = Object.fromEntries( - Object.entries(claudeCode.modelMap).map(([source, model]) => [source, migrateAgentReference(model)]), - ); + if (oldDisabledSelectors.size > 0 && config.disabledModels) { + config.disabledModels = [...new Set(config.disabledModels.map(model => ( + oldDisabledSelectors.has(model) ? newDisabledModel : model + )))]; } - if (claudeCode.intercept?.modelMap) { - claudeCode.intercept = { - ...claudeCode.intercept, - modelMap: Object.fromEntries( - Object.entries(claudeCode.intercept.modelMap).map(([pickerId, route]) => [pickerId, migrateAgentReference(route)]), - ), - }; + for (const [providerName, patch] of sidecarPatches) { + const provider = config.providers?.[providerName]; + if (!provider) continue; + const capabilities = mergeModelCapabilities(provider.modelCapabilities, patch); + if (capabilities === undefined) delete provider.modelCapabilities; + else provider.modelCapabilities = capabilities; } - config.claudeCode = claudeCode; + }, deps.saveConfigPreservingClaudeCode ?? saveConfigPreservingClaudeCode); + } catch (error) { + if (!(error instanceof ConfigWritePublishedError)) { + return jsonResponse({ error: "combo could not be saved" }, 500); } + publicationError = true; } - if (oldDisabledSelectors.size > 0 && config.disabledModels) { - config.disabledModels = [...new Set(config.disabledModels.map(model => ( - oldDisabledSelectors.has(model) ? newDisabledModel : model - )))]; - } - saveConfigPreservingClaudeCode(config); reconcileLiveStateStores(); clearComboSelectionState(id); clearComboTargetCooldowns(id); @@ -350,7 +453,9 @@ export async function handleComboRoutes(ctx: ManagementContext): Promise { )).toBe(false); }); - test("returns true only when every complete target advertises image", () => { + test("returns true when every known target advertises image", () => { const models = [ { provider: "a", id: "m1", inputModalities: ["text", "image"] }, { provider: "b", id: "m2", inputModalities: ["text", "image"] }, @@ -857,20 +857,90 @@ describe("comboImagesSupported", () => { )).toBe(true); }); - test("returns false when any target is missing from the catalog or lacks image", () => { + test("allows text-only members for sidecar coverage but fails closed on unknown, silent, and text-free rows", () => { const models = [ { provider: "a", id: "m1", inputModalities: ["text", "image"] }, { provider: "b", id: "m2", inputModalities: ["text"] }, ]; + // Known text-only rows can be declared text-only on save; only catalog-absent rows block. expect(comboImagesSupported( [{ provider: "a", model: "m1" }, { provider: "b", model: "m2" }], models, + )).toBe(true); + // No known modalities: the runtime cannot prove text input, so enrollment is refused. + expect(comboImagesSupported( + [{ provider: "a", model: "m1" }, { provider: "c", model: "m3" }], + [...models, { provider: "c", id: "m3" }], + )).toBe(false); + // Audio-only rows have no text input; the sidecar cannot cover them. + expect(comboImagesSupported( + [{ provider: "a", model: "m1" }, { provider: "c", model: "m3" }], + [...models, { provider: "c", id: "m3", inputModalities: ["audio"] }], )).toBe(false); expect(comboImagesSupported( [{ provider: "a", model: "m1" }, { provider: "b", model: "ghost" }], models, )).toBe(false); }); + + test("classifies from the declared modalities so enrollment survives reload", () => { + // After enrollment the catalog WIDENS the row to image, but the declaration + // still says text-only: reload must keep it a sidecar member, not native vision. + const reloaded = [ + { provider: "a", id: "m1", inputModalities: ["text", "image"], inputModalitiesDeclared: ["text"] }, + ]; + expect(comboImagesSupported([{ provider: "a", model: "m1" }], reloaded)).toBe(true); + expect(comboVisionSidecarTargets([{ provider: "a", model: "m1" }], reloaded)) + .toEqual([{ provider: "a", model: "m1" }]); + }); +}); + +describe("comboVisionSidecarTargets", () => { + test("returns only known text-only members, deduplicated", () => { + const models = [ + { provider: "a", id: "m1", inputModalities: ["text", "image"] }, + { provider: "b", id: "m2", inputModalities: ["text"] }, + { provider: "c", id: "m3" }, + ]; + expect(comboVisionSidecarTargets( + [ + { provider: "a", model: "m1" }, + { provider: "b", model: "m2" }, + { provider: "b", model: "m2" }, + { provider: "c", model: "m3" }, + ], + models, + )).toEqual([ + { provider: "b", model: "m2" }, + ]); + }); + + test("ignores incomplete or catalog-absent targets", () => { + expect(comboVisionSidecarTargets( + [{ provider: "", model: "" }, { provider: "b", model: "ghost" }], + [{ provider: "b", id: "m2", inputModalities: ["text"] }], + )).toEqual([]); + }); + + test("comboImageBlockedTargets names the members that refuse coverage", () => { + const models = [ + { provider: "a", id: "m1", inputModalities: ["text", "image"] }, + { provider: "b", id: "m2" }, + { provider: "c", id: "m3", inputModalities: ["audio"] }, + ]; + expect(comboImageBlockedTargets( + [ + { provider: "a", model: "m1" }, + { provider: "b", model: "m2" }, + { provider: "c", model: "m3" }, + { provider: "b", model: "ghost" }, + ], + models, + )).toEqual([ + { provider: "b", model: "m2" }, + { provider: "c", model: "m3" }, + ]); + }); }); describe("combo imageInput draft persistence", () => { @@ -900,4 +970,15 @@ describe("combo imageInput draft persistence", () => { const disabled = { ...auto, imageInput: "disabled" as const }; expect(toPutBody(disabled).combo.imageInput).toBe("disabled"); }); + + test("toPutBody carries visionSidecarTargets as a top-level request-only field", () => { + const auto = emptyDraft("x"); + auto.targets = [{ provider: "a", model: "m1" }]; + const sidecar = [{ provider: "a", model: "m1" }]; + const body = toPutBody(auto, { visionSidecarTargets: sidecar }); + expect(body.visionSidecarTargets).toEqual(sidecar); + expect(body.combo).not.toHaveProperty("visionSidecarTargets"); + // Empty lists stay off the wire. + expect(toPutBody(auto, { visionSidecarTargets: [] })).not.toHaveProperty("visionSidecarTargets"); + }); }); diff --git a/tests/routing/combo-management-api.test.ts b/tests/routing/combo-management-api.test.ts index 2251559454..72eff49f18 100644 --- a/tests/routing/combo-management-api.test.ts +++ b/tests/routing/combo-management-api.test.ts @@ -39,6 +39,7 @@ import { handleManagementAPI } from "../../src/server/management-api"; import { handleResponses } from "../../src/server/responses"; import type { OcxConfig } from "../../src/types"; import { syncCatalogModels } from "../../src/codex/catalog"; +import { isModelVisionSidecarConsumer } from "../../src/vision/eligibility"; import { injectClaudeAgentDefs } from "../../src/claude/agents-inject"; import { catalogConvergenceFactory } from "../helpers/catalog-convergence"; import { acquireOwnedSpendHome } from "../helpers/owned-spend-home"; @@ -484,6 +485,213 @@ describe("combo management API", () => { }); }); + test("PUT visionSidecarTargets declares exact members text-only and never persists the request field", async () => { + await withTempHome(async () => { + const base = baseConfig(); + const config = baseConfig({ + providers: { + ...base.providers, + b: { + ...base.providers.b!, + modelCapabilities: { m2: { contextTier: "long_context" } }, + }, + }, + combos: undefined, + }); + saveConfig(config); + const response = await comboApi(config, "PUT", "/api/combos", { + id: "mixed", + visionSidecarTargets: [ + { provider: "b", model: "m2" }, + { provider: "b", model: "m2" }, // exact duplicates are deduplicated + ], + combo: { + targets: [ + { provider: "a", model: "m1" }, + { provider: "b", model: "m2" }, + ], + }, + }); + expect(response?.status).toBe(200); + // Exact text-only declaration added; the sibling capability axis survives. + expect(config.providers?.b?.modelCapabilities).toEqual({ + m2: { contextTier: "long_context", inputModalities: ["text"] }, + }); + // Untouched member provider gains no declaration. + expect(config.providers?.a?.modelCapabilities).toBeUndefined(); + // Request-only field never leaks into the persisted combo. + expect(config.combos?.mixed).not.toHaveProperty("visionSidecarTargets"); + }); + }); + + test("PUT rejects invalid visionSidecarTargets without mutating config", async () => { + await withTempHome(async () => { + const config = baseConfig({ combos: undefined }); + saveConfig(config); + const before = readFileSync(getConfigPath(), "utf8"); + const bodies: unknown[] = [ + { id: "x", visionSidecarTargets: "nope", combo: VALID_COMBO }, + { id: "x", visionSidecarTargets: [{ provider: "a" }], combo: VALID_COMBO }, + // Not a target of the submitted combo. + { id: "x", visionSidecarTargets: [{ provider: "c", model: "m3" }], combo: VALID_COMBO }, + // Enrollment only makes sense while images are accepted. + { + id: "x", + visionSidecarTargets: [{ provider: "a", model: "m1" }], + combo: { targets: [{ provider: "a", model: "m1" }], imageInput: "disabled" }, + }, + ]; + for (const body of bodies) { + const response = await comboApi(config, "PUT", "/api/combos", body); + expect(response?.status).toBe(400); + } + expect(readFileSync(getConfigPath(), "utf8")).toBe(before); + expect(config.providers?.a?.modelCapabilities).toBeUndefined(); + }); + }); + + test("PUT rejects already-image-capable and audio-only sidecar targets instead of overwriting", async () => { + await withTempHome(async () => { + const cases: Array<{ name: string; config: OcxConfig }> = [ + { + name: "capability axis already declares image", + config: baseConfig({ + providers: { + ...baseConfig().providers, + b: { ...baseConfig().providers.b!, modelCapabilities: { m2: { inputModalities: ["text", "image"] } } }, + }, + combos: undefined, + }), + }, + { + name: "capability axis declares audio only", + config: baseConfig({ + providers: { + ...baseConfig().providers, + b: { ...baseConfig().providers.b!, modelCapabilities: { m2: { inputModalities: ["audio"] } } }, + }, + combos: undefined, + }), + }, + { + name: "custom row declares audio only", + config: baseConfig({ + customModels: [{ id: "custom-b-m2", provider: "b", modelId: "m2", inputModalities: ["audio"] }], + combos: undefined, + }), + }, + ]; + for (const { config } of cases) { + saveConfig(config); + const before = readFileSync(getConfigPath(), "utf8"); + const response = await comboApi(config, "PUT", "/api/combos", { + id: "mixed", + visionSidecarTargets: [{ provider: "b", model: "m2" }], + combo: { + targets: [ + { provider: "a", model: "m1" }, + { provider: "b", model: "m2" }, + ], + }, + }); + expect(response?.status).toBe(400); + expect(config.combos).toBeUndefined(); + expect(readFileSync(getConfigPath(), "utf8")).toBe(before); + } + }); + }); + + test("PUT treats an already-covered member as a no-op write", async () => { + await withTempHome(async () => { + const exactTextOnly = baseConfig({ + providers: { + ...baseConfig().providers, + b: { ...baseConfig().providers.b!, modelCapabilities: { m2: { inputModalities: ["text"] } } }, + }, + combos: undefined, + }); + saveConfig(exactTextOnly); + const viaCapability = await comboApi(exactTextOnly, "PUT", "/api/combos", { + id: "mixed", + visionSidecarTargets: [{ provider: "b", model: "m2" }], + combo: { targets: [{ provider: "b", model: "m2" }] }, + }); + expect(viaCapability?.status).toBe(200); + // The operator's declaration is preserved verbatim, never rewritten. + expect(exactTextOnly.providers?.b?.modelCapabilities).toEqual({ m2: { inputModalities: ["text"] } }); + expect(isModelVisionSidecarConsumer(exactTextOnly.providers!.b!, "m2")).toBe(true); + + const viaNoVision = baseConfig({ + providers: { + ...baseConfig().providers, + b: { ...baseConfig().providers.b!, noVisionModels: ["m2"] }, + }, + combos: undefined, + }); + saveConfig(viaNoVision); + const response = await comboApi(viaNoVision, "PUT", "/api/combos", { + id: "mixed", + visionSidecarTargets: [{ provider: "b", model: "m2" }], + combo: { targets: [{ provider: "b", model: "m2" }] }, + }); + expect(response?.status).toBe(200); + expect(viaNoVision.providers?.b?.modelCapabilities).toBeUndefined(); + expect(isModelVisionSidecarConsumer(viaNoVision.providers!.b!, "m2")).toBe(true); + }); + }); + + test("PUT enrolls the member so the runtime treats it as a sidecar consumer", async () => { + await withTempHome(async () => { + const config = baseConfig({ combos: undefined }); + saveConfig(config); + const response = await comboApi(config, "PUT", "/api/combos", { + id: "mixed", + visionSidecarTargets: [{ provider: "b", model: "m2" }], + combo: { targets: [{ provider: "b", model: "m2" }] }, + }); + expect(response?.status).toBe(200); + // The exact declaration the catalog's consumer check reads: text without image. + expect(isModelVisionSidecarConsumer(config.providers!.b!, "m2")).toBe(true); + const persisted = JSON.parse(readFileSync(getConfigPath(), "utf8")) as OcxConfig; + expect(persisted.providers?.b?.modelCapabilities?.m2?.inputModalities).toEqual(["text"]); + }); + }); + + test("PUT rolls back combo, declarations, and migrated references when the save fails", async () => { + await withTempHome(async () => { + const config = baseConfig({ + combos: { + old: { strategy: "failover", targets: [{ provider: "a", model: "m1" }, { provider: "b", model: "m2" }] }, + }, + subagentModels: ["combo/old"], + }); + saveConfig(config); + const before = readFileSync(getConfigPath(), "utf8"); + const req = new Request("http://localhost/api/combos", { + method: "PUT", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + id: "new", + renameFrom: "old", + visionSidecarTargets: [{ provider: "b", model: "m2" }], + combo: { targets: [{ provider: "a", model: "m1" }, { provider: "b", model: "m2" }] }, + }), + }); + const response = await handleManagementAPI(req, new URL(req.url), config, { + createManagementConvergeCodex: catalogConvergenceFactory(), + saveConfigPreservingClaudeCode: () => { throw new Error("disk full"); }, + }); + expect(response?.status).toBe(500); + expect(await responseJson(response)).toEqual({ error: "combo could not be saved" }); + // Nothing moved: the combo keeps its old key, references stay unmigrated, + // and the sidecar declaration was never applied. + expect(Object.keys(config.combos!)).toEqual(["old"]); + expect(config.subagentModels).toEqual(["combo/old"]); + expect(config.providers?.b?.modelCapabilities).toBeUndefined(); + expect(readFileSync(getConfigPath(), "utf8")).toBe(before); + }); + }); + test("reasoningEffortMode survives a management round-trip and stays sparse when strict", async () => { await withTempHome(async () => { const config = baseConfig({ combos: undefined });