Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Loading
Sorry, something went wrong. Reload?
Sorry, we cannot display this file.
Sorry, this file is invalid so it cannot be displayed.
3 changes: 1 addition & 2 deletions gui/src/pages/Models.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -296,9 +296,8 @@ export default function Models({ apiBase, restartEpoch = 0, catalogSyncedAt, rep
pickerFlight.current?.controller.abort();
pickerFlight.current?.clear();
pickerFlight.current = null;
cancelAppServerRead();
};
}, [apiBase, catalogActive, cancelAppServerRead]);
}, [apiBase, catalogActive]);
useLayoutEffect(() => {
// Pin inferred Custom before any late GET can switch mode and unmount its draft.
if (catalogActive && pickerDraft === null && pickerMode === "custom") setPickerDraft("custom");
Expand Down
25 changes: 25 additions & 0 deletions gui/tests/models-status-toast.test.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -613,6 +613,31 @@ test("leaving Models aborts its pending picker save", async () => {
expect(container.querySelector(".action-toast")).toBeNull();
});

test("changing Models tabs preserves the pending app-server status read", async () => {
const baseFetch = globalThis.fetch;
let statusSignal: AbortSignal | null | undefined;
let releaseStatus!: (response: Response) => void;
globalThis.fetch = (async (input, init) => {
if (String(input).endsWith("/api/system/codex-app-server")) {
statusSignal = init?.signal;
return new Promise<Response>(resolve => { releaseStatus = resolve; });
}
return baseFetch(input, init);
}) as typeof fetch;

await mountModelsForRefreshWarning();
await waitForModelsFeedback(() => releaseStatus !== undefined);
const combosTab = [...container.querySelectorAll<HTMLButtonElement>('[role="tab"]')]
.find(button => button.textContent?.startsWith("Combos"));
expect(combosTab).toBeDefined();
await act(async () => { combosTab!.click(); });

expect(statusSignal?.aborted).toBe(false);
await act(async () => { releaseStatus(Response.json({ state: "stale", runningCount: 1 })); });
await waitForModelsFeedback(() => container.querySelector(".codex-stale-banner") !== null);
expect(container.querySelector(".codex-stale-banner")).not.toBeNull();
});


function holdPostSaveAppServerRead() {
const baseFetch = globalThis.fetch;
Expand Down
1 change: 1 addition & 0 deletions scripts/test-layout/layout.json
Original file line number Diff line number Diff line change
Expand Up @@ -335,6 +335,7 @@
"catalog-zero-credit-picker.test.ts": "codex-integration",
"chat-completions-deferred-tools.test.ts": "responses",
"chat-completions-endpoint.test.ts": "responses",
"chat-native-spend.test.ts": "responses",
"chat-conversation-affinity.test.ts": "responses",
"chat-inbound-reasoning-none.test.ts": "responses",
"chat-inbound-reasoning-replay.test.ts": "responses",
Expand Down
17 changes: 12 additions & 5 deletions src/cli/access.ts
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,13 @@ const USAGE = `Usage:
*/
function formatKeyRows(payload: Record<string, unknown>, keys: Array<Record<string, unknown>>): string[] {
const cells: string[][] = [["ID", "NAME", "PREFIX", "REQ 7D", "TOTAL", "LAST USED"]];
// A string that does not parse is not attribution data: treat it like an absent
// field so malformed payloads still render "unavailable" instead of usage values.
const attributionSince = typeof payload.attributionSince === "string"
&& !Number.isNaN(Date.parse(payload.attributionSince))
? payload.attributionSince
: undefined;
const usageAvailable = attributionSince !== undefined;
for (const entry of keys) {
const usage = (entry.usage ?? {}) as Record<string, unknown>;
const ambiguous = usage.ambiguous === true;
Expand All @@ -47,16 +54,16 @@ function formatKeyRows(payload: Record<string, unknown>, keys: Array<Record<stri
String(entry.name ?? ""),
String(entry.prefix ?? ""),
// One marker spanning both numeric columns: the union guarantees neither exists.
ambiguous ? "ambiguous" : num(usage.requests7d),
ambiguous ? "" : num(usage.totalRequests),
ambiguous ? "" : (typeof usage.lastUsedAt === "string" ? usage.lastUsedAt : "never"),
!usageAvailable ? "unavailable" : ambiguous ? "ambiguous" : num(usage.requests7d),
!usageAvailable || ambiguous ? "" : num(usage.totalRequests),
!usageAvailable || ambiguous ? "" : (typeof usage.lastUsedAt === "string" ? usage.lastUsedAt : "never"),
]);
}
const widths = cells[0]!.map((_, column) => Math.max(...cells.map(row => (row[column] ?? "").length)));
const lines = cells.map(row => row.map((cell, i) => (cell ?? "").padEnd(widths[i]!)).join(" ").trimEnd());
const footer: string[] = [];
if (typeof payload.attributionSince === "string") {
footer.push(`attribution since ${payload.attributionSince}`);
if (attributionSince !== undefined) {
footer.push(`attribution since ${attributionSince}`);
}
if (payload.historyTruncated === true) {
footer.push("older history truncated");
Expand Down
11 changes: 8 additions & 3 deletions src/providers/label.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,8 @@ export function canonicalUsageProviderLabel(provider: string): string {
return provider === "chatgpt" || provider === "openai-multi" ? "openai" : provider;
}

const LEGACY_MAIN_ACCOUNT_PROVIDER_LABELS = new Set(["openai-main", "chatgpt-main", "openai-multi-main"]);

export function usesApiKeyAccount(provider: Pick<OcxProviderConfig, "authMode" | "_apiKeyAttempt">): boolean {
return provider.authMode === "key"
|| (provider.authMode === undefined && !!provider._apiKeyAttempt?.reference);
Expand All @@ -29,11 +31,14 @@ export function baseProviderLabel(provider: string): string {
const cut = provider.lastIndexOf("-");
if (cut <= 0) return canonicalUsageProviderLabel(provider);
const suffix = provider.slice(cut + 1);
// `-main` is the legacy log label for the main Codex account (MAIN_CODEX_ACCOUNT_ID). New entries
// log under the base provider name, but historical `<provider>-main` entries must still collapse.
// `-main` was the legacy log label for the main Codex account (MAIN_CODEX_ACCOUNT_ID). Restrict
// that compatibility mapping to the known Codex provider labels so configured providers whose
// names naturally end in `-main` remain distinct.
// ChatGPT auth-pool and OpenAI passthrough are the same Codex/OpenAI usage surface, so display
// summaries normalize them to one `openai` row after recognized main/pool suffixes are removed.
if (suffix === "main") return canonicalUsageProviderLabel(provider.slice(0, cut));
if (LEGACY_MAIN_ACCOUNT_PROVIDER_LABELS.has(provider)) {
return canonicalUsageProviderLabel(provider.slice(0, cut));
}
return CODEX_ACCOUNT_LOG_LABEL_RE.test(suffix) ? canonicalUsageProviderLabel(provider.slice(0, cut)) : provider;
}

Expand Down
5 changes: 4 additions & 1 deletion src/routing/history/indexer.ts
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@ import { getConfigDir } from "../../config";
import { recordOwnedConfigPath } from "../../lib/config-ownership";
import {
currentUsageLogRevision,
encodePersistedRequestedModel,
normalizeUsageEntryForTest,
usageLogPath,
type PersistedUsageEntry,
Expand Down Expand Up @@ -510,7 +511,9 @@ function queryRows(
};
if (filters.provider !== undefined) add("provider = ?", filters.provider);
if (filters.model !== undefined) add("model = ?", filters.model);
if (filters.requestedModel !== undefined) add("requested_model = ?", filters.requestedModel);
// Rows store the bounded encoded form, so the lookup value must be encoded the
// same way — short selectors encode to themselves and still match verbatim.
if (filters.requestedModel !== undefined) add("requested_model = ?", encodePersistedRequestedModel(filters.requestedModel));
if (filters.status !== undefined) add("status = ?", filters.status);
if (filters.conversationId !== undefined) add("conversation_id = ?", filters.conversationId);
if (filters.surface !== undefined) add("surface = ?", filters.surface);
Expand Down
13 changes: 12 additions & 1 deletion src/server/chat-completions.ts
Original file line number Diff line number Diff line change
Expand Up @@ -199,7 +199,18 @@ async function handleChatCompletionsWithBudget(
}
// Combos must enter the Responses routing path so child selection, forced default
// effort, failover, and per-attempt telemetry run before any native Chat send.
if (!route.combo && !effortRow && isNativeChatRouteEligible(route, chatBody, config)) chatNativeRoute = route;
if (!route.combo && !effortRow && isNativeChatRouteEligible(route, chatBody, config)) {
chatNativeRoute = route;
if (logCtx.usageLogInputTokens === undefined) {
const parts = [JSON.stringify(chatBody.messages ?? [])];
if (chatBody.tools !== undefined) parts.push(JSON.stringify(chatBody.tools));
logCtx.usageLogInputTokens = Math.max(1, estimateTokens(parts.join("\n"), requestedModel));
}
const outputCeiling = chatBody.max_completion_tokens ?? chatBody.max_tokens;
if (typeof outputCeiling === "number" && outputCeiling > 0) {
logCtx.spendOutputCeilingTokens = Math.trunc(outputCeiling);
}
}
} catch (err) {
if (err instanceof AdmissionModelDeniedError) {
logCtx.requestedModel = requestedModel;
Expand Down
15 changes: 15 additions & 0 deletions src/server/chat-native.ts
Original file line number Diff line number Diff line change
Expand Up @@ -48,6 +48,7 @@ import {
transientRetryPolicyFor,
} from "../providers/key-failover";
import { fastPolicyForModel } from "../providers/service-tier";
import { stampApiKeyAccountLabel } from "../providers/label";
import { providerApiKeySelectionIsCurrent, resolveCurrentProviderApiKeyTransport } from "../providers/api-key-selection";
import { enrichOpenCodeZenFreeTierMessage } from "../providers/opencode-zen-rate-limit";
import type { OcxProviderTransport } from "../providers/xai-transport";
Expand All @@ -68,12 +69,16 @@ import {
} from "./request-log";
import { jsonCompletionSse, nativeChatSse, structuredError, usageFromChat } from "./chat-native-sse";
import { registerTurn, unregisterTurn } from "./lifecycle";
import { attachRequestSpendTracker } from "./responses/request-spend";
import { workflowRefusalResponse } from "./workflow-refusal";

type Rec = Record<string, unknown>;

const MAX_NATIVE_CHAT_JSON_BYTES = 32 * 1024 * 1024;
const MAX_NATIVE_CHAT_ERROR_BYTES = 64 * 1024;

class NativeChatSpendRefusal extends Error {}

const chatEffortSnapshots = new WeakMap<Rec, {
inputModel: string;
providerName: string;
Expand Down Expand Up @@ -262,6 +267,8 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
const proactiveKeyProvider = selectProactiveApiKeyTransport(config, route.providerName, route.provider);
if (proactiveKeyProvider) route.provider = proactiveKeyProvider;
let activeProvider: OcxProviderConfig = route.provider;
stampApiKeyAccountLabel(logCtx, route.providerName, activeProvider);
const spendTracker = attachRequestSpendTracker(req, logCtx);
let activeAdapter: ProviderAdapter = createOpenAIChatAdapter(activeProvider);
let activeRequest: AdapterRequest;
let retainedRequestBytes = 0;
Expand Down Expand Up @@ -339,6 +346,7 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
throw new Error("Provider key selection is no longer available for native Chat");
}
activeProvider = current;
stampApiKeyAccountLabel(logCtx, route.providerName, activeProvider);
activeAdapter = createOpenAIChatAdapter(current);
activeRequest.releaseBodyObservation?.();
releaseRetainedRequest();
Expand All @@ -353,6 +361,7 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
const encoding = new Headers(init.headers).get("accept-encoding");
if (!headers.has("accept-encoding") && encoding) headers.set("accept-encoding", encoding);
if (init.signal?.aborted) throw init.signal.reason;
if (!spendTracker.charge()) throw new NativeChatSpendRefusal();
noteProviderAttemptSend(logCtx, route.providerName, activeProvider, logCtx.usageLogInputTokens, transportRecovery ?? recovery);
// A reselected provider transport is still a physical send: the connection policy
// and manual-redirect ownership wrap the selected implementation (#4992).
Expand Down Expand Up @@ -432,6 +441,7 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
if (!transientSendAvailable()) break;
try { void response.body?.cancel().catch(() => {}); } catch { /* already closed */ }
activeProvider = rotated;
stampApiKeyAccountLabel(logCtx, route.providerName, activeProvider);
activeAdapter = createOpenAIChatAdapter(activeProvider);
releaseRetainedRequest();
activeRequest = buildActiveRequest();
Expand All @@ -443,6 +453,11 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
cleanupAbort();
upstream.abort();
if (req.signal.aborted) return fail(499, "Client cancelled request", "client_cancelled");
if (error instanceof NativeChatSpendRefusal) {
const refusal = workflowRefusalResponse("workflow-spend-exhausted", logCtx);
finishLog(429);
return refusal;
}
if (isTranslatorBudgetExceededError(error)) {
return fail(413, "request translation buffer exceeded the safe limit", "request_too_large", "translation_buffer_limit");
}
Expand Down
29 changes: 28 additions & 1 deletion src/usage/log.ts
Original file line number Diff line number Diff line change
Expand Up @@ -222,8 +222,33 @@ export interface PersistedRequestSpend extends RequestSpendTotals {
}

const MAX_PERSISTED_MOVE_REASONS = 8;
// Model selectors are NOT length-bound at admission: configured and discovered
// model ids reach MODEL_DISCOVERY_MAX_MODEL_ID_LENGTH, and the wire `model`
// field is raw client input. Persisting a plain prefix would merge selectors
// that share it, so over-long selectors persist as prefix + a digest of the
// FULL selector — bounded, deterministic, and still exact-matchable.
const MAX_PERSISTED_REQUESTED_MODEL_LEN = 130;
const REQUESTED_MODEL_DIGEST_HEX_LEN = 16;
const LOGICAL_REQUEST_ID_RE = /^[A-Za-z0-9_.:-]{1,64}$/;

/**
* Persisted form of the wire model selector. Selectors within the bound persist
* verbatim; longer selectors persist as a prefix plus a short digest of the full
* value, so two distinct selectors that share the prefix never collapse into one
* persisted identity. Exact-match readers (`requested_model = ?`) must encode
* lookup input through this same function. Idempotent — encoded forms fit the
* bound — which matters because rows are normalized again on read.
*/
export function encodePersistedRequestedModel(selector: string): string {
if (selector.length <= MAX_PERSISTED_REQUESTED_MODEL_LEN) return selector;
const digest = createHash("sha256")
.update(selector)
.digest("hex")
.slice(0, REQUESTED_MODEL_DIGEST_HEX_LEN);
const prefixLen = MAX_PERSISTED_REQUESTED_MODEL_LEN - REQUESTED_MODEL_DIGEST_HEX_LEN - 1;
return `${selector.slice(0, prefixLen)}~${digest}`;
}

export function isLogicalRequestId(value: unknown): value is string {
return typeof value === "string" && LOGICAL_REQUEST_ID_RE.test(value);
}
Expand Down Expand Up @@ -856,7 +881,9 @@ function normalizeUsageEntry(entry: PersistedUsageEntry): PersistedUsageEntry {
? { conversationId: entry.conversationId.trim().slice(0, 128) }
: {}),
...(entry.resolvedModel ? { resolvedModel: entry.resolvedModel } : {}),
...(entry.requestedModel ? { requestedModel: entry.requestedModel } : {}),
...(typeof entry.requestedModel === "string" && entry.requestedModel
? { requestedModel: encodePersistedRequestedModel(entry.requestedModel) }
: {}),
...(shadowCallRewrittenFrom ? { shadowCallRewrittenFrom } : {}),
...(typeof entry.requestedEffort === "string" && entry.requestedEffort
? { requestedEffort: capMetadataString(entry.requestedEffort) }
Expand Down
7 changes: 7 additions & 0 deletions structure/gui-and-management-api.md
Original file line number Diff line number Diff line change
Expand Up @@ -540,6 +540,11 @@ status, so an unexpected management response cannot add raw upstream material.

> Decision record: [ADR-0078](decisions/ADR-0078-usage-accounting.md)

Requested selectors longer than 130 characters persist as a prefix plus a digest of the complete
selector; the request-history exact-match filter applies the same idempotent encoding. Serving-model
identities remain unchanged. Only historical Codex `openai`, `chatgpt` and `openai-multi` main labels
collapse for reporting; configured provider names ending in `-main` remain separate. CLI access-key
usage is unavailable without a valid attribution timestamp, rather than a measured zero or never-used key.
`src/usage/log.ts` writes append-only JSONL to `~/.opencodex/usage.jsonl` with file mode `0o600`
inside an owner-only `0o700` directory. Consecutive appends reuse the directory and permission
check for at most one second; the first append at or after that boundary attempts to reapply both
Expand Down Expand Up @@ -739,6 +744,8 @@ converge the Codex catalog once and return its disposition. The Models UI owns a
picker data resource so failure cannot erase the ordinary model inventory; Apply publishes through
the resource's generation fence, and Most used reads usage only on explicit Apply. Stored mode
survives availability drift, while complete/native custom orders await explicit replacement.
The Models app-server status read is owned by its API-base/restart effect, not the picker tab;
switching to Combos preserves a pending read and its existing stale-state banner.

The shared atomic replacement publisher also identifies explicit Remote Workspace file writes as `remote-workspace`. Remote Workspace uses a separate, explicitly enabled server surface with structural WebSocket callbacks and awaited per-server cleanup; [its contract](remote-workspace.md) owns that integration and records its isolated owner and support limits.

Expand Down
8 changes: 5 additions & 3 deletions structure/transports/responses.md
Original file line number Diff line number Diff line change
Expand Up @@ -1229,10 +1229,12 @@ target's own recovery decision, while the physical-send total is what binds ever
The request's send budget bounds how many times it may reach upstream; the spend ledger bounds
what those sends may cost, and it is the only bound here that survives a restart. Its production
caller is `request-spend.ts`, installed on the execution budget at genuine ingress in `core.ts`
and parked on the log context so `addFinalRequestLog` can settle it.
and parked on the log context so `addFinalRequestLog` can settle it. Native Chat installs the
same tracker before its independent physical-send ladder and charges it immediately before each
dispatch, so taking that fast path cannot bypass root, identity, or provider-pool ceilings.

It books by observing the budget's own send counter rather than by being called from each
dispatch site. That counter moves exactly once per physical send — a reservation increments it, a
The Responses path books by observing its budget's own send counter rather than calling each
dispatch site; Native Chat directly charges messages, tool definitions and the output ceiling. That counter moves exactly once per physical send — a reservation increments it, a
refund decrements it, and an externally reported send settles against a booking already counted —
so one ledger entry per increment is one entry per send, and a dispatch path added later cannot
forget to book. The previous attempt at this wiring shipped the whole reserve/dispatch/settle
Expand Down
Loading
Loading