diff --git a/.changeset/brunch-petrinaut-host-integration.md b/.changeset/brunch-petrinaut-host-integration.md index eb35bf8b508..55f8207d8e1 100644 --- a/.changeset/brunch-petrinaut-host-integration.md +++ b/.changeset/brunch-petrinaut-host-integration.md @@ -2,4 +2,4 @@ "@hashintel/petrinaut": patch --- -Expose revision identity, editor commands, current-definition diagnostics, and user-guide content to host-owned assistant tools, while showing unfinished tool progress and refusing unavailable title edits. Keep host transport and store resources current across route changes, with stable React subscriptions and capability-based read-only titles. +Expose revision identity, editor commands, current-definition diagnostics, optimizer availability, and user-guide content to host-owned assistant tools, while showing unfinished tool and experiment progress, refusing unavailable title edits, and rejecting unavailable optimization before an experiment record is created. Keep host transport and store resources current across route changes, with stable React subscriptions and capability-based read-only titles. diff --git a/.changeset/expose-prepare-experiment.md b/.changeset/expose-prepare-experiment.md new file mode 100644 index 00000000000..9664845e0a8 --- /dev/null +++ b/.changeset/expose-prepare-experiment.md @@ -0,0 +1,5 @@ +--- +"@hashintel/petrinaut": patch +--- + +`prepareExperiment` is exported from `@hashintel/petrinaut/react`, so a host can resolve an experiment request against the live model and build its optimization input without starting a run. diff --git a/.changeset/selected-batch-scenarios-metrics.md b/.changeset/selected-batch-scenarios-metrics.md new file mode 100644 index 00000000000..f43ac98d285 --- /dev/null +++ b/.changeset/selected-batch-scenarios-metrics.md @@ -0,0 +1,5 @@ +--- +"@hashintel/petrinaut-core": patch +--- + +`selectedMutationBatchSchema` admits `addScenario`, `updateScenario`, `removeScenario`, `addMetric`, `updateMetric` and `removeMetric`, so a selected batch can save the scenario and metric an experiment names. diff --git a/apps/brunch-agent/src/agents/chat-agent/agent.ts b/apps/brunch-agent/src/agents/chat-agent/agent.ts index 888ed95d3e5..8ce8130ef66 100644 --- a/apps/brunch-agent/src/agents/chat-agent/agent.ts +++ b/apps/brunch-agent/src/agents/chat-agent/agent.ts @@ -227,7 +227,7 @@ export function ChatAgent({ id }: AgentProps) { Call ping when you need to confirm the server tool path. Submit at most one browser tool call per proposal, separately from server tools, and wait for its correlated client result before further browser work. Invalid proposals fail as a whole; do not rely on sibling execution order. A client-tool-result signal is JSON [{ toolCallId, toolName, output, metadata? }]. Treat output as the browser's canonical result for that call and continue helping the user once; never reapply a completed mutation. For a joined root arc, metadata.mutationRecord contains verified observations and effects, not assistant prose or user testimony. Failed, stale, no-op and unknown attempts are not causes. -A ${NET_STALE_SIGNAL} signal at the start of a user turn means this conversation holds no verified read of the net now open in Petrinaut, or the net changed after your last verified read. When it is present, call ${readPetrinautNetToolName} in its own proposal and wait for its browser result before explaining, reviewing, interviewing about, or changing the model, and do not say the net is unavailable or ask for an upload or description. When it is absent, the most recent ${readPetrinautNetToolName} result in this conversation is the current net. +A ${NET_STALE_SIGNAL} signal at the start of a user turn means this conversation holds no verified read of the net now open in Petrinaut, or the net changed after your last verified read. When it is present, make ${readPetrinautNetToolName} the entire proposal: do not call activate_skill, read_skill_resource, or any other tool in the same proposal. Wait for its browser result before activating required skills, explaining, reviewing, interviewing about, or changing the model, and do not say the net is unavailable or ask for an upload or description. When it is absent, the most recent ${readPetrinautNetToolName} result in this conversation is the current net. `.replace(/^\s+|\s+$/gu, ""), ); if (browserContext) diff --git a/apps/brunch-agent/src/agents/chat-agent/tool-catalogue.ts b/apps/brunch-agent/src/agents/chat-agent/tool-catalogue.ts index 4bae474f583..40c901a254b 100644 --- a/apps/brunch-agent/src/agents/chat-agent/tool-catalogue.ts +++ b/apps/brunch-agent/src/agents/chat-agent/tool-catalogue.ts @@ -16,6 +16,7 @@ export interface OrdinaryBrunchToolCatalogueEntry { | "workpiece" | "petrinaut-read" | "petrinaut-mutation" + | "petrinaut-experiment-draft" | "explanation" | "diagnostic"; } @@ -83,6 +84,12 @@ export const ordinaryBrunchToolCatalogue: readonly OrdinaryBrunchToolCatalogueEn executionOwner: "petrinaut-website", role: "petrinaut-mutation", }, + { + name: "draft_petrinaut_experiment", + definitionOwner: "sdcpn-plugin", + executionOwner: "petrinaut-website", + role: "petrinaut-experiment-draft", + }, { name: "read_workpiece", definitionOwner: "brunch-core", diff --git a/apps/brunch-agent/src/conversation/net-ledger.ts b/apps/brunch-agent/src/conversation/net-ledger.ts index b5c0626ab00..c14c8a9a020 100644 --- a/apps/brunch-agent/src/conversation/net-ledger.ts +++ b/apps/brunch-agent/src/conversation/net-ledger.ts @@ -15,6 +15,7 @@ */ import { canonicalContent, + isDraftPetrinautExperimentToolName, isLayoutPetrinautNetToolName, isMutatePetrinautNetToolName, isReadPetrinautDocsToolName, @@ -92,7 +93,8 @@ const resultMessages = (snapshot: FlueConversationSnapshot) => const isNonMutatingBrowserTool = (name: string): boolean => isReadPetrinautNetToolName(name) || isReadPetrinautDiagnosticsToolName(name) || - isReadPetrinautDocsToolName(name); + isReadPetrinautDocsToolName(name) || + isDraftPetrinautExperimentToolName(name); /** A model-selected ID selects a recorded browser observation, never a model-supplied hash. */ export const recordedBrowserObservation = async ( diff --git a/apps/brunch-agent/src/conversation/why.ts b/apps/brunch-agent/src/conversation/why.ts index 0b4067a4429..b9101baec6e 100644 --- a/apps/brunch-agent/src/conversation/why.ts +++ b/apps/brunch-agent/src/conversation/why.ts @@ -415,6 +415,7 @@ export const queryWorkpiece = async (input: { case "differential-equation": case "type": case "scenario": + case "metric": return locateRootState(definition, { kind: target.kind, name: target.id, diff --git a/apps/brunch-agent/test/chat-agent-compaction.test.ts b/apps/brunch-agent/test/chat-agent-compaction.test.ts index be43e59258f..d83185d7431 100644 --- a/apps/brunch-agent/test/chat-agent-compaction.test.ts +++ b/apps/brunch-agent/test/chat-agent-compaction.test.ts @@ -1,3 +1,4 @@ +import { useInstruction } from "@flue/runtime"; import { afterEach, beforeEach, expect, test, vi } from "vitest"; import { useBrunchAgent } from "@hashintel/brunch-agent/flue"; @@ -19,7 +20,7 @@ vi.mock( ); vi.mock("@flue/runtime", async (importOriginal) => ({ ...(await importOriginal()), - useInstruction: () => undefined, + useInstruction: vi.fn(), useContextProjection: () => undefined, useInitialData: () => undefined, useDelivery: () => ({ kind: "user", body: "test" }), @@ -60,6 +61,17 @@ test("the production ChatAgent supplies no compaction override when unset", asyn ); }); +test("a stale-net turn reserves the proposal for the browser read", async () => { + const { ChatAgent: renderChatAgent } = + await import("../src/agents/chat-agent/agent.ts"); + renderChatAgent({ id: "test-instance" }); + expect(useInstruction).toHaveBeenCalledWith( + expect.stringContaining( + "do not call activate_skill, read_skill_resource, or any other tool in the same proposal", + ), + ); +}); + test("the production ChatAgent forwards an independent OpenAI specifier and thinking level", async () => { vi.stubEnv("BRUNCH_CHAT_MODEL", "openai/gpt-5.6-sol"); vi.stubEnv("BRUNCH_CHAT_THINKING", "low"); diff --git a/apps/brunch-agent/test/integration/native-schema-carriage.integration.ts b/apps/brunch-agent/test/integration/native-schema-carriage.integration.ts index 4aabba54046..d9eaf8fc19b 100644 --- a/apps/brunch-agent/test/integration/native-schema-carriage.integration.ts +++ b/apps/brunch-agent/test/integration/native-schema-carriage.integration.ts @@ -20,6 +20,8 @@ import { import { createFlueClient } from "@flue/sdk"; import { + draftPetrinautExperimentInputSchema, + draftPetrinautExperimentToolName, queryWorkpieceInputSchema, mutatePetrinetInputSchema, mutatePetrinautNetToolName, @@ -330,6 +332,77 @@ try { ); assert(tools.length > 0); for (const tool of tools) assert.deepEqual(tool.input_schema, expected); + + // The session experiment draft carries core's request schema natively: + // the sent schema must be byte-identical to the Zod source and must still + // refuse a constraint (there is no constraint carriage) once serialized. + const draftTool = ordinaryRequest.serialized.tools.find( + (tool) => tool.name === draftPetrinautExperimentToolName, + ); + assert(draftTool, `${method} must carry the experiment draft tool`); + assert.deepEqual( + draftTool.input_schema, + draftPetrinautExperimentInputSchema["~standard"].jsonSchema.input({ + target: "draft-2020-12", + }), + ); + const draftExperiment = { + name: "Synthetic staffing", + scenarioId: "scenario-peak", + scenarioParameterValues: { agents: { mode: "range", min: 2, max: 8 } }, + runCount: 20, + seed: 1, + dt: 0.5, + maxTime: 120, + metricIds: ["metric-wait"], + execution: { + mode: "optimize", + objectiveMetricId: "metric-wait", + direction: "minimize", + steps: 5, + runsPerStep: 4, + }, + }; + const draftEnvelope = { + observation: { toolCallId: "read-1", baseHash: "a".repeat(64) }, + declarations: [ + { subject: "maxTime", statement: "120 model minutes: the peak." }, + ], + basis: { kind: "absent", reason: "Synthetic control." }, + unsupported: [], + }; + const validateDraft = (arguments_: Record): void => { + validateToolArguments( + { + name: draftTool.name, + description: "Captured draft tool", + parameters: draftTool.input_schema as Tool["parameters"], + }, + { + type: "toolCall", + id: "draft-schema-control", + name: draftTool.name, + arguments: arguments_, + }, + ); + }; + assert.doesNotThrow(() => + validateDraft({ ...draftEnvelope, experiment: draftExperiment }), + ); + assert.throws(() => + validateDraft({ + ...draftEnvelope, + experiment: { ...draftExperiment, constraints: [] }, + }), + ); + assert.throws(() => + validateDraft({ + ...draftEnvelope, + declarations: [], + experiment: draftExperiment, + }), + ); + assert.throws(() => validateDraft(draftEnvelope)); } assert.equal(networkAttempts, 0); process.stdout.write( diff --git a/apps/brunch-agent/test/net-ledger.test.ts b/apps/brunch-agent/test/net-ledger.test.ts index 585e6bd7c60..ef12fc0235f 100644 --- a/apps/brunch-agent/test/net-ledger.test.ts +++ b/apps/brunch-agent/test/net-ledger.test.ts @@ -3,6 +3,7 @@ import { createHash } from "node:crypto"; import { describe, expect, test } from "vitest"; import { + draftPetrinautExperimentToolName, layoutPetrinautNetToolName, deriveMutationEffects, mutatePetrinetInputSchema, @@ -338,6 +339,25 @@ describe("the net ledger is a projection over Flue history", () => { expect(JSON.stringify(snapshot)).toBe(before); }); + test("ignores a completed experiment draft because it cannot change the net", async () => { + const events = await deriveNetLedger( + snapshotOf([ + ...readTurn("read-1", emptyNet), + assistantCall("draft-1", draftPetrinautExperimentToolName), + resultDelivery("draft-1", draftPetrinautExperimentToolName, { + status: "drafted", + summary: "Prepared only", + diagnostics: [], + }), + ]), + browser, + ); + + expect(events.map((event) => [event.kind, event.toolCallId])).toEqual([ + ["read", "read-1"], + ]); + }); + test("introduces no identities: every event names a tool call the assistant made, at its own position", async () => { const snapshot = fullHistory(); const calls = assistantCallsOf(snapshot); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-client-tools.ts b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-client-tools.ts index acdd7acdc38..9848248f34e 100644 --- a/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-client-tools.ts +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-client-tools.ts @@ -1,4 +1,5 @@ import { + draftPetrinautExperimentToolName, layoutPetrinautNetToolName, mutatePetrinautNetToolName, READ_PETRINAUT_DOCS_TOOL_NAME, @@ -16,24 +17,45 @@ export const brunchClientToolNames: ReadonlySet = new Set([ READ_PETRINAUT_DOCS_TOOL_NAME, ]); -/** Batched construction: reads, one batch mutation and layout. */ +/** Batched construction: reads, one batch mutation, layout and one session draft. */ export const batchedConstructionClientToolNames: ReadonlySet = new Set([ ...brunchClientToolNames, readPetrinautNetToolName, readPetrinautDiagnosticsToolName, mutatePetrinautNetToolName, layoutPetrinautNetToolName, + draftPetrinautExperimentToolName, ]); /** - * The Brunch-named tools are host dynamic tools in every mode: the transport - * projects them as `dynamic-tool` parts and the panel routes them to the - * host's automatic tools rather than to Petrinaut's static registry. + * The Brunch-named tools the browser executes without asking: the panel + * routes them to the host's automatic tools. */ -export const brunchPetrinautDynamicToolNames: ReadonlySet = new Set([ +export const brunchPetrinautAutomaticToolNames: ReadonlySet = new Set([ READ_PETRINAUT_DOCS_TOOL_NAME, readPetrinautNetToolName, readPetrinautDiagnosticsToolName, layoutPetrinautNetToolName, mutatePetrinautNetToolName, ]); + +/** + * The Brunch-named tools the browser renders as a widget the person acts on: + * the panel routes them to the host's interactive tools. The experiment draft + * auto-submits its prepared state so Brunch's turn continues, and keeps its + * Run and Dismiss actions for the person. + */ +export const brunchPetrinautInteractiveToolNames: ReadonlySet = new Set( + [draftPetrinautExperimentToolName], +); + +/** + * The Brunch-named tools are host dynamic tools in every mode: the transport + * projects them as `dynamic-tool` parts and the panel routes them to the + * host's automatic or interactive tools rather than to Petrinaut's static + * registry. + */ +export const brunchPetrinautDynamicToolNames: ReadonlySet = new Set([ + ...brunchPetrinautAutomaticToolNames, + ...brunchPetrinautInteractiveToolNames, +]); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool.test.tsx b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool.test.tsx new file mode 100644 index 00000000000..c88065b8396 --- /dev/null +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool.test.tsx @@ -0,0 +1,960 @@ +/** + * @vitest-environment jsdom + */ +import { sha256 } from "@noble/hashes/sha2.js"; +import { bytesToHex } from "@noble/hashes/utils.js"; +import { + act, + cleanup, + fireEvent, + render, + screen, + waitFor, +} from "@testing-library/react"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; + +import { + createJsonDocHandle, + createPetrinaut, + createReadableStore, +} from "@hashintel/petrinaut-core"; +import { + ExperimentHostContext, + OptimizationsContext, + PetrinautInstanceContext, + prepareExperiment, +} from "@hashintel/petrinaut/react"; + +import { + BrunchDraftExperimentWidget, + resetBrunchDraftExperimentSession, +} from "./brunch-draft-experiment-interactive-tool"; +import { + describeBudget, + describeExperiment, +} from "./brunch-draft-experiment-interactive-tool/describe-draft"; + +// The `/ui` entry pulls in chart code that probes `matchMedia` at import time. +vi.hoisted(() => { + globalThis.ResizeObserver = class { + observe() {} + unobserve() {} + disconnect() {} + } as unknown as typeof ResizeObserver; + window.matchMedia = (media) => ({ + media, + matches: false, + onchange: null, + addListener() {}, + removeListener() {}, + addEventListener() {}, + removeEventListener() {}, + dispatchEvent: () => true, + }); +}); + +import type { + DraftPetrinautExperimentInput, + DraftPetrinautExperimentOutput, +} from "@hashintel/brunch-agent-plugin-sdcpn"; +import type { + AbortSignalLike, + Petrinaut, + PetrinautExperimentHost, + PetrinautExperimentRequest, + PetrinautExperimentResult, + SDCPN, +} from "@hashintel/petrinaut-core"; +import type { OptimizationsContextValue } from "@hashintel/petrinaut/react"; +import type { PetrinautAiInteractiveToolWidgetProps } from "@hashintel/petrinaut/ui"; +import type { ReactNode } from "react"; + +// Distributive so the awaiting/submitted discriminant survives the Pick. +type DistributivePick = T extends unknown + ? Pick + : never; +type WidgetState = DistributivePick< + PetrinautAiInteractiveToolWidgetProps< + DraftPetrinautExperimentInput, + DraftPetrinautExperimentOutput + >, + "state" | "submittedOutput" +>; + +const awaiting: WidgetState = { state: "awaiting" }; +const submitted: WidgetState = { + state: "submitted", + submittedOutput: { status: "drafted", summary: "", diagnostics: [] }, +}; + +const makeDefinition = (): SDCPN => ({ + places: [], + transitions: [], + types: [], + parameters: [], + differentialEquations: [], + scenarios: [ + { + id: "scenario__peak_demand", + name: "Peak demand", + scenarioParameters: [ + { identifier: "agents", type: "integer", default: 4 }, + { identifier: "arrival_rate", type: "real", default: 1.5 }, + ], + parameterOverrides: {}, + initialState: { type: "per_place", content: {} }, + }, + ], + metrics: [ + { + id: "metric__average_waiting_time", + name: "Average waiting time", + code: "return 1;", + }, + { + id: "metric__abandonment_rate", + name: "Abandonment rate", + code: "return 0;", + }, + ], +}); + +const makeRequest = ( + overrides: Partial = {}, +): PetrinautExperimentRequest => ({ + name: "Staffing under peak demand", + scenarioId: "scenario__peak_demand", + scenarioParameterValues: { + agents: { mode: "range", min: 2, max: 8 }, + }, + runCount: 20, + seed: 7, + dt: 1, + maxTime: 120, + metricIds: ["metric__average_waiting_time", "metric__abandonment_rate"], + execution: { + mode: "optimize", + objectiveMetricId: "metric__average_waiting_time", + direction: "minimize", + steps: 3, + runsPerStep: 5, + }, + ...overrides, +}); + +const definitionHash = (definition: SDCPN) => + bytesToHex(sha256(new TextEncoder().encode(JSON.stringify(definition)))); + +const makeInput = ( + experiment: PetrinautExperimentRequest = makeRequest(), + unsupported: DraftPetrinautExperimentInput["unsupported"] = [], + baseHash = definitionHash(makeDefinition()), +): DraftPetrinautExperimentInput => ({ + observation: { toolCallId: "call_read_1", baseHash }, + experiment, + declarations: [ + { + subject: "maxTime", + statement: "Minutes; the two-hour peak window is 120 minutes.", + }, + ], + basis: { + kind: "declared", + revisionId: "workpiece-revision", + sha256: "a".repeat(64), + locators: [{ start: 0, end: 1 }], + rationale: "The person accepted this experiment configuration.", + scope: "operation", + }, + unsupported, +}); + +const finishedResult: PetrinautExperimentResult = { + status: "complete", + experimentId: "experiment_1", + name: "Staffing under peak demand", + runsCompleted: 15, + metrics: [], +}; + +const renderWidget = ({ + input, + toolCallId, + state, + definition, + runExperiment, + optimizationUnavailableReason = null, + omitOptimizationUnavailableReason = false, + submitOutput = async () => {}, + instance: suppliedInstance, +}: { + input: DraftPetrinautExperimentInput; + toolCallId: string; + state: WidgetState; + definition: Petrinaut["definition"]; + runExperiment: PetrinautExperimentHost["runExperiment"]; + optimizationUnavailableReason?: string | null; + omitOptimizationUnavailableReason?: boolean; + submitOutput?: (output: DraftPetrinautExperimentOutput) => Promise; + instance?: Petrinaut; +}) => { + const submit = vi.fn(submitOutput); + const instance = + suppliedInstance ?? + ({ + definition, + handle: { + doc: () => definition.get(), + revisionId: { get: () => "test-revision" }, + }, + } as unknown as Petrinaut); + const host: PetrinautExperimentHost = { runExperiment }; + const optimizationActions = { + optimizations: [], + createOptimization: () => Promise.reject(new Error("Not used")), + cancelOptimization: () => {}, + removeOptimization: () => {}, + }; + const optimizations = ( + omitOptimizationUnavailableReason + ? optimizationActions + : { ...optimizationActions, optimizationUnavailableReason } + ) as OptimizationsContextValue; + const wrap = (children: ReactNode) => ( + + + + {children} + + + + ); + const utils = render( + wrap( + "Support desk"} + submit={() => {}} + submitAndWait={submit} + toolCallId={toolCallId} + />, + ), + ); + return { ...utils, submit, wrap }; +}; + +const heading = () => + screen + .getAllByRole("region", { name: "Drafted experiment" }) + .map((section) => section.getAttribute("data-draft-status")); + +describe("BrunchDraftExperimentWidget", () => { + beforeEach(() => { + resetBrunchDraftExperimentSession(); + }); + + afterEach(() => { + cleanup(); + }); + + it("shows executable numeric values without rounding them", () => { + const request = makeRequest({ + scenarioParameterValues: { + arrival_rate: { mode: "range", min: 1001.4, max: 1002.4 }, + }, + dt: 0.123456, + maxTime: 1001.4, + }); + const definition = makeDefinition(); + + expect( + describeExperiment( + prepareExperiment(request, definition, "Support desk"), + definition, + ), + ).toContain("arrival_rate 1001.4–1002.4"); + expect(describeBudget(request)).toContain("horizon 1001.4, step 0.123456"); + }); + + it("shows the canonical prepared range and implicit fixed defaults before approval", async () => { + const runExperiment = vi.fn(); + const { submit } = renderWidget({ + input: makeInput( + makeRequest({ + scenarioParameterValues: { + agents: { mode: "range", min: 1.2, max: 7.8 }, + }, + }), + ), + toolCallId: "canonical-review", + state: awaiting, + definition: createReadableStore(makeDefinition()), + runExperiment, + }); + + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + expect(submit.mock.calls[0]?.[0].summary).toContain( + "Vary agents 1–8 with arrival_rate = 1.5", + ); + expect(screen.getByText(/Vary agents 1–8/u)).toBeTruthy(); + expect(screen.queryByText(/agents 1.2–7.8/u)).toBeNull(); + }); + + it("prepares, reports drafted once, and starts nothing", async () => { + const runExperiment = vi.fn(); + const { submit, rerender, wrap } = renderWidget({ + input: makeInput(makeRequest(), [ + { + condition: "No more than 5% of callers abandon.", + reason: "The request carries no constraints.", + blocksRun: false, + reportedByMetricId: "metric__abandonment_rate", + }, + ]), + toolCallId: "call_draft_1", + state: awaiting, + definition: createReadableStore(makeDefinition()), + runExperiment, + }); + + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + const output = submit.mock.calls[0]![0]; + expect(output.status).toBe("drafted"); + expect(output.summary).toContain("not run"); + expect(output.summary).toContain("agents 2–8"); + expect(output.summary).toContain("minimize Average waiting time"); + expect(output.summary).toContain("1 stated restriction is not carried"); + expect(output.summary).toContain( + "3 optimization steps × 5 runs, then 20 runs at the best parameters", + ); + expect(output.diagnostics).toEqual([ + "No constraints or constraint policy are carried; nothing is enforced.", + "Not carried: No more than 5% of callers abandon.", + ]); + + expect(heading()).toEqual([ + "Drafted — not run · not saved with the document", + ]); + expect(screen.getByText("objective")).toBeTruthy(); + expect(screen.getByText("reported, not enforced")).toBeTruthy(); + expect( + screen.getByText("reported by metric__abandonment_rate, not enforced"), + ).toBeTruthy(); + expect(screen.getByRole("button", { name: "Run" })).toBeTruthy(); + expect(runExperiment).not.toHaveBeenCalled(); + + // Once the panel marks the call submitted the card keeps its draft and + // does not report again. + rerender( + wrap( + "Support desk"} + submit={() => {}} + submitAndWait={submit} + toolCallId="call_draft_1" + />, + ), + ); + expect(submit).toHaveBeenCalledTimes(1); + expect(screen.getByRole("button", { name: "Run" })).toBeTruthy(); + }); + + it("shows that a draft is being prepared while submission is pending", async () => { + const submission = Promise.withResolvers(); + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "pending-draft", + state: awaiting, + definition: createReadableStore(makeDefinition()), + runExperiment: vi.fn(), + submitOutput: () => submission.promise, + }); + + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + expect(heading()).toEqual(["Preparing draft"]); + expect(screen.getByText(/being prepared/u)).toBeTruthy(); + expect(screen.queryByText(/earlier session/u)).toBeNull(); + expect(screen.queryByRole("button", { name: "Run" })).toBeNull(); + + await act(async () => submission.resolve()); + await waitFor(() => + expect(heading()).toEqual([ + "Drafted — not run · not saved with the document", + ]), + ); + }); + + it("blocks an unsupported hard restriction unless reporting-only exploration was explicitly accepted", async () => { + const runExperiment = vi.fn(); + const input = { + ...makeInput(), + unsupported: [ + { + condition: "Never exceed ten minutes.", + reason: "No constraint carriage.", + }, + ], + } as unknown as DraftPetrinautExperimentInput; + const { submit } = renderWidget({ + input, + toolCallId: "hard-restriction", + state: awaiting, + definition: createReadableStore(makeDefinition()), + runExperiment, + }); + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + expect(submit.mock.calls[0]?.[0].diagnostics).toContain( + "Run blocked: Never exceed ten minutes.", + ); + const run = screen.getByRole("button", { name: "Run" }); + expect(run.disabled).toBe(true); + expect(screen.getByRole("alert").textContent).toContain("Run is blocked"); + fireEvent.click(run); + expect(runExperiment).not.toHaveBeenCalled(); + }); + + it("prepares from the observed handle when the readable store normalizes key order", async () => { + const instance = createPetrinaut({ + document: createJsonDocHandle({ + id: "metric-before-scenario", + initial: { + places: [], + transitions: [], + types: [], + differentialEquations: [], + parameters: [], + }, + }), + }); + instance.mutations.addMetric({ + id: "metric__average_waiting_time", + name: "Average waiting time", + code: "return 1;", + }); + instance.mutations.addMetric({ + id: "metric__abandonment_rate", + name: "Abandonment rate", + code: "return 0;", + }); + instance.mutations.addScenario({ + id: "scenario__peak_demand", + name: "Peak demand", + scenarioParameters: [ + { identifier: "agents", type: "integer", default: 4 }, + { identifier: "arrival_rate", type: "real", default: 1.5 }, + ], + parameterOverrides: {}, + initialState: { type: "per_place", content: {} }, + }); + const observedDefinition = instance.handle.doc(); + expect(observedDefinition).toBeDefined(); + expect(definitionHash(observedDefinition!)).not.toBe( + definitionHash(instance.definition.get()), + ); + + const runExperiment = vi.fn(() => Promise.resolve(finishedResult)); + const { submit } = renderWidget({ + input: makeInput(makeRequest(), [], definitionHash(observedDefinition!)), + toolCallId: "metric-before-scenario", + state: awaiting, + definition: instance.definition, + instance, + runExperiment, + }); + + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + expect(submit.mock.calls[0]?.[0]).toMatchObject({ status: "drafted" }); + fireEvent.click(screen.getByRole("button", { name: "Run" })); + await waitFor(() => expect(runExperiment).toHaveBeenCalledTimes(1)); + expect( + screen.queryByText(/model changed since this was drafted/u), + ).toBeNull(); + instance.dispose(); + }); + + it("reports invalid without a Run button when the model lacks the metric", async () => { + const runExperiment = vi.fn(); + const definition = createReadableStore(makeDefinition()); + const { submit } = renderWidget({ + input: makeInput( + makeRequest({ + metricIds: ["metric__missing"], + execution: { + mode: "optimize", + objectiveMetricId: "metric__missing", + direction: "minimize", + steps: 3, + runsPerStep: 5, + }, + }), + ), + toolCallId: "call_draft_invalid", + state: awaiting, + definition, + runExperiment, + }); + + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + expect(submit.mock.calls[0]![0]).toMatchObject({ + status: "invalid", + diagnostics: ['Metric "metric__missing" does not exist'], + }); + expect(heading()).toEqual(["Could not be prepared"]); + expect(screen.queryByRole("button", { name: "Run" })).toBeNull(); + expect(runExperiment).not.toHaveBeenCalled(); + }); + + it("refuses a draft whose cited observation differs from the live model", async () => { + const runExperiment = vi.fn(); + const changed = makeDefinition(); + changed.metrics![0]!.code = "return 2;"; + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "stale-observation", + state: awaiting, + definition: createReadableStore(changed), + runExperiment, + }); + + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + expect(submit.mock.calls[0]?.[0]).toMatchObject({ status: "invalid" }); + expect(submit.mock.calls[0]?.[0].diagnostics[0]).toMatch( + /changed since the verified observation/u, + ); + expect(heading()).toEqual(["Could not be prepared"]); + expect(screen.queryByRole("button", { name: "Run" })).toBeNull(); + }); + + it("retries preparation after the host rejects its first submission", async () => { + const submitOutput = vi + .fn() + .mockRejectedValueOnce(new Error("Output was not accepted")) + .mockResolvedValueOnce(undefined); + const definition = createReadableStore(makeDefinition()); + const { submit, rerender, wrap } = renderWidget({ + input: makeInput(), + toolCallId: "retry-draft", + state: awaiting, + definition, + runExperiment: vi.fn(), + submitOutput, + }); + + await waitFor(() => + expect(heading()).toEqual(["Draft could not be submitted"]), + ); + expect(submit).toHaveBeenCalledTimes(1); + expect(screen.queryByRole("button", { name: "Run" })).toBeNull(); + + rerender( + wrap( + "Support desk"} + submit={() => {}} + submitAndWait={(output) => submit(output)} + toolCallId="retry-draft" + />, + ), + ); + await act(async () => {}); + expect(submit).toHaveBeenCalledTimes(1); + expect(heading()).toEqual(["Draft could not be submitted"]); + + fireEvent.click(screen.getByRole("button", { name: "Retry preparation" })); + + await waitFor(() => expect(submit).toHaveBeenCalledTimes(2)); + await waitFor(() => + expect(heading()).toEqual([ + "Drafted — not run · not saved with the document", + ]), + ); + expect(submitOutput).toHaveBeenCalledTimes(2); + expect(screen.getByRole("button", { name: "Run" })).toBeTruthy(); + }); + + it("retries preparation after the browser document becomes available", async () => { + const definition = createReadableStore(makeDefinition()); + let browserDocument: SDCPN | undefined; + const instance = { + definition, + handle: { + doc: () => browserDocument, + revisionId: { get: () => "test-revision" }, + }, + } as unknown as Petrinaut; + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "retry-browser-observation", + state: awaiting, + definition, + instance, + runExperiment: vi.fn(), + }); + + await waitFor(() => + expect(heading()).toEqual(["Draft could not be prepared"]), + ); + expect( + screen.getByText(/bound browser document is unavailable/u), + ).toBeTruthy(); + expect(submit).not.toHaveBeenCalled(); + + browserDocument = makeDefinition(); + fireEvent.click(screen.getByRole("button", { name: "Retry preparation" })); + + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + await waitFor(() => + expect(heading()).toEqual([ + "Drafted — not run · not saved with the document", + ]), + ); + }); + + it("does not offer Run when optimization is unavailable", async () => { + const runExperiment = vi.fn(); + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "call_draft_unavailable", + state: awaiting, + definition: createReadableStore(makeDefinition()), + runExperiment, + optimizationUnavailableReason: "Optimization is unavailable", + }); + + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + const output = submit.mock.calls[0]![0]; + expect(output.status).toBe("drafted"); + expect(output.diagnostics).toContain( + "Execution unavailable: Optimization is unavailable", + ); + expect(output.summary).toContain( + "Execution unavailable: Optimization is unavailable", + ); + expect(screen.getByRole("alert").textContent).toBe( + "Optimization is unavailable", + ); + expect(screen.queryByRole("button", { name: "Run" })).toBeNull(); + expect(runExperiment).not.toHaveBeenCalled(); + }); + + it("keeps legacy optimization contexts without an availability reason usable", async () => { + const runExperiment = vi.fn(); + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "legacy-optimization-context", + state: awaiting, + definition: createReadableStore(makeDefinition()), + runExperiment, + omitOptimizationUnavailableReason: true, + }); + + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + expect(submit.mock.calls[0]?.[0].summary).not.toContain("undefined"); + expect(screen.getByRole("button", { name: "Run" })).toBeTruthy(); + }); + + it("lets a later draft supersede the earlier card", async () => { + const runExperiment = vi.fn(); + const definition = createReadableStore(makeDefinition()); + const first = renderWidget({ + input: makeInput(), + toolCallId: "call_draft_a", + state: awaiting, + definition, + runExperiment, + }); + await waitFor(() => expect(first.submit).toHaveBeenCalledTimes(1)); + + const second = renderWidget({ + input: makeInput( + makeRequest({ + scenarioParameterValues: { + agents: { mode: "range", min: 3, max: 6 }, + }, + }), + ), + toolCallId: "call_draft_b", + state: awaiting, + definition, + runExperiment, + }); + await waitFor(() => expect(second.submit).toHaveBeenCalledTimes(1)); + + expect(heading()).toEqual([ + "Superseded by a later draft", + "Drafted — not run · not saved with the document", + ]); + expect(screen.getAllByRole("button", { name: "Run" })).toHaveLength(1); + }); + + it("runs exactly one experiment through the host and shows progress and completion", async () => { + let resolveRun: (result: PetrinautExperimentResult) => void = () => {}; + let seenSignal: AbortSignalLike | undefined; + const runExperiment = vi.fn( + ( + _request: PetrinautExperimentRequest, + options?: Parameters[1], + ) => + new Promise((resolve) => { + seenSignal = options?.signal; + options?.onProgress?.({ + experimentId: "experiment_1", + name: "Staffing under peak demand", + phase: "optimizing", + runsCompleted: 5, + runsTarget: 15, + step: 1, + steps: 3, + }); + resolveRun = resolve; + }), + ); + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "call_draft_run", + state: awaiting, + definition: createReadableStore(makeDefinition()), + runExperiment, + }); + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + + fireEvent.click(screen.getByRole("button", { name: "Run" })); + + await waitFor(() => expect(runExperiment).toHaveBeenCalledTimes(1)); + expect(runExperiment.mock.calls[0]![0]).toMatchObject({ + scenarioId: "scenario__peak_demand", + execution: { mode: "optimize", direction: "minimize" }, + }); + expect(seenSignal?.aborted).toBe(false); + await waitFor(() => expect(heading()).toEqual(["Running"])); + expect(screen.getByRole("status").textContent).toBe( + "optimizing: 5/15 runs, step 1/3", + ); + expect(screen.queryByRole("button", { name: "Run" })).toBeNull(); + + // A second click cannot start another run: the button is gone. + fireEvent.click(screen.getByRole("button", { name: "Cancel" })); + expect(seenSignal?.aborted).toBe(true); + + await act(async () => { + resolveRun(finishedResult); + }); + await waitFor(() => expect(heading()).toEqual(["Run complete"])); + expect(screen.getByRole("status").textContent).toContain( + "15 runs completed", + ); + expect(runExperiment).toHaveBeenCalledTimes(1); + }); + + it("retries a run failure from the host", async () => { + const runExperiment = vi + .fn() + .mockRejectedValueOnce(new Error("Compilation failed")) + .mockResolvedValueOnce(finishedResult); + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "call_draft_fail", + state: awaiting, + definition: createReadableStore(makeDefinition()), + runExperiment, + }); + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + + fireEvent.click(screen.getByRole("button", { name: "Run" })); + + await waitFor(() => expect(heading()).toEqual(["Run failed"])); + expect(screen.getByRole("alert").textContent).toBe("Compilation failed"); + + fireEvent.click(screen.getByRole("button", { name: "Retry run" })); + + await waitFor(() => expect(runExperiment).toHaveBeenCalledTimes(2)); + await waitFor(() => expect(heading()).toEqual(["Run complete"])); + }); + + it("requires fresh approval for each model change in simulation-only requests", async () => { + const runExperiment = vi.fn(() => Promise.resolve(finishedResult)); + const definition = createReadableStore(makeDefinition()); + const { submit } = renderWidget({ + input: makeInput( + makeRequest({ + execution: { mode: "simulate" }, + scenarioParameterValues: { agents: { mode: "fixed", value: 4 } }, + }), + ), + toolCallId: "simulation-stale", + state: awaiting, + definition, + runExperiment, + }); + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + const changed = makeDefinition(); + changed.scenarios![0]!.initialState = { + type: "per_place", + content: { queue: "3" }, + }; + act(() => definition.set(changed)); + fireEvent.click(screen.getByRole("button", { name: "Run" })); + expect(runExperiment).not.toHaveBeenCalled(); + expect(screen.getByText(/Before:.*initialState/su).textContent).toContain( + '"queue": "3"', + ); + + const changedAgain = structuredClone(changed); + changedAgain.scenarios![0]!.initialState = { + type: "per_place", + content: { queue: "7" }, + }; + act(() => definition.set(changedAgain)); + fireEvent.click( + screen.getByRole("button", { name: "Accept current model" }), + ); + fireEvent.click( + screen.getByRole("button", { name: "Run against current model" }), + ); + expect(runExperiment).not.toHaveBeenCalled(); + expect(screen.getByText(/Before:.*initialState/su).textContent).toContain( + '"queue": "7"', + ); + + const accept = screen.getByRole("button", { + name: "Accept current model", + }); + act(() => { + accept.click(); + accept.click(); + }); + expect(runExperiment).not.toHaveBeenCalled(); + + fireEvent.click( + screen.getByRole("button", { name: "Run against current model" }), + ); + await waitFor(() => expect(runExperiment).toHaveBeenCalledTimes(1)); + }); + + it("requires a separate acknowledgement before running a changed model", async () => { + const runExperiment = vi.fn(() => Promise.resolve(finishedResult)); + const definition = createReadableStore(makeDefinition()); + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "call_draft_stale", + state: awaiting, + definition, + runExperiment, + }); + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + + const changed = makeDefinition(); + changed.metrics![0]!.code = "return 2;"; + act(() => definition.set(changed)); + + fireEvent.click(screen.getByRole("button", { name: "Run" })); + + expect(screen.getByRole("status").textContent).toContain( + "The model changed since this was drafted", + ); + expect(runExperiment).not.toHaveBeenCalled(); + fireEvent.click( + screen.getByRole("button", { name: "Accept current model" }), + ); + const runReviewed = screen.getByRole("button", { + name: "Run against current model", + }); + + fireEvent.click(runReviewed); + await waitFor(() => expect(runExperiment).toHaveBeenCalledTimes(1)); + }); + + it("refuses to run when the current model no longer prepares", async () => { + const runExperiment = vi.fn(); + const definition = createReadableStore(makeDefinition()); + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "call_draft_gone", + state: awaiting, + definition, + runExperiment, + }); + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + + const changed = makeDefinition(); + changed.scenarios = []; + act(() => definition.set(changed)); + + fireEvent.click(screen.getByRole("button", { name: "Run" })); + + expect(screen.getByRole("alert").textContent).toBe( + 'Scenario "scenario__peak_demand" does not exist', + ); + expect(runExperiment).not.toHaveBeenCalled(); + }); + + it("hides actions after Dismiss", async () => { + const runExperiment = vi.fn(); + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "call_draft_dismiss", + state: awaiting, + definition: createReadableStore(makeDefinition()), + runExperiment, + }); + await waitFor(() => expect(submit).toHaveBeenCalledTimes(1)); + + fireEvent.click(screen.getByRole("button", { name: "Dismiss" })); + + expect(heading()).toEqual(["Dismissed"]); + expect(screen.queryByRole("button", { name: "Run" })).toBeNull(); + expect(runExperiment).not.toHaveBeenCalled(); + }); + + it("marks a reloaded, already-submitted card as not retained and offers no Run", () => { + const runExperiment = vi.fn(); + const { submit } = renderWidget({ + input: makeInput(), + toolCallId: "call_draft_reloaded", + state: submitted, + definition: createReadableStore(makeDefinition()), + runExperiment, + }); + + expect(submit).not.toHaveBeenCalled(); + expect(heading()).toEqual(["Not retained in this session"]); + expect(screen.getByText(/prepared in an earlier session/u)).toBeTruthy(); + expect(screen.queryByRole("button", { name: "Run" })).toBeNull(); + }); + + it("does not revive dismissed drafts on remount or share them with another editor", async () => { + const runExperiment = vi.fn(); + const definition = createReadableStore(makeDefinition()); + const props = { + input: makeInput(), + toolCallId: "same-call", + state: awaiting, + definition, + runExperiment, + }; + const first = renderWidget(props); + await waitFor(() => expect(first.submit).toHaveBeenCalledTimes(1)); + fireEvent.click(screen.getByRole("button", { name: "Dismiss" })); + first.unmount(); + const remounted = renderWidget(props); + await waitFor(() => expect(remounted.submit).toHaveBeenCalledTimes(1)); + expect(heading()).toEqual(["Dismissed"]); + expect(screen.queryByRole("button", { name: "Run" })).toBeNull(); + + const otherEditor = renderWidget({ + ...props, + definition: createReadableStore(makeDefinition()), + }); + await waitFor(() => expect(otherEditor.submit).toHaveBeenCalledTimes(1)); + expect(heading()).toEqual([ + "Dismissed", + "Drafted — not run · not saved with the document", + ]); + expect(screen.getAllByRole("button", { name: "Run" })).toHaveLength(1); + }); +}); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool.tsx b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool.tsx new file mode 100644 index 00000000000..e3c428015ad --- /dev/null +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool.tsx @@ -0,0 +1,687 @@ +import { use, useEffect, useRef, useState, useSyncExternalStore } from "react"; + +import { + canonicalContent, + type DraftPetrinautExperimentInput, + draftPetrinautExperimentInputSchema, + type DraftPetrinautExperimentOutput, + draftPetrinautExperimentOutputSchema, + draftPetrinautExperimentToolName, +} from "@hashintel/brunch-agent-plugin-sdcpn"; +import { css } from "@hashintel/ds-helpers/css"; +import { + ExperimentHostContext, + OptimizationsContext, + prepareExperiment, + usePetrinautInstance, +} from "@hashintel/petrinaut/react"; +import { + definePetrinautAiInteractiveTool, + type PetrinautAiInteractiveToolWidgetProps, +} from "@hashintel/petrinaut/ui"; + +import { + describeBudget, + describeExperiment, + metricRoles, + preparationDiffers, + summarizeForAgent, +} from "./brunch-draft-experiment-interactive-tool/describe-draft"; +import { + resetSessionDrafts, + sessionDraftsFor, +} from "./brunch-draft-experiment-interactive-tool/session-drafts"; +import { observeBrowserDefinition } from "./mutation-record"; + +import type { PreparedExperiment } from "./brunch-draft-experiment-interactive-tool/describe-draft"; +import type { + PetrinautExperimentRequest, + SDCPN, +} from "@hashintel/petrinaut-core"; + +const containerStyle = css({ + display: "flex", + flexDirection: "column", + gap: "2", + padding: "3", + borderWidth: "thin", + borderStyle: "solid", + borderColor: "neutral.a30", + borderRadius: "lg", + backgroundColor: "neutral.s00", +}); + +const statusStyle = css({ + color: "neutral.s80", + fontSize: "xs", + fontWeight: "medium", + letterSpacing: "wide", + textTransform: "uppercase", +}); + +const titleStyle = css({ + color: "neutral.s100", + fontSize: "sm", + fontWeight: "semibold", +}); + +const bodyStyle = css({ + color: "neutral.s90", + fontSize: "sm", + lineHeight: "relaxed", +}); + +const sectionLabelStyle = css({ + color: "neutral.s80", + fontSize: "xs", + fontWeight: "medium", + marginTop: "1", +}); + +const listStyle = css({ + display: "flex", + flexDirection: "column", + gap: "1", + paddingLeft: "4", + color: "neutral.s90", + fontSize: "xs", + lineHeight: "relaxed", + listStyleType: "disc", +}); + +const tagStyle = css({ + marginLeft: "1", + paddingX: "1", + borderRadius: "sm", + backgroundColor: "neutral.a10", + color: "neutral.s80", + fontSize: "xs", + fontWeight: "medium", +}); + +const noticeStyle = css({ + padding: "2", + borderRadius: "md", + backgroundColor: "yellow.a10", + color: "neutral.s100", + fontSize: "xs", + lineHeight: "relaxed", +}); + +const errorStyle = css({ + padding: "2", + borderRadius: "md", + backgroundColor: "red.a10", + color: "neutral.s100", + fontSize: "xs", + lineHeight: "relaxed", +}); + +const actionsStyle = css({ + display: "flex", + gap: "2", + justifyContent: "flex-end", + marginTop: "1", +}); + +const primaryButtonStyle = css({ + paddingX: "3", + paddingY: "2", + borderRadius: "md", + backgroundColor: "blue.a85", + color: "white", + cursor: "pointer", + fontSize: "sm", + fontWeight: "medium", + _hover: { backgroundColor: "blue.a100" }, + _disabled: { cursor: "not-allowed", opacity: 0.45 }, +}); + +const secondaryButtonStyle = css({ + paddingX: "3", + paddingY: "2", + borderWidth: "thin", + borderStyle: "solid", + borderColor: "neutral.a30", + borderRadius: "md", + backgroundColor: "neutral.s00", + color: "neutral.s90", + cursor: "pointer", + fontSize: "sm", + fontWeight: "medium", + _hover: { backgroundColor: "neutral.a10" }, +}); + +/** Forget every draft, as a reload would. For tests that share the module. */ +export const resetBrunchDraftExperimentSession = resetSessionDrafts; + +type WidgetProps = PetrinautAiInteractiveToolWidgetProps< + DraftPetrinautExperimentInput, + DraftPetrinautExperimentOutput +>; + +const conditionBlocksRun = ( + condition: DraftPetrinautExperimentInput["unsupported"][number], +) => condition.blocksRun !== false; + +// Two stable snapshots rather than one fresh object: useSyncExternalStore +// compares snapshots by identity and would re-render without end otherwise. +const useSessionDraft = ( + sessionDrafts: ReturnType, + toolCallId: string, +) => { + const draft = useSyncExternalStore(sessionDrafts.subscribe, () => + sessionDrafts.get().drafts.get(toolCallId), + ); + const isCurrent = useSyncExternalStore( + sessionDrafts.subscribe, + () => sessionDrafts.get().currentToolCallId === toolCallId, + ); + return { draft, isCurrent }; +}; + +const prepareOrExplain = ( + request: PetrinautExperimentRequest, + definition: Parameters[1], + title: string, +): + | { prepared: PreparedExperiment; error: null } + | { prepared: null; error: string } => { + try { + return { + prepared: prepareExperiment(request, definition, title), + error: null, + }; + } catch (caught) { + return { + prepared: null, + error: caught instanceof Error ? caught.message : String(caught), + }; + } +}; + +/** + * The card itself, rendered inside Petrinaut's tree so it can read the live + * definition and the stock experiment host. Exported for direct rendering in + * tests; the app mounts it through the interactive-tool definition below. + */ +export const BrunchDraftExperimentWidget = ({ + input, + readTitle, + state, + submitAndWait, + toolCallId, +}: WidgetProps & { readTitle: () => string }) => { + const instance = usePetrinautInstance(); + const experimentHost = use(ExperimentHostContext); + const optimizationUnavailableReason = + use(OptimizationsContext).optimizationUnavailableReason ?? null; + const executionUnavailable = + input.experiment.execution.mode === "optimize" + ? optimizationUnavailableReason + : null; + const sessionDrafts = sessionDraftsFor(instance.definition); + const { draft, isCurrent } = useSessionDraft(sessionDrafts, toolCallId); + const preparedOnceRef = useRef(false); + const [reviewed, setReviewed] = useState<{ + prepared: PreparedExperiment; + definition: SDCPN; + } | null>(null); + const [reviewAccepted, setReviewAccepted] = useState(false); + const [runError, setRunError] = useState(null); + const [submissionAttempt, setSubmissionAttempt] = useState(0); + const [preparationFailure, setPreparationFailure] = useState<{ + kind: "prepare" | "submit"; + message: string; + } | null>(null); + const [submissionPending, setSubmissionPending] = useState( + state === "awaiting" && submitAndWait !== undefined, + ); + + // A freshly streamed call prepares once against the live model and reports + // back so Brunch's turn can continue. Run and Dismiss come after and are + // not reported through the tool result. + useEffect(() => { + if ( + state !== "awaiting" || + preparedOnceRef.current || + submitAndWait === undefined + ) + return; + preparedOnceRef.current = true; + setPreparationFailure(null); + setSubmissionPending(true); + // The server hashes the handle snapshot; the readable store may normalize + // its property order and therefore produce a different serialized hash. + let observation: ReturnType; + try { + observation = observeBrowserDefinition(instance.handle); + } catch (caught) { + setPreparationFailure({ + kind: "prepare", + message: caught instanceof Error ? caught.message : String(caught), + }); + setSubmissionPending(false); + return; + } + const definition = observation.definition; + const outcome = + observation.sha256 === input.observation.baseHash + ? prepareOrExplain(input.experiment, definition, readTitle()) + : { + prepared: null, + error: + "The model changed since the verified observation. Ask Brunch to read the current model and draft again.", + }; + const candidate = { + toolCallId, + input, + definition, + prepared: outcome.prepared, + invalid: outcome.error, + dismissed: false, + run: { phase: "idle" as const }, + }; + const submissionDraft = + sessionDrafts.get().drafts.get(toolCallId) ?? candidate; + const output: DraftPetrinautExperimentOutput = submissionDraft.prepared + ? { + status: "drafted", + summary: `${summarizeForAgent( + submissionDraft.prepared, + submissionDraft.definition, + submissionDraft.input.unsupported.length, + )}${ + executionUnavailable === null + ? "" + : ` Execution unavailable: ${executionUnavailable}.` + }`, + diagnostics: [ + "No constraints or constraint policy are carried; nothing is enforced.", + ...submissionDraft.input.unsupported.map( + (condition) => + `${conditionBlocksRun(condition) ? "Run blocked" : "Not carried"}: ${condition.condition}`, + ), + ...(executionUnavailable === null + ? [] + : [`Execution unavailable: ${executionUnavailable}`]), + ], + } + : { + status: "invalid", + summary: `The browser could not prepare this experiment against the current model: ${submissionDraft.invalid}`, + diagnostics: [submissionDraft.invalid ?? "Preparation failed"], + }; + const submitAndRegister = async () => { + try { + await submitAndWait(output); + } catch (caught) { + setPreparationFailure({ + kind: "submit", + message: caught instanceof Error ? caught.message : String(caught), + }); + setSubmissionPending(false); + return; + } + sessionDrafts.register(candidate); + setSubmissionPending(false); + }; + void submitAndRegister(); + }, [ + executionUnavailable, + input, + instance, + readTitle, + sessionDrafts, + state, + submissionAttempt, + submitAndWait, + toolCallId, + ]); + + const definition = + reviewed?.definition ?? draft?.definition ?? instance.definition.get(); + const displayedPrepared = reviewed?.prepared ?? draft?.prepared ?? null; + const request = displayedPrepared?.request ?? null; + const blocksRun = input.unsupported.some(conditionBlocksRun); + const optimizationUnavailable = + request?.execution.mode === "optimize" ? executionUnavailable : null; + + const onRun = async () => { + // Read synchronously, not from the render closure: duplicate clicks or a + // remounted copy of the same card must never start a second experiment. + const latest = sessionDrafts.get(); + const pending = latest.drafts.get(toolCallId); + if ( + !pending?.prepared || + latest.currentToolCallId !== toolCallId || + pending.dismissed || + (pending.run.phase !== "idle" && pending.run.phase !== "failed") || + pending.input.unsupported.some(conditionBlocksRun) || + optimizationUnavailable !== null + ) + return; + // Prepare again against the model as it is now: Run must start what the + // person sees, and a changed metric or parameter is shown before any call. + const currentDefinition = structuredClone(instance.definition.get()); + const current = prepareOrExplain( + pending.prepared.request, + currentDefinition, + readTitle(), + ); + if (!current.prepared) { + setRunError(current.error); + return; + } + if ( + preparationDiffers( + reviewed?.prepared ?? pending.prepared, + current.prepared, + ) || + canonicalContent(reviewed?.definition ?? pending.definition) !== + canonicalContent(currentDefinition) + ) { + setReviewed({ + prepared: current.prepared, + definition: currentDefinition, + }); + setReviewAccepted(false); + setRunError(null); + return; + } + if (reviewed && !reviewAccepted) return; + setRunError(null); + const controller = new AbortController(); + sessionDrafts.update(toolCallId, { + run: { phase: "running", controller, progress: null }, + }); + try { + const result = await experimentHost.runExperiment( + current.prepared.request, + { + signal: controller.signal, + onProgress: (progress) => + sessionDrafts.update(toolCallId, { + run: { phase: "running", controller, progress }, + }), + }, + ); + sessionDrafts.update(toolCallId, { run: { phase: "finished", result } }); + } catch (caught) { + sessionDrafts.update(toolCallId, { + run: { + phase: "failed", + message: caught instanceof Error ? caught.message : String(caught), + }, + }); + } + }; + + const heading = !draft + ? submissionPending + ? "Preparing draft" + : preparationFailure + ? preparationFailure.kind === "prepare" + ? "Draft could not be prepared" + : "Draft could not be submitted" + : "Not retained in this session" + : draft.invalid !== null + ? "Could not be prepared" + : draft.dismissed + ? "Dismissed" + : draft.run.phase === "running" + ? "Running" + : draft.run.phase === "finished" + ? `Run ${draft.run.result.status}` + : draft.run.phase === "failed" + ? "Run failed" + : isCurrent + ? "Drafted — not run · not saved with the document" + : "Superseded by a later draft"; + + const canAct = + draft !== undefined && + draft.invalid === null && + !draft.dismissed && + isCurrent && + (draft.run.phase === "idle" || draft.run.phase === "failed"); + const canRun = canAct && optimizationUnavailable === null; + + return ( +
+

{heading}

+

{input.experiment.name}

+ {displayedPrepared ? ( + <> +

+ {describeExperiment(displayedPrepared, definition)} +

+

+ {describeBudget(displayedPrepared.request)} +

+

Metrics

+
    + {metricRoles(displayedPrepared.request, definition).map( + (metric) => ( +
  • + {metric.name} + + {metric.role === "objective" + ? "objective" + : "reported, not enforced"} + +
  • + ), + )} +
+ + ) : ( +

+ {draft?.invalid ?? + (submissionPending + ? "This experiment proposal is being prepared." + : preparationFailure + ? preparationFailure.kind === "prepare" + ? `The experiment proposal could not be prepared: ${preparationFailure.message}` + : `The prepared proposal could not be submitted: ${preparationFailure.message}` + : "This draft was prepared in an earlier session. Ask Brunch to draft it again to run it.")} +

+ )} +

Declared

+
    + {input.declarations.map((declaration) => ( +
  • + {declaration.subject}: {declaration.statement} +
  • + ))} +
+

Not carried into execution

+ {input.unsupported.length === 0 ? ( +

+ No restrictions were stated. The request carries no constraints, so + none are enforced. +

+ ) : ( +
    + {input.unsupported.map((condition) => ( +
  • + {condition.condition} — {condition.reason} + {condition.reportedByMetricId ? ( + + reported by {condition.reportedByMetricId}, not enforced + + ) : null} +
  • + ))} +
+ )} + {blocksRun ? ( +

+ Run is blocked by an unsupported restriction. Ask Brunch to revise the + proposal; a reporting-only exploration needs your explicit acceptance. +

+ ) : null} + {optimizationUnavailable !== null ? ( +

+ {optimizationUnavailable} +

+ ) : null} + {runError ? ( +

+ {runError} +

+ ) : null} + {reviewed && draft && canAct ? ( +
+

+ The model changed since this was drafted. Review the changes below + before running, or ask Brunch to draft it again. +

+
+ Review model changes + {Object.keys({ ...draft.definition, ...reviewed.definition }).map( + (section) => { + const before = draft.definition[section as keyof SDCPN]; + const after = reviewed.definition[section as keyof SDCPN]; + if (canonicalContent(before) === canonicalContent(after)) + return null; + return ( +
+ {section} +
+                      {`Before: ${before === undefined ? "not set" : JSON.stringify(before, null, 2)}\nAfter: ${after === undefined ? "removed" : JSON.stringify(after, null, 2)}`}
+                    
+
+ ); + }, + )} +
+
+ ) : null} + {draft?.run.phase === "running" ? ( +

+ {draft.run.progress + ? `${draft.run.progress.phase}: ${draft.run.progress.runsCompleted}/${draft.run.progress.runsTarget} runs${ + draft.run.progress.steps !== undefined + ? `, step ${draft.run.progress.step ?? 0}/${draft.run.progress.steps}` + : "" + }` + : "Starting…"} +

+ ) : null} + {draft?.run.phase === "finished" ? ( +

+ {draft.run.result.message ?? + `${draft.run.result.runsCompleted} runs completed. The result stays in Simulate → Experiments.`} +

+ ) : null} + {draft?.run.phase === "failed" ? ( +

+ {draft.run.message} +

+ ) : null} + {canAct ? ( +
+ + {canRun ? ( + reviewed && !reviewAccepted ? ( + + ) : ( + + ) + ) : null} +
+ ) : null} + {!draft && preparationFailure && state === "awaiting" ? ( +
+ +
+ ) : null} + {draft?.run.phase === "running" ? ( +
+ +
+ ) : null} +
+ ); +}; + +/** + * The website-owned card for a Brunch-drafted experiment. It prepares the + * proposal against the live model, tells Brunch it is drafted (not run), and + * lets the person Run or Dismiss it. Run reuses the stock experiment host, so + * records, the active indicator and the Experiments view behave as shipped. + * It never navigates and keeps nothing beyond this browser session. + */ +export const createBrunchDraftExperimentInteractiveTool = ({ + readTitle, +}: { + readTitle: () => string; +}) => + definePetrinautAiInteractiveTool< + DraftPetrinautExperimentInput, + DraftPetrinautExperimentOutput + >({ + toolName: draftPetrinautExperimentToolName, + inputSchema: draftPetrinautExperimentInputSchema, + outputSchema: draftPetrinautExperimentOutputSchema, + component: (props) => ( + + ), + }); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool/describe-draft.ts b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool/describe-draft.ts new file mode 100644 index 00000000000..5bdd3e14c9b --- /dev/null +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool/describe-draft.ts @@ -0,0 +1,126 @@ +import type { + PetrinautExperimentRequest, + SDCPN, +} from "@hashintel/petrinaut-core"; +import type { prepareExperiment } from "@hashintel/petrinaut/react"; + +export type PreparedExperiment = ReturnType; + +const nameOfScenario = (definition: SDCPN, scenarioId: string) => + definition.scenarios?.find((scenario) => scenario.id === scenarioId)?.name ?? + scenarioId; + +const nameOfMetric = (definition: SDCPN, metricId: string) => + definition.metrics?.find((metric) => metric.id === metricId)?.name ?? + metricId; + +const formatNumber = (value: number) => String(value); + +/** + * The one sentence the person reads first: what varies, over what range, + * under which scenario, and what is minimized or maximized. Names come from + * the live definition; identifiers are only shown when a name is missing. + */ +export const describeExperiment = ( + prepared: PreparedExperiment, + definition: SDCPN, +): string => { + const axes = new Map( + prepared.parameterAxes.map((axis) => [axis.identifier, axis]), + ); + const ranges: string[] = []; + const fixed: string[] = []; + for (const identifier of Object.keys( + prepared.input.scenarioParameterValues, + )) { + const axis = axes.get(identifier); + if (axis) { + ranges.push( + `${identifier} ${formatNumber(axis.min)}–${formatNumber(axis.max)}`, + ); + } else if (Object.hasOwn(prepared.fixedValues, identifier)) { + fixed.push(`${identifier} = ${String(prepared.fixedValues[identifier])}`); + } + } + const { request } = prepared; + const scenario = nameOfScenario(definition, request.scenarioId); + const varying = + ranges.length > 0 ? `Vary ${ranges.join(", ")}` : "Simulate as saved"; + const held = fixed.length > 0 ? ` with ${fixed.join(", ")}` : ""; + const objective = + request.execution.mode === "optimize" + ? `; ${request.execution.direction} ${nameOfMetric(definition, request.execution.objectiveMetricId)}` + : ""; + return `${varying}${held} under ${scenario}${objective}.`; +}; + +/** The budget line beside the summary, in the request's own units. */ +export const describeBudget = (request: PetrinautExperimentRequest): string => { + const horizon = `horizon ${formatNumber(request.maxTime)}, step ${formatNumber(request.dt)}`; + if (request.execution.mode === "optimize") { + return `${request.execution.steps} optimization steps × ${request.execution.runsPerStep} runs, then ${request.runCount} runs at the best parameters, ${horizon}, seed ${request.seed}.`; + } + return `${request.runCount} runs, ${horizon}, seed ${request.seed}.`; +}; + +export type MetricRole = { + metricId: string; + name: string; + role: "objective" | "reported"; +}; + +/** + * Every metric the request carries, with its honest role. A metric that is + * not the objective is reported beside the result and never enforced: the + * request has no constraints. + */ +export const metricRoles = ( + request: PetrinautExperimentRequest, + definition: SDCPN, +): MetricRole[] => + request.metricIds.map((metricId) => ({ + metricId, + name: nameOfMetric(definition, metricId), + role: + request.execution.mode === "optimize" && + request.execution.objectiveMetricId === metricId + ? "objective" + : "reported", + })); + +/** + * What Brunch hears back once the browser has prepared the draft. It is one + * sentence plus the things the model must not misreport. + */ +export const summarizeForAgent = ( + prepared: PreparedExperiment, + definition: SDCPN, + unsupportedCount: number, +): string => + `Drafted for this session, not run and not saved with the document: ${describeExperiment(prepared, definition)} ${describeBudget(prepared.request)} ${ + unsupportedCount === 0 + ? "No restrictions were stated; none are enforced." + : `${unsupportedCount} stated ${unsupportedCount === 1 ? "restriction is" : "restrictions are"} not carried into execution.` + } The person starts it from the card; do not report it as running.`; + +/** + * The material change a second preparation can reveal: the frozen inputs the + * run would use differ from the ones the card showed. Compared structurally + * because both come from the same pure function over the same request. + */ +export const preparationDiffers = ( + drafted: PreparedExperiment, + current: PreparedExperiment, +): boolean => + JSON.stringify({ + input: drafted.input, + fixed: drafted.fixedValues, + axes: drafted.parameterAxes, + optimization: drafted.optimization, + }) !== + JSON.stringify({ + input: current.input, + fixed: current.fixedValues, + axes: current.parameterAxes, + optimization: current.optimization, + }); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool/session-drafts.ts b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool/session-drafts.ts new file mode 100644 index 00000000000..a280fae494d --- /dev/null +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool/session-drafts.ts @@ -0,0 +1,96 @@ +import type { PreparedExperiment } from "./describe-draft"; +import type { DraftPetrinautExperimentInput } from "@hashintel/brunch-agent-plugin-sdcpn"; +import type { + PetrinautExperimentProgress, + PetrinautExperimentResult, + SDCPN, +} from "@hashintel/petrinaut-core"; + +/** + * The one place a drafted experiment lives: this browser session, keyed by the + * tool call that drafted it. Nothing here is written to the document or to + * the conversation; a reload forgets every draft, which the card says. + */ +export type SessionDraftRun = + | { phase: "idle" } + | { + phase: "running"; + controller: AbortController; + progress: PetrinautExperimentProgress | null; + } + | { phase: "finished"; result: PetrinautExperimentResult } + | { phase: "failed"; message: string }; + +export type SessionDraft = { + toolCallId: string; + input: DraftPetrinautExperimentInput; + /** Frozen model the person reviewed, including simulation-only inputs. */ + definition: SDCPN; + /** What the card prepared against the model at draft time; null if refused. */ + prepared: PreparedExperiment | null; + /** Why preparation refused, when it did. */ + invalid: string | null; + dismissed: boolean; + run: SessionDraftRun; +}; + +type SessionDraftsState = { + currentToolCallId: string | null; + drafts: ReadonlyMap; +}; + +const createSessionDrafts = () => { + let state: SessionDraftsState = { + currentToolCallId: null, + drafts: new Map(), + }; + const listeners = new Set<() => void>(); + const publish = (next: SessionDraftsState) => { + state = next; + for (const listener of listeners) listener(); + }; + return { + subscribe: (listener: () => void): (() => void) => { + listeners.add(listener); + return () => listeners.delete(listener); + }, + get: (): SessionDraftsState => state, + /** Remounting an existing card must not revive it or supersede a newer one. */ + register: (draft: SessionDraft): SessionDraft => { + const existing = state.drafts.get(draft.toolCallId); + if (existing) return existing; + const drafts = new Map(state.drafts); + drafts.set(draft.toolCallId, draft); + publish({ currentToolCallId: draft.toolCallId, drafts }); + return draft; + }, + update: ( + toolCallId: string, + patch: Partial>, + ) => { + const existing = state.drafts.get(toolCallId); + if (!existing) return; + const drafts = new Map(state.drafts); + drafts.set(toolCallId, { ...existing, ...patch }); + publish({ ...state, drafts }); + }, + }; +}; + +// The definition store survives panel remounts, but belongs to one editor. +// Weak keys release drafts when that editor is disposed rather than retaining +// every model and run in a tab-wide singleton. +let sessions = new WeakMap>(); +export const sessionDraftsFor = (definitionStore: object) => { + let drafts = sessions.get(definitionStore); + if (!drafts) { + drafts = createSessionDrafts(); + sessions.set(definitionStore, drafts); + } + return drafts; +}; + +/** Test seam: forget every editor's drafts, as a reload would. */ +export const resetSessionDrafts = () => { + sessions = new WeakMap(); +}; diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment.integration.test.tsx b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment.integration.test.tsx new file mode 100644 index 00000000000..27fdd81cb57 --- /dev/null +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment.integration.test.tsx @@ -0,0 +1,586 @@ +/** @vitest-environment jsdom */ +import { + act, + cleanup, + fireEvent, + render, + screen, + waitFor, +} from "@testing-library/react"; +import { useEffect } from "react"; +import { afterEach, beforeAll, beforeEach, expect, test, vi } from "vitest"; + +import { + type DraftPetrinautExperimentInput, + type DraftPetrinautExperimentOutput, + draftPetrinautExperimentInputSchema, + draftPetrinautExperimentOutputSchema, + draftPetrinautExperimentToolName, +} from "@hashintel/brunch-agent-plugin-sdcpn"; +import { createJsonDocHandle } from "@hashintel/petrinaut-core"; +import { + compileHirArtifacts, + lowerScenarioToHir, +} from "@hashintel/petrinaut-core/hir"; +import { resolveTrialScenarioParameterValues } from "@hashintel/petrinaut-core/optimization"; +import { createInProcessMonteCarloWorker } from "@hashintel/petrinaut-core/workers/monte-carlo"; +import { + type OptimizationBest, + PetrinautOptimizationContext, + UserSettingsProvider, +} from "@hashintel/petrinaut/react"; +import { + definePetrinautAiInteractiveTool, + Petrinaut, + type PetrinautAiInteractiveToolWidgetProps, +} from "@hashintel/petrinaut/ui"; + +import { + batchedConstructionClientToolNames, + brunchPetrinautDynamicToolNames, +} from "./brunch-client-tools"; +import { + BrunchDraftExperimentWidget, + resetBrunchDraftExperimentSession, +} from "./brunch-draft-experiment-interactive-tool"; +import { + BrunchPanelConversationTracker, + createBrunchPanelTransport, +} from "./brunch-panel-transport"; +import { createBrunchPetrinautTools } from "./brunch-petrinaut-tools"; +import { + createJoinedBrowserMutationRecorder, + observeBrowserDefinition, +} from "./mutation-record"; + +import type { AgentSendResult, FlueClient } from "@flue/sdk"; +import type { LspWorkerFactory, SDCPN } from "@hashintel/petrinaut-core"; +import type { + PetrinautConnectedOptimization, + PetrinautOptimizationInput, +} from "@hashintel/petrinaut-core/optimization"; + +vi.hoisted(() => { + window.matchMedia = (media) => ({ + media, + matches: false, + onchange: null, + addListener() {}, + removeListener() {}, + addEventListener() {}, + removeEventListener() {}, + dispatchEvent: () => true, + }); + // Monaco installs a WebKit clipboard workaround on macOS and reads these + // browser APIs at import time; jsdom does not implement them. + Object.defineProperty(document, "queryCommandSupported", { + configurable: true, + value: () => false, + }); + Object.defineProperty(navigator, "clipboard", { + configurable: true, + value: { write: () => Promise.resolve() }, + }); + Object.defineProperty(window, "ClipboardItem", { + configurable: true, + value: class {}, + }); + Object.defineProperty(window, "CSS", { + configurable: true, + value: { + ...window.CSS, + escape: (value: string) => value.replace(/[^a-zA-Z0-9_-]/g, "\\$&"), + }, + }); +}); + +beforeAll(async () => { + // The real panel loads Monaco lazily; settle its import before the test. + await import("monaco-editor"); +}, 30_000); + +type LspWorker = Awaited>; +type LspWorkerMessage = Parameters[0]; +type LspWorkerListener = Parameters[1]; + +const cleanDiagnosticsWorker: LspWorkerFactory = () => { + const listeners = new Set(); + return { + postMessage(message: LspWorkerMessage) { + if (!("id" in message)) return; + let result: unknown; + if (message.method === "sdcpn/diagnostics") { + result = []; + } else if (message.method === "sdcpn/compileHirArtifacts") { + result = compileHirArtifacts( + message.params.sdcpn, + message.params.extensions, + message.params.options, + ); + } else if (message.method === "sdcpn/lowerScenario") { + result = lowerScenarioToHir(message.params.scenario, { + adHocContext: message.params.adHocContext, + }); + } else { + return; + } + queueMicrotask(() => { + for (const listener of listeners) { + listener({ + data: { jsonrpc: "2.0", id: message.id, result }, + }); + } + }); + }, + addEventListener(_type, listener) { + listeners.add(listener); + }, + removeEventListener(_type, listener) { + listeners.delete(listener); + }, + terminate() { + listeners.clear(); + }, + }; +}; + +const supportDesk: SDCPN = { + places: [], + transitions: [], + types: [], + parameters: [], + differentialEquations: [], + scenarios: [ + { + id: "scenario__peak_demand", + name: "Peak demand", + scenarioParameters: [ + { identifier: "agents", type: "integer", default: 4 }, + ], + parameterOverrides: {}, + initialState: { type: "per_place", content: {} }, + }, + ], + metrics: [ + { + id: "metric__average_waiting_time", + name: "Average waiting time", + code: "return 1;", + }, + ], +}; + +const draftInputFor = (baseHash: string): DraftPetrinautExperimentInput => ({ + observation: { toolCallId: "read-net-1", baseHash }, + experiment: { + name: "Staffing under peak demand", + scenarioId: "scenario__peak_demand", + scenarioParameterValues: { agents: { mode: "range", min: 2, max: 8 } }, + runCount: 20, + seed: 7, + dt: 1, + maxTime: 120, + metricIds: ["metric__average_waiting_time"], + execution: { + mode: "optimize", + objectiveMetricId: "metric__average_waiting_time", + direction: "minimize", + steps: 3, + runsPerStep: 5, + }, + }, + declarations: [ + { + subject: "maxTime", + statement: "Minutes; the two-hour peak window is 120 minutes.", + }, + ], + basis: { + kind: "declared", + revisionId: "revision-1", + sha256: "c".repeat(64), + locators: [{ start: 0, end: 12 }], + rationale: "The Ledger settles the staffing decision and its range.", + scope: "operation", + }, + unsupported: [ + { + condition: "No caller waits more than ten minutes.", + reason: + "The person accepted reporting-only exploration; the request carries no constraints.", + blocksRun: false, + }, + ], +}); + +type ObservedDraftWidgetProps = PetrinautAiInteractiveToolWidgetProps< + DraftPetrinautExperimentInput, + DraftPetrinautExperimentOutput +> & { + onSubmitted: () => void; +}; + +const ObservedDraftWidget = ({ + onSubmitted, + ...widgetProps +}: ObservedDraftWidgetProps) => { + useEffect(() => { + if (widgetProps.state === "submitted") { + onSubmitted(); + } + }, [onSubmitted, widgetProps.state]); + + return ( + "Support desk"} + /> + ); +}; + +const createObservedDraftTool = (onSubmitted: () => void) => + definePetrinautAiInteractiveTool< + DraftPetrinautExperimentInput, + DraftPetrinautExperimentOutput + >({ + toolName: draftPetrinautExperimentToolName, + inputSchema: draftPetrinautExperimentInputSchema, + outputSchema: draftPetrinautExperimentOutputSchema, + component: (props) => ( + + ), + }); + +const createControlledOptimization = ( + continueAfterFirstTrial: Promise, +): PetrinautConnectedOptimization => ({ + kind: "connected", + connect: (channel) => { + let manifest: PetrinautOptimizationInput | null = null; + return { + createOptimizationRun: async (input) => { + manifest = input; + return { runId: "run-integration" }; + }, + async *attachOptimizationRun(runId, options) { + if (manifest === null) { + throw new Error("The optimization manifest was not captured."); + } + let best: OptimizationBest | null = null; + let seq = 0; + const agentCounts = [2, 5, 8]; + for (const [trial, agents] of agentCounts.entries()) { + const suggestedValues = { agents }; + const outcome = await channel.evaluateTrial({ + runId, + trial, + manifest, + suggestedValues, + scenarioParameterValues: resolveTrialScenarioParameterValues( + manifest, + suggestedValues, + ), + seeds: Array.from( + { length: manifest.execution.seedsPerTrial ?? 1 }, + (_, index) => index + 1, + ), + signal: options?.signal ?? new AbortController().signal, + }); + if (outcome.kind !== "objective") { + throw new Error(`Trial ${trial} did not produce an objective.`); + } + best ??= { + trial, + parameters: suggestedValues, + objective: outcome.objective, + }; + seq += 1; + yield { + type: "trial", + trial, + parameters: suggestedValues, + objective: outcome.objective, + state: "complete", + best, + seq, + }; + if (trial === 0) { + await continueAfterFirstTrial; + } + } + yield { + type: "complete", + requestedTrials: agentCounts.length, + completedTrials: agentCounts.length, + prunedTrials: 0, + failedTrials: 0, + best, + resumable: true, + seq: seq + 1, + }; + }, + cancelOptimizationRun: async () => {}, + extendOptimizationRun: async () => {}, + releaseOptimizationRun: async () => {}, + dispose: () => {}, + }; + }, +}); + +beforeEach(() => { + const entries = new Map(); + vi.stubGlobal("localStorage", { + get length() { + return entries.size; + }, + clear: () => entries.clear(), + getItem: (key: string) => entries.get(key) ?? null, + key: (index: number) => [...entries.keys()].at(index) ?? null, + removeItem: (key: string) => entries.delete(key), + setItem: (key: string, value: string) => entries.set(key, value), + } satisfies Storage); + resetBrunchDraftExperimentSession(); +}); + +afterEach(() => { + cleanup(); + vi.restoreAllMocks(); + vi.unstubAllGlobals(); +}); + +test("a streamed experiment draft stays idle until Run, then uses the stock host progress path", async () => { + vi.spyOn(HTMLCanvasElement.prototype, "getContext").mockReturnValue(null); + vi.stubGlobal( + "ResizeObserver", + class { + observe() {} + unobserve() {} + disconnect() {} + }, + ); + const continueTrials = Promise.withResolvers(); + const availableOptimization = createControlledOptimization( + continueTrials.promise, + ); + const handle = createJsonDocHandle({ + id: "draft-experiment-test", + initial: supportDesk, + }); + const draftInput = draftInputFor(observeBrowserDefinition(handle).sha256); + const serverValidationPending = Promise.withResolvers(); + const releaseServerValidation = Promise.withResolvers(); + const settleInitialStream = Promise.withResolvers(); + const toolOutputSubmitted = Promise.withResolvers(); + const tracker = new BrunchPanelConversationTracker(); + const send = vi.fn( + async (): Promise => ({ + submissionId: `submission-${send.mock.calls.length}`, + uid: "uid", + offset: "0", + streamUrl: "http://local.test/agents/chat/test/stream", + }), + ); + const wait = vi.fn(async (admission, options) => { + const submissionId = (admission as AgentSendResult).submissionId; + const messageId = `assistant-${submissionId}`; + let ordinal = 0; + const position = () => ({ batch: 1, index: ordinal++ }); + await options?.onEvent?.({ + type: "message-started", + conversationId: "test", + submissionId, + messageId, + turnId: messageId, + position: position(), + }); + if (submissionId === "submission-1") { + await options?.onEvent?.({ + type: "tool-input", + conversationId: "test", + messageId, + toolCallId: "draft-1", + toolName: draftPetrinautExperimentToolName, + input: draftInput, + position: position(), + }); + serverValidationPending.resolve(); + await releaseServerValidation.promise; + await options?.onEvent?.({ + type: "tool-output", + conversationId: "test", + toolCallId: "draft-1", + output: { awaiting: "client" }, + position: position(), + }); + await settleInitialStream.promise; + } else { + await options?.onEvent?.({ + type: "message-delta", + conversationId: "test", + messageId, + kind: "text", + delta: "Drafted; press Run when you want it started.", + position: position(), + }); + } + await options?.onEvent?.({ + type: "message-completed", + conversationId: "test", + messageId, + position: position(), + }); + await options?.onEvent?.({ + type: "submission-settled", + conversationId: "test", + submissionId, + outcome: "completed", + position: position(), + }); + }); + const client = { send, wait } as Pick< + FlueClient, + "send" | "wait" + > as FlueClient; + const mutationRecorder = createJoinedBrowserMutationRecorder({ + handle, + binding: { + conversationId: "test", + documentId: handle.id, + incarnationId: "incarnation", + }, + }); + + render( + + + "Support desk", + }), + interactiveTools: [ + createObservedDraftTool(toolOutputSubmitted.resolve), + ], + conversationId: "test", + requestStop: async () => "already-settled", + transport: createBrunchPanelTransport( + Promise.resolve(client), + tracker, + { + clientToolNames: batchedConstructionClientToolNames, + dynamicClientToolNames: brunchPetrinautDynamicToolNames, + validatedClientToolNames: + mutationRecorder.validatedClientToolNames, + }, + ), + }} + /> + + , + ); + + const showPanel = await screen.findByRole("button", { + name: "Show AI assistant", + }); + await act(async () => fireEvent.click(showPanel)); + const composer = await screen.findByRole("textbox", { + name: "Message AI assistant", + }); + fireEvent.change(composer, { + target: { value: "We need to decide how many agents to schedule." }, + }); + await act(async () => { + fireEvent.click(screen.getByRole("button", { name: "Send message" })); + }); + + await serverValidationPending.promise; + await act(async () => {}); + expect( + screen.queryByRole("region", { name: "Drafted experiment" }), + ).toBeNull(); + + // Release the browser input only after Brunch has accepted the draft, then + // hold the first response open until its host-owned widget has submitted. + // This is the ordering that used to let AI SDK's stream-end continuation + // race Petrinaut's ready-state continuation. + await act(async () => releaseServerValidation.resolve()); + await toolOutputSubmitted.promise; + await act(async () => settleInitialStream.resolve()); + + // The card is rendered inside Petrinaut's tree, prepared against the live + // model, and reads as drafted with Run available. + const card = await screen.findByRole("region", { + name: "Drafted experiment", + }); + await waitFor(() => + expect(card.getAttribute("data-draft-status")).toBe( + "Drafted — not run · not saved with the document", + ), + ); + const simulate = screen.getByRole("radio", { + name: "Simulate", + }); + expect(simulate.checked).toBe(false); + expect( + document.querySelector("[data-draft-experiment-indicator]"), + ).toBeNull(); + expect(card.textContent).toContain("Vary agents 2–8 under Peak demand"); + expect(card.textContent).toContain("minimize Average waiting time"); + expect(card.textContent).toContain("No caller waits more than ten minutes."); + expect(screen.getByRole("button", { name: "Run" })).toBeTruthy(); + expect(screen.getByRole("button", { name: "Dismiss" })).toBeTruthy(); + + // The prepared result went back to Brunch as a client-tool result, and the + // follow-up turn rendered. + await waitFor(() => expect(send.mock.calls.length).toBeGreaterThanOrEqual(2)); + const followUp = send.mock.calls[1]![0].message; + expect(followUp.kind).toBe("signal"); + const results = JSON.parse((followUp as { body: string }).body) as Array<{ + toolCallId: string; + toolName: string; + output: unknown; + }>; + expect(results).toHaveLength(1); + expect(results[0]).toMatchObject({ + toolCallId: "draft-1", + toolName: draftPetrinautExperimentToolName, + output: { status: "drafted" }, + }); + expect((results[0]!.output as { diagnostics: string[] }).diagnostics).toEqual( + [ + "No constraints or constraint policy are carried; nothing is enforced.", + "Not carried: No caller waits more than ten minutes.", + ], + ); + await screen.findByText("Drafted; press Run when you want it started."); + expect(send).toHaveBeenCalledTimes(2); + + // Nothing ran: the stock active-experiments indicator is absent, and the + // card still offers Run. + expect( + screen.queryByRole("button", { name: /active Monte Carlo simulation/u }), + ).toBeNull(); + await act(async () => { + fireEvent.click(screen.getByRole("button", { name: "Run" })); + }); + expect(simulate.checked).toBe(false); + await screen.findByRole("button", { + name: "Show 1 active Monte Carlo simulations", + }); + await screen.findByText("optimizing: 5/5 runs, step 2/3"); + await act(async () => { + continueTrials.resolve(); + }); + await waitFor(() => + expect(card.getAttribute("data-draft-status")).toBe("Run complete"), + ); + expect( + screen.queryByRole("button", { name: /active Monte Carlo simulation/u }), + ).toBeNull(); + expect(screen.getByRole("radio", { name: "Simulate" })).toBe(simulate); + expect(simulate.checked).toBe(false); + expect(send).toHaveBeenCalledTimes(2); +}, 30_000); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-petrinaut-tools.test.ts b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-petrinaut-tools.test.ts index 8bf4ac55ad6..f70a365ab28 100644 --- a/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-petrinaut-tools.test.ts +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/brunch-petrinaut-tools.test.ts @@ -10,7 +10,7 @@ import { } from "@hashintel/petrinaut-core"; import { petrinautDocsContent } from "@hashintel/petrinaut/ui"; -import { brunchPetrinautDynamicToolNames } from "./brunch-client-tools"; +import { brunchPetrinautAutomaticToolNames } from "./brunch-client-tools"; import { createBrunchPetrinautTools } from "./brunch-petrinaut-tools"; import { observeBrowserDefinition } from "./mutation-record"; @@ -102,7 +102,7 @@ const toolNamed = ( }; describe("Brunch-named Petrinaut tools", () => { - test("mounts every Brunch-named tool as a dynamic host tool", () => { + test("mounts every Brunch-named automatic tool as a dynamic host tool", () => { const tools = createBrunchPetrinautTools({ readTitle: () => "Net", mutation: { @@ -114,7 +114,7 @@ describe("Brunch-named Petrinaut tools", () => { }, }); expect(tools.map(({ toolName }) => toolName).toSorted()).toEqual( - [...brunchPetrinautDynamicToolNames].toSorted(), + [...brunchPetrinautAutomaticToolNames].toSorted(), ); expect(toolNamed(tools, "layout_petrinaut_net").visibility).toBe("hidden"); expect( @@ -243,6 +243,95 @@ describe("Brunch-named Petrinaut tools", () => { expect(output).toEqual(expect.objectContaining({ applied: true })); }); + test("shows a saved scenario and metric in the next net read after the batch that saved them", async () => { + const instance = instanceFor(twoNodeNet); + const tools = createBrunchPetrinautTools({ + readTitle: () => "Support desk", + mutation: { + binding: { + conversationId: "c", + documentId: "brunch-tools", + incarnationId: "i", + }, + }, + }); + const readNet = toolNamed(tools, "read_petrinaut_net"); + const before = readNet.execute(paramsFor(instance, {})) as { + definition: SDCPN; + observation: { sha256: string }; + }; + expect(before.definition.scenarios).toBeUndefined(); + expect(before.definition.metrics).toBeUndefined(); + + const basis = { + basisId: "basis-1", + basis: { kind: "absent" as const, reason: "Loopback tracer" }, + }; + const output = (await toolNamed(tools, "mutate_petrinaut_net").execute( + paramsFor(instance, { + observation: { + toolCallId: "call-1", + baseHash: before.observation.sha256, + }, + bases: [basis], + operations: [ + { + basisId: basis.basisId, + operationId: "add-peak", + type: "addScenario", + input: { + id: "peak-demand", + name: "Peak demand", + scenarioParameters: [ + { identifier: "active_agents", type: "integer", default: 4 }, + ], + initialState: { + type: "per_place", + content: { queue: "scenario.active_agents" }, + }, + }, + }, + { + basisId: basis.basisId, + operationId: "add-wait", + type: "addMetric", + input: { + id: "average-wait", + name: "Average waiting time", + code: "return state.places.Queue.count;", + }, + }, + ], + }), + )) as { outcomes: { status: string }[] }; + expect(output.outcomes.map(({ status }) => status)).toEqual([ + "applied", + "applied", + ]); + + const after = readNet.execute(paramsFor(instance, {})) as { + definition: SDCPN; + observation: { sha256: string }; + }; + expect(after.observation.sha256).not.toBe(before.observation.sha256); + expect(after.definition.scenarios).toEqual([ + expect.objectContaining({ + id: "peak-demand", + name: "Peak demand", + scenarioParameters: [ + { identifier: "active_agents", type: "integer", default: 4 }, + ], + }), + ]); + expect(after.definition.metrics).toEqual([ + expect.objectContaining({ + id: "average-wait", + name: "Average waiting time", + }), + ]); + instance.dispose(); + }); + test("does not wait on the host for a tool that left the document unchanged", async () => { const instance = instanceFor(twoNodeNet); const settleDocumentRevision = vi.fn(() => new Promise(() => {})); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/live-pending-tool.integration.test.ts b/apps/petrinaut-website/src/main/app/local-storage-demo/live-pending-tool.integration.test.ts index 680722f0d8e..8ae2dd6b93e 100644 --- a/apps/petrinaut-website/src/main/app/local-storage-demo/live-pending-tool.integration.test.ts +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/live-pending-tool.integration.test.ts @@ -286,6 +286,7 @@ test("bounds a native OpenAI tool row without replaying completed tool work", as expect(requests.at(2)?.model).toBe(requests.at(1)?.model); expect(requests.at(2)?.reasoning).toEqual(requests.at(1)?.reasoning); await Promise.all(attempts.map((attempt) => attempt.cancelled)); + await consumed.catch(() => {}); const retryToolCallId = attempts.at(1)?.toolCallId; const terminalMessage = observedMessages.findLast((message) => message.parts.some( @@ -321,7 +322,7 @@ test("bounds a native OpenAI tool row without replaying completed tool work", as expect(erroredRows).toHaveLength(2); expect( erroredRows.every( - (toolRow) => toolRow.getAttribute("aria-busy") !== "true", + (erroredRow) => erroredRow.getAttribute("aria-busy") !== "true", ), ).toBe(true); expect(stall.chronology()).toEqual([ @@ -330,7 +331,6 @@ test("bounds a native OpenAI tool row without replaying completed tool work", as { kind: "started", toolCallId: attempts.at(1)?.toolCallId }, { kind: "cancelled", toolCallId: attempts.at(1)?.toolCallId }, ]); - await consumed.catch(() => {}); const afterFailure = await client.history(); expect( diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/local-storage-demo-app.test.tsx b/apps/petrinaut-website/src/main/app/local-storage-demo/local-storage-demo-app.test.tsx index a05c8de6c3f..bab6abb80ab 100644 --- a/apps/petrinaut-website/src/main/app/local-storage-demo/local-storage-demo-app.test.tsx +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/local-storage-demo-app.test.tsx @@ -520,7 +520,14 @@ describe("local storage demo Brunch voice integration", () => { expect(aiAssistant.requestStop).toBeTypeOf("function"); expect([...brunchClientToolNames]).toEqual(["read_petrinaut_docs"]); expect(aiAssistant.executeMutation).toBeTypeOf("function"); - expect(aiAssistant.interactiveTools).toEqual([]); + // The only interactive tool is the session experiment draft card; the + // voice-only brunch_ask widget is never mounted here. + expect( + aiAssistant.interactiveTools?.map(({ toolName }) => toolName), + ).toEqual(["draft_petrinaut_experiment"]); + expect(editorProps.current?.slots).not.toHaveProperty( + "simulateModeIndicator", + ); expect(aiAssistant.resolveToolPresentation).toBeTypeOf("function"); expect(aiAssistant.workingLabel).toBe("Brunch is working"); expect( @@ -1377,6 +1384,7 @@ describe("local storage demo Brunch controls", () => { incarnationId, }); expect([...(transportOptions.clientToolNames ?? [])].toSorted()).toEqual([ + "draft_petrinaut_experiment", "layout_petrinaut_net", "mutate_petrinaut_net", "read_petrinaut_diagnostics", @@ -1389,6 +1397,7 @@ describe("local storage demo Brunch controls", () => { "read_petrinaut_diagnostics", "layout_petrinaut_net", "mutate_petrinaut_net", + "draft_petrinaut_experiment", ]); }); }); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/local-storage-demo-app.tsx b/apps/petrinaut-website/src/main/app/local-storage-demo/local-storage-demo-app.tsx index fde0ddf5766..842cb45a5d3 100644 --- a/apps/petrinaut-website/src/main/app/local-storage-demo/local-storage-demo-app.tsx +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/local-storage-demo-app.tsx @@ -79,6 +79,7 @@ import { brunchPetrinautDynamicToolNames, } from "./brunch-client-tools"; import { ordinaryConstructionConversationIdFrom } from "./brunch-conversation-id"; +import { createBrunchDraftExperimentInteractiveTool } from "./brunch-draft-experiment-interactive-tool"; import { BrunchPanelConversationTracker, type BrunchPanelAdmissionTarget, @@ -836,7 +837,17 @@ export const LocalStorageDemoApp = ({ }), }), }), - interactiveTools: [], + // The drafted-experiment card is the one Brunch tool the person answers + // in the panel: it prepares against the live model, reports "drafted", + // and waits for Run or Dismiss. Session-only; nothing is persisted. + interactiveTools: + flueClientPromise === null + ? [] + : [ + createBrunchDraftExperimentInteractiveTool({ + readTitle: () => currentNetTitle, + }), + ], transport: petrinautAiChatTransport, ...(mutationRecorder === undefined ? {} diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/mutate-petrinet-tool.test.ts b/apps/petrinaut-website/src/main/app/local-storage-demo/mutate-petrinet-tool.test.ts index 197561bf656..9c767bea92d 100644 --- a/apps/petrinaut-website/src/main/app/local-storage-demo/mutate-petrinet-tool.test.ts +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/mutate-petrinet-tool.test.ts @@ -248,6 +248,66 @@ const stateRemovals = [ }, ]; +/** The saved scenario and metric an experiment names, then edited and removed by ID. */ +const experimentEntities = [ + { + operationId: "add-peak", + type: "addScenario" as const, + input: { + id: "peak-demand", + name: "Peak demand", + scenarioParameters: [ + { + identifier: "active_agents", + type: "integer" as const, + default: 4, + }, + ], + initialState: { + type: "per_place" as const, + content: { queue: "scenario.active_agents" }, + }, + }, + }, + { + operationId: "add-wait", + type: "addMetric" as const, + input: { + id: "average-wait", + name: "Average wait", + code: "return state.places.Queue.count;", + }, + }, + { + operationId: "describe-peak", + type: "updateScenario" as const, + input: { + scenarioId: "peak-demand", + update: { description: "Monday morning arrivals" }, + }, + }, + { + operationId: "rename-wait", + type: "updateMetric" as const, + input: { + metricId: "average-wait", + update: { name: "Average waiting time" }, + }, + }, +]; +const experimentEntityRemovals = [ + { + operationId: "drop-wait", + type: "removeMetric" as const, + input: { metricId: "average-wait" }, + }, + { + operationId: "drop-peak", + type: "removeScenario" as const, + input: { scenarioId: "peak-demand" }, + }, +]; + const run = ( tool: ReturnType, instance: ReturnType, @@ -280,6 +340,8 @@ const inputFor = ( | (typeof edits)[number] | typeof colouredPlace | (typeof stateRemovals)[number] + | (typeof experimentEntities)[number] + | (typeof experimentEntityRemovals)[number] )[], ) => { const observed = observeBrowserDefinition(instance.handle); @@ -489,6 +551,105 @@ describe("mutate_petrinet automatic host tool", () => { instance.dispose(); }); + test("saves, edits and removes the scenario and metric an experiment names, each verified at the receiving boundary", async () => { + const instance = createInstance(); + const recorder = createBrowserMutationRecorder({ + handle: instance.handle, + binding, + requestFor: (toolCallId) => { + throw new Error(`Unexpected requestFor(${toolCallId})`); + }, + }); + const tool = createMutatePetrinetAutomaticTool(binding, { + retainAttempt: recorder.retainAttempt, + }); + await run(tool, instance, inputFor(instance, [place]), "batch-place"); + + const saved = mutatePetrinetOutputSchema.parse( + await run( + tool, + instance, + inputFor(instance, experimentEntities), + "batch-experiment-entities", + ), + ); + expect(saved.outcomes.map(({ status }) => status)).toEqual( + experimentEntities.map(() => "applied"), + ); + expect(instance.definition.get()).toMatchObject({ + scenarios: [ + { + id: "peak-demand", + description: "Monday morning arrivals", + scenarioParameters: [ + { identifier: "active_agents", type: "integer" }, + ], + initialState: { + type: "per_place", + content: { queue: "scenario.active_agents" }, + }, + }, + ], + metrics: [{ id: "average-wait", name: "Average waiting time" }], + }); + // The first scenario and metric each land as one entity creation; the + // edits attribute only their requested field. + const directPaths = (operationId: string) => + saved.outcomes + .filter((outcome) => outcome.operationId === operationId) + .flatMap((outcome) => + outcome.status === "applied" + ? outcome.effects + .filter((effect) => effect.classification === "direct") + .map((effect) => effect.path) + : [], + ) + .sort(); + expect(directPaths("add-wait")).toEqual([ + "/metrics/0/code", + "/metrics/0/id", + "/metrics/0/name", + ]); + expect(directPaths("describe-peak")).toEqual(["/scenarios/0/description"]); + expect(directPaths("rename-wait")).toEqual(["/metrics/0/name"]); + + const removed = mutatePetrinetOutputSchema.parse( + await run( + tool, + instance, + inputFor(instance, experimentEntityRemovals), + "batch-experiment-removals", + ), + ); + expect(removed.outcomes.map(({ status }) => status)).toEqual( + experimentEntityRemovals.map(() => "applied"), + ); + expect(instance.definition.get()).toMatchObject({ + scenarios: [], + metrics: [], + }); + + const records = recorder + .records() + .filter((record) => + record.attempts.some((attempt) => + attempt.request.toolCallId.startsWith("batch-experiment-"), + ), + ); + expect(records).toHaveLength( + experimentEntities.length + experimentEntityRemovals.length, + ); + const verified = await Promise.all( + records.flatMap((record) => + record.attempts.map((attempt) => verifyMutationAttempt(attempt)), + ), + ); + expect(verified.map(({ outcome }) => outcome)).toEqual( + verified.map(() => "applied"), + ); + instance.dispose(); + }); + test("applies invalid dynamics without treating structural success as compiler-clean", async () => { const instance = createInstance(); const tool = createMutatePetrinetAutomaticTool(binding); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/mutate-petrinet-tool.ts b/apps/petrinaut-website/src/main/app/local-storage-demo/mutate-petrinet-tool.ts index 95fb05fc334..bc8b6790d7a 100644 --- a/apps/petrinaut-website/src/main/app/local-storage-demo/mutate-petrinet-tool.ts +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/mutate-petrinet-tool.ts @@ -49,6 +49,12 @@ const executeCanonicalMutation = ( | "removeTypeElement" | "removeParameter" | "removeDifferentialEquation" + | "addScenario" + | "updateScenario" + | "removeScenario" + | "addMetric" + | "updateMetric" + | "removeMetric" >, operation: SelectedMutationOperation, ): void => { @@ -119,6 +125,24 @@ const executeCanonicalMutation = ( case "removeDifferentialEquation": mutations.removeDifferentialEquation(operation.input); break; + case "addScenario": + mutations.addScenario(operation.input); + break; + case "updateScenario": + mutations.updateScenario(operation.input); + break; + case "removeScenario": + mutations.removeScenario(operation.input); + break; + case "addMetric": + mutations.addMetric(operation.input); + break; + case "updateMetric": + mutations.updateMetric(operation.input); + break; + case "removeMetric": + mutations.removeMetric(operation.input); + break; default: { operation satisfies never; throw new Error("Unsupported mutate_petrinaut_net operation"); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/mutation-record.test.ts b/apps/petrinaut-website/src/main/app/local-storage-demo/mutation-record.test.ts index cddd4b484c5..431aff8993f 100644 --- a/apps/petrinaut-website/src/main/app/local-storage-demo/mutation-record.test.ts +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/mutation-record.test.ts @@ -1,6 +1,7 @@ import { expect, test } from "vitest"; import { + draftPetrinautExperimentToolName, mutatePetrinautNetToolName, readPetrinautNetToolName, } from "@hashintel/brunch-agent-plugin-sdcpn"; @@ -34,10 +35,11 @@ const createRecorder = () => { return { handle, recorder }; }; -test("validates only the canonical batched mutation tool", () => { +test("waits for server validation before releasing provenance-bound client tools", () => { const { recorder } = createRecorder(); expect([...recorder.validatedClientToolNames]).toEqual([ mutatePetrinautNetToolName, + draftPetrinautExperimentToolName, ]); }); diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/mutation-record.ts b/apps/petrinaut-website/src/main/app/local-storage-demo/mutation-record.ts index 7483209d4d8..97fff183288 100644 --- a/apps/petrinaut-website/src/main/app/local-storage-demo/mutation-record.ts +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/mutation-record.ts @@ -7,6 +7,7 @@ import { classifyMutationOutcome, deriveLayoutEffects, deriveMutationEffects, + draftPetrinautExperimentToolName, isLayoutPetrinautNetToolName, isMutatePetrinautNetToolName, isReadPetrinautNetToolName, @@ -24,12 +25,15 @@ import { } from "@hashintel/brunch-agent-plugin-sdcpn"; import type { FlueChatTransportOptions } from "@hashintel/brunch-agent-transport-aisdk"; -import type { PetrinautDocHandle } from "@hashintel/petrinaut-core"; +import type { PetrinautDocHandle, SDCPN } from "@hashintel/petrinaut-core"; import type { PetrinautAiAssistant } from "@hashintel/petrinaut/ui"; type MutationExecutor = NonNullable; type MutationOutput = ReturnType; +export const hashBrowserDefinition = (definition: SDCPN): string => + bytesToHex(sha256(new TextEncoder().encode(JSON.stringify(definition)))); + /** Read the bound handle, never the request or React's last rendered snapshot. */ export const observeBrowserDefinition = ( handle: PetrinautDocHandle, @@ -39,9 +43,7 @@ export const observeBrowserDefinition = ( const definition = structuredClone(live); return { definition, - sha256: bytesToHex( - sha256(new TextEncoder().encode(JSON.stringify(definition))), - ), + sha256: hashBrowserDefinition(definition), revisionId: handle.revisionId.get(), }; }; @@ -441,6 +443,9 @@ export const createJoinedBrowserMutationRecorder = (input: { mapClientToolInput, clientToolResultMetadata, clientToolResultOutput, - validatedClientToolNames: new Set([mutatePetrinautNetToolName]), + validatedClientToolNames: new Set([ + mutatePetrinautNetToolName, + draftPetrinautExperimentToolName, + ]), }; }; diff --git a/apps/petrinaut-website/src/main/app/local-storage-demo/use-persisted-state.ts b/apps/petrinaut-website/src/main/app/local-storage-demo/use-persisted-state.ts index fddda3c337b..78e0ead4182 100644 --- a/apps/petrinaut-website/src/main/app/local-storage-demo/use-persisted-state.ts +++ b/apps/petrinaut-website/src/main/app/local-storage-demo/use-persisted-state.ts @@ -46,6 +46,8 @@ export const usePersistedState = ({ setValue(next); }; refresh(); + // Hydrate from external storage after mount, then expose its readiness. + // eslint-disable-next-line react-hooks-js/set-state-in-effect -- Tracks completion of the localStorage synchronization above, including mode changes. setLoadedMode(enabled); if (!enabled || storageKey === undefined) return; const onStorage = (event: StorageEvent) => { diff --git a/libs/@hashintel/brunch-agent/MISSION.md b/libs/@hashintel/brunch-agent/MISSION.md index c8c160ca8db..ee65e634818 100644 --- a/libs/@hashintel/brunch-agent/MISSION.md +++ b/libs/@hashintel/brunch-agent/MISSION.md @@ -1,137 +1,154 @@ -# FE-1650 — Expose Brunch and Voice in Labs +# Mission 8 — Propose and run a Ledger-derived experiment, the user pressing Run (FE-1745) ## Status -Implementation-complete and verified on -`kostandin/fe-1650-brunch-voice-settings` for -[FE-1650](https://linear.app/hash/issue/FE-1650/let-demo-users-select-brunch-and-opt-into-voice-from-settings); -draft [PR #9793](https://github.com/hashintel/hash/pull/9793) is open for review -with no known implementation blocker. +Live, not accepted, on `kostandin/fe-1745-ledger-derived-experiment`, rebased onto `main` after Lu's `ln/fe-1573-mission-7e-express` merged in [PR #9761](https://github.com/hashintel/hash/pull/9761) as an engineering partial, with WP-E accepted and re-closed but the mission not accepted or product-scalable. Lu authorized this second live mission on its own branch, the Linear issue [FE-1745](https://linear.app/hash/issue/FE-1745), and the merge posture against the 7e branch on 2026-09-16. Mission 7e is inherited through `main` as the branch base but not consumed as this mission; this branch carries exactly one live mission. -The demo now exposes the existing Stock-or-Brunch choice and a separate, -default-off Voice preference in **Settings → Labs**. Both choices persist, the -complete assistant configurations and provider histories remain isolated, and -effective Voice remains gated on Brunch selection, the user preference, and -server-reported capability. +**Established base.** Mission 7d/7e provide a settled Markdown Ledger with evidence-validated `mutate_workpiece` settlements, fresh-base `read_petrinaut_net` observations with verified identity, browser-executed construction tools through the website wrappers with a durability barrier, and the batched-construction skill flow. Chris's execution host (FE-1666, FE-1667, FE-1668) provides `petrinautExperimentRequestSchema`, `PetrinautExperimentHost.runExperiment`, `ExperimentsProvider` records, progress and cancellation, the Simulate → Experiments view, and the top-bar active-experiments indicator with navigation to details. The [Mission 8 draft](docs/mission-drafts/8-experiment-configuration-from-the-ledger.md) holds the terrain read, the Ledger-to-experiment correspondence table and the accepted design A in its refined form; every terrain claim there was re-verified at this fork. -This cut adds exactly two website-owned controls through one optional -host-content slot in Petrinaut. It does not redesign assistant switching, -active-turn lifecycle, provider history, Voice transport, server policy, or -Petrinaut's settings model. No paid Voice or model run was authorized or used. +**Observed gap.** Nothing in Brunch turns a settled decision, measure, tunable quantity and regime into an experiment. `createExperiment` is a stock-assistant tool, not in Brunch's catalogue; `prepareExperiment` is pure and separable but not exported from `@hashintel/petrinaut/react`; the mounted `mutate_petrinaut_net` union has no scenario or metric operations, so Brunch cannot construct the saved scenario and saved metric every experiment binds to; the skill has no readiness guidance, so the model waits for the words "experiment" or "optimize". The 2026-09-16 demo feedback asked for Brunch to recognise the opportunity itself. -The next authorized action is draft-PR review. The real demo route has already -shown both controls, persistence across reload, and Voice exposure only when -all three gates permit it; the retained capture is a local candidate, not -deployed-environment acceptance. +**Where this stands.** Work packages A–E are on the branch ([PR #9762](https://github.com/hashintel/hash/pull/9762)): scenario and metric operations, readiness guidance, the draft tool and its carriage, the website widget, and the support-desk case with the recorded run. Canonical preparation remains exposed provisionally as a named `prepareExperiment` export pending Chris's boundary preference. The direct corrective cut now also exposes optimizer availability to the host path, refuses an unavailable optimization before creating an experiment record, and keeps a host-owned run active while its sweep is idle between optimizer steps; model timing, metric mechanics and optimizer search semantics remain untouched. What remains is Chris's boundary choice, Lu's spend authorization for one paid persona run of the support-desk case (the behavioural half of Ledger-driven readiness), and the owners' witness of the demo. ### Owner decisions -- **2026-09-20 — Kostandin, location:** place the controls under **Settings → Labs**. -- **2026-09-20 — Kostandin, choices:** expose both **Use Brunch** and a separate, default-off **Enable Voice** preference. -- **2026-09-20 — Kostandin, reduced scope:** use one optional React-node Labs slot rather than a typed settings DSL; keep the current immediate switching lifecycle and add no busy-state API; add only minimal tests for the new behavior. -- **Inherited product decisions:** Stock remains the ordinary-route fallback, explicit stored assistant choices remain authoritative without migration, the command-palette switch remains available, and worked-model routes continue to force Brunch. +- **2026-09-16 — Lu, design:** design A in its refined form (Brunch drafts, Petrinaut's own preparation validates and holds, the user presses Run) is the practical approach now. Experiments are routine modelling entities for the user and therefore for the agent — a prompt and skill principle, not a document-model change. Design E is an option to open with Chris, not a destination. +- **2026-09-16 — Kostandin, A1 submission shape:** the prepared proposal auto-submits so Brunch can continue the turn; the session-only widget stays available for Run or Dismiss. The widget does not navigate; the stock indicator and its navigation are reused. +- **2026-09-16 — Kostandin, constraint scope:** the first demo is constraint-free. Restrictions land in `unsupported`; an observed metric beside them is optional and must be labelled "reported, not enforced". A hard restriction is never weakened or encoded as a penalty. +- **2026-09-16 — Kostandin, widget scope:** a minimal website-owned "Drafted — not run" widget is authorized. No persistence, draft list, draft status, drawer prefill or broader experiment lifecycle. +- **2026-09-16 — Kostandin, demo:** the support-desk staffing scenario is the first demo case. +- **2026-09-16 — Kostandin, direct corrective scope:** fix only the draft-to-run experiment path in this PR. The card must not offer Run when its optimizer is unavailable; the host must reject that condition before creating an experiment record; and a host-owned optimization must remain visible in the shipped active-experiments indicator between steps. Do not change the support model, time units, wait metric, finite-range search semantics or Petrinaut Core. +- **2026-09-16 — Chris, upstream delta:** canonical experiment preparation is exposed without execution; no constraint carriage or UI change is requested in this cut. The boundary — a named `prepareExperiment` export from `@hashintel/petrinaut/react` or `prepareExperiment(request)` on `PetrinautExperimentHost` — is pending Chris's preference. This branch carries the named export provisionally because it is the smaller change and the widget can move to the host method without changing anything else; the commit is dropped or replaced if Chris prefers the host method. +- **2026-09-16 — Lu, lifecycle:** second live mission on a separate branch; issue FE-1745; PR base is the 7e branch until it merges, and the `MISSION.md` conflict at that point is Lu's call. The consumed draft stays in place as the residual home for material this cut does not admit (constraints, the Inventory example, design E, drawer prefill), because those items have no other home yet; it remains a non-authority draft. ## Imperative -A demo user can discover and select Brunch, then explicitly opt into Voice, from **Settings → Labs** without knowing the command-palette shortcut. Stock remains the default and Voice remains off until the user enables it. +During ordinary modelling — not after the user asks for optimization — when the Ledger settles a decision, a measure with direction, a tunable quantity with a user-stated range and unit, and an operating regime and horizon, and the net carries the saved scenario, typed scenario parameter and saved metric those need, Brunch proposes the experiment that measures them in the same breath as it proposes a place or a scenario. The proposal is prepared by Petrinaut against the live model, shown unrun in the panel, and started only when the user presses Run. Missing scenarios and metrics are constructed, not invented; missing operational facts are asked for; restrictions the request cannot carry are listed, not dropped. -Why now: changing the ordinary-route default to Stock made Voice undiscoverable to anyone who does not know the hidden **Use Brunch** command. +Why now: the execution substrate exists and works from the stock assistant; the demo failed only on the Brunch side, where the user had to know to ask. Another demo without proactive recognition would repeat the known gap. ## Throughline ```text -user opens Settings → Labs -→ Petrinaut renders one optional host-provided React node -→ website renders Use Brunch and Enable Voice controls -→ Use Brunch updates the existing petrinaut-website:assistant preference -→ the existing brunchSelected branch switches the complete assistant configuration -→ Enable Voice updates a separate browser-local, default-off preference -→ effective Voice requires Brunch selected + Voice enabled + server capability -→ Petrinaut receives renderVoiceMode only when all three are true -→ reload restores both preferences without starting microphone capture or playback +user discusses an operational decision +→ Brunch settles the Ledger; the settlement that completes readiness carries + the decision, measure + direction, tunable + range + unit, regime + horizon, + restrictions and what the result must not claim +→ read_petrinaut_net confirms scenario, scenario-parameter and metric identities +→ mutate_petrinaut_net adds the missing saved scenario / metric (new operations) +→ draft_petrinaut_experiment: server tool awaits the client; payload keeps the + core PetrinautExperimentRequest separable from Brunch's declarations, basis + and unsupported envelope +→ website dynamic interactive tool renders the widget inside the Petrinaut tree +→ widget calls prepareExperiment(request, liveDefinition, title); auto-submits + {status: drafted | invalid, summary, diagnostics}; Brunch's turn continues +→ widget shows "Drafted — not run · not saved with the document", declarations, + unsupported list, Run and Dismiss; a later draft supersedes the earlier card +→ user presses Run → re-prepare against the model now, show any difference, + experimentHost.runExperiment(request) +→ stock ExperimentsProvider record, progress, Cancel; "N active" indicator; + Simulate → Experiments updates live; completion clears the indicator and + leaves the result ``` -The ordinary route uses both controls. A worked-model route still forces Brunch and cannot change assistant provider; its Voice preference may still be changed. Missing Brunch configuration and unavailable Voice remain visibly unavailable rather than implying a working service. +### Cold-start reads + +Paths are relative to this context root unless prefixed `../../../`. + +- **Planning record:** [Mission 8 draft](docs/mission-drafts/8-experiment-configuration-from-the-ledger.md) — terrain at HEAD, the correspondence table, the design assessment and the implementer's entry. [Draft Mission 11](docs/mission-drafts/11-optimisation-handoff.md) owns anything a consumer runs, receives or judges; this mission ends at one user-started run. +- **Petrinaut experiment terrain:** [`prepare-experiment.ts`](../../petrinaut/src/react/experiment-host/prepare-experiment.ts), [`run-experiment.ts`](../../petrinaut/src/react/experiment-host/run-experiment.ts), [`experiment-host/provider.tsx`](../../petrinaut/src/react/experiment-host/provider.tsx), [`react/index.ts`](../../petrinaut/src/react/index.ts), core [`experiments/host.ts`](../petrinaut-core/src/experiments/host.ts), and the interactive-widget extension [`ai-interactive-tool.ts`](../../petrinaut/src/ui/types/ai-interactive-tool.ts) with its rendering in the assistant panel's tool list. +- **Brunch tool carriage:** [`packages/plugin-sdcpn/src/flue.ts`](packages/plugin-sdcpn/src/flue.ts), [`construction-tool-names.ts`](packages/plugin-sdcpn/src/construction-tool-names.ts), [`mutate-petrinet.ts`](packages/plugin-sdcpn/src/mutate-petrinet.ts) (operation union), [`tools/mutate-petrinet.ts`](packages/plugin-sdcpn/src/tools/mutate-petrinet.ts), [`mutation-record.ts`](packages/plugin-sdcpn/src/mutation-record.ts), and the app catalogue [`tool-catalogue.ts`](../../../apps/brunch-agent/src/agents/chat-agent/tool-catalogue.ts) with its carriage check `yarn test:native-schema`. +- **Website execution:** [`brunch-client-tools.ts`](../../../apps/petrinaut-website/src/main/app/local-storage-demo/brunch-client-tools.ts), [`brunch-panel-transport.ts`](../../../apps/petrinaut-website/src/main/app/local-storage-demo/brunch-panel-transport.ts), [`brunch-petrinaut-tools.ts`](../../../apps/petrinaut-website/src/main/app/local-storage-demo/brunch-petrinaut-tools.ts), [`mutate-petrinet-tool.ts`](../../../apps/petrinaut-website/src/main/app/local-storage-demo/mutate-petrinet-tool.ts), the widget precedent [`brunch-ask-interactive-tool.tsx`](../../../apps/petrinaut-website/src/main/app/local-storage-demo/brunch-ask-interactive-tool.tsx), and the mount in `local-storage-demo-app.tsx`. +- **Guidance:** [`sdcpn-modelling/SKILL.md`](packages/plugin-sdcpn/src/skills/sdcpn-modelling/SKILL.md) (Construct disposition is the readiness hook), its [`references/`](packages/plugin-sdcpn/src/skills/sdcpn-modelling/references/), [`templates/workpiece.md`](packages/plugin-sdcpn/src/skills/sdcpn-modelling/templates/workpiece.md), and the packaging test [`sdcpn-modelling-skill.test.ts`](packages/plugin-sdcpn/test/sdcpn-modelling-skill.test.ts). +- **Evaluation:** persona cases under [`evaluations/cases/`](evaluations/cases/) (Inventory is the existing case); the launcher in [`persona/launch.ts`](../../../apps/brunch-agent/src/evaluations/persona/launch.ts). Read [execution safety](evaluations/README.md#execution-safety) before any paid run; none is authorized by this document. + +### Work packages + +#### WP-A — Scenario and metric construction + +Add `addScenario`, `updateScenario`, `removeScenario`, `addMetric`, `updateMetric` and `removeMetric` to the `mutate_petrinaut_net` operation union, the server mutation record and the website executor, each carrying the same declared basis as existing operations. Scenario parameters carry their declared `type`, because the sweep domain derives from it; a count is an integer-typed parameter, never a rounded real. The skill teaches scenarios and metrics as ordinary construction, so the experiment's prerequisites are built when the Ledger's conditions are settled, not when the experiment is proposed. + +#### WP-B — Readiness guidance + +One skill reference (`experiment-configuration.md`) carrying the correspondence table, the routine-entity principle with its honesty rule ("drafted for this session", not "added to the model"), the readiness rule, the mandatory declarations, the refusals and the shape of the one-sentence proposal. Readiness is Ledger meaning and model executability together: the Ledger states the decision, the measure and its direction, the tunable quantity with a user-stated range and unit, and the regime and horizon; the model has the saved scenario, an integer- or real-typed scenario parameter for the tunable, and a saved metric for the measure. Structure alone never triggers; the model alone never supplies the objective. Propose once when readiness is first reached or when its meaningful configuration changes; a prior draft is not repeated for an unchanged configuration. Run, Dismiss and completion happen after the tool result and are not reported to the conversation, so the skill does not treat them as observed; a refusal stated in conversation remains authoritative. When a fact is missing, ask the smallest focused question. Restrictions the request cannot carry go to `unsupported`, stated to the user before Run is offered. + +#### WP-C — The draft tool and its carriage + +One Brunch client tool, `draft_petrinaut_experiment`, mounted in batched-construction mode and awaiting the client like the existing five. Its payload is `{ experiment: PetrinautExperimentRequest, declarations, basis, unsupported }`: core's request schema unchanged and separable; Brunch adds only provenance and disclosure. It is named in the website's dynamic and batched-construction client tool sets, the plugin's `draft-experiment.ts` and the app catalogue, so native schema carriage is checked by the existing test. + +#### WP-D — The website-owned widget + +One `definePetrinautAiInteractiveTool` definition. On mount it calls `prepareExperiment` against the live definition and auto-submits `{status, summary, diagnostics}` so the model's turn continues. It renders the prepared proposal as "Drafted — not run · not saved with the document" with declarations, the unsupported list, Run and Dismiss. One current proposal exists per session; an earlier card shows "Superseded". Run re-prepares, surfaces any material difference, then calls `experimentHost.runExperiment` and shows the stock progress and Cancel. It does not navigate, persist, list, or resolve identities itself. + +On this branch: `apps/petrinaut-website/src/main/app/local-storage-demo/brunch-draft-experiment-interactive-tool.tsx` exports the definition factory (`createBrunchDraftExperimentInteractiveTool({ readTitle })`, mounted in `local-storage-demo-app.tsx` as the panel's only interactive tool when a Flue client exists) and the card component; its private subtree holds the session store (`session-drafts.ts`, scoped to the editor's definition store and keyed by tool call, current = last newly registered) and the pure describers (`describe-draft.ts`: the one-sentence summary, budget line, metric roles, agent summary, and the structural comparison Run uses). Input and output schemas are the plugin's own Zod schemas, so the widget cannot drift from the server tool. The card reads the host's optimizer availability: an unavailable draft remains dismissible, reports the blocker in its submitted diagnostics and renders no Run button. A rejected auto-submission leaves the candidate unregistered, reports the failure on the current card and offers Retry preparation; only a host-accepted retry registers the draft and exposes Run. + +Review fixes are covered by `brunch-draft-experiment-interactive-tool.test.tsx`: dismissed cards stay dismissed on remount, separate editors do not share drafts, duplicate Run actions start once, and each successive model change requires fresh approval, including simulation-only initial state. The card compares the full model snapshot and exposes changed sections before confirmation. Unsupported restrictions block Run by default; `blocksRun: false` is reserved for explicitly accepted reporting-only exploration, not implicit acceptance from a reporting metric. `draft-experiment.test.ts` rejects a reporting metric absent from the request. The support-desk persona's hard ten-minute rule therefore blocks the run unless the person accepts that exploration; the existing scripted recording does not establish that acceptance or constraint enforcement. + +#### WP-E — The support-desk demo + +A support-desk staffing case (`evaluations/cases/support-desk-staffing/`): agents 2–8 as an integer scenario parameter on a saved "Peak demand" scenario, average waiting time as a saved metric, two-hour horizon, minimize. The persona never says "experiment" or "optimize". One agent-browser recording of the flow from proposal through Run to completion is the reviewable artifact. + +On this branch: the case (`opening-message.md`, `situation-pack.md`; `yarn brunch:persona --list-cases` lists `support-desk-staffing`) gives Priya Nandakumar of Brightwater Utilities a 10:00–12:00 weekday peak at about 50 calls an hour, six-minute handle times, an integer rota of two to eight agents, average waiting time to minimize, and a hard ten-minute rule the persona holds even though the request cannot carry it. The persona is forbidden the words "experiment", "optimi-" and "sweep" and approves a run only when asked. No paid persona run has been made; the behavioural half of "Readiness is Ledger-driven" is still owed. The recording was made against a scripted transport that streams one `draft_petrinaut_experiment` call with the case's request into the real panel, widget and `runExperiment` on a hand-built support-desk net (calls carry their remaining handle time; one Finish per call), so it proves the panel-to-execution path and the stock indicator, popover, navigation, live Experiments drawer and completion — not Brunch's recognition. Two prerequisites for an optimize run in the browser: the route must mount `BrowserOptimizationProvider` (the website's `/` and `/ai-experiments` do) and the person must have **Parameter sweeps** and **In-browser optimization** on in Settings → Simulation; otherwise the draft reports "Optimization is unavailable", offers no Run and creates no experiment record. ## Proof -The narrow proof obligations and their current dispositions are: - -- **Optional host surface — PASS.** The focused Petrinaut settings suite shows - supplied Labs content, confirms an unsupplied host remains unchanged, and - exercises first/last enabled host-control entry, vertical traversal, - built-in/host boundary crossing, and ArrowLeft return: - `NODE_OPTIONS=--no-experimental-webstorage yarn workspace - @hashintel/petrinaut test:unit --run - src/ui/views/Editor/editor-view/user-settings.test.tsx` — 26/26 passed. The - Node option disables Node 26's experimental web-storage global so jsdom owns - `localStorage`; without it, the environment fails at `localStorage.clear()` - before test behavior runs. -- **Website behavior — PASS.** `NODE_OPTIONS=--no-experimental-webstorage yarn - workspace @apps/petrinaut-website test:unit --run - src/main/app/local-storage-demo/assistant-labs-settings.test.tsx - src/main/app/local-storage-demo/local-storage-demo-app.test.tsx` — 47/47 - passed. Coverage includes assistant-preference loading, missing Brunch - configuration, forced Brunch, Voice-preference loading, Voice-capability - loading, invalid Voice storage reading off, both rendered Labs controls - writing their own keys, remount restoration, default-off Voice, and the - three-part effective Voice gate. -- **Visible path — LOCAL CANDIDATE COMPLETE.** The retained, ignored capture at - `.superpowers/sdd/task-2-ui-capture/` contains five inspected 1440×900 states: - `01-stock-selected.png` (Stock; Voice disabled), - `02-brunch-voice-unavailable.png` (Brunch; unavailable), - `03-brunch-voice-available.png` (available and off), - `04-voice-enabled.png` (enabled), and `05-reload-persisted.png` (both choices - restored). Every artifact named by `manifest.json` exists and its exact byte - count matches the manifest, including the five screenshots, final - screenshot, JSON/text logs, HAR, and video. `errors.json` is empty; - `audio-tap-status.json` reports zero input/output sources and chunks. The HAR - contains only `localhost`, blocked `127.0.0.1:9`, and null origins; media- or - provider-named matches are local Vite source-module GETs, not provider or - media-session requests. No paid provider was contacted and no media session - started. Limitation: the optional 262,144-byte VP8 video encoder fell behind; - the manifest recorded 3.56 seconds, but the prematurely ended file no longer - yields a format duration to `ffprobe`. The five screenshots are the visual - proof. -- **Package integrity — PASS.** `npx turbo run build lint:tsc lint:eslint - --filter '@hashintel/petrinaut' --filter '@apps/petrinaut-website'` completed - 18/18 tasks; `yarn workspace @local/petrinaut-arch-docs lint:arch-docs` - passed with 84 layers, 440 edges, 924 files, 85 generated pages, and 44 - authored pages; `yarn lint:format` passed across 6,108 files. - -Existing tests remain the owners for transport routing, history isolation, conversation identity, bundle-route behavior, Voice lifecycle, and server policy. This mission does not duplicate them merely because the same selection state gains another control. +### Claim discipline + +A drafted proposal proves configuration correspondence, not experiment credibility or optimizer results. Reusing the stock indicator, records and Experiments view proves integration, not their correctness. A passing persona run proves one case, not portfolio breadth. This mission does not claim constraint enforcement of any kind, persistence of proposals, Mission 11's consumer handoff, or any change to the meaning of Petrinaut's optimizer. + +### Visible product advance + +**Release-note sentence:** While you model a decision with Brunch, it notices when the question can be tested, drafts the experiment for you, and runs it when you say so. + +**Product-manager script:** open the website demo with Brunch, turn on **Parameter sweeps** and **In-browser optimization** in Settings → Simulation, say "Help me model our support operation; we need to decide how many agents to schedule while keeping waiting times low at peak," answer Brunch's questions about the range, the measure and the peak period without using the word "experiment", watch Brunch add the scenario and metric, then see a card saying it has drafted an experiment varying agents 2–8 on Peak demand to minimize average waiting time — not run. Press Run. The top bar shows "1 active"; Simulate → Experiments shows the run progressing; on completion the indicator clears and the result stays. + +What was impossible before: the user had to know that an experiment existed, leave the conversation, create the scenario and metric by hand and author the experiment in the drawer. + +### Proof obligations and dispositions + +| Required result | Oracle | Current disposition | +| --- | --- | --- | +| Scenario and metric operations construct, update and remove with declared basis and persist through the durability barrier | `mutate-petrinet.test.ts` and `root-state.test.ts` extended with one fixture per operation, including an integer-typed scenario parameter; website `mutate-petrinet-tool.test.ts` executes each; one loopback case shows the saved scenario and metric present in the next `read_petrinaut_net` | Met on this branch. Plugin admits `addScenario`, `updateScenario`, `removeScenario`, `addMetric`, `updateMetric`, `removeMetric` (union 22 → 28); core `selectedMutationOperationSchema` mirrors them; the website executor applies each and the recorder classifies each `applied`; `brunch-petrinaut-tools.test.ts` loopback reads both back with an advanced observation hash. Caveat: the positional JSON diff attributes only a final-index array removal as a direct deletion, so an earlier-index `removeScenario`/`removeMetric` (like `removeParameter` before it) shows as derived shifts; the outcome still classifies `applied`. | +| Readiness is Ledger-driven | Skill packaging test asserts the reference is mounted and names the readiness conjunction and refusals; a fixture Ledger with parameters and metrics but no stated decision yields no proposal in a deterministic guidance check; the support-desk persona transcript contains no "experiment"/"optimi" token in persona lines yet Brunch proposes | Guidance half met on this branch: `references/experiment-configuration.md` is packaged, hooked into the Construct disposition of `SKILL.md`, and `sdcpn-modelling-skill.test.ts` asserts the conjunction, the workpiece sections it reads, the request fields it teaches, the no-constraint disclosure, the once-only rule and the refusals as text. The "no stated decision → no proposal" rule is asserted as guidance text, not as observed model behaviour; the behavioural half waits on the WP-E persona transcript. | +| Tool payload validates and carries natively | Schema fixture test: one fixture per correspondence row validates or lands in `unsupported` with the expected reason; `yarn test:native-schema` passes with the new tool in the ordinary catalogue | Met on this branch. `draft-experiment.test.ts` (18 tests) validates an integer range with a declared basis and a restriction in `unsupported` carrying `reportedByMetricId`, accepts an absent basis, and rejects an objective outside `metricIds`, `dt` above `maxTime`, budgets above the host's bound, a fixed-only optimization, an inverted range, duplicate metrics, empty `declarations`, a missing or malformed observation and — because there is no constraint carriage — any `constraints` field on the request; the server tool defers `{ awaiting: "client" }` only after the basis and the exact cited observation hash check, and refuses before the workpiece is settled. `yarn test:native-schema` passes with the tool in the ordinary catalogue and now also asserts the serialized `input_schema` equals the Zod source and still refuses `constraints`, empty `declarations` and a missing `experiment` (`NATIVE_SCHEMA_CARRIAGE {"passed":true}`). The serialized `scenarioParameterValues` record uses `additionalProperties` with a nested `oneOf`, which the adapter accepted; only top-level `oneOf`/`anyOf`/`allOf` are forbidden. | +| No execution before Run | Website integration: the tool call completes, the widget renders the drafted state, the experiments context holds no record, and the durability-barrier trace shows no `runExperiment` | Met at the widget boundary on this branch. `brunch-draft-experiment-interactive-tool.test.tsx` renders the card under a stubbed `ExperimentHostContext`: an awaiting call prepares once, submits `{status: "drafted"}` exactly once (a later `submitted` re-render submits nothing more), shows "Drafted — not run · not saved with the document" with Run and Dismiss, and the host's `runExperiment` is never called; a rejected submission exposes Retry preparation without Run, and only a successful retry registers the draft and exposes Run. A missing metric submits `{status: "invalid"}` with the preparation error and offers no Run. The unavailable-optimizer case submits the blocker, offers only Dismiss and never calls the host. `run-experiment.test.ts` proves the same host-level preflight returns `experimentId: null` without calling either `createExperiment` or `createOptimization`. Met through the real panel and transport: `brunch-draft-experiment.integration.test.tsx` renders `` with the Brunch panel transport and a fake Flue client that streams one `draft_petrinaut_experiment` tool-input; the card appears inside the tree reading "Drafted — not run", the second `send` carries the client-tool result signal with `{status: "drafted"}` and the two no-enforcement diagnostics, Brunch's follow-up text renders, the stock "active Monte Carlo simulations" indicator is absent, and Run is still offered. `local-storage-demo-app.test.tsx` asserts the tool is the panel's only interactive tool. Not covered: the durability-barrier trace, which does not observe experiment calls. | +| Second draft supersedes | Website integration: two tool calls leave one current proposal; the earlier card reads "Superseded" | Met at the widget boundary: two cards registered in one editor session leave one Run button; the earlier reads "Superseded by a later draft". Each editor has its own session store keyed by tool call; a reload forgets it, and a `submitted` card whose draft is absent reads "Not retained in this session" with no Run (tested). | +| Run starts exactly one experiment with the approved request against the model at that moment | Website integration: pressing Run creates one record whose request equals the approved request; editing the metric or parameter type between draft and Run surfaces the difference before any call | Met at the widget boundary against a stubbed host: Run calls `runExperiment` once with the prepared request and an abort signal, shows the host's progress line ("optimizing: 5/15 runs, step 1/3"), Cancel aborts the signal, completion reads "Run complete" with the run count, and a rejected run reads "Run failed" with the message. Changing the metric code between draft and Run shows "The model changed since this was drafted" and makes no call; the second press ("Run against current model") runs. Removing the scenario shows the preparation error and makes no call. The real-panel integration now continues after its idle assertion: Run creates the stock sweep through the real `ExperimentHostProvider` and `ExperimentsProvider`, an in-process Monte Carlo worker evaluates the connected optimizer's steps, the card reports step progress and reaches "Run complete". Depends on the `prepareExperiment` export on this branch. | +| Restrictions are disclosed, never enforced by claim | Fixture test: a restriction lands in `unsupported`; widget render shows it under "Not carried into execution"; any observed metric is labelled "reported, not enforced" | Met on this branch. The plugin fixture test carries a restriction with `reportedByMetricId`; the widget test renders it under "Not carried into execution" tagged "reported by , not enforced", tags the non-objective metric "reported, not enforced" and the objective "objective", and the auto-submitted diagnostics always lead with "No constraints or constraint policy are carried; nothing is enforced." followed by one `Run blocked:` or `Not carried:` line per restriction. Run is disabled unless every unsupported condition has explicit reporting-only acceptance. | +| Stock indicator, live Experiments update and completion behave as shipped | Real panel integration plus the agent-browser recording of the demo | Met for the direct path. `brunch-draft-experiment.integration.test.tsx` pauses a real host-owned optimization between steps and observes the stock top-bar "1 active" control and the card's live step progress, then releases it, observes "Run complete" and verifies the indicator clears. `context.test.ts` pins the activity rule: an ordinary idle sweep remains inactive, while an idle sweep with `requestActive` remains active until its host request settles. The 2026-09-16 agent-browser recording additionally proves the popover, navigation and live Experiments drawer against the scripted route. | +| Mission readiness | Kostandin's and Lu's witness of the support-desk demo: proposal without a request for one, unrun card, one run on Run, indicator appears and clears, result remains, no restriction dropped | Open. | ## Constraints -- Petrinaut gains only an optional `settingsLabs?: ReactNode`-style host slot and remains unaware of Stock, Brunch, Voice policy, website storage keys, or deployment configuration. -- The website owns the controls, labels, persistence, capability interpretation, and callbacks. -- Keep `petrinaut-website:assistant` and its current parsing/default behavior unchanged. Add one separate Voice preference whose missing or invalid value is off. -- Keep the command-palette provider switch and current immediate switching behavior. Do not add an assistant busy callback, idle-only transition policy, implicit durable stop, or lifecycle coordinator. -- Preserve the current complete-configuration switch. Do not splice histories, move Brunch messages into the Stock store, recreate the current net, or make Voice available under Stock. -- A Voice preference is not capability or authorization. Existing server checks remain authoritative; enabling the preference must not start media or bypass disclosure. -- Worked-model routes continue to force Brunch and preserve the ordinary-route assistant preference. -- Do not revive `brunchDemoMode`, build a generic settings schema, move website policy into `UserSettings`, or clean up unrelated assistant/Voice code. -- Synthetic tests must not reach paid providers. -- Update the affected user/configuration docs and add the required `@hashintel/petrinaut` patch changeset for the public slot. +- No Brunch experiment schema; core's `PetrinautExperimentRequest` is used unchanged and stays separable from the envelope. +- No persistence of proposals, no draft list, no draft status, no drawer prefill, no navigation from the widget, no change to optimizer search semantics, result presentation or the stock experiment card. The only indicator change is that a host-owned request remains active while its sweep is idle between optimizer steps. +- Never encode a hard restriction as an objective penalty; never state that a constraint is enforced when it does not reach execution. +- Never infer the objective from model structure alone; never invent bounds, units or thresholds; never trigger on the presence of parameters or metrics. +- Never call the draft tool as a way to run: execution begins only on the user's Run. +- The website wrapper and widget do not resolve identities or check types themselves; that is `prepareExperiment`'s job. +- Changes in Chris's package are limited to the provisional named preparation export and the owner-authorized direct availability, preflight and host-owned activity corrections; Petrinaut Core remains unchanged. +- Every Linear write needs Lu's explicit approval; every paid run needs explicit spend authorization. +- Do not edit or duplicate the Mission 8 draft's content here beyond the pointers this document needs. ## Fog-line -- The exact host-control spacing and copy should follow the existing Labs visual language, but this cut does not create a reusable host-settings component library. -- Immediate provider switching during active work is inherited behavior. An observed orphaned-work or media-cleanup failure may justify a separate lifecycle cut; this mission does not anticipate one with a new API. -- Deployed Brunch and Voice environment values are operational capability, not acceptance evidence for the browser-local controls. +- Whether the model recognises readiness reliably from the Markdown Ledger sections without typed condition slots (the spine's rule: types only after repeated retrieval failure). +- Whether auto-submit plus a persisting widget reads clearly in the panel, or the card needs an explicit "waiting for you" state. +- How one more construction step interacts with the settlement-cost pressure Mission 7e is measuring. +- Runtime failures after an available optimizer has started: which errors should preserve a failed record rather than a cancelled one. ## Stop or reorient -- Stop if the implementation requires Petrinaut to understand Brunch or website policy. -- Stop if enabling Voice can bypass server capability, disclosure, or a user gesture. -- Stop if the new control path switches anything other than the existing assistant preference, loses the current net, or mixes provider histories. -- Reassess before replacing the React-node slot with a generalized settings DSL or exposing new assistant lifecycle state. -- Do not broaden focused regression coverage into duplicate routing, history, lifecycle, or special-route suites. +- Stop WP-C/WP-D if the client-tool path cannot complete a tool call before the user acts; check the panel's interactive-tool submission path first. +- Stop the demo case if its load-bearing restriction cannot be stated honestly as unsupported and the user's meaning would be misrepresented. +- Stop if the wrapper or widget starts resolving ids, checking types or holding experiment semantics; that belongs upstream. +- Stop and return to Lu if anything beyond the preparation export or the direct availability, preflight and host-owned activity corrections is needed in Chris's package. +- If Chris lands a first-class experiment definition (design E) in this window, drop the session store and make the payload a `mutate_petrinaut_net` operation; guidance and WP-A survive. ## Deferred -- Idle-only provider changes, durable stop-before-switch, and any public busy-state API. -- A generalized typed settings-extension system or reusable host settings components. -- Stock-history semantics inside worked-model routes. -- Cross-tab live synchronization of the two preferences. -- Cleanup or replacement of the dormant `brunchDemoMode` setting and its unreachable picker. +- **Constraint carriage:** request-schema and preparation support for parameter and state constraints, the α policy and the faithful-run leaf remain in the [Mission 8 draft](docs/mission-drafts/8-experiment-configuration-from-the-ledger.md) as its `ORACLE GAP`; this cut states the limitation and does not resolve it. +- **Inventory example and avoid-state proof:** the draft's Inventory objective and avoid-state throughline remain there at full fidelity; the support-desk case does not replace them. +- **Design E and drawer prefill:** remain options in the draft's design assessment, selected only by its reversal evidence. +- **Reading results back into conversation and any consumer handoff:** [Draft Mission 11](docs/mission-drafts/11-optimisation-handoff.md). +- **Mission 7e's own Deferred items** remain with Lu's branch and the [future spine](MISSION.next.md); this mission neither adopts nor discards them. diff --git a/libs/@hashintel/brunch-agent/evaluations/README.md b/libs/@hashintel/brunch-agent/evaluations/README.md index 98877fb0b50..64d3c6e7290 100644 --- a/libs/@hashintel/brunch-agent/evaluations/README.md +++ b/libs/@hashintel/brunch-agent/evaluations/README.md @@ -21,6 +21,7 @@ Case selection does not authorize paid execution; follow the mission's allocatio Current process-model-elicitation assets: - `cases/inventory-purchasing/` — the current from-scratch flagship; its hand-built reference net remains evaluator-only. +- `cases/support-desk-staffing/` — one bounded staffing decision (agents 2–8 on a two-hour weekday peak, average waiting time, one unenforceable ten-minute rule) for Mission 8's Ledger-derived experiment proposal; the persona never asks for an experiment. - `cases/vestera-scheduling/` and `oracles/vestera-scheduling/` — the executed Vestera exemplar and its case-specific retrospective and prospective ledgers, plus the filled runbook IR used by the supported headless construction command. diff --git a/libs/@hashintel/brunch-agent/evaluations/cases/support-desk-staffing/opening-message.md b/libs/@hashintel/brunch-agent/evaluations/cases/support-desk-staffing/opening-message.md new file mode 100644 index 00000000000..84975998074 --- /dev/null +++ b/libs/@hashintel/brunch-agent/evaluations/cases/support-desk-staffing/opening-message.md @@ -0,0 +1,11 @@ +# Opening message — Brightwater support desk staffing + +**Sources and authorship.** Brightwater Utilities, Priya Nandakumar, the desk +and every number in this case are authored persona synthesis for local +development. They are not claims about a real company or employee. + +The first user message the interviewer receives. + +--- + +I'm Priya Nandakumar and I run the customer support desk at Brightwater Utilities. Help me model our support operation. We need to decide how many agents to schedule on the phones while keeping customer waiting times low during our peak demand, and I'd like to see that decision tested rather than argued about in a meeting. Please interview me about how the desk works and build the model with me as we go. diff --git a/libs/@hashintel/brunch-agent/evaluations/cases/support-desk-staffing/situation-pack.md b/libs/@hashintel/brunch-agent/evaluations/cases/support-desk-staffing/situation-pack.md new file mode 100644 index 00000000000..c459b285206 --- /dev/null +++ b/libs/@hashintel/brunch-agent/evaluations/cases/support-desk-staffing/situation-pack.md @@ -0,0 +1,170 @@ +# Situation pack — Brightwater support desk staffing + +**Private to the simulated interviewee.** This file is the system prompt for +the agent playing the user. Do not reveal or quote it to Brunch. Speak as the +operational participant; never coach Brunch about tools, schemas, Petri nets, +workpiece structure or expected model IDs. + +**Sources and authorship.** Brightwater Utilities, Priya Nandakumar, the desk, +the phone-system exports and every number below are authored persona +synthesis for local development, composed before any run. They are not claims +about a real company, employee or dataset. Material marked _(assumption)_ is a +working assumption Priya's team made, not measured evidence. Material marked +_(doesn't know)_ must not be invented. Material marked _(believes)_ is a +genuine working belief that yields to probing. + +**What this case is for.** One bounded decision — how many agents to put on +the phones in the weekday peak — with a stated measure, a stated range and +unit, a stated peak regime and horizon, and one restriction that a bounded +simulation cannot enforce. Priya never asks for an experiment, an +optimisation or a sweep; she describes the decision and answers questions. The +case exists to see whether the interviewer recognises, from what she says, +that the decision can be tested, proposes the test itself, and leaves her to +start it. + +## Role instructions + +You are role-playing **Priya Nandakumar**, customer support operations manager +at the fictional **Brightwater Utilities**, a regional water and energy +retailer. You own the phone desk's rota, its service numbers and the +relationship with the agents' team leads. You have asked for help building a +model of the desk because the staffing argument for the weekday peak keeps +recurring and nobody can show what a different headcount would do to waiting +times. + +Behavioural rules, in priority order: + +1. **Answer only what is asked.** A few sentences at a time. Volunteer at most + one adjacent fact where a desk manager naturally would. +2. **Speak desk language.** Calls, callers, queue, agents on the phones, the + peak, handle time, hold, hang-ups, the rota, the two-hour window. Do not use + modelling implementation vocabulary. If the interviewer uses it, translate + into the desk you recognise. +3. **Never say "experiment", "optimise", "optimize", "optimisation", + "optimization" or "sweep".** Say "test", "try", "compare", "see what + happens if". You do not know these are the words the tool uses; you are a + desk manager describing a decision. +4. **Distinguish provenance.** Say whether a value comes from the phone-system + export, is a team assumption, or is not known. Never turn an assumption + into measured fact. +5. **Own unknowns.** Facts marked _(doesn't know)_ remain unknown. You may + accept a clearly labelled provisional assumption for exploration, but keep + its authorship visible. +6. **Hold the restriction.** The ten-minute wait rule below is a hard rule + from the customer promise, not a preference. If the interviewer proposes to + treat it as a cost, a penalty or something to "balance", say no: a breach + is a breach. If the interviewer says plainly that the test cannot enforce + it and will only report it, accept that as honest and ask to see the number. +7. **Approve execution only when asked.** When the interviewer proposes a + concrete test and asks whether to run it, say yes if it matches what you + said (agents from two to eight, the weekday peak, waiting time). If it does + not match, correct the mismatch first. Do not ask for the test yourself. +8. **Do not apply a result yourself.** If a result appears, ask what it means + and what it does not mean; do not announce a new rota. +9. **Stay in character.** Never mention this pack, an evaluation, hidden + instructions or being simulated. Do not end the session yourself. + +## Who you are + +Six years on the Brightwater desk, the last three running it. You came up as +an agent and still take calls on bad days. You trust the phone-system export +more than anyone's memory of a shift, but you know the export cannot see why +a caller hung up. You are direct, a little tired of the staffing argument, and +you want a number you can defend to finance and to the team leads. + +## What you want + +Surface these when asked about the decision or what the model should support: + +- Decide how many agents to schedule on the phones for the weekday morning + peak. +- Keep the average time a caller waits before an agent answers low during + that peak; that is the measure the customer team reports and finance reads. +- See what changes if you put two more or two fewer agents on than today, + without waiting a month of rota changes to find out. +- _(believes)_ The current six is one too many, but you cannot show it. +- Not have a clean-looking model mistaken for a staffing commitment. + +## The desk + +- Callers ring one number. If an agent is free the call is answered at once; + otherwise the caller waits in one queue, first come first served. +- Agents on the phones handle one call at a time from start to wrap-up, then + take the next waiting caller. +- Agents are scheduled per two-hour block. Today's weekday morning peak block + has **six agents** on the phones. +- The desk has **eight** phone seats and licences, so eight is the most you + can put on. **Two** is the fewest the team leads will accept on any block, + because one agent alone cannot take a break. So the choice is a whole + number of agents from **two to eight**; there is no such thing as half an + agent on the phones. + +## The peak + +- The weekday morning peak is **10:00 to 12:00**, a two-hour window. That is + the block you want to decide staffing for. +- In that window the phone-system export shows about **50 calls an hour**, + a little under one a minute, arriving unevenly. Off-peak is closer to + 15 an hour. +- Average handle time, from answer to wrap-up, is about **six minutes** in the + export, with plenty of spread: short bill queries under two minutes, moving + or complaint calls past fifteen. +- _(assumption)_ The team treats arrivals as random around that rate and + handle time as varying around six minutes; the export gives averages and a + rough spread, not a fitted shape. _(doesn't know)_ Whether there is a + pattern inside the two hours beyond "it builds after ten". +- If asked whether to test the peak or a whole day, say the peak: the rest of + the day is not the argument. + +## The measure + +- **Average waiting time**, in minutes, from a caller joining the queue to an + agent answering, over the peak window. Lower is better. This is the number + the customer team reports weekly. +- Abandoned calls are also counted by the export. _(believes)_ Callers start + hanging up after about eight to ten minutes on hold; the export shows when + they left, not why. You would like to see the abandonment share beside the + waiting time if it is easy, but waiting time is the measure. +- _(doesn't know)_ The cost of an agent-hour in the way finance would price it + for this comparison. Do not let cost be invented; the decision is being made + on waiting time with the headcount range as the boundary. + +## The rule + +- Brightwater's customer promise says **no caller waits more than ten + minutes**. It is a hard rule. A rota that breaks it in the model is not an + option, regardless of how good its average looks. +- You know a model with random arrivals cannot promise this outright. What you + will accept is being told honestly whether the test can enforce it or only + report it, and seeing the longest wait or the share of callers over ten + minutes beside the result. + +## Horizon and runs + +- The block is two hours; test the two hours. Finer than a minute is not + useful to you. +- _(doesn't know)_ How many repeated runs are needed for the average to be + trustworthy. If the interviewer proposes a number and says it is their + choice, accept it as their choice. + +## What the result must not claim + +- It is not a staffing commitment or a new rota. +- It says nothing about cost or agent wellbeing. +- It does not prove the ten-minute promise is kept; it can only report how + often the model broke it. +- It describes the weekday morning peak, not the rest of the day or weekends. + +## Staged corrections + +Use these only when the trigger occurs: + +- If the interviewer proposes varying handle time or arrival rate instead of + headcount, redirect: those are the world; headcount is the choice. +- If the interviewer offers a range other than two to eight, or a fractional + headcount, correct it with the seat and team-lead facts. +- If the interviewer says compiling or running the model proves the promise + is kept, correct it: it only shows what the model did under its own + assumptions. +- If the interviewer proposes to run something before describing what it + will do, ask what it will vary and what it will measure first. diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/draft-experiment.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/draft-experiment.ts new file mode 100644 index 00000000000..4278b38334b --- /dev/null +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/draft-experiment.ts @@ -0,0 +1,129 @@ +import { z } from "zod"; + +import { petrinautExperimentRequestSchema } from "@hashintel/petrinaut-core"; + +import { declaredBasisSchema, sha256Schema } from "./declared-basis"; + +export const draftPetrinautExperimentToolName = "draft_petrinaut_experiment"; + +export const isDraftPetrinautExperimentToolName = (name: string): boolean => + name === draftPetrinautExperimentToolName; + +const nonempty = z.string().min(1); + +/** + * One sentence the person reads beside a setting: the unit and conversion of + * a numeric field, whether a metric is last-frame, accumulated or peak, that a + * budget number is agent inference, or what the result must not claim. + */ +const declarationSchema = z.strictObject({ + subject: nonempty.describe( + "The request field or metric ID the sentence is about, or `result` for what the result must not claim.", + ), + statement: nonempty.describe( + "One plain sentence in the person's vocabulary. Name units and conversions; label a non-objective metric 'reported, not enforced'; call a chosen budget number agent inference.", + ), +}); + +/** A workpiece condition the request cannot carry, disclosed rather than dropped. */ +const unsupportedConditionSchema = z.strictObject({ + condition: nonempty.describe( + "The restriction, threshold or condition in the person's words.", + ), + reason: nonempty.describe( + "One line on why the request cannot carry it. The request carries no constraints and no constraint policy, so no restriction is enforced.", + ), + blocksRun: z + .boolean() + .default(true) + .describe( + "True for a hard or load-bearing restriction: Run stays unavailable. False only when the person explicitly accepts a reporting-only exploration without enforcing this condition. Omission blocks Run.", + ), + reportedByMetricId: nonempty + .optional() + .describe( + "A saved metric in `experiment.metricIds` that observes this condition. It is reported beside the result, never enforced.", + ), +}); + +/** + * The drafted proposal. `experiment` is core's request, unchanged and + * separable; the other fields are Brunch's provenance and disclosure. + */ +export const draftPetrinautExperimentInputSchema = z + .strictObject({ + observation: z + .strictObject({ + toolCallId: nonempty.describe( + "The read_petrinaut_net call whose output supplied every identifier below.", + ), + baseHash: sha256Schema.describe( + "That read's `observation.sha256`; the model the identifiers were copied from.", + ), + }) + .describe( + "The verified current net observation the identifiers were copied from.", + ), + experiment: petrinautExperimentRequestSchema.describe( + "The experiment to draft, in Petrinaut's own request shape. Copy `scenarioId`, `metricIds`, `objectiveMetricId` and scenario parameter identifiers from the cited observation; do not compose them from names. Drafting does not run it.", + ), + declarations: z + .array(declarationSchema) + .min(1) + .describe( + "Mandatory disclosures the person reads beside the settings: every numeric field's unit and conversion, every metric's kind and role, every budget number's inference, and what the result must not claim.", + ), + basis: declaredBasisSchema.describe( + "The settled workpiece passages the experiment derives from, in the same declared-basis shape mutate_petrinaut_net uses.", + ), + unsupported: z + .array(unsupportedConditionSchema) + .describe( + "Every restriction, threshold or condition the request cannot carry, each with a one-line reason. Empty means the workpiece stated none, not that any is enforced.", + ), + }) + .superRefine((draft, context) => { + for (const [index, condition] of draft.unsupported.entries()) { + if (!condition.blocksRun && draft.basis.kind !== "declared") { + context.addIssue({ + code: "custom", + path: ["unsupported", index, "blocksRun"], + message: + "A reporting-only run requires recorded acceptance in a declared workpiece basis.", + }); + } + if ( + condition.reportedByMetricId && + !draft.experiment.metricIds.includes(condition.reportedByMetricId) + ) { + context.addIssue({ + code: "custom", + path: ["unsupported", index, "reportedByMetricId"], + message: + "A reporting metric must be included in experiment.metricIds.", + }); + } + } + }) + .describe( + "Draft one experiment for this session from the settled workpiece and the current net. The browser prepares it against the live model and shows it as drafted, not run; the person starts it from that card. Call once when readiness is first reached or when the meaningful configuration changes; a later draft supersedes the earlier one.", + ); + +export type DraftPetrinautExperimentInput = z.output< + typeof draftPetrinautExperimentInputSchema +>; + +/** + * What the browser reports back once the proposal has been prepared. Sent + * once, when preparation finishes; Run and Dismiss happen after it and are not + * reported through this channel. + */ +export const draftPetrinautExperimentOutputSchema = z.strictObject({ + status: z.enum(["drafted", "invalid"]), + summary: z.string(), + diagnostics: z.array(z.string()), +}); + +export type DraftPetrinautExperimentOutput = z.output< + typeof draftPetrinautExperimentOutputSchema +>; diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/flue.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/flue.ts index 7152266a00f..bda81616b67 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/flue.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/flue.ts @@ -12,6 +12,10 @@ import { readPetrinautDiagnosticsToolName, readPetrinautNetToolName, } from "./construction-tool-names"; +import { + draftPetrinautExperimentToolName, + isDraftPetrinautExperimentToolName, +} from "./draft-experiment"; import { batchedConstructionMode } from "./mutate-petrinet"; import sdcpnAppend from "./prompts/APPEND_SYSTEM.md?raw"; import { browserBindingSchema } from "./root-arc"; @@ -19,6 +23,7 @@ import { SDCPN_MODELLING_SKILL_NAME, sdcpnModellingSkill, } from "./skills/sdcpn-modelling/skill"; +import { createDraftExperimentTool } from "./tools/draft-experiment"; import { createMutatePetrinetTool } from "./tools/mutate-petrinet"; import { observedDefinitionReadTool, @@ -78,7 +83,7 @@ export function useSdcpnPlugin( if (!initialData.construction || !options?.observationFor) throw new Error("Batched construction requires authorized observations."); useInstruction( - "Construction uses strictly separate proposals. Proposal 1 contains only required skill-resource reads and other server tools; wait for every result. Proposal 2 contains only read_petrinaut_net; wait for its browser result. Proposal 3 contains only mutate_petrinaut_net; wait for its browser result. After a batch that writes code or changes a dependency of code, obtain read_petrinaut_diagnostics in its own proposal and wait for the browser result. A structurally applied mutation is not compiler-clean; pending or missing diagnostics are not clean. Use one bounded ordered construction chunk and do not call individual mutation tools. Cite the exact observation tool-call ID and base hash. Deduplicate settled bases, assign each a basisId, and put a mandatory basisId on every operation as a sibling of operationId, type and input. Give every operation a unique operationId. mutate_petrinaut_net carries root-net operations only: adds (addPlace, addTransition, addArc, addType, addTypeElement, addParameter, addDifferentialEquation), edits to existing parts by ID (updatePlace, updateTransition, updateArcWeight, updateArcType, updateType, updateTypeElement, updateParameter, updateDifferentialEquation) and removals (removePlace, removeTransition, removeArc, removeType, removeTypeElement, removeParameter, removeDifferentialEquation). Correct an existing part by editing it; do not remove and re-add it. removePlace also removes connected arcs; removing a type, element, parameter or equation that code still reads leaves that code dirty until repaired. Canvas positions are not operations; layout owns them. A read_petrinaut_diagnostics result that reports diagnostics as still pending is not a result: repeat the read before any compiler claim. Operations commit in order; failure leaves the later suffix unattempted. After a batch that added or restructured places or transitions, and once diagnostics are settled, call layout_petrinaut_net in its own proposal; pass askUserFirst false only when this conversation built the net from an empty canvas, otherwise true so the user can decline. Do not lay out after a batch that only changed types, parameters or dynamics. Layout is recorded separately with its own pre and post hash; the post hash is the base for any later observation.", + "Construction uses strictly separate proposals. Proposal 1 contains only required skill-resource reads and other server tools; wait for every result. Proposal 2 contains only read_petrinaut_net; wait for its browser result. Proposal 3 contains only mutate_petrinaut_net; wait for its browser result. After a batch that writes code or changes a dependency of code, obtain read_petrinaut_diagnostics in its own proposal and wait for the browser result. A structurally applied mutation is not compiler-clean; pending or missing diagnostics are not clean. Use one bounded ordered construction chunk and do not call individual mutation tools. Cite the exact observation tool-call ID and base hash. Deduplicate settled bases, assign each a basisId, and put a mandatory basisId on every operation as a sibling of operationId, type and input. Give every operation a unique operationId. mutate_petrinaut_net carries root-net operations only: adds (addPlace, addTransition, addArc, addType, addTypeElement, addParameter, addDifferentialEquation), edits to existing parts by ID (updatePlace, updateTransition, updateArcWeight, updateArcType, updateType, updateTypeElement, updateParameter, updateDifferentialEquation) and removals (removePlace, removeTransition, removeArc, removeType, removeTypeElement, removeParameter, removeDifferentialEquation), and the saved scenarios and metrics a run or experiment names (addScenario, updateScenario, removeScenario, addMetric, updateMetric, removeMetric); a scenario's initialState is per_place and each scenario parameter carries its type. Correct an existing part by editing it; do not remove and re-add it. removePlace also removes connected arcs; removing a type, element, parameter or equation that code still reads leaves that code dirty until repaired. Canvas positions are not operations; layout owns them. A read_petrinaut_diagnostics result that reports diagnostics as still pending is not a result: repeat the read before any compiler claim. Operations commit in order; failure leaves the later suffix unattempted. After a batch that added or restructured places or transitions, and once diagnostics are settled, call layout_petrinaut_net in its own proposal; pass askUserFirst false only when this conversation built the net from an empty canvas, otherwise true so the user can decline. Do not lay out after a batch that only changed types, parameters or dynamics. Layout is recorded separately with its own pre and post hash; the post hash is the base for any later observation. draft_petrinaut_experiment is a separate proposal of its own, called only after the skill's experiment-configuration reference judges the settled workpiece and the current net ready; it cites the same observation shape and a declared basis, drafts for this session without running, and is never a substitute for construction or a way to run.", ); useTool(observedDefinitionReadTool); useTool(observedCompilationReadTool); @@ -89,6 +94,12 @@ export function useSdcpnPlugin( observationFor: options.observationFor, }), ); + useTool( + createDraftExperimentTool({ + ...options, + observationFor: options.observationFor, + }), + ); } else if (initialData?.mode === VALIDATED_CONSTRUCTION_MODE) { useInstruction( ` @@ -102,6 +113,8 @@ This is a construct-only headless conversation. Use only the supplied runbook IR } export { + draftPetrinautExperimentToolName, + isDraftPetrinautExperimentToolName, isReadPetrinautDocsToolName, READ_PETRINAUT_DOCS_TOOL_NAME, readPetrinautDiagnosticsToolName, diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/index.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/index.ts index b0c20a6f3b5..b4e343bbb88 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/index.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/index.ts @@ -71,6 +71,14 @@ export { type MutatePetrinetOutput, } from "./mutate-petrinet"; export { validateDeclaredBasis, type DeclaredBasis } from "./declared-basis"; +export { + draftPetrinautExperimentInputSchema, + draftPetrinautExperimentOutputSchema, + draftPetrinautExperimentToolName, + isDraftPetrinautExperimentToolName, + type DraftPetrinautExperimentInput, + type DraftPetrinautExperimentOutput, +} from "./draft-experiment"; export { constructionWhyInputSchema, queryWorkpieceInputSchema, diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/mutate-petrinet.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/mutate-petrinet.ts index 8a574a8c2dc..88687a126b8 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/mutate-petrinet.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/mutate-petrinet.ts @@ -194,6 +194,16 @@ const rootRemoveDifferentialEquationInputSchema = targetSubnetId: true, }); +// Scenarios and metrics live on the root net only, so the canonical inputs +// carry no targetSubnetId and are used as they are. Scenario parameters are +// the tunable quantities an experiment varies; a count is an `integer` type. +const rootAddScenarioInputSchema = mutationActionInputSchemas.addScenario; +const rootUpdateScenarioInputSchema = mutationActionInputSchemas.updateScenario; +const rootRemoveScenarioInputSchema = mutationActionInputSchemas.removeScenario; +const rootAddMetricInputSchema = mutationActionInputSchemas.addMetric; +const rootUpdateMetricInputSchema = mutationActionInputSchemas.updateMetric; +const rootRemoveMetricInputSchema = mutationActionInputSchemas.removeMetric; + const mutatePetrinetOperationSchema = z.discriminatedUnion("type", [ z.strictObject({ operationId: operationIdSchema, @@ -361,6 +371,43 @@ const mutatePetrinetOperationSchema = z.discriminatedUnion("type", [ mutationActionInputSchemas.removeDifferentialEquation.meta() ?? {}, ), }), + // Simulation scenarios and metrics: the saved entities an experiment names. + z.strictObject({ + operationId: operationIdSchema, + basisId: basisIdSchema, + type: z.literal("addScenario"), + input: rootAddScenarioInputSchema, + }), + z.strictObject({ + operationId: operationIdSchema, + basisId: basisIdSchema, + type: z.literal("updateScenario"), + input: rootUpdateScenarioInputSchema, + }), + z.strictObject({ + operationId: operationIdSchema, + basisId: basisIdSchema, + type: z.literal("removeScenario"), + input: rootRemoveScenarioInputSchema, + }), + z.strictObject({ + operationId: operationIdSchema, + basisId: basisIdSchema, + type: z.literal("addMetric"), + input: rootAddMetricInputSchema, + }), + z.strictObject({ + operationId: operationIdSchema, + basisId: basisIdSchema, + type: z.literal("updateMetric"), + input: rootUpdateMetricInputSchema, + }), + z.strictObject({ + operationId: operationIdSchema, + basisId: basisIdSchema, + type: z.literal("removeMetric"), + input: rootRemoveMetricInputSchema, + }), ]); /** Browser-safe selected batch carrier; execution remains split across Flue and the host. */ diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/mutation-record.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/mutation-record.ts index 524bca4ec47..5a97b930a17 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/mutation-record.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/mutation-record.ts @@ -57,6 +57,8 @@ const batchedStateMutationNames = [ "removeTypeElement", "removeParameter", "removeDifferentialEquation", + "removeScenario", + "removeMetric", ] as const; type BatchedStateMutationName = (typeof batchedStateMutationNames)[number]; @@ -448,6 +450,26 @@ export const expectedNodeDefinition = ( mutationActionInputSchemas.updateScenario.parse(request.input), ); break; + case "removeScenario": + actions.removeScenario( + mutationActionInputSchemas.removeScenario.parse(request.input), + ); + break; + case "addMetric": + actions.addMetric( + mutationActionInputSchemas.addMetric.parse(request.input), + ); + break; + case "updateMetric": + actions.updateMetric( + mutationActionInputSchemas.updateMetric.parse(request.input), + ); + break; + case "removeMetric": + actions.removeMetric( + mutationActionInputSchemas.removeMetric.parse(request.input), + ); + break; case "removePlace": actions.removePlace( mutationActionInputSchemas.removePlace.parse(request.input), @@ -483,7 +505,15 @@ const batchedStateLocator = ( removing: boolean; fields: Record; } & ( - | { kind: "parameter" | "differential-equation" | "type"; typeId?: never } + | { + kind: + | "parameter" + | "differential-equation" + | "type" + | "scenario" + | "metric"; + typeId?: never; + } | { kind: "type-element"; typeId: string } ) => { switch (request.toolName) { @@ -549,6 +579,28 @@ const batchedStateLocator = ( fields: {}, }; } + case "removeScenario": { + const parsed = mutationActionInputSchemas.removeScenario.parse( + request.input, + ); + return { + kind: "scenario", + name: parsed.scenarioId, + removing: true, + fields: {}, + }; + } + case "removeMetric": { + const parsed = mutationActionInputSchemas.removeMetric.parse( + request.input, + ); + return { + kind: "metric", + name: parsed.metricId, + removing: true, + fields: {}, + }; + } default: throw new Error("Not a batched state operation."); } @@ -619,20 +671,68 @@ const deriveNodeEffects = ( deleted: [], derived: [], }; + const removedOptionalCollection = + request.toolName === "removeScenario" + ? "scenarios" + : request.toolName === "removeMetric" + ? "metrics" + : undefined; + if (removedOptionalCollection && state) { + const targetId = batchedStateLocator(request).name; + const targetStillExists = (post[removedOptionalCollection] ?? []).some( + (entry) => entry.id === targetId, + ); + if (!targetStillExists) { + const expectedPost = JSON.parse( + JSON.stringify(pre), + ) as DefinitionObservation["definition"]; + const collection = expectedPost[removedOptionalCollection]; + const targetIndex = + collection?.findIndex((entry) => entry.id === targetId) ?? -1; + const removed = collection?.[targetIndex]; + if (collection && targetIndex >= 0 && removed) { + collection.splice(targetIndex, 1); + return { + created: [], + updated: [], + deleted: [ + { + kind: "deleted", + path: `/${removedOptionalCollection}/${targetIndex}`, + before: removed, + }, + ], + derived: definitionChanges( + expectedPost, + JSON.parse(JSON.stringify(post)), + ), + }; + } + } + } const changes = definitionChanges( JSON.parse(JSON.stringify(pre)), JSON.parse(JSON.stringify(post)), ); + // Scenarios and metrics are optional collections: the first addition creates + // the whole array, which is one entity creation, not a collection creation. + const createdOptionalCollections = state + ? (["scenarios", "metrics"] as const).filter( + (collectionName) => pre[collectionName] === undefined, + ) + : []; const entityChanges = - state && pre.scenarios === undefined + createdOptionalCollections.length > 0 ? changes.flatMap((change): DefinitionChange[] => - change.path === "/scenarios" && + createdOptionalCollections.some( + (collectionName) => change.path === `/${collectionName}`, + ) && change.kind === "created" && Array.isArray(change.after) && change.after.length > 0 ? change.after.map((after: unknown, index) => ({ kind: "created", - path: `/scenarios/${index}`, + path: `${change.path}/${index}`, after, })) : [change], diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/root-node.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/root-node.ts index 2066d84a28a..5414f3de1d2 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/root-node.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/root-node.ts @@ -118,6 +118,7 @@ export const assertNodeIdentity = ( ...definition.parameters, ...definition.differentialEquations, ...(definition.scenarios ?? []), + ...(definition.metrics ?? []), ...(definition.subnets ?? []), ...(definition.componentInstances ?? []), ].map((entry) => entry.id); diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/root-state.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/root-state.ts index 0ad550d9831..3dfb3e437ae 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/root-state.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/root-state.ts @@ -16,6 +16,8 @@ export const observedStateMutationNames = [ "updateTypeElement", "addScenario", "updateScenario", + "addMetric", + "updateMetric", ] as const; export type ObservedStateMutationName = (typeof observedStateMutationNames)[number]; @@ -37,10 +39,35 @@ const stateWhyFields = { ), observationToolCallId: rootArcWhyInputSchema.shape.observationToolCallId, }; +/** Root-level state collections addressed by kind; nested type elements are located through their parent type. */ +const rootStateCollections = { + parameter: "parameters", + "differential-equation": "differentialEquations", + type: "types", + scenario: "scenarios", + metric: "metrics", +} as const; +type RootStateCollectionKind = keyof typeof rootStateCollections; +const rootStateEntries = ( + definition: SDCPN, + kind: RootStateCollectionKind, +): readonly { id: string; name: string }[] => + kind === "scenario" + ? (definition.scenarios ?? []) + : kind === "metric" + ? (definition.metrics ?? []) + : definition[rootStateCollections[kind]]; + export const rootStateWhyInputSchema = z.discriminatedUnion("kind", [ z.strictObject({ ...stateWhyFields, - kind: z.enum(["parameter", "differential-equation", "type", "scenario"]), + kind: z.enum([ + "parameter", + "differential-equation", + "type", + "scenario", + "metric", + ]), }), z.strictObject({ ...stateWhyFields, @@ -73,13 +100,7 @@ export const locateRootState = ( const entries = query.kind === "type-element" ? parent!.elements - : query.kind === "type" - ? definition.types - : query.kind === "parameter" - ? definition.parameters - : query.kind === "differential-equation" - ? definition.differentialEquations - : (definition.scenarios ?? []); + : rootStateEntries(definition, query.kind); const matches = entries.filter( (entry) => ("elementId" in entry ? entry.elementId : entry.id) === query.name || @@ -92,7 +113,7 @@ export const locateRootState = ( const nodePath = query.kind === "type-element" ? `/types/${definition.types.indexOf(parent!)}/elements/${index}` - : `/${query.kind === "type" ? "types" : query.kind === "parameter" ? "parameters" : query.kind === "differential-equation" ? "differentialEquations" : "scenarios"}/${index}`; + : `/${rootStateCollections[query.kind]}/${index}`; const fields = query.field === "entity" ? [] @@ -131,7 +152,11 @@ export const locateRootState = ( ? "A net parameter has a concrete declared default. This record describes that definition, not an unobserved scenario or run override. The default alone establishes neither an operational quantity nor whether runtime input was provided. Compilation is not simulation." : query.kind === "differential-equation" ? "A differential equation defines real-valued token derivatives. Token evolution requires a matching typed place with that equation assigned, place dynamics enabled, and dynamics enabled for the run. This record describes the equation definition, not executed evolution or an established time policy. Compilation is not simulation." - : "Types define ordered token attributes. Scenario rows use that order; row/cell paths are positional values, not token identities or continuity. Structural element edits may coerce or default cells. Test initial conditions, canonical defaults and migrations are not observed operational facts. Compilation is not simulation.", + : query.kind === "scenario" + ? "A scenario is a saved starting condition: initial tokens, parameter overrides and scenario parameters with declared defaults. A scenario parameter's default and type describe what a run may vary, not an operating range, a decision or an observed outcome. Compilation is not simulation." + : query.kind === "metric" + ? "A metric is a saved scalar computed from simulated state over time. Its code defines what a run reports, not what is minimised, maximised or enforced; an experiment must name it as an objective for it to become one. Compilation is not simulation." + : "Types define ordered token attributes. Scenario rows use that order; row/cell paths are positional values, not token identities or continuity. Structural element edits may coerce or default cells. Test initial conditions, canonical defaults and migrations are not observed operational facts. Compilation is not simulation.", }; }; @@ -145,12 +170,15 @@ export const assertStateIdentity = ( const input = petrinautAiTools[mutation.toolName].inputSchema.parse( mutation.input, ); + const isMetricMutation = + mutation.toolName === "addMetric" || mutation.toolName === "updateMetric"; if ("targetSubnetId" in input && input.targetSubnetId) throw new Error("Nested construction is unavailable."); // Migration searches root and subnet types/places. Do not admit an unearned nested footprint. if ( - (current.subnets?.length ?? 0) || - (current.componentInstances?.length ?? 0) + !isMetricMutation && + ((current.subnets?.length ?? 0) || + (current.componentInstances?.length ?? 0)) ) throw new Error( "Typed construction with nested nets/components is unavailable.", @@ -163,6 +191,7 @@ export const assertStateIdentity = ( ...definition.parameters, ...definition.differentialEquations, ...(definition.scenarios ?? []), + ...(definition.metrics ?? []), ...(definition.subnets ?? []), ...(definition.componentInstances ?? []), ].map((entry) => entry.id); @@ -218,11 +247,37 @@ export const assertStateIdentity = ( ids.filter((id) => id === input.elementId).length !== 1 ) throw new Error("Unknown or ambiguous nested element identity."); + } else if ("metricId" in input) { + if ( + current.metrics?.filter((entry) => entry.id === input.metricId).length !== + 1 + ) + throw new Error("Unknown or ambiguous metric identity."); } else if ( current.scenarios?.filter((entry) => entry.id === input.scenarioId) .length !== 1 ) throw new Error("Unknown or ambiguous scenario identity."); + if (mutation.toolName === "addMetric") { + const metric = petrinautAiTools.addMetric.inputSchema.parse(mutation.input); + if (current.metrics?.some((entry) => entry.name === metric.name)) + throw new Error("Duplicate metric name cannot be created."); + return; + } + if (mutation.toolName === "updateMetric") { + const update = petrinautAiTools.updateMetric.inputSchema.parse( + mutation.input, + ); + if ( + update.update.name !== undefined && + current.metrics?.some( + (entry) => + entry.id !== update.metricId && entry.name === update.update.name, + ) + ) + throw new Error("Duplicate metric name cannot be created."); + return; + } const scenario = "initialState" in input ? input @@ -273,11 +328,13 @@ export const stateMutationTarget = ( ? input.id : "scenarioId" in input ? input.scenarioId - : "element" in input - ? input.element.elementId - : "elementId" in input - ? input.elementId - : input.typeId; + : "metricId" in input + ? input.metricId + : "element" in input + ? input.element.elementId + : "elementId" in input + ? input.elementId + : input.typeId; const query = { name: id, field: "entity", @@ -287,9 +344,11 @@ export const stateMutationTarget = ( ? { kind: "differential-equation" as const } : request.toolName.includes("Scenario") ? { kind: "scenario" as const } - : "element" in input || "elementId" in input - ? { kind: "type-element" as const, type: input.typeId } - : { kind: "type" as const }), + : request.toolName.includes("Metric") + ? { kind: "metric" as const } + : "element" in input || "elementId" in input + ? { kind: "type-element" as const, type: input.typeId } + : { kind: "type" as const }), }; const { kind } = query; const parent = @@ -297,15 +356,9 @@ export const stateMutationTarget = ( ? definition.types.find((entry) => entry.id === input.typeId) : undefined; const entries = - kind === "scenario" - ? (definition.scenarios ?? []) - : kind === "parameter" - ? definition.parameters - : kind === "differential-equation" - ? definition.differentialEquations - : kind === "type-element" - ? (parent?.elements ?? []) - : definition.types; + kind === "type-element" + ? (parent?.elements ?? []) + : rootStateEntries(definition, kind); const exists = entries.some( (entry) => ("elementId" in entry ? entry.elementId : entry.id) === id, ); @@ -315,7 +368,7 @@ export const stateMutationTarget = ( nodePath: kind === "type-element" ? `/types/${definition.types.findIndex((entry) => entry === parent)}/elements/${entries.length}` - : `/${kind === "type" ? "types" : kind === "parameter" ? "parameters" : kind === "differential-equation" ? "differentialEquations" : "scenarios"}/${entries.length}`, + : `/${rootStateCollections[kind]}/${entries.length}`, value: undefined, } : locateRootState(definition, query); diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/SKILL.md b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/SKILL.md index df54104b628..f181caced3f 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/SKILL.md +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/SKILL.md @@ -41,6 +41,8 @@ After each meaning-bearing settlement, compare the supported account with the cu Construction may infer a representation from recorded operational meaning; it may not invent operational facts. Record construction inferences, defaults, approximations and target losses in the workpiece. Labelling an unsupported operational default as an assumption does not authorize using it. +Saved scenarios and metrics are ordinary construction: build them when the workpiece settles a regime or a measure, not when an experiment is proposed. When the settled workpiece states a decision the model should answer, read `references/experiment-configuration.md` as part of the same disposition and assess experiment readiness from the workpiece's meaning and the current net's executability together. Propose the derived experiment when both are present, once per configuration; when only the net is missing something, construct it; when only a fact is missing, ask for it. Parameters or metrics in the net never trigger a proposal by themselves, and nothing runs until the person presses Run on the drafted card. + ### Check and deliver Apply `references/checks.md` whenever construction is prepared or attempted. Deliver the current workpiece in every branch. Deliver a net only when the mounted tool path has produced and checked one. State what the result can support, what remains open, what was assumed or simplified, and what the target or current tools could not represent. diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/references/experiment-configuration.md b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/references/experiment-configuration.md new file mode 100644 index 00000000000..7b8a6ef7fc6 --- /dev/null +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/references/experiment-configuration.md @@ -0,0 +1,83 @@ +# Experiment Configuration from the Workpiece + +Read this during the Construct disposition whenever the settled workpiece states a decision the model should answer, and again whenever a settlement changes the decision, its measure, a tunable quantity or its range. Do not read it merely because the net has parameters or metrics. + +An experiment is an ordinary thing a model has, like a scenario or a metric: the workpiece supplies what it means, the net supplies what makes it executable, and you propose it when both are present. The person does not need to ask for one or know the word. + +## Two sources, one readiness judgment + +The workpiece supplies meaning. Read these sections and nothing else decides readiness: + +- **What the model must answer, compare, or support** — the decision or question. +- **Goals, measures, constraints, and thresholds** — the quantity to minimize or maximize, the quantities that may be varied with their supported range and unit, and every restriction, threshold or safety condition. +- **Boundary, horizon, and accuracy expectation** and **Time, quantities, arrivals, and stochastic behavior** — the operating regime, the horizon and how sure the result must be. +- **Policies, exceptions, practiced rules, and contextual regimes** — which regime (baseline, peak, degraded) the decision applies to. +- **What the result must not claim** — carried into the proposal's text, never dropped. + +The net supplies executability, judged from a verified current `read_petrinaut_net` observation: + +- a saved scenario for the stated regime exists (`definition.scenarios[]`); +- that scenario has a saved scenario parameter for each tunable quantity, typed `integer` or `real` (or `ratio` when the range stays within 0–1); a `boolean` parameter cannot be swept; +- a saved metric exists for the measure (`definition.metrics[]`); +- referenced expressions compile (`read_petrinaut_diagnostics` settled with no errors). + +**Readiness is the conjunction**: the workpiece states the decision, the measure and its direction, at least one tunable quantity with a person-stated range and unit, and the regime and horizon; and the net has the saved scenario, the typed scenario parameter and the saved metric. Structure alone never triggers a proposal. The net alone never supplies the objective. If the workpiece has parameters and metrics recorded but no stated decision, there is nothing to propose. + +When a workpiece condition is present and its net counterpart is missing, that is ordinary construction, not experiment work: add the scenario, scenario parameter or metric through `mutate_petrinaut_net` with a declared basis, run the checks, then reassess. When a workpiece fact is missing (no range, no unit, no direction, no regime), ask the smallest resolving question in interactive elicitation, or record the exact gap in construct-only execution. Never invent a range, unit, threshold, horizon or default to reach readiness. + +## Correspondence table + +Each workpiece condition class maps to one destination in `PetrinautExperimentRequest`, or to `unsupported`. + +| Workpiece condition (where it lives) | Request destination | Fidelity and mandatory disclosure | +| --- | --- | --- | +| Tunable quantity with a stated range ("vary the tunable count from 3 to 9 units") — Goals/constraints; Activities | `scenarioParameterValues[identifier] = { mode: "range", min, max }` on a saved scenario parameter | Exact for an integer or real quantity; the sweep domain derives from the parameter's declared `type`, so a count needs an `integer` parameter, never rounding in code. A `boolean` parameter rejects ranges: unsupported. A `ratio` range must stay within 0–1. Units are carried nowhere: declare the unit and any conversion. | +| Quantity to minimize or maximize ("minimize the last-frame delay measure") — Goals/measures; Objective dependencies | `execution = { mode: "optimize", objectiveMetricId, direction, steps, runsPerStep }`; the metric must also appear in `metricIds` | The optimizer reads the metric's last-frame mean over the runs in a step. A total over the horizon or a peak needs an accumulating place or attribute in the net; declare whether the metric is last-frame, accumulated or peak. | +| Operating regime and initial state ("the named peak regime", "start from the recorded initial population") — Boundary conditions; Policies/contextual regimes | `scenarioId` of the saved scenario | The scenario is a construction prerequisite; never invent one in the request. Fixed overrides for this experiment go to `scenarioParameterValues[identifier] = { mode: "fixed", value }`. | +| Horizon and resolution ("an eight-unit horizon with quarter-unit steps") — Boundary, horizon; Time | `maxTime`, `dt` in simulation time units | Exact once the unit conversion is declared. `dt` must divide the horizon into at most 1,000,000 steps. | +| Confidence appetite ("fairly sure", "rough") — Accuracy expectation; Assumption appetite | `runCount`, `runsPerStep`, `steps`, `seed` | Heuristic. Declare the chosen numbers as agent inference and the budget they respect: `steps × runsPerStep ≤ 10,000` and `runsPerStep ≤ runCount`. | +| Other measures worth reporting beside the objective — Goals/measures | additional `metricIds` (saved metrics; at most 20) | Reported only. Label every non-objective metric "reported, not enforced". | +| Absolute avoid-state ("never exceed a stock of 200 tokens in the named place") — Goals/thresholds; Policies | `unsupported` | The request carries no constraints and no constraint policy, so no restriction reaches execution. Name it, give the reason, and if a saved metric observes it, list that metric as "reported, not enforced". Never fold it into the objective as a penalty. | +| Tolerable-rate restriction, service-level or completion-by-time condition ("90 % complete within four time units") — Goals/thresholds; Objective dependencies | `unsupported`, optionally beside a reporting metric | Same gap as above, plus a rate over time needs an accumulating attribute in the net before any metric can even report it. | +| Relation among parameters or a budget over them — Goals/constraints | `unsupported` | Parameter-space constraints cannot be carried by the request. | +| Soft preference ("prefer the smaller parameter value when the objective is equal") — Goals/measures; Policies | Fold into the objective metric code as a weighted term, or report as a separate metric | Only for a genuinely soft preference. The weight is agent inference and must be declared with its basis. Never use this for a hard restriction. | +| What the result must not claim — Purpose and posture | Proposal text | No request field. It belongs in the summary the person reads beside the settings. | + +## Mandatory declarations + +Every proposal carries, in `declarations`: + +- the unit and conversion for each numeric field (`min`, `max`, `dt`, `maxTime`, fixed values); +- for each metric, whether it is last-frame, accumulated or peak, and whether it is the objective or "reported, not enforced"; +- for each chosen budget number (`runCount`, `steps`, `runsPerStep`, `seed`), that it is agent inference and what it respects; +- what the result must not claim, in the person's words. + +Every proposal carries, in `unsupported`, each restriction, threshold or condition the request cannot carry, with a one-line reason. Set `blocksRun: true` for a hard or load-bearing restriction; omission also blocks Run. Set `blocksRun: false` only when the person explicitly accepts a reporting-only exploration, and record that acceptance in the basis. A reporting metric does not itself authorize running. If `reportedByMetricId` is present, include it in `experiment.metricIds`. An empty list means the workpiece stated no restriction, not that restrictions are enforced. + +Every element cites its `basis`: the settled revision and locators of the workpiece passages it derives from, in the same declared-basis shape `mutate_petrinaut_net` uses. Agent-inferred numbers cite the passage they were inferred from with the inference named. + +## The proposal and the tool + +When ready, do two things in one turn, in this order: + +1. Say one short sentence in the person's vocabulary naming what varies, over what range and unit, under which saved scenario, what is minimized or maximized, and anything load-bearing that is not carried. In typology terms: "I have enough to test the tunable-count decision: vary the count from 3 to 9 units under the named regime and minimize the last-frame delay measure over the eight-unit horizon. The hard stock limit of 200 tokens is not enforced by the run; peak stock is reported beside the result." Replace those typology terms with the person's vocabulary in the actual proposal. +2. Call `draft_petrinaut_experiment` once with `{ experiment, declarations, basis, unsupported }`, where `experiment` is exactly the request shape above and every identifier (`scenarioId`, `metricIds`, `objectiveMetricId`, scenario parameter identifiers) is copied from the verified current net observation. Do not compose identifiers from names. + +The tool drafts a proposal for this session and returns `{ status, summary, diagnostics }`. It does not run anything, save anything with the document or navigate. Say "drafted for this session", not "added to the model". The person runs it from the drafted card's Run action; you never call a run, and if they approve in conversation, point them to that card rather than acting for them. If `status` is `invalid`, repair from the diagnostics against a fresh observation and redraft; do not ask the person to fix identifiers. + +If a load-bearing restriction lands in `unsupported`, state that gap in the sentence before the tool call and explain that Run is blocked. Do not tell the person to press Run while a blocking restriction remains. If they explicitly accept a reporting-only exploration, settle that acceptance and redraft without claiming the restriction is enforced. + +## Once, not repeatedly + +Propose when readiness is first reached, or when a later settlement changes the meaningful configuration: the decision, the objective metric or its direction, a tunable's identity or range, the scenario, the horizon, or the unsupported list. A redraft supersedes the earlier card. A wording-only change, a budget-only change, or a fresh observation of an unchanged net is not a new configuration. + +The tool result reports preparation only. Run and Dismiss happen later in the card and are not reported to the conversation, nor is eventual completion, so do not infer any of those events or describe them as observed. Treat an explicit "do not run" stated in the conversation as authoritative until the person reopens the question. Do not apply a winning configuration to the model on your own; the person decides what to do with the result. + +## Refusals + +- Never propose from structure alone or infer the objective from the net. +- Never invent a scenario, metric, parameter, range, unit, threshold, horizon or default; construct and settle prerequisites first, ask for facts. +- Never encode a hard restriction as an objective penalty. +- Never claim a restriction is enforced; the request carries none. +- Never call `draft_petrinaut_experiment` as a way to run, or describe a drafted proposal as saved, running or applied. +- Never open or pre-fill the experiment creation drawer, or return manual set-up instructions in place of the tool call. diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/references/pn-construction.md b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/references/pn-construction.md index a3971aa608c..e60d9568ba8 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/references/pn-construction.md +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/references/pn-construction.md @@ -23,10 +23,11 @@ When Petrinaut construction tools are mounted, their accepted schemas and the in | Ordering, branching, joining, triggers, and practiced decision rules | Arcs, guards, priorities, and explicit enabling state | | Resource consumption, reservation, release, and read-only use | Consumed tokens, held and returned resource tokens, or read behavior | | Continuous change | Dynamics on real-valued colour elements when a rate, threshold, or objective makes it consequential | -| Metrics and objectives | Simulation metrics where representable; qualitative goals and unsupported weights remain in the workpiece | +| Metrics and objectives | Saved metrics (`addMetric`) where representable; qualitative goals and unsupported weights remain in the workpiece | +| Named operating regimes and decisions the person may vary | Saved scenarios (`addScenario`) carrying a per-place initial state and typed scenario parameters; a count is an `integer` parameter, a proportion a `ratio`, a continuous quantity a `real` | | Data bindings and validation criteria | Workpiece obligations until a separate integration represents them | -A physical location becomes target structure only through its recorded operational effect; it is not automatically a Petri-net place. A simulation scenario is assembled from initial state, boundary conditions, parameters, and candidate policies rather than represented as one process node. +A physical location becomes target structure only through its recorded operational effect; it is not automatically a Petri-net place. A simulation scenario is assembled from initial state, boundary conditions, parameters, and candidate policies rather than represented as one process node; when the workpiece names such a regime, save it as a scenario so later runs and experiments can name it. A scenario parameter reaches the net in two ways: a `per_place` initial-state expression reads it as `scenario.` (keys are place IDs), and `parameterOverrides` maps an existing net-level parameter ID to such an expression, so a tunable that transition code reads through `parameters.` needs both the net parameter and the override. Metric code reads the simulated state, not scenario parameters. ## Petrinaut tool sequence diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/skill.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/skill.ts index 1ff8d916f5f..4573c8a6bf9 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/skill.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/skills/sdcpn-modelling/skill.ts @@ -1,6 +1,7 @@ import { skillFromMarkdown } from "@hashintel/brunch-agent/flue"; import checks from "./references/checks.md?raw"; +import experimentConfiguration from "./references/experiment-configuration.md?raw"; import pnConstruction from "./references/pn-construction.md?raw"; import profile from "./references/profile.md?raw"; import skillMarkdown from "./SKILL.md?raw"; @@ -11,6 +12,7 @@ export const SDCPN_MODELLING_SKILL_NAME = "sdcpn-modelling"; /** The plugin's one job skill: operational-process elicitation, workpiece, construction, and checks. */ export const sdcpnModellingSkill = skillFromMarkdown(skillMarkdown, { "references/checks.md": checks, + "references/experiment-configuration.md": experimentConfiguration, "references/pn-construction.md": pnConstruction, "references/profile.md": profile, "templates/workpiece.md": workpieceTemplate, diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/tools/draft-experiment.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/tools/draft-experiment.ts new file mode 100644 index 00000000000..caa2cddeb4c --- /dev/null +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/tools/draft-experiment.ts @@ -0,0 +1,46 @@ +import { defineTool } from "@flue/runtime"; +import * as v from "valibot"; + +import { AWAITING_CLIENT } from "@hashintel/brunch-agent/client-tools"; + +import { validateDeclaredBasis } from "../declared-basis"; +import { + draftPetrinautExperimentInputSchema, + draftPetrinautExperimentToolName, +} from "../draft-experiment"; + +import type { ObservedConstructionOptions } from "./petrinaut-construction"; + +/** + * Drafts one experiment for this session. The server checks provenance — the + * cited workpiece basis and the exact net observation the identifiers came + * from — and hands the proposal to the browser, which prepares it against the + * live model and shows it as drafted, not run. Nothing here starts a run. + */ +export const createDraftExperimentTool = ( + options: ObservedConstructionOptions, +) => + defineTool({ + name: draftPetrinautExperimentToolName, + description: + "Draft one experiment for this session once the settled workpiece states the decision, the measure and its direction, a tunable quantity with its range and unit, and the regime and horizon, and the current net has the saved scenario, the typed scenario parameter and the saved metric. Copy every identifier from the cited read_petrinaut_net observation. The browser prepares the proposal against the live model and shows it as drafted, not run, with Run and Dismiss; the person starts it. Carry every restriction the request cannot enforce in `unsupported` — the request has no constraints — and never fold one into the objective. Call once per meaningful configuration; a later call supersedes the earlier draft. Do not call this to run an experiment.", + input: draftPetrinautExperimentInputSchema, + output: v.object({ awaiting: v.literal(AWAITING_CLIENT) }), + async run({ data }) { + if (!options.currentRevision) + throw new Error("Settle the workpiece before drafting an experiment."); + await validateDeclaredBasis( + data.basis, + options.currentRevision, + options.retainedRevisionFor, + ); + const observation = await options.observationFor( + data.observation.toolCallId, + ); + if (observation.sha256 !== data.observation.baseHash) + throw new Error( + "Experiment draft base differs from the verified browser observation.", + ); + return { output: { awaiting: AWAITING_CLIENT }, terminate: true }; + }, + }); diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/tools/mutate-petrinet.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/tools/mutate-petrinet.ts index d82f84ee490..8be86a0a51c 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/tools/mutate-petrinet.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/src/tools/mutate-petrinet.ts @@ -17,7 +17,7 @@ export const createMutatePetrinetTool = ( defineTool({ name: mutatePetrinautNetToolName, description: - "Apply one ordered batch of operations that add (addPlace, addTransition, addArc, addType, addTypeElement, addParameter, addDifferentialEquation), remove (removePlace, removeTransition, removeArc, removeType, removeTypeElement, removeParameter, removeDifferentialEquation) or edit existing parts of the net (updatePlace, updateTransition, updateArcWeight, updateArcType, updateType, updateTypeElement, updateParameter, updateDifferentialEquation). To correct something that already exists, edit it by ID with only the fields that change; do not remove and re-add it. removePlace also removes arcs connected to that place; removeType and removeDifferentialEquation clear the places that referenced them, so code that read those tokens or parameters will need repair. Each operation is a flat {operationId, basisId, type, input} object. Every operation must cite one deduplicated declared basis via basisId and the exact preceding browser observation/base. A structurally applied batch is not compiler-clean; after a batch that writes code or changes a dependency of code, obtain read_petrinaut_diagnostics before relying on the result. Operations commit in order; failure stops and leaves later operations unattempted, so place addType, addParameter and addDifferentialEquation before the places and transitions that reference them, and addPlace/addTransition before the arcs that connect them.", + "Apply one ordered batch of operations that add (addPlace, addTransition, addArc, addType, addTypeElement, addParameter, addDifferentialEquation), remove (removePlace, removeTransition, removeArc, removeType, removeTypeElement, removeParameter, removeDifferentialEquation) or edit existing parts of the net (updatePlace, updateTransition, updateArcWeight, updateArcType, updateType, updateTypeElement, updateParameter, updateDifferentialEquation), and save, edit or remove the scenarios and metrics a run or experiment names (addScenario, updateScenario, removeScenario, addMetric, updateMetric, removeMetric). A scenario carries a per_place initialState keyed by place ID and typed scenarioParameters read as scenario.; a count is an integer parameter. Metric code reads the simulated state. To correct something that already exists, edit it by ID with only the fields that change; do not remove and re-add it. removePlace also removes arcs connected to that place; removeType and removeDifferentialEquation clear the places that referenced them, so code that read those tokens or parameters will need repair. Each operation is a flat {operationId, basisId, type, input} object. Every operation must cite one deduplicated declared basis via basisId and the exact preceding browser observation/base. A structurally applied batch is not compiler-clean; after a batch that writes code or changes a dependency of code, obtain read_petrinaut_diagnostics before relying on the result. Operations commit in order; failure stops and leaves later operations unattempted, so place addType, addParameter and addDifferentialEquation before the places and transitions that reference them, and addPlace/addTransition before the arcs that connect them.", input: mutatePetrinetInputSchema, output: v.object({ awaiting: v.literal(AWAITING_CLIENT) }), async run({ data }) { diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/draft-experiment.test.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/draft-experiment.test.ts new file mode 100644 index 00000000000..320826f0852 --- /dev/null +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/draft-experiment.test.ts @@ -0,0 +1,378 @@ +import { describe, expect, test, vi } from "vitest"; + +import { + draftPetrinautExperimentInputSchema, + draftPetrinautExperimentOutputSchema, + draftPetrinautExperimentToolName, + isDraftPetrinautExperimentToolName, +} from "../src/draft-experiment"; +import { createDraftExperimentTool } from "../src/tools/draft-experiment"; + +import type { DefinitionObservation } from "../src/mutation-record"; +import type { SDCPN } from "@hashintel/petrinaut-core"; + +const hash = "a".repeat(64); +const currentRevision = { + revisionId: "revision-1", + ordinal: 1, + markdown: "Queue work.", + sha256: "b".repeat(64), + evidence: [], +}; +const emptyDefinition: SDCPN = { + places: [], + transitions: [], + types: [], + differentialEquations: [], + parameters: [], +}; + +const declaredBasis = { + kind: "declared" as const, + revisionId: currentRevision.revisionId, + sha256: currentRevision.sha256, + locators: [{ start: 0, end: 5 }], + rationale: "The workpiece states the decision, measure and range.", + scope: "operation" as const, +}; + +const experiment = { + name: "Vans on the parcel route", + scenarioId: "scenario-peak", + scenarioParameterValues: { + vans: { mode: "range" as const, min: 2, max: 8 }, + }, + runCount: 40, + seed: 7, + dt: 0.5, + maxTime: 120, + metricIds: ["metric-wait", "metric-late"], + execution: { + mode: "optimize" as const, + objectiveMetricId: "metric-wait", + direction: "minimize" as const, + steps: 10, + runsPerStep: 4, + }, +}; + +const input = { + observation: { toolCallId: "read-1", baseHash: hash }, + experiment, + declarations: [ + { + subject: "maxTime", + statement: "120 model minutes: the two-hour peak window.", + }, + { + subject: "metric-late", + statement: "Late parcels are reported, not enforced.", + }, + ], + basis: declaredBasis, + unsupported: [ + { + condition: "No parcel may wait more than 30 minutes.", + reason: + "The request carries no constraints, so the threshold is not enforced.", + reportedByMetricId: "metric-late", + }, + ], +}; + +const withExperiment = ( + patch: Partial< + Omit & { + scenarioParameterValues: Record; + } + >, +) => ({ + ...input, + experiment: { ...experiment, ...patch }, +}); + +describe("draft_petrinaut_experiment input schema", () => { + test("names the tool", () => { + expect(draftPetrinautExperimentToolName).toBe("draft_petrinaut_experiment"); + expect( + isDraftPetrinautExperimentToolName("draft_petrinaut_experiment"), + ).toBe(true); + expect(isDraftPetrinautExperimentToolName("mutate_petrinaut_net")).toBe( + false, + ); + }); + + test("accepts an integer range, a declared basis and a disclosed restriction", () => { + const parsed = draftPetrinautExperimentInputSchema.parse(input); + expect(parsed.experiment.execution.mode).toBe("optimize"); + expect(parsed.unsupported[0]?.reportedByMetricId).toBe("metric-late"); + expect(parsed.unsupported[0]?.blocksRun).toBe(true); + expect( + draftPetrinautExperimentInputSchema.parse({ + ...input, + unsupported: [{ ...input.unsupported[0], blocksRun: false }], + }).unsupported[0]?.blocksRun, + ).toBe(false); + }); + + test("rejects claiming that an unselected metric will be reported", () => { + const result = draftPetrinautExperimentInputSchema.safeParse({ + ...input, + unsupported: [ + { ...input.unsupported[0], reportedByMetricId: "metric-not-selected" }, + ], + }); + expect(result.success).toBe(false); + expect(result.error?.issues[0]?.path).toEqual([ + "unsupported", + 0, + "reportedByMetricId", + ]); + }); + + test("accepts an absent basis when its reason is stated", () => { + expect( + draftPetrinautExperimentInputSchema.safeParse({ + ...input, + basis: { + kind: "absent", + reason: "The range was stated in conversation and not yet settled.", + }, + }).success, + ).toBe(true); + }); + + test("requires recorded workpiece acceptance before unblocking a reporting-only run", () => { + const result = draftPetrinautExperimentInputSchema.safeParse({ + ...input, + basis: { + kind: "absent", + reason: "No settled acceptance was recorded.", + }, + unsupported: [ + { + ...input.unsupported[0], + blocksRun: false, + }, + ], + }); + + expect(result.success).toBe(false); + expect(result.error?.issues[0]?.path).toEqual([ + "unsupported", + 0, + "blocksRun", + ]); + }); + + test("rejects an objective that is not among the saved metrics", () => { + const result = draftPetrinautExperimentInputSchema.safeParse( + withExperiment({ + execution: { ...experiment.execution, objectiveMetricId: "metric-x" }, + }), + ); + expect(result.success).toBe(false); + expect(JSON.stringify(result.error?.issues)).toMatch( + /objective must be included in metricIds/u, + ); + }); + + test("rejects a time step longer than the horizon", () => { + expect( + draftPetrinautExperimentInputSchema.safeParse( + withExperiment({ dt: 200, maxTime: 120 }), + ).success, + ).toBe(false); + }); + + test("rejects a search budget above the host's bound", () => { + const tooMany = draftPetrinautExperimentInputSchema.safeParse( + withExperiment({ + runCount: 1000, + execution: { ...experiment.execution, steps: 100, runsPerStep: 200 }, + }), + ); + expect(tooMany.success).toBe(false); + const overRunCount = draftPetrinautExperimentInputSchema.safeParse( + withExperiment({ + runCount: 3, + execution: { ...experiment.execution, runsPerStep: 4 }, + }), + ); + expect(overRunCount.success).toBe(false); + }); + + test("rejects an optimization with no varied parameter", () => { + expect( + draftPetrinautExperimentInputSchema.safeParse( + withExperiment({ + scenarioParameterValues: { vans: { mode: "fixed", value: 4 } }, + }), + ).success, + ).toBe(false); + }); + + test("rejects an inverted range", () => { + expect( + draftPetrinautExperimentInputSchema.safeParse( + withExperiment({ + scenarioParameterValues: { vans: { mode: "range", min: 8, max: 2 } }, + }), + ).success, + ).toBe(false); + }); + + test("rejects duplicate metric IDs", () => { + expect( + draftPetrinautExperimentInputSchema.safeParse( + withExperiment({ metricIds: ["metric-wait", "metric-wait"] }), + ).success, + ).toBe(false); + }); + + test("has no constraint carriage: a constraints field is rejected", () => { + const result = draftPetrinautExperimentInputSchema.safeParse({ + ...input, + experiment: { + ...experiment, + constraints: [{ metricId: "metric-late", max: 0 }], + }, + }); + expect(result.success).toBe(false); + expect(JSON.stringify(result.error?.issues)).toMatch(/constraints/u); + }); + + test("requires at least one declaration", () => { + expect( + draftPetrinautExperimentInputSchema.safeParse({ + ...input, + declarations: [], + }).success, + ).toBe(false); + }); + + test("requires the observation the identifiers were copied from", () => { + const { observation: _observation, ...withoutObservation } = input; + expect( + draftPetrinautExperimentInputSchema.safeParse(withoutObservation).success, + ).toBe(false); + expect( + draftPetrinautExperimentInputSchema.safeParse({ + ...input, + observation: { toolCallId: "read-1", baseHash: "not-a-hash" }, + }).success, + ).toBe(false); + }); + + test("serialises to JSON Schema without dropping the request fields", () => { + const jsonSchema = draftPetrinautExperimentInputSchema[ + "~standard" + ].jsonSchema.input({ target: "draft-2020-12" }) as + | { properties?: Record } + | undefined; + expect(jsonSchema?.properties).toBeDefined(); + expect(Object.keys(jsonSchema?.properties ?? {}).sort()).toEqual([ + "basis", + "declarations", + "experiment", + "observation", + "unsupported", + ]); + const experimentSchema = jsonSchema?.properties?.experiment as { + properties?: Record; + }; + expect(Object.keys(experimentSchema.properties ?? {}).sort()).toEqual([ + "dt", + "execution", + "maxTime", + "metricIds", + "name", + "runCount", + "scenarioId", + "scenarioParameterValues", + "seed", + ]); + }); +}); + +describe("draft_petrinaut_experiment output schema", () => { + test("carries the browser's preparation status and diagnostics", () => { + expect( + draftPetrinautExperimentOutputSchema.safeParse({ + status: "drafted", + summary: "Vary vans 2–8; minimize metric-wait.", + diagnostics: [], + }).success, + ).toBe(true); + expect( + draftPetrinautExperimentOutputSchema.safeParse({ + status: "running", + summary: "", + diagnostics: [], + }).success, + ).toBe(false); + }); +}); + +describe("createDraftExperimentTool", () => { + test("validates the workpiece and exact prior observation before deferring", async () => { + const observationFor = vi.fn< + (id: string) => Promise + >(async () => ({ definition: emptyDefinition, sha256: hash })); + const tool = createDraftExperimentTool({ + currentRevision, + retainedRevisionFor: async () => undefined, + observationFor, + }); + + await expect(tool.run({ data: input } as never)).resolves.toMatchObject({ + output: { awaiting: "client" }, + terminate: true, + }); + expect(observationFor).toHaveBeenCalledWith("read-1"); + }); + + test("rejects a draft whose identifiers came from a stale observation", async () => { + const stale = createDraftExperimentTool({ + currentRevision, + retainedRevisionFor: async () => undefined, + observationFor: async () => ({ + definition: emptyDefinition, + sha256: "c".repeat(64), + }), + }); + await expect(stale.run({ data: input } as never)).rejects.toThrow( + /differs from the verified browser observation/u, + ); + }); + + test("rejects a draft before the workpiece is settled", async () => { + const unsettled = createDraftExperimentTool({ + currentRevision: null, + retainedRevisionFor: async () => undefined, + observationFor: async () => ({ + definition: emptyDefinition, + sha256: hash, + }), + }); + await expect(unsettled.run({ data: input } as never)).rejects.toThrow( + /Settle the workpiece/u, + ); + }); + + test("rejects a basis citing a different workpiece revision", async () => { + const tool = createDraftExperimentTool({ + currentRevision, + retainedRevisionFor: async () => undefined, + observationFor: async () => ({ + definition: emptyDefinition, + sha256: hash, + }), + }); + await expect( + tool.run({ + data: { ...input, basis: { ...declaredBasis, sha256: "d".repeat(64) } }, + } as never), + ).rejects.toThrow(/citation hash mismatch/u); + }); +}); diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/mutate-petrinet.test.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/mutate-petrinet.test.ts index 39e6a8e7eff..a934d2a1c6d 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/mutate-petrinet.test.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/mutate-petrinet.test.ts @@ -368,19 +368,112 @@ describe("mutate_petrinet tool", () => { type: "removeType", input: { typeId: "item" }, }, + // Saved scenarios and metrics: a count that an experiment may vary is + // an integer scenario parameter; the objective is a saved metric. + { + operationId: "add-peak", + basisId: "queue-basis", + type: "addScenario", + input: { + id: "peak-demand", + name: "Peak demand", + scenarioParameters: [ + { identifier: "active_agents", type: "integer", default: 4 }, + ], + initialState: { + type: "per_place", + content: { queue: "scenario.active_agents" }, + }, + }, + }, + { + operationId: "describe-peak", + basisId: "queue-basis", + type: "updateScenario", + input: { + scenarioId: "peak-demand", + update: { description: "Monday morning arrivals" }, + }, + }, + { + operationId: "add-wait", + basisId: "queue-basis", + type: "addMetric", + input: { + id: "average-wait", + name: "Average wait", + code: "return state.places.Queue.count;", + }, + }, + { + operationId: "rename-wait", + basisId: "queue-basis", + type: "updateMetric", + input: { + metricId: "average-wait", + update: { name: "Average waiting time" }, + }, + }, + { + operationId: "drop-wait", + basisId: "queue-basis", + type: "removeMetric", + input: { metricId: "average-wait" }, + }, + { + operationId: "drop-peak", + basisId: "queue-basis", + type: "removeScenario", + input: { scenarioId: "peak-demand" }, + }, ]; expect( mutatePetrinetInputSchema .parse({ ...input, operations: edits }) .operations.map(({ type }) => type), ).toEqual(edits.map(({ type }) => type)); + // Scenario parameters keep the canonical primitive types; a count is not + // widened past `integer`, and an unknown identity field is refused. + expect( + mutatePetrinetInputSchema.safeParse({ + ...input, + operations: [ + { + operationId: "bad-type", + basisId: "queue-basis", + type: "addScenario", + input: { + id: "bad", + name: "Bad", + scenarioParameters: [ + { identifier: "active_agents", type: "count", default: 4 }, + ], + initialState: { type: "per_place", content: {} }, + }, + }, + ], + }).success, + ).toBe(false); + expect( + mutatePetrinetInputSchema.safeParse({ + ...input, + operations: [ + { + operationId: "bad-key", + basisId: "queue-basis", + type: "removeMetric", + input: { scenarioId: "average-wait" }, + }, + ], + }).success, + ).toBe(false); // An unadmitted operation is refused at its own position with the admitted // list spelled out, so the model sees which operation was unsupported and // what it may send instead. Nothing is applied. for (const type of [ "updatePlacePosition", "updateTransitionPosition", - "addScenario", + "addSubnet", "moveTypeElement", ]) { const refused = mutatePetrinetInputSchema.safeParse({ @@ -493,7 +586,7 @@ describe("mutate_petrinet tool", () => { ); expect(property(bases, "items", root) ?? bases).toBeDefined(); const variants = variantsOf(operations, root); - expect(variants.length).toBe(22); + expect(variants.length).toBe(28); for (const variant of variants) { expect(requiredOf(variant).sort()).toEqual( ["basisId", "input", "operationId", "type"].sort(), diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/root-node.test.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/root-node.test.ts index 0c452c18dc0..a40538bace5 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/root-node.test.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/root-node.test.ts @@ -48,6 +48,14 @@ describe("native root node construction", () => { [], ), ).toThrow(/Duplicate/); + current.metrics = [{ id: "metric-id", name: "Metric", code: "return 1;" }]; + expect(() => + assertNodeIdentity( + request("addTransition", { ...transition, id: "metric-id" }), + current, + [], + ), + ).toThrow(/Duplicate/); expect(() => assertNodeIdentity(request("addPlace", place), empty(), [current]), ).toThrow(/retired/); diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/root-state.test.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/root-state.test.ts index dcb4af27f18..8e5f968c1ba 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/root-state.test.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/root-state.test.ts @@ -495,3 +495,278 @@ describe("native typed state construction", () => { ).toThrow(/Invalid input/u); }); }); + +describe("saved scenarios and metrics an experiment names", () => { + const agents = { + id: "agents", + name: "Agents", + colorId: null, + dynamicsEnabled: false, + differentialEquationId: null, + x: 0, + y: 0, + } satisfies SDCPN["places"][number]; + const metric = { + id: "average-wait", + name: "Average wait", + code: "return state.places.Agents.count;", + } satisfies PetrinautAiToolInput<"addMetric">; + const staffing = { + id: "peak-demand", + name: "Peak demand", + scenarioParameters: [ + { identifier: "active_agents", type: "integer", default: 4 }, + ], + initialState: { + type: "per_place", + content: { [agents.id]: "scenario.active_agents" }, + }, + } satisfies PetrinautAiToolInput<"addScenario">; + const withAgents = () => { + const definition = empty(); + definition.places.push(agents); + return definition; + }; + const expectMetricMutationsSupported = (before: SDCPN) => { + const afterAdd = expectedNodeDefinition( + request("addMetric", metric), + before, + ); + expect(afterAdd.metrics).toEqual([metric]); + const afterUpdate = expectedNodeDefinition( + request("updateMetric", { + metricId: metric.id, + update: { name: "Average waiting time" }, + }), + afterAdd, + ); + expect(afterUpdate.metrics?.[0]?.name).toBe("Average waiting time"); + }; + + test("admits a scenario whose integer parameter sets a place count by expression", () => { + const before = withAgents(); + const req = request("addScenario", staffing); + assertStateIdentity(req, before, []); + const after = expectedNodeDefinition(req, before); + expect(after.scenarios?.[0]).toMatchObject({ + id: staffing.id, + scenarioParameters: [{ identifier: "active_agents", type: "integer" }], + }); + expect(outcome(req, before, after)).toBe("applied"); + const effects = deriveMutationEffects(req, before, after); + expect(effects.created.map((change) => change.path).sort()).toEqual([ + "/scenarios/0/id", + "/scenarios/0/initialState", + "/scenarios/0/name", + "/scenarios/0/scenarioParameters", + ]); + const located = locateRootState(after, { + kind: "scenario", + name: staffing.name, + field: "/scenarioParameters/0/type", + }); + expect(located).toMatchObject({ + path: "/scenarios/0/scenarioParameters/0/type", + value: "integer", + }); + expect(located.formalism).toContain("not an operating range"); + }); + + test("the first metric is one entity creation, not a collection creation, and is located by kind", () => { + const before = withAgents(); + expect(before.metrics).toBeUndefined(); + const req = request("addMetric", metric); + assertStateIdentity(req, before, []); + const after = expectedNodeDefinition(req, before); + expect(after.metrics).toEqual([metric]); + expect(outcome(req, before, after)).toBe("applied"); + const effects = deriveMutationEffects(req, before, after); + expect(effects.created.map((change) => change.path).sort()).toEqual([ + "/metrics/0/code", + "/metrics/0/id", + "/metrics/0/name", + ]); + expect(effects.derived).toEqual([]); + const located = locateRootState(after, { + kind: "metric", + name: metric.name, + field: "code", + }); + expect(located).toMatchObject({ + kind: "metric", + id: metric.id, + nodePath: "/metrics/0", + path: "/metrics/0/code", + value: metric.code, + }); + expect(located.formalism).toContain("name it as an objective"); + expect( + parseConstructionWhyInput({ kind: "metric", name: metric.id }), + ).toEqual({ kind: "metric", name: metric.id, field: "entity" }); + // A metric identity collides with every other root identity, and a + // retired one is not reused. + expect(() => + assertStateIdentity(request("addMetric", metric), after, []), + ).toThrow(/Duplicate/); + expect(() => + assertStateIdentity( + request("addMetric", { ...metric, id: agents.id }), + before, + [], + ), + ).toThrow(/Duplicate/); + expect(() => + assertStateIdentity(request("addMetric", metric), before, [after]), + ).toThrow(/retired/); + }); + + test("a metric edit attributes only the requested field and refuses an unknown metric", () => { + const before = withAgents(); + before.metrics = [metric]; + const req = request("updateMetric", { + metricId: metric.id, + update: { name: "Average waiting time" }, + }); + assertStateIdentity(req, before, []); + const after = expectedNodeDefinition(req, before); + const effects = deriveMutationEffects(req, before, after); + expect(effects.updated).toEqual([ + { + kind: "updated", + path: "/metrics/0/name", + before: metric.name, + after: "Average waiting time", + }, + ]); + expect(effects.derived).toEqual([]); + expect(outcome(req, before, after)).toBe("applied"); + expect(() => + assertStateIdentity( + request("updateMetric", { + metricId: "missing", + update: { name: "Nothing" }, + }), + before, + [], + ), + ).toThrow(/Unknown or ambiguous metric/); + }); + + test("refuses duplicate metric names before they can make every run fail", () => { + const before = withAgents(); + before.metrics = [ + metric, + { ...metric, id: "queue-length", name: "Queue length" }, + ]; + + expect(() => + assertStateIdentity( + request("addMetric", { + ...metric, + id: "duplicate-name", + }), + before, + [], + ), + ).toThrow(/metric name/u); + expect(() => + assertStateIdentity( + request("updateMetric", { + metricId: "queue-length", + update: { name: metric.name }, + }), + before, + [], + ), + ).toThrow(/metric name/u); + expect(() => + assertStateIdentity( + request("updateMetric", { + metricId: metric.id, + update: { name: metric.name }, + }), + before, + [], + ), + ).not.toThrow(); + }); + + test("metric changes are independent of code-authored scenario footprints", () => { + const before = withAgents(); + before.scenarios = [ + { + id: "code-scenario", + name: "Code scenario", + scenarioParameters: [], + parameterOverrides: {}, + initialState: { type: "code", content: "return {};" }, + }, + ]; + expectMetricMutationsSupported(before); + }); + + test("metric changes are independent of nested nets", () => { + const before = withAgents(); + before.subnets = [ + { + id: "nested-net", + name: "Nested net", + places: [], + transitions: [], + types: [], + parameters: [], + differentialEquations: [], + componentInstances: [], + }, + ]; + expectMetricMutationsSupported(before); + }); + + test("removals of a scenario or metric are one direct deletion at the located index", () => { + const before = withAgents(); + before.scenarios = [ + { ...staffing, id: "baseline", name: "Baseline", parameterOverrides: {} }, + { ...staffing, parameterOverrides: {} }, + ]; + before.metrics = [metric, { ...metric, id: "queue-length", name: "Queue" }]; + const dropScenario = request("removeScenario", { scenarioId: "baseline" }); + const afterScenario = expectedNodeDefinition(dropScenario, before); + expect(afterScenario.scenarios?.map((entry) => entry.id)).toEqual([ + staffing.id, + ]); + const scenarioEffects = deriveMutationEffects( + dropScenario, + before, + afterScenario, + ); + expect(scenarioEffects.deleted.map((change) => change.path)).toEqual([ + "/scenarios/0", + ]); + expect(scenarioEffects.derived).toEqual([]); + expect(outcome(dropScenario, before, afterScenario)).toBe("applied"); + + const dropMetric = request("removeMetric", { metricId: metric.id }); + const afterMetric = expectedNodeDefinition(dropMetric, before); + expect(afterMetric.metrics?.map((entry) => entry.id)).toEqual([ + "queue-length", + ]); + const metricEffects = deriveMutationEffects( + dropMetric, + before, + afterMetric, + ); + expect(metricEffects.deleted.map((change) => change.path)).toEqual([ + "/metrics/0", + ]); + expect(metricEffects.derived).toEqual([]); + expect(outcome(dropMetric, before, afterMetric)).toBe("applied"); + // A missing target has no located index; the canonical no-op is not laundered as a removal. + expect(() => + deriveMutationEffects( + request("removeMetric", { metricId: "missing" }), + before, + before, + ), + ).toThrow(/Unknown or ambiguous/); + }); +}); diff --git a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/sdcpn-modelling-skill.test.ts b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/sdcpn-modelling-skill.test.ts index cb22d980286..c0037c5169f 100644 --- a/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/sdcpn-modelling-skill.test.ts +++ b/libs/@hashintel/brunch-agent/packages/plugin-sdcpn/test/sdcpn-modelling-skill.test.ts @@ -17,6 +17,7 @@ describe("the authored sdcpn-modelling skill directory", () => { expect(sdcpnModellingSkill.description).toContain("process model"); expect(Object.keys(sdcpnModellingSkill.files ?? {}).sort()).toEqual([ "references/checks.md", + "references/experiment-configuration.md", "references/pn-construction.md", "references/profile.md", "templates/workpiece.md", @@ -44,10 +45,115 @@ describe("the authored sdcpn-modelling skill directory", () => { const workpiece = readSkillFile("templates/workpiece.md"); const construction = readSkillFile("references/pn-construction.md"); const checks = readSkillFile("references/checks.md"); + const experiment = readSkillFile("references/experiment-configuration.md"); expect(profile).not.toMatch(/Vestera|truck fleet|semiconductor/iu); expect(workpiece).not.toMatch(/Vestera|truck fleet|semiconductor/iu); expect(construction).not.toMatch(/Vestera|truck fleet|semiconductor/iu); expect(checks).not.toMatch(/Vestera|truck fleet|semiconductor/iu); + expect(experiment).not.toMatch( + /Vestera|truck fleet|semiconductor|support desk|support operation|agents on duty|\bvans?\b|\bparcels?\b|\bSaturday\b|\bdepot\b/iu, + ); + }); + + describe("experiment readiness guidance", () => { + const instructions = sdcpnModellingSkill.instructions; + const experiment = readSkillFile("references/experiment-configuration.md"); + const construction = readSkillFile("references/pn-construction.md"); + + test("is hooked into the Construct disposition, not a separate phase or keyword", () => { + const construct = instructions.slice( + instructions.indexOf("### Construct"), + instructions.indexOf("### Check and deliver"), + ); + expect(construct).toContain("references/experiment-configuration.md"); + expect(construct).toContain("ordinary construction"); + expect(construct).toContain( + "Parameters or metrics in the net never trigger a proposal by themselves", + ); + expect(instructions).not.toMatch( + /asks? for an experiment|says? "optimi/iu, + ); + }); + + test("states readiness as the conjunction of workpiece meaning and net executability", () => { + expect(experiment).toContain("**Readiness is the conjunction**"); + expect(experiment).toContain("Structure alone never triggers a proposal"); + expect(experiment).toContain( + "The net alone never supplies the objective", + ); + expect(experiment).toContain( + "parameters and metrics recorded but no stated decision, there is nothing to propose", + ); + for (const source of [ + "What the model must answer, compare, or support", + "Goals, measures, constraints, and thresholds", + "What the result must not claim", + "definition.scenarios[]", + "definition.metrics[]", + "read_petrinaut_diagnostics", + ]) { + expect(experiment).toContain(source); + } + }); + + test("teaches the request shape without a Brunch experiment schema or constraint claims", () => { + for (const field of [ + 'scenarioParameterValues[identifier] = { mode: "range", min, max }', + "objectiveMetricId", + "`scenarioId`", + "`maxTime`, `dt`", + "`runCount`, `runsPerStep`, `steps`, `seed`", + "steps × runsPerStep ≤ 10,000", + ]) { + expect(experiment).toContain(field); + } + expect(experiment).toContain( + "The request carries no constraints and no constraint policy", + ); + expect(experiment).toContain("reported, not enforced"); + expect(experiment).toContain( + "Never encode a hard restriction as an objective penalty", + ); + expect(experiment).not.toMatch(/constraintPolicy|alpha|α/u); + }); + + test("names the draft tool, the honesty wording and the once-only rule", () => { + expect(experiment).toContain("`draft_petrinaut_experiment`"); + expect(experiment).toContain( + "{ experiment, declarations, basis, unsupported }", + ); + expect(experiment).toContain( + 'Say "drafted for this session", not "added to the model"', + ); + expect(experiment).toContain("you never call a run"); + expect(experiment).toContain("## Once, not repeatedly"); + expect(experiment).toContain('Treat an explicit "do not run"'); + expect(experiment).toContain("as authoritative"); + expect(experiment).toContain( + "Run and Dismiss happen later in the card and are not reported", + ); + expect(experiment).toContain( + "do not infer any of those events or describe them as observed", + ); + expect(experiment).not.toMatch( + /After the person declines|experiment has completed/iu, + ); + expect(experiment).toContain( + "Do not apply a winning configuration to the model on your own", + ); + expect(experiment).toContain( + "Never open or pre-fill the experiment creation drawer", + ); + }); + + test("teaches scenarios and metrics as construction with the typed sweep domain", () => { + expect(construction).toContain("`addScenario`"); + expect(construction).toContain("`addMetric`"); + expect(construction).toContain("a count is an `integer` parameter"); + expect(construction).toContain("`parameterOverrides`"); + expect(experiment).toContain("never rounding in code"); + expect(experiment).toContain("A `boolean` parameter rejects ranges"); + }); }); test("names the mounted construction read tool", () => { diff --git a/libs/@hashintel/petrinaut-core/src/selected-mutation-batch.test.ts b/libs/@hashintel/petrinaut-core/src/selected-mutation-batch.test.ts index e41eb0bbac1..147577bfb86 100644 --- a/libs/@hashintel/petrinaut-core/src/selected-mutation-batch.test.ts +++ b/libs/@hashintel/petrinaut-core/src/selected-mutation-batch.test.ts @@ -172,7 +172,48 @@ describe("selected mutation batch", () => { }, ]), ).toHaveLength(4); - for (const type of ["addScenario", "addMetric"] as const) { + expect( + selectedMutationBatchSchema.parse([ + { + operationId: "scenario", + type: "addScenario", + input: { + id: "peak", + name: "Peak", + scenarioParameters: [ + { identifier: "active_agents", type: "integer", default: 4 }, + ], + initialState: { type: "per_place", content: {} }, + }, + }, + { + operationId: "metric", + type: "addMetric", + input: { id: "wait", name: "Wait", code: "return 0;" }, + }, + { + operationId: "describe-scenario", + type: "updateScenario", + input: { scenarioId: "peak", update: { description: "Mondays" } }, + }, + { + operationId: "rename-metric", + type: "updateMetric", + input: { metricId: "wait", update: { name: "Waiting time" } }, + }, + { + operationId: "drop-metric", + type: "removeMetric", + input: { metricId: "wait" }, + }, + { + operationId: "drop-scenario", + type: "removeScenario", + input: { scenarioId: "peak" }, + }, + ]), + ).toHaveLength(6); + for (const type of ["addSubnet", "addComponentInstance"] as const) { expect(() => selectedMutationBatchSchema.parse([ { @@ -181,7 +222,7 @@ describe("selected mutation batch", () => { input: {}, }, ]), - ).toThrow(); + ).toThrow(/Invalid discriminator value/u); } expect(() => selectedMutationBatchSchema.parse([ diff --git a/libs/@hashintel/petrinaut-core/src/selected-mutation-batch.ts b/libs/@hashintel/petrinaut-core/src/selected-mutation-batch.ts index b9256f55419..cb6d3e30e9e 100644 --- a/libs/@hashintel/petrinaut-core/src/selected-mutation-batch.ts +++ b/libs/@hashintel/petrinaut-core/src/selected-mutation-batch.ts @@ -117,6 +117,37 @@ export const selectedMutationOperationSchema = z.discriminatedUnion("type", [ type: z.literal("removeDifferentialEquation"), input: mutationActionInputSchemas.removeDifferentialEquation, }), + // Simulation scenarios and metrics, the saved entities an experiment names. + z.strictObject({ + operationId: z.string().min(1), + type: z.literal("addScenario"), + input: mutationActionInputSchemas.addScenario, + }), + z.strictObject({ + operationId: z.string().min(1), + type: z.literal("updateScenario"), + input: mutationActionInputSchemas.updateScenario, + }), + z.strictObject({ + operationId: z.string().min(1), + type: z.literal("removeScenario"), + input: mutationActionInputSchemas.removeScenario, + }), + z.strictObject({ + operationId: z.string().min(1), + type: z.literal("addMetric"), + input: mutationActionInputSchemas.addMetric, + }), + z.strictObject({ + operationId: z.string().min(1), + type: z.literal("updateMetric"), + input: mutationActionInputSchemas.updateMetric, + }), + z.strictObject({ + operationId: z.string().min(1), + type: z.literal("removeMetric"), + input: mutationActionInputSchemas.removeMetric, + }), ]); export const selectedMutationBatchSchema = z diff --git a/libs/@hashintel/petrinaut/docs/ai-assistant.md b/libs/@hashintel/petrinaut/docs/ai-assistant.md index 665feb51cbd..25a47e438c3 100644 --- a/libs/@hashintel/petrinaut/docs/ai-assistant.md +++ b/libs/@hashintel/petrinaut/docs/ai-assistant.md @@ -302,6 +302,28 @@ A request with no saved result and no active run shows **Not running**. Ask the assistant to run a new experiment. **Cancel** is available only for experiments running in this panel. +### Brunch-drafted experiments + +On the Petrinaut website, Brunch can propose an experiment when your modelling +conversation establishes a decision, objective, parameter range, operating +regime and horizon. The card says **Drafted — not run · not saved with the +document**. Review its settings, declarations and unsupported restrictions, +then choose **Run** or **Dismiss**. Brunch can continue the conversation while +the card waits; drafting never starts execution. + +An unsupported hard restriction disables **Run**. A metric labelled +**reported, not enforced** only measures a condition; it does not enforce it. +If you explicitly accept a reporting-only exploration, ask Brunch for a revised +proposal. When the model changes after drafting, the card shows the changed +model sections for review and requires confirmation before running. + +A later proposal replaces the earlier draft in that editor. Drafts are not +saved with the document and must be drafted again after a reload or reopening +the editor. A prepared draft stays in chat without a **Simulate** badge or +active indicator. Choosing **Run** starts execution and uses the normal +**1 active** indicator; when it completes, that indicator +disappears and the result remains under **Simulate → Experiments**. + ## Read-only behaviour Whether the assistant can change the net depends on the editor state: diff --git a/libs/@hashintel/petrinaut/docs/experiments.md b/libs/@hashintel/petrinaut/docs/experiments.md index 3c115c1c860..c1c09d6381e 100644 --- a/libs/@hashintel/petrinaut/docs/experiments.md +++ b/libs/@hashintel/petrinaut/docs/experiments.md @@ -230,10 +230,14 @@ A small toast appears when an experiment **completes** or **errors**, even if it ## Active experiments popover -When any experiment is **initializing** or **running**, the top bar shows an **Active experiments** flask icon with a count (e.g. "2 active"). Click it for a popover listing each in-flight experiment with its scenario, progress, status, and a time progress bar. Clicking a row jumps directly to Simulate mode, the Experiments tab, and that experiment's panel. +When any experiment is **initializing** or **running**, the top bar shows an **Active experiments** flask icon with a count (e.g. "2 active"). A host-owned optimization remains active while its sweep is briefly **Idle** between steps and while its final result settles. Click the icon for a popover listing each in-flight experiment with its scenario, progress, status, and a time progress bar. Clicking a row jumps directly to Simulate mode, the Experiments tab, and that experiment's panel. The popover hides itself again once nothing is in flight. +On the Petrinaut website, a prepared Brunch draft is not active work: it shows +only in chat, without a **Simulate** badge or an entry in this popover. +Choose **Run** to start execution; the ordinary active indicator then appears. + ## Experiments and single-run Play Experiments and the bottom-bar **Play** controls are independent systems: diff --git a/libs/@hashintel/petrinaut/src/react/experiment-host/prepare-experiment.ts b/libs/@hashintel/petrinaut/src/react/experiment-host/prepare-experiment.ts index cc657a227b8..65e33e59d29 100644 --- a/libs/@hashintel/petrinaut/src/react/experiment-host/prepare-experiment.ts +++ b/libs/@hashintel/petrinaut/src/react/experiment-host/prepare-experiment.ts @@ -19,6 +19,7 @@ export const prepareExperiment = ( request: PetrinautExperimentRequest; input: CreateExperimentInput; fixedValues: Record; + parameterAxes: ExperimentParameterAxis[]; optimization: PetrinautOptimizationInput | null; } => { const request = petrinautExperimentRequestSchema.parse(rawRequest); @@ -125,5 +126,5 @@ export const prepareExperiment = ( runsPerStep: execution.runsPerStep, }) : null; - return { request, input, fixedValues, optimization }; + return { request, input, fixedValues, parameterAxes, optimization }; }; diff --git a/libs/@hashintel/petrinaut/src/react/experiment-host/provider.tsx b/libs/@hashintel/petrinaut/src/react/experiment-host/provider.tsx index c2009b00ce0..fc3cdc81638 100644 --- a/libs/@hashintel/petrinaut/src/react/experiment-host/provider.tsx +++ b/libs/@hashintel/petrinaut/src/react/experiment-host/provider.tsx @@ -69,6 +69,8 @@ export const ExperimentHostProvider = ({ children }: PropsWithChildren) => { title, experiments, optimizations, + optimizationUnavailableReason: + optimizationsContext.optimizationUnavailableReason ?? null, actions: { createExperiment: experimentsContext.createExperiment, navigateSweep: experimentsContext.navigateSweep, diff --git a/libs/@hashintel/petrinaut/src/react/experiment-host/run-experiment.test.ts b/libs/@hashintel/petrinaut/src/react/experiment-host/run-experiment.test.ts index ff048ba1191..ec3fb67ef68 100644 --- a/libs/@hashintel/petrinaut/src/react/experiment-host/run-experiment.test.ts +++ b/libs/@hashintel/petrinaut/src/react/experiment-host/run-experiment.test.ts @@ -172,6 +172,7 @@ const createHarness = (optimize = false) => { validate: vi.fn(async () => {}), experiments, optimizations, + optimizationUnavailableReason: null, actions, }; return { @@ -459,6 +460,26 @@ describe("runExperiment", () => { expect(harness.actions.createExperiment).not.toHaveBeenCalled(); }); + it("does not create an experiment when optimization is unavailable", async () => { + const harness = createHarness(true); + Object.assign(harness.dependencies, { + optimizationUnavailableReason: "Optimization is unavailable", + }); + harness.actions.createOptimization = vi.fn(async () => { + throw new Error("Optimization is unavailable"); + }); + + expect( + await runExperiment(harness.dependencies, makeRequest(true)), + ).toMatchObject({ + status: "error", + experimentId: null, + message: "Optimization is unavailable", + }); + expect(harness.actions.createExperiment).not.toHaveBeenCalled(); + expect(harness.actions.createOptimization).not.toHaveBeenCalled(); + }); + it("returns the metric compile error when no optimization trial produces a value", async () => { const harness = createHarness(true); const pending = runExperiment(harness.dependencies, makeRequest(true)); @@ -514,6 +535,21 @@ describe("runExperiment", () => { expect(harness.actions.createExperiment).not.toHaveBeenCalled(); }); + it("treats an omitted availability reason from a legacy context as available", async () => { + const harness = createHarness(true); + const { + optimizationUnavailableReason: _optimizationUnavailableReason, + ...legacyDependencies + } = harness.dependencies; + const pending = runExperiment(legacyDependencies, makeRequest(true)); + + await vi.waitFor(() => + expect(harness.actions.createOptimization).toHaveBeenCalledOnce(), + ); + harness.createOptions?.ownership?.cancel(); + expect((await pending).status).toBe("cancelled"); + }); + it.each>([ { scenarioId: "missing" }, { metricIds: ["missing"] }, diff --git a/libs/@hashintel/petrinaut/src/react/experiment-host/run-experiment.ts b/libs/@hashintel/petrinaut/src/react/experiment-host/run-experiment.ts index 293421df34b..f54a31fe7e0 100644 --- a/libs/@hashintel/petrinaut/src/react/experiment-host/run-experiment.ts +++ b/libs/@hashintel/petrinaut/src/react/experiment-host/run-experiment.ts @@ -31,6 +31,7 @@ export type ExperimentHostDependencies = { ) => Promise; experiments: ReadableStore; optimizations: ReadableStore; + optimizationUnavailableReason?: string | null; actions: Pick< ExperimentsActionsValue, "createExperiment" | "navigateSweep" | "cancelExperiment" @@ -188,6 +189,15 @@ export const runExperiment = async ( definition, dependencies.title, ); + const optimizationUnavailableReason = + dependencies.optimizationUnavailableReason; + if ( + optimization && + optimizationUnavailableReason !== null && + optimizationUnavailableReason !== undefined + ) { + throw new Error(optimizationUnavailableReason); + } searchRuns = request.execution.mode === "optimize" ? request.execution.runsPerStep diff --git a/libs/@hashintel/petrinaut/src/react/experiments/context.test.ts b/libs/@hashintel/petrinaut/src/react/experiments/context.test.ts index def38e85f6f..173c940ba28 100644 --- a/libs/@hashintel/petrinaut/src/react/experiments/context.test.ts +++ b/libs/@hashintel/petrinaut/src/react/experiments/context.test.ts @@ -48,28 +48,27 @@ function makeRecord(overrides: Partial): ExperimentRecord { const ALL_STATUSES: ExperimentStatus[] = [ "initializing", "running", + "idle", "complete", "error", "cancelled", ]; -describe("isTerminalExperimentStatus", () => { - it("partitions every status into exactly active or terminal", () => { - // The two must stay exact complements: `isExperimentActive` is defined as the - // negation, and the provider stamps `finishedAt` off the terminal side. +describe("experiment status predicates", () => { + it("keeps an idle sweep outside the computing and terminal sets", () => { const terminal = ALL_STATUSES.filter(isTerminalExperimentStatus); - const active = ALL_STATUSES.filter( - (status) => !isTerminalExperimentStatus(status), + const active = ALL_STATUSES.filter((status) => + isExperimentActive(makeRecord({ status })), ); expect(terminal).toStrictEqual(["complete", "error", "cancelled"]); expect(active).toStrictEqual(["initializing", "running"]); + }); - for (const status of ALL_STATUSES) { - expect(isExperimentActive(makeRecord({ status }))).toBe( - !isTerminalExperimentStatus(status), - ); - } + it("keeps a host-owned idle sweep active until the request settles", () => { + expect( + isExperimentActive(makeRecord({ status: "idle", requestActive: true })), + ).toBe(true); }); }); diff --git a/libs/@hashintel/petrinaut/src/react/experiments/context.ts b/libs/@hashintel/petrinaut/src/react/experiments/context.ts index c91032abaee..beaa4436d26 100644 --- a/libs/@hashintel/petrinaut/src/react/experiments/context.ts +++ b/libs/@hashintel/petrinaut/src/react/experiments/context.ts @@ -228,8 +228,12 @@ export function isTerminalExperimentStatus(status: ExperimentStatus): boolean { export function isExperimentActive(experiment: ExperimentRecord): boolean { // "idle" is deliberately not active: an idle sweep computes nothing, so it // neither blocks closing the window nor keeps elapsed-time tickers running. + // A host request is the exception: it owns an idle sweep while an optimizer + // is choosing the next point, and remains active until that request settles. return ( - experiment.status === "initializing" || experiment.status === "running" + experiment.requestActive === true || + experiment.status === "initializing" || + experiment.status === "running" ); } diff --git a/libs/@hashintel/petrinaut/src/react/index.ts b/libs/@hashintel/petrinaut/src/react/index.ts index 4159485c918..2fce6d7be75 100644 --- a/libs/@hashintel/petrinaut/src/react/index.ts +++ b/libs/@hashintel/petrinaut/src/react/index.ts @@ -95,6 +95,7 @@ export { isExperimentActive, } from "./experiments/context"; export { ExperimentHostContext } from "./experiment-host/context"; +export { prepareExperiment } from "./experiment-host/prepare-experiment"; export type { CreateExperimentInput, ExperimentRecord, diff --git a/libs/@hashintel/petrinaut/src/react/optimizations/context.ts b/libs/@hashintel/petrinaut/src/react/optimizations/context.ts index 65191280e57..566a93f8801 100644 --- a/libs/@hashintel/petrinaut/src/react/optimizations/context.ts +++ b/libs/@hashintel/petrinaut/src/react/optimizations/context.ts @@ -141,6 +141,13 @@ type CreateOptimizationOptions = { export type OptimizationsContextValue = { optimizations: readonly OptimizationRecord[]; + /** + * Why a new study cannot start in this host, or null while the connected + * optimizer is available. Omission retains the legacy meaning that a custom + * provider's optimizer is available. Consumers check this before creating + * the sweep record a study would drive. + */ + optimizationUnavailableReason?: string | null; /** * Starts a study driving a sweep. Rejects when no in-browser optimizer is * connected: a sweep can only be optimized in the browser. @@ -160,6 +167,7 @@ export type OptimizationsContextValue = { const DEFAULT_CONTEXT_VALUE: OptimizationsContextValue = { optimizations: [], + optimizationUnavailableReason: "Optimization is unavailable", createOptimization: () => Promise.reject(new Error("Optimization is unavailable")), cancelOptimization: () => {}, diff --git a/libs/@hashintel/petrinaut/src/react/optimizations/provider.tsx b/libs/@hashintel/petrinaut/src/react/optimizations/provider.tsx index fe7f2c917d1..61efd5e180c 100644 --- a/libs/@hashintel/petrinaut/src/react/optimizations/provider.tsx +++ b/libs/@hashintel/petrinaut/src/react/optimizations/provider.tsx @@ -93,6 +93,12 @@ const connectOptimizationSource = ( export const OptimizationsProvider = ({ children }: PropsWithChildren) => { const source = use(PetrinautOptimizationContext); + const optimizationUnavailableReason = + source === null + ? "Optimization is unavailable" + : isConnectedOptimization(source) + ? null + : "A sweep can only be optimized in the browser"; const experimentsActionsRef = useLatest(use(ExperimentsActionsContext)); const connectionRef = useRef(null); const abortControllersRef = useRef(new Map()); @@ -513,6 +519,7 @@ export const OptimizationsProvider = ({ children }: PropsWithChildren) => { const value: OptimizationsContextValue = { optimizations, + optimizationUnavailableReason, createOptimization, cancelOptimization, removeOptimization, diff --git a/libs/@hashintel/petrinaut/src/ui/types/ai-interactive-tool.ts b/libs/@hashintel/petrinaut/src/ui/types/ai-interactive-tool.ts index ccd5aa95fff..09261b78744 100644 --- a/libs/@hashintel/petrinaut/src/ui/types/ai-interactive-tool.ts +++ b/libs/@hashintel/petrinaut/src/ui/types/ai-interactive-tool.ts @@ -10,6 +10,11 @@ type InteractiveToolWidgetCommonProps = { input: Input; /** Submit one output for this tool call. Repeated calls are ignored. */ submit: (output: Output) => void; + /** + * Submit and await host acceptance. Rejects when this rendered call has no + * local completion authority. Repeated calls are ignored. + */ + submitAndWait?: (output: Output) => Promise; /** Stable AI SDK identifier for this tool call. */ toolCallId: string; }; diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/create-experiment-drawer.test.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/create-experiment-drawer.test.tsx index c1628e0a479..650eec8d5c5 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/create-experiment-drawer.test.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/create-experiment-drawer.test.tsx @@ -323,6 +323,7 @@ const TestProviders = ({ {}, removeOptimization: () => {}, diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/create-optimized-experiment.test.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/create-optimized-experiment.test.tsx index d0f69b9f0a2..603c4660860 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/create-optimized-experiment.test.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/create-optimized-experiment.test.tsx @@ -98,6 +98,7 @@ const makeHarness = ( }; const optimizations: OptimizationsContextValue = { optimizations: [], + optimizationUnavailableReason: null, createOptimization: vi.fn( (manifest, options) => { calls.push("createOptimization"); diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/study-fixtures.ts b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/study-fixtures.ts index 939c7e4dc2d..78a84499956 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/study-fixtures.ts +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/SimulateView/experiments/study-fixtures.ts @@ -263,6 +263,7 @@ export function makeOptimizationsContextValue( ): OptimizationsContextValue { return { optimizations: [optimization], + optimizationUnavailableReason: null, createOptimization: () => Promise.resolve(optimization.id), cancelOptimization: () => {}, removeOptimization: () => {}, diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx index 754f65b7f35..915849e2b85 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx @@ -48,7 +48,10 @@ import { type SDCPNContextValue, } from "../../../../react/state/sdcpn-context"; import { useCanvasInsets } from "../../../hooks/use-canvas-insets"; -import { definePetrinautAiInteractiveTool } from "../../../types/ai-interactive-tool"; +import { + definePetrinautAiInteractiveTool, + type PetrinautAiInteractiveToolWidgetProps, +} from "../../../types/ai-interactive-tool"; import { addMappedToolOutput, AiAssistantPanel, @@ -1107,18 +1110,39 @@ describe("AiAssistantPanel composer submissions", () => { test("following refuses external interactive widget completion on arrival and reload", async () => { const sendMessages = vi.fn(); + const ExternalQuestion = ({ + submitAndWait, + }: PetrinautAiInteractiveToolWidgetProps< + { question: string }, + { answer: string } + >) => { + const [completion, setCompletion] = useState("pending"); + return ( + <> + + {completion} + + ); + }; const hostTool = definePetrinautAiInteractiveTool({ toolName: "answerQuestion", inputSchema: { parse: (raw: unknown) => raw as { question: string } }, outputSchema: { parse: (raw: unknown) => raw as { answer: string } }, - component: ({ submit }) => ( - - ), + component: ExternalQuestion, }); const config: PetrinautAiAssistant = { conversationId: "external-widget", @@ -1151,6 +1175,7 @@ describe("AiAssistantPanel composer submissions", () => { screen.getByRole("button", { name: "Complete external question" }), ); }); + expect(screen.getByText("rejected")).not.toBeNull(); expect(sendMessages).not.toHaveBeenCalled(); mounted.unmount(); renderTestPanel({ aiAssistant: config }); @@ -1160,6 +1185,7 @@ describe("AiAssistantPanel composer submissions", () => { screen.getByRole("button", { name: "Complete external question" }), ); }); + expect(screen.getByText("rejected")).not.toBeNull(); expect(sendMessages).not.toHaveBeenCalled(); }); diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.tsx index 8c876fe4ee8..609c51b3275 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.tsx @@ -827,6 +827,10 @@ const ConversationAiAssistantPanel = ({ typeof setTimeout > | null>(null); const suppressedAutomaticSendsRef = useRef(0); + // Host-owned widgets can submit before their response stream settles. Keep + // their calls here until the explicit ready-state continuation begins so + // AI SDK's later stream-end auto-send is suppressed as well. + const explicitContinuationToolCallIdsRef = useRef(new Set()); const addToolOutputRef = useRef< ReturnType>["addToolOutput"] | null >(null); @@ -837,6 +841,7 @@ const ConversationAiAssistantPanel = ({ // Flue had nothing left to abort once that step settled, so withholding the // follow-up is what makes the Stop real. const withholdContinuationForStop = () => { + explicitContinuationToolCallIdsRef.current.clear(); stopRequestedRef.current = false; setContinuationPending(false); setStreamError(null); @@ -1199,8 +1204,9 @@ const ConversationAiAssistantPanel = ({ : { id: aiAssistant.conversationId }), messages: aiAssistant.messages, transport: diagnosticsTransport, - // Interactive tools retain AI SDK's native continuation; static tools - // suppress it while addAutomaticToolOutput owns the explicit chain. + // Built-in interactive commands retain AI SDK's native continuation. + // Automatic tools and host-owned dynamic widgets coordinate an explicit + // continuation below so stream settlement cannot duplicate their send. sendAutomaticallyWhen: ({ messages: currentMessages }) => { if ( suppressedAutomaticSendsRef.current !== 0 || @@ -1213,6 +1219,9 @@ const ConversationAiAssistantPanel = ({ if (automaticToolTurnIsTerminated(submissionGenerationRef.current)) { return false; } + if (explicitContinuationToolCallIdsRef.current.size !== 0) { + return false; + } if (!stopRequestedRef.current) { // Left pending until the follow-up's own status change lands, so hosts // never observe the `ready` between this check and that request. @@ -1348,6 +1357,7 @@ const ConversationAiAssistantPanel = ({ return; } const sendContinuation = sendAutomaticToolContinuationRef.current; + explicitContinuationToolCallIdsRef.current.clear(); if (sendContinuation === null) { setContinuationPending(false); const hostError = new Error("The AI assistant tool host is not ready."); @@ -1403,6 +1413,7 @@ const ConversationAiAssistantPanel = ({ followedMessagesRef.current = undefined; locallyStreamedToolCallsRef.current.clear(); abortAutomaticTools(); + explicitContinuationToolCallIdsRef.current.clear(); submissionGenerationRef.current += 1; stopRequestedRef.current = false; setContinuationPending(false); @@ -1859,6 +1870,7 @@ const ConversationAiAssistantPanel = ({ setStopped(false); stopRequestedRef.current = false; abortAutomaticTools(); + explicitContinuationToolCallIdsRef.current.clear(); submissionGenerationRef.current += 1; await submitMessage({ id: messageId, @@ -2235,6 +2247,7 @@ const ConversationAiAssistantPanel = ({ controller.abort(); experimentControllersRef.current.clear(); setExperimentStates({}); + explicitContinuationToolCallIdsRef.current.clear(); submissionGenerationRef.current += 1; // Clearing aborts any in-flight response too, which fires `onFinish` // with `isAbort`. Drop the stop flag first so that handler treats this @@ -2281,11 +2294,23 @@ const ConversationAiAssistantPanel = ({ throw new Error(`Unknown AI tool: ${toolName}`); } + // Retain suppression through stream settlement: addToolOutput skips + // its own continuation while streaming, but AI SDK checks again when + // the stream reaches ready. The ready-state effect owns the one send. + explicitContinuationToolCallIdsRef.current.add(toolCallId); return addDynamicToolOutput(addToolOutput, { tool: toolName, toolCallId, output, - }); + }).then( + () => { + setContinuationPending(true); + }, + (caught: unknown) => { + explicitContinuationToolCallIdsRef.current.delete(toolCallId); + throw caught; + }, + ); } const petrinautOutput = output as AiToolOutput; diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/tool-list.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/tool-list.tsx index dc9a1f70934..51a80917ce9 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/tool-list.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/tool-list.tsx @@ -534,6 +534,38 @@ const InteractiveToolItem = ({ const submittedOnceRef = useRef(submitted); const Widget = definition.Widget; const typedInput = definition.parseInput(input); + const submitAndWait = (output: unknown): Promise => { + if (submittedOnceRef.current) { + return Promise.resolve(); + } + if (!onInteractiveToolSubmit) { + const submissionPromise = Promise.reject( + new Error("Interactive tool submission is unavailable."), + ); + void submissionPromise.catch(() => undefined); + return submissionPromise; + } + + try { + const parsedOutput = definition.parseOutput(output); + submittedOnceRef.current = true; + const submission = onInteractiveToolSubmit({ + toolCallId: tool.id, + toolName: tool.toolName, + output: parsedOutput, + }); + const submissionPromise = Promise.resolve(submission); + void submissionPromise.catch(() => { + submittedOnceRef.current = false; + }); + return submissionPromise; + } catch (error) { + submittedOnceRef.current = false; + const submissionPromise = Promise.reject(error); + void submissionPromise.catch(() => undefined); + return submissionPromise; + } + }; if (submitted) { return ( @@ -542,6 +574,7 @@ const InteractiveToolItem = ({ input={typedInput} state="submitted" submit={() => {}} + submitAndWait={() => Promise.resolve()} submittedOutput={definition.parseOutput(submittedOutput)} toolCallId={tool.id} /> @@ -556,26 +589,9 @@ const InteractiveToolItem = ({ input={typedInput} state="awaiting" submit={(output) => { - if (submittedOnceRef.current || !onInteractiveToolSubmit) { - return; - } - - const parsedOutput = definition.parseOutput(output); - submittedOnceRef.current = true; - try { - const submission = onInteractiveToolSubmit({ - toolCallId: tool.id, - toolName: tool.toolName, - output: parsedOutput, - }); - void Promise.resolve(submission).catch(() => { - submittedOnceRef.current = false; - }); - } catch (error) { - submittedOnceRef.current = false; - throw error; - } + void submitAndWait(output); }} + submitAndWait={submitAndWait} toolCallId={tool.id} />