Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .changeset/browser-experiment-host.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
"@hashintel/petrinaut": patch
---

Expose `ExperimentHostContext.runExperiment` with validation, progress callbacks, cancellation, and captured results for up to 100,000 runs.
5 changes: 5 additions & 0 deletions .changeset/experiment-ai-tool.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
"@hashintel/petrinaut-core": patch
---

Register the createExperiment tool with its validated request schema for AI integrations.
5 changes: 5 additions & 0 deletions .changeset/small-experiments-chat.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
"@hashintel/petrinaut": patch
---

Run experiments and optimizations from AI chat with progress, cancellation, and a link to metric distributions. Keep the conversation open while inspecting experiments.
5 changes: 5 additions & 0 deletions .changeset/typed-experiment-host.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
"@hashintel/petrinaut-core": patch
---

Add validated request and result schemas and a `runExperiment` host interface for simulations and optimizations with up to 100,000 final runs.
6 changes: 0 additions & 6 deletions apps/brunch-agent/src/agents/chat-agent/tool-catalogue.ts
Original file line number Diff line number Diff line change
Expand Up @@ -47,12 +47,6 @@ export const ordinaryBrunchToolCatalogue: readonly OrdinaryBrunchToolCatalogueEn
executionOwner: "flue",
role: "substrate",
},
{
name: "brunch_mark_question",
definitionOwner: "brunch-core",
executionOwner: "brunch-app",
role: "workpiece",
},
{
name: "mutate_workpiece",
definitionOwner: "brunch-core",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@ import {
clientToolHistoryFrom,
clientToolResultSignal,
} from "@hashintel/brunch-agent-transport-aisdk";
import { BRUNCH_QUESTION_TOOL_NAMES } from "@hashintel/brunch-agent/question-marker";
import {
petrinautAiTools,
type PetrinautAiToolInput,
Expand Down Expand Up @@ -159,8 +160,10 @@ try {
});
assert.deepEqual(generatedAddType.parameters, canonicalSchema);
assert(
generatedTools.some((tool) => tool.name === "brunch_mark_question"),
"Question marker missing",
!generatedTools.some((tool) =>
BRUNCH_QUESTION_TOOL_NAMES.some((name) => name === tool.name),
),
"Legacy question marker must not be mounted",
);
assert.deepEqual(headless.definition().types, [
petrinautAiTools.addType.inputSchema.parse(nestedType),
Expand Down
3 changes: 2 additions & 1 deletion apps/brunch-agent/test/compiler-feedback.integration.ts
Original file line number Diff line number Diff line change
Expand Up @@ -34,7 +34,8 @@ import { openBrowserFixture } from "./browser-fixture.ts";
import { browserResultFrom } from "./browser-result.ts";
import { nativeSchemaProvider } from "./native-schema-provider.ts";

const cleanCompilation = "No errors or warnings found in net function code.";
const cleanCompilation =
"No errors or warnings found in net function code. Scenario and metric compilation is checked when creating an experiment.";
const output = mkdtempSync(join(tmpdir(), "m7c-compiler-feedback-"));
const save = (name: string, value: unknown) =>
writeFileSync(join(output, `${name}.json`), JSON.stringify(value, null, 2));
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,7 @@ import {
VALIDATED_CONSTRUCTION_MODE,
} from "@hashintel/brunch-agent-plugin-sdcpn/flue";
import { snapshotToUiMessages } from "@hashintel/brunch-agent-transport-aisdk";
import { BRUNCH_QUESTION_TOOL_NAME } from "@hashintel/brunch-agent/question-marker";
import { BRUNCH_QUESTION_TOOL_NAMES } from "@hashintel/brunch-agent/question-marker";

import {
CLIENT_TOOL_RESULT_SIGNAL,
Expand Down Expand Up @@ -70,7 +70,7 @@ const browserNames: ReadonlySet<string> = new Set([
const project = (history: FlueConversationSnapshot) =>
snapshotToUiMessages(history, {
clientToolNames: browserNames,
hiddenToolNames: new Set([BRUNCH_QUESTION_TOOL_NAME]),
hiddenToolNames: new Set(BRUNCH_QUESTION_TOOL_NAMES),
});
const faux = fauxProvider({
provider: "anthropic",
Expand Down Expand Up @@ -152,20 +152,10 @@ const run = async () => {
const observations = [];
try {
for (const names of [
[BRUNCH_QUESTION_TOOL_NAME, "addType"],
["addType", BRUNCH_QUESTION_TOOL_NAME],
["mutate_workpiece", "addType"],
["addType", "mutate_workpiece"],
[BRUNCH_QUESTION_TOOL_NAME, "mutate_workpiece", "addType"],
[BRUNCH_QUESTION_TOOL_NAME, "addType", "mutate_workpiece"],
["mutate_workpiece", BRUNCH_QUESTION_TOOL_NAME, "addType"],
["mutate_workpiece", "addType", BRUNCH_QUESTION_TOOL_NAME],
["addType", BRUNCH_QUESTION_TOOL_NAME, "mutate_workpiece"],
["addType", "mutate_workpiece", BRUNCH_QUESTION_TOOL_NAME],
["addType", "unmounted_admission_probe"],
["addType"],
[BRUNCH_QUESTION_TOOL_NAME],
["mutate_workpiece", BRUNCH_QUESTION_TOOL_NAME],
]) {
caseId = names.join("-");
const client = clientFor();
Expand Down Expand Up @@ -310,12 +300,7 @@ const run = async () => {
const message = fauxAssistantMessage(
[
fauxText(text),
...(abort
? [makeCall("addType")]
: [
makeCall("mutate_workpiece"),
makeCall(BRUNCH_QUESTION_TOOL_NAME),
]),
...(abort ? [makeCall("addType")] : [makeCall("mutate_workpiece")]),
Comment thread
cursor[bot] marked this conversation as resolved.
],
{ stopReason: "toolUse" },
);
Expand Down Expand Up @@ -391,7 +376,7 @@ const run = async () => {
});
}
const rejected = observations.find(
(observation) => observation.caseId === "brunch_mark_question-addType",
(observation) => observation.caseId === "mutate_workpiece-addType",
)!;
const priorIds = new Set(
rejected.seeded.messages.map((message) => message.id),
Expand Down
16 changes: 3 additions & 13 deletions apps/brunch-agent/test/integration/admission-controls.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -31,12 +31,12 @@ beforeAll(async () => {
});

test("production rejects every mixed proposal before publishing or partially executing it", () => {
expect(result.observations).toHaveLength(14);
expect(result.observations).toHaveLength(4);
const mixed = result.observations.filter(
({ generated }) =>
generated.length > 1 && generated.some((call) => call.name === "addType"),
);
expect(mixed).toHaveLength(11);
expect(mixed).toHaveLength(3);
for (const observation of mixed) {
expect(observation.pendingMutationIds).toEqual([]);
expect(observation.after).toEqual(observation.before);
Expand Down Expand Up @@ -138,7 +138,7 @@ test("every failed submission is attributable from the server output by stage, s
}
});

test("production still settles revisions and noninteractive markers without browser results", () => {
test("production settles revisions without browser results", () => {
for (const observation of result.observations) {
expect(observation.seed.error).toBeNull();
const revision = observation.seeded.messages
Expand All @@ -151,16 +151,6 @@ test("production still settles revisions and noninteractive markers without brow
output: { revisionId: `${observation.caseId}-old-revision`, ordinal: 1 },
});
}
for (const caseId of [
"brunch_mark_question",
"mutate_workpiece-brunch_mark_question",
]) {
const observation = result.observations.find(
(entry) => entry.caseId === caseId,
)!;
expect(observation.attempt.error).toBeNull();
expect(observation.providerCallsBeforeClientResult).toBe(2);
}
});

test("an independently admitted browser mutation waits for its correlated result and does not reapply", () => {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -37,7 +37,7 @@ test.skipIf(!enabled)(
expect(summary.mode).toBe("batched-construction");
expect(summary.dirtyCompilation).toContain("definitelyNotDefined");
expect(summary.cleanCompilation).toBe(
"No errors or warnings found in net function code.",
"No errors or warnings found in net function code. Scenario and metric compilation is checked when creating an experiment.",
);
expect(summary.repairHash).toMatch(/^[a-f0-9]{64}$/u);
expect(summary.layoutHash).toMatch(/^[a-f0-9]{64}$/u);
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ import {
snapshotToUiMessages,
CLIENT_TOOL_RESULT_SIGNAL,
} from "@hashintel/brunch-agent-transport-aisdk";
import { BRUNCH_QUESTION_TOOL_NAME } from "@hashintel/brunch-agent/question-marker";
import { BRUNCH_QUESTION_TOOL_NAMES } from "@hashintel/brunch-agent/question-marker";

import {
agentOwnershipHeaders,
Expand Down Expand Up @@ -350,7 +350,7 @@ const tools = (name: string, input: Record<string, unknown>, id: string) =>
const project = (snapshot: FlueConversationSnapshot) =>
snapshotToUiMessages(snapshot, {
clientToolNames: new Set([READ_PETRINAUT_DOCS_TOOL_NAME]),
hiddenToolNames: new Set([BRUNCH_QUESTION_TOOL_NAME]),
hiddenToolNames: new Set(BRUNCH_QUESTION_TOOL_NAMES),
});
const status = async (operation: () => Promise<unknown>) => {
try {
Expand Down Expand Up @@ -431,11 +431,6 @@ try {
);
responses.push(
tools("ping", { note: "a4-early-ping" }, "a4-ping-early"),
tools(
BRUNCH_QUESTION_TOOL_NAME,
{ question: "Which synthetic record follows?" },
"a4-question",
),
tools(
READ_PETRINAUT_DOCS_TOOL_NAME,
{ doc: "ai-assistant" },
Expand Down Expand Up @@ -515,13 +510,7 @@ try {
.filter((part) => part.type === "dynamic-tool");
assert.deepEqual(
publicTools.map((part) => part.toolCallId),
[
"a4-ping-early",
"a4-question",
"a4-doc-early",
"a4-ping-middle",
"a4-doc-late",
],
["a4-ping-early", "a4-doc-early", "a4-ping-middle", "a4-doc-late"],
);
for (const suffix of ["early", "middle"]) {
const ping = publicTools.find(
Expand All @@ -531,15 +520,11 @@ try {
assert.deepEqual(ping.input, { note: `a4-${suffix}-ping` });
assert.deepEqual(ping.output, { ok: true, note: `a4-${suffix}-ping` });
}
const marker = publicTools.find(
(part) => part.toolCallId === "a4-question",
);
assert(marker?.state === "output-available");
assert.deepEqual(marker.output, { marked: true });
assert(
before.messages
!before.messages
.flatMap((message) => message.parts)
.some((part) => part.type === "data-brunch-question"),
"New responses must not create question markers",
);
const clientResults = clientToolHistoryFrom(before.messages).results;
assert.deepEqual(
Expand Down Expand Up @@ -702,15 +687,15 @@ try {
? compactions.some(
(event) =>
!event.isError &&
event.messagesBefore === 20 &&
event.messagesBefore === 18 &&
event.messagesAfter === 3,
)
: compactions.some(
(event) =>
!event.isError && event.messagesAfter < event.messagesBefore,
),
silentOverflow
? "Silent overflow must fold the known 20-message window to 3"
? "Silent overflow must fold the known 18-message window to 3"
: "Actual successful folding must reduce runtime context messages",
);
assert(
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,7 @@ export interface PetrinautChatResult {
readonly resumedStatus: number;
readonly resumedText: string;
readonly resumedFinish: UIMessageChunk | undefined;
readonly questionResponseProviderCalls: number;
readonly questionMarkerLive: unknown;
readonly questionMarkerHistory: unknown;
readonly questionToolVisibleLive: boolean;
Expand Down
35 changes: 19 additions & 16 deletions apps/brunch-agent/test/integration/petrinaut-chat.integration.ts
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@ import {
fauxText,
fauxThinking,
fauxToolCall,
type Provider,
} from "@earendil-works/pi-ai";
import { createFlueClient, FlueApiError } from "@flue/sdk";

Expand All @@ -21,7 +22,7 @@ import {
import { ELICITATION_SKILL_NAME } from "@hashintel/brunch-agent/flue";
import {
BRUNCH_QUESTION_DATA_NAME,
BRUNCH_QUESTION_TOOL_NAME,
BRUNCH_QUESTION_TOOL_NAMES,
} from "@hashintel/brunch-agent/question-marker";

import { PING_TOOL_NAME } from "../../src/agents/chat-agent/tools/ping.ts";
Expand Down Expand Up @@ -113,13 +114,22 @@ const questionToolVisibleInHistory = (
): boolean =>
messages
.flatMap((message) => message.parts)
.some((part) => part.type === `tool-${BRUNCH_QUESTION_TOOL_NAME}`);
.some((part) =>
BRUNCH_QUESTION_TOOL_NAMES.some((name) => part.type === `tool-${name}`),
);

const faux = fauxProvider({
provider: "anthropic",
models: [{ id: CHAT_MODEL_ID, reasoning: true }],
});
installFauxProvider(faux.provider);
let providerCallCount = 0;
installFauxProvider({
...faux.provider,
streamSimple(model, context, options) {
providerCallCount += 1;
return faux.provider.streamSimple(model, context, options);
},
} satisfies Provider);
const application = await loadBuiltBrunchApplication();

try {
Expand All @@ -134,14 +144,14 @@ try {
const panelTransport = createFlueChatTransport({
client: historyClient,
clientToolNames,
hiddenToolNames: new Set([BRUNCH_QUESTION_TOOL_NAME]),
hiddenToolNames: new Set(BRUNCH_QUESTION_TOOL_NAMES),
});
const projectHistory = (
snapshot: Awaited<ReturnType<typeof historyClient.history>>,
) =>
snapshotToUiMessages(snapshot, {
clientToolNames,
hiddenToolNames: new Set([BRUNCH_QUESTION_TOOL_NAME]),
hiddenToolNames: new Set(BRUNCH_QUESTION_TOOL_NAMES),
});

if (process.env.BRUNCH_RESUME_PHASE === "1") {
Expand Down Expand Up @@ -237,16 +247,6 @@ try {
],
{ stopReason: "toolUse" },
),
fauxAssistantMessage(
[
fauxToolCall(
BRUNCH_QUESTION_TOOL_NAME,
{ question },
{ id: "tool-question-1" },
),
],
{ stopReason: "toolUse" },
),
fauxAssistantMessage([
fauxText(
`The guide says the assistant can read its own documentation pages. ${question}`,
Expand Down Expand Up @@ -343,6 +343,7 @@ try {
],
},
] as UIMessage[];
const questionResponseCallStart = providerCallCount;
const resumedChunks = await chunksFrom(
await panelTransport.sendMessages({
trigger: "submit-message",
Expand Down Expand Up @@ -451,12 +452,14 @@ try {
.map((chunk) => chunk.delta)
.join(""),
resumedFinish: resumedChunks.at(-1),
questionResponseProviderCalls:
providerCallCount - questionResponseCallStart,
questionMarkerLive: questionMarkerFromChunks(resumedChunks),
questionMarkerHistory: questionMarkerFromHistory(historyMessages),
questionToolVisibleLive: resumedChunks.some(
(chunk) =>
chunk.type === "tool-input-available" &&
chunk.toolName === BRUNCH_QUESTION_TOOL_NAME,
BRUNCH_QUESTION_TOOL_NAMES.some((name) => name === chunk.toolName),
),
questionToolVisibleHistory: questionToolVisibleInHistory(historyMessages),
historyUserEntryCount: userEntryIds.length,
Expand Down
Loading
Loading