Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion apps/brunch-agent/src/agents/chat-agent/agent.ts
Original file line number Diff line number Diff line change
Expand Up @@ -40,7 +40,9 @@ export function ChatAgent() {
) {
useInstruction(`Voice response presentation for this delivery only:
Write the complete canonical on-screen response normally, with the same content and detail you would provide for typed delivery. Do not shorten or reshape it for speech: Realtime rephrases the completed response later.
Present any marked question in its exact wording so its authoritative text remains available for exact delivery.
Before presenting a direct question for the person to answer, call brunch_mark_question with its exact text; then present the marked question in its exact wording in ordinary assistant prose.
A source-attributed or repeated question still needs a marker when you ask the person to answer it. Do not mark quoted questions you are only discussing, rhetorical questions, or headings.
The marker supplies question_text for exact Voice delivery. An unmarked question may be omitted from the spoken rephrasing.
These are presentation instructions only. Retain all domain, evidence, workpiece, and tool obligations.`);
}

Expand Down
32 changes: 28 additions & 4 deletions apps/brunch-agent/test/voice-context.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@ test("ChatAgent scopes its fixed Voice instructions to the current delivery", as
models: [{ id: CHAT_MODEL_ID }],
});
provider.setResponses(
Array.from({ length: 5 }, () => (context) => {
Array.from({ length: 6 }, () => (context) => {
prompts.push(context.systemPrompt ?? "");
return fauxAssistantMessage([fauxText("Canonical answer.")]);
}),
Expand Down Expand Up @@ -49,6 +49,15 @@ test("ChatAgent scopes its fixed Voice instructions to the current delivery", as
},
})
.then((receipt) => handle.read(receipt));
await handle
.dispatch({
message: {
kind: "user",
body: "Explain the source's unresolved question.",
context: { responseMode: "voice" },
},
})
.then((receipt) => handle.read(receipt));
await handle
.dispatch({ message: { kind: "user", body: "Typed again." } })
.then((receipt) => handle.read(receipt));
Expand All @@ -61,8 +70,8 @@ test("ChatAgent scopes its fixed Voice instructions to the current delivery", as
},
})
.then((receipt) => handle.read(receipt));
expect(prompts).toHaveLength(5);
expect(prompts[0]).not.toContain("Voice response style");
expect(prompts).toHaveLength(6);
expect(prompts[0]).not.toContain("Voice response presentation");
expect(prompts[1]).toContain("Voice response presentation");
expect(prompts[1]).toContain(
"complete canonical on-screen response normally",
Expand All @@ -75,8 +84,23 @@ test("ChatAgent scopes its fixed Voice instructions to the current delivery", as
expect(prompts[1]).not.toContain("one or two spoken sentences");
expect(prompts[1]).not.toContain("offers to read long responses");
expect(prompts[2]).toBe(prompts[1]);
expect(prompts[3]).toBe(prompts[0]);
expect(prompts[3]).toBe(prompts[1]);
expect(prompts[4]).toBe(prompts[0]);
expect(prompts[5]).toBe(prompts[0]);
// These assertions pin producer instructions, not faux-provider compliance.
const markerInstruction =
"Before presenting a direct question for the person to answer, call brunch_mark_question with its exact text";
expect(prompts[0]).not.toContain(markerInstruction);
expect(prompts[1]).toContain(markerInstruction);
expect(prompts[1]).toContain(
"A source-attributed or repeated question still needs a marker when you ask the person to answer it",
);
expect(prompts[1]).toContain(
"Do not mark quoted questions you are only discussing, rhetorical questions, or headings",
);
expect(prompts[1]).toContain(
"An unmarked question may be omitted from the spoken rephrasing",
);
expect(prompts.join("\n")).not.toContain("UNTRUSTED_CONTEXT");
} finally {
await runtime.stop();
Expand Down
4 changes: 3 additions & 1 deletion apps/petrinaut-website/package.json
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,8 @@
"lint:eslint": "oxlint --type-aware --report-unused-disable-directives-severity=error .",
"lint:tsc": "tsgo --noEmit && tsgo --noEmit --project api/tsconfig.json",
"preview": "vite preview",
"test:unit": "vitest run --passWithNoTests"
"test:unit": "vitest run --passWithNoTests",
"voice:e2e": "node --experimental-strip-types scripts/voice-e2e/run.ts"
},
"dependencies": {
"@ai-sdk/openai": "3.0.63",
Expand Down Expand Up @@ -52,6 +53,7 @@
"oxc-transform-react": "0.145.0",
"oxlint": "1.63.0",
"oxlint-tsgolint": "0.22.1",
"playwright": "1.58.2",
"sharp": "0.35.3",
"vite": "8.2.2",
"vitest": "4.1.10"
Expand Down
146 changes: 146 additions & 0 deletions apps/petrinaut-website/scripts/voice-e2e/README.md

Large diffs are not rendered by default.

174 changes: 174 additions & 0 deletions apps/petrinaut-website/scripts/voice-e2e/record-remote-audio.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,174 @@
// Injected before the application. Observe native events without replacing its
// listeners, ontrack property, provider messages, or microphone stream.
(() => {
/** @type {Blob[]} */
const chunks = [];
/** @type {import('./trace-checks.ts').InputCommit[]} */
const commits = [];
/** @type {import('./trace-checks.ts').LatencyMark[]} */
const latency = [];
/** @type {MediaRecorder | undefined} */
let recorder;
/** @type {number | undefined} */
let startedAt;
/** @type {number | undefined} */
let finishedAt;
/** @type {number | undefined} */
let microphoneRequestedAt;
/** @type {Promise<void> | undefined} */
let stopped;
/** @type {Promise<string> | undefined} */
let recording;
/** @type {string | undefined} */
let error;
const observedChannels = new WeakSet();

const originalMeasure = Performance.prototype.measure;
Performance.prototype.measure = function observeMeasure(...args) {
const measure = originalMeasure.apply(this, args);
if (measure.name.startsWith("voice-interview:")) {
const detail = /** @type {unknown} */ (measure.detail);
if (
typeof detail === "object" &&
detail !== null &&
"correlationId" in detail &&
typeof detail.correlationId === "string"
) {
latency.push({
name: measure.name.slice("voice-interview:".length),
elapsedMs: measure.duration,
correlationId: detail.correlationId,
observedAtMs: performance.now(),
});
}
}
return measure;
};

const originalGetUserMedia = navigator.mediaDevices.getUserMedia;
navigator.mediaDevices.getUserMedia = function observeMicrophone(...args) {
microphoneRequestedAt ??= performance.now();
return originalGetUserMedia.apply(this, args);
};

/** @param {RTCDataChannel} channel */
const observeChannel = (channel) => {
if (observedChannels.has(channel)) return;
observedChannels.add(channel);
channel.addEventListener("message", (event) => {
if (typeof event.data !== "string") return;
try {
const message = /** @type {unknown} */ (JSON.parse(event.data));
// Only the approved commit identity leaves this listener. Never retain
// provider payloads, SDP, transcripts, usage blobs, or audio messages.
if (
typeof message === "object" &&
message !== null &&
"type" in message &&
message.type === "input_audio_buffer.committed" &&
"item_id" in message &&
typeof message.item_id === "string"
) {
const itemId = message.item_id;
if (!commits.some((commit) => commit.itemId === itemId))
commits.push({ itemId, observedAtMs: performance.now() });
}
} catch {
// Ignore unrelated/non-JSON messages without exposing their contents.
}
});
};

const NativePeerConnection = window.RTCPeerConnection;
window.RTCPeerConnection = new Proxy(NativePeerConnection, {
construct(target, args, newTarget) {
const peer = /** @type {RTCPeerConnection} */ (
Reflect.construct(target, args, newTarget)
);
const createDataChannel = peer.createDataChannel;
peer.createDataChannel = function observeDataChannel(...parameters) {
const channel = createDataChannel.apply(this, parameters);
observeChannel(channel);
return channel;
};
peer.addEventListener("datachannel", (event) =>
observeChannel(event.channel),
);
peer.addEventListener("track", (event) => {
if (event.track.kind !== "audio" || recorder) return;
try {
// A track event can have no streams. Record the track itself.
const capture = new MediaRecorder(new MediaStream([event.track]), {
mimeType: "audio/webm;codecs=opus",
});
recorder = capture;
capture.addEventListener("dataavailable", (data) => {
if (data.data.size > 0) chunks.push(data.data);
});
stopped = new Promise((resolveStopped) => {
capture.addEventListener(
"stop",
() => {
finishedAt = performance.now();
resolveStopped();
},
{ once: true },
);
capture.addEventListener(
"error",
() => {
error = "remote-audio-recorder-failed";
finishedAt = performance.now();
resolveStopped();
},
{ once: true },
);
});
capture.start(250);
startedAt = performance.now();
} catch {
error = "remote-audio-recorder-unavailable";
}
});
return peer;
},
});

/** @type {Window & { __voiceE2E?: unknown }} */ (window).__voiceE2E = {
commits,
latency,
get error() {
return error;
},
get microphoneRequestedAt() {
return microphoneRequestedAt;
},
get recordedMs() {
return startedAt === undefined
? 0
: (finishedAt ?? performance.now()) - startedAt;
},
stopRecording: () => {
recording ??= (async () => {
if (!recorder) return "";
if (recorder.state !== "inactive") recorder.stop();
await stopped;
const blob = new Blob(chunks, { type: "audio/webm;codecs=opus" });
/** @type {Promise<string>} */
const encoded = new Promise((resolveBase64, reject) => {
const reader = new FileReader();
reader.onload = () =>
resolveBase64(
typeof reader.result === "string"
? (reader.result.split(",")[1] ?? "")
: "",
);
reader.onerror = () => reject(new Error("recording-read-failed"));
reader.readAsDataURL(blob);
});
return encoded;
})();
return recording;
},
};
})();
Loading
Loading