From c0dcbdd58748c7076a25dd33b9b7761c6a2efd5e Mon Sep 17 00:00:00 2001 From: Kostandin Angjellari Date: Tue, 15 Sep 2026 15:41:42 +0200 Subject: [PATCH 01/14] Authorize FE-1722 Voice controls Co-authored-by: Cursor --- apps/petrinaut-website/MISSION.md | 49 +- libs/@hashintel/brunch-agent/MISSION.md | 618 ++++++------------- libs/@hashintel/brunch-agent/MISSION.next.md | 24 +- 3 files changed, 223 insertions(+), 468 deletions(-) diff --git a/apps/petrinaut-website/MISSION.md b/apps/petrinaut-website/MISSION.md index 4478f01f342..591a81d6f58 100644 --- a/apps/petrinaut-website/MISSION.md +++ b/apps/petrinaut-website/MISSION.md @@ -1,29 +1,30 @@ -# GPT-Live capture mitigation +# Improve Brunch Voice controls The child branch's sole execution authority is the -[Brunch mission](../../libs/@hashintel/brunch-agent/MISSION.md). -This file is a pointer, not a second mission. +[Brunch mission](../../libs/@hashintel/brunch-agent/MISSION.md). This file is a +pointer, not a second mission. -FE-1712 permits explicit browser capture preferences, semantic VAD with medium -eagerness on the separate Live transcription session, and provider-free checks. -It also permits a 500 ms Speaking-indicator hold and a patient-listening Live -instruction. Submission timing, separate finalized transcription and Realtime stay -unchanged. The indicator does not control playback or establish turn completion. -Acoustic benefit remains Kostandin's matched speaker/headphone witness; no -deterministic feedback prevention or migration-readiness claim is established. -The sole mission specifies bounded headless probe allocations and their results; -natural human turn boundaries still require the owner witness. No new publication authority. +FE-1722 selects the reduced Voice-control cut: one compact dock, direct +microphone mute, one secondary audio popover and the existing conversation +panel for output. Both Live and Realtime expose canonical Stop only while +Brunch is submitted or streaming; End remains separate Voice teardown. +Show/Hide conversation changes visibility only. -The publication base is restacked FE-1664 at -[6d188da42f](https://github.com/hashintel/hash/commit/6d188da42f86b2d6ef3d211ec55058685185e8c7). -Its integration contract and earlier standalone comparisons are retained in the -[future spine](../../libs/@hashintel/brunch-agent/MISSION.next.md#voice-feedback-follow-up). -The parent has removed its superseded `PR_DESCRIPTION.md` draft; its PR body on -GitHub is its authority. This child's local Git branch description mirrors its mission. +Live microphone mute gates the existing shared capture track without silencing +playback. Realtime preserves its current microphone gating. Both providers gain +session-local speaker mute and normalized volume, reset for every new session. +Read-full-response, repeat-question and interruption-by-speaking remain +Realtime-only. Speaker settings do not redefine Speaking, and a +provider-finalized partial transcript after mid-utterance mute is allowed. -Kostandin authorizes pushing this child and opening its draft PR against FE-1664. -The authorized conflict fix preserves the parent's consent and Thinking dock behavior -and refreshes this draft's proof record. No other issue/PR changes, agent -microphone access, merge or deployment are authorized. The sole provider exception -is the bounded synthetic transcription probe specified in the mission. -Delegation-driven invocation and transcript filtering remain deferred. +The branch is stacked on FE-1712 at +[`377be52823`](https://github.com/hashintel/hash/commit/377be52823a65fcb3100d7b09e2ca471c374001e). +Its capture preferences, semantic VAD, patient listening, 500 ms output hold and +unfinished owner-held obligations remain inherited and unchanged. Device +switching, voice and speed selection, helmet animation and persistence remain +deferred in the +[future spine](../../libs/@hashintel/brunch-agent/MISSION.next.md#voice-control-follow-up). + +No FE-1722 product implementation or verification is claimed by this authority +cut. No agent microphone/provider session, push, PR, merge, deployment, Linear +write or other tracker change is authorized. diff --git a/libs/@hashintel/brunch-agent/MISSION.md b/libs/@hashintel/brunch-agent/MISSION.md index 04648879d0f..9eda1c6eda5 100644 --- a/libs/@hashintel/brunch-agent/MISSION.md +++ b/libs/@hashintel/brunch-agent/MISSION.md @@ -1,466 +1,206 @@ -# Stabilize GPT-Live full-duplex voice feedback +# Improve Brunch Voice controls ## Status -Live capture and transcription turn-boundary mission for -[FE-1712](https://linear.app/hash/issue/FE-1712/stabilize-gpt-live-full-duplex-voice-feedback). -Publication base: restacked FE-1664 at -[6d188da42f](https://github.com/hashintel/hash/commit/6d188da42f86b2d6ef3d211ec55058685185e8c7), -not `origin/main`. The original comparison revision is -[3cf4ca6b1f](https://github.com/hashintel/hash/commit/3cf4ca6b1f75f78cb2e086463517c02affd6ce54); -the previous publication base was -[006cbced7f](https://github.com/hashintel/hash/commit/006cbced7f10263f8b5f3cc305ee1ca6b722b9ce). -Only this child's commits were rebased; FE-1664 and its PR are not modified by -this mission. The latest restack applied cleanly; earlier restacks only conflicted -in this mission and its website pointer. The parent's consent, Thinking dock and -separation of temporary status notices from durable Voice warnings are preserved -without expanding the capture-only cut. The inherited `PR_DESCRIPTION.md` draft -remains removed. -The separate authority commit is -[466034cfe1](https://github.com/hashintel/hash/commit/466034cfe138e4f1f7befdcde3d16cd9f6037905). -The capture-only implementation is prepared: its assertion failed before the -change, and 317 targeted tests, website typechecking, lint and build now pass. -The semantic-VAD recut is implemented locally and provider-free checks pass. -The corrected medium probe completed all three synthetic transcripts, with the -correction retained in one item and continuous silence verified. Earlier correction -verdicts remain invalid because those harnesses stopped sending after the last clip. -The accepted 500 ms Speaking-indicator hold and patient Live listening prompt are -implemented and provider-free checks pass. Next: Kostandin tries a fresh Live session -to judge hesitation, self-corrections and the indicator's feel. Human conversational -latency and the physical speaker/headphone witness remain owner-held; no additional -automatic tuning or provider run. Commit and push of this preparation are authorized. -Acoustic benefit, natural turn boundaries and mission acceptance remain unproved. -All three provider allocations are consumed; no further provider run. Publication -of the prepared work is authorized below. The child is restacked on the parent's -current head; the inherited root `PR_DESCRIPTION.md` that failed CI Markdown lint -and formatting is gone with the parent. Repository-wide format and Markdown lint -pass locally; the GitHub Lint workflow had not run since the parent conflict began. +Live execution authority for +[FE-1722](https://linear.app/hash/issue/FE-1722/improve-brunch-voice-controls). +This branch is stacked on +[`origin/kostandin/fe-1712-stabilize-gpt-live-full-duplex-voice-feedback` at +`377be52823`](https://github.com/hashintel/hash/commit/377be52823a65fcb3100d7b09e2ca471c374001e), +not `origin/main`. FE-1712's implementation and evidence are inherited from +that pinned parent; its unfinished speech, acoustic and recovery obligations +are not accepted or replaced here. + +No FE-1722 product implementation or product proof exists yet. The next +authorized move, after this separate authority commit, is the provider-free +implementation of the reduced Voice-control cut below. Kostandin owns the +subsequent browser witness. No microphone or provider session is authorized for +an agent. Parent publication permissions do not transfer to this child; no +push, PR, merge, deployment or tracker write is authorized by this mission. ## Imperative -Reduce the risk that assistant playback becomes fresh user input while preserving -genuine interruptions and existing Realtime support. First determine whether -requesting the browser processing already used by Realtime improves Live's capture. -This is a mitigation hypothesis, not deterministic feedback-loop prevention. -Also reduce premature single-word submissions reported by Kostandin: use semantic -turn detection on the separate transcription session rather than silence alone. -Reduce rapid Speaking/Thinking/Listening flicker and ask Live to allow hesitation -and self-correction without taking over the person's unfinished thought. +Make an active Brunch Voice session compact and predictable without changing +who owns capture, canonical work or playback. Keep microphone mute immediately +available, move secondary audio controls into one popover, expose the canonical +Stop action only while Brunch is working, and use the existing conversation +panel for visible output. ## Throughline -Consented Start → one microphone capture → Live and `gpt-4o-transcribe` WebRTC -sessions → committed/completed input ordering → existing composer/Flue admission -→ Brunch settlement → frozen commentary → native Live playback. Playback can still -re-enter capture; filtering canonical input alone would not prevent Live reacting. - -Protected source: FE-1664 at the pinned base above. Its complete integration, -canonical ownership, admission, delivery and recovery contracts remain inherited -behavior, not accepted proof. Permitted deltas are Live's `getUserMedia` preferences -and the separate transcription session's turn-detection configuration below, -plus the accepted indicator hold and listening-prompt recut below. -The prior mission and future obligations remain discoverable through -[the future spine](MISSION.next.md#voice-feedback-follow-up). - -Preserve the parent's newer consent and dock contracts: concise OpenAI voice and -transcription disclosure, permission checkbox, Start voice and Cancel; stationary -dock/viewport controls with consent above them; Voice setup before Start. Separate -transcription still incurs additional provider usage. During submitted/streaming -Brunch work, show Thinking only while Live is connected, not stopped and not playing -output. Connection/error states take precedence, and playback remains Speaking. -These are inherited local UI semantics, not progress speech, `session.thinking.append`, -new invocation or completion proof. Parent controller tests own the transitions; -parent desktop/mobile witnesses own the visual layout. Only the output-activity -hold changes; status precedence and layout remain unchanged. - -Cold-start paths in `apps/petrinaut-website/src/main/app/voice-interview/`: - -- `live-conversation.ts`: change `{ audio: true }` to explicit - `autoGainControl: true`, `echoCancellation: true`, `noiseSuppression: true`. - Retain one capture and all connection/cleanup behavior. These are preferences, - not required capabilities; add no unsupported-device refusal or fallback retry. -- `live-conversation.test.ts`: extend the existing “starts Live and transcription - WebRTC from one consented capture and connects only when both are usable” test - with the exact requested preferences. Its existing assertions own shared track - identity and dual readiness. Watch the new assertion fail before implementation. -- `openai-realtime-session.ts`: reference for the preferences; leave it unchanged. - No shared helper is warranted for this small literal. - -Turn-boundary recut in `apps/petrinaut-website/src/server/voice/`: - -- `openai-transcription-session.ts`: replace `server_vad` with - `{ type: "semantic_vad", eagerness: "medium" }` for `gpt-4o-transcribe` only. - No silence timer, transcript aggregation, admission change or fallback retry. -- `openai-transcription-session.test.ts`: update the existing exact outbound - session-body assertion first; observe failure on server VAD, then pass on the - selected semantic configuration. Preserve model, scoped credential and raw SDP. -- `openai-voice-policy.ts` and Realtime routes remain unchanged. PR #9619 already - used semantic VAD with medium eagerness; this comparison now matches that setting. - -Patient-listening recut, relative to the website's `src/`: - -- `main/app/voice-interview/live-conversation.ts`: increase the existing local - output-activity hold from 300 to 500 ms. Keep the 100 ms sampler, immediate - activity onset and teardown/recovery behavior. This is display telemetry only, - never a playback-completion signal or a submission delay. -- Extend its existing telemetry test first: 400 and 499 ms stay active; the next - sample at 500 ms clears activity. A later audio burst restarts the hold; Stop - during the hold mutes playback immediately and late samples cannot revive it. -- `server/voice/openai-live-session.ts`: keep sparse backchannels and add a short - instruction to listen through thinking pauses and self-corrections rather than - take over unfinished thoughts. Preserve Brunch authority and interruption policy. -- Run the Live transport, controller, session-creation, transcription, bridge and - Realtime regression suites plus website build/typecheck/lint. Review prompt - delivery in the existing request test; a phrase-inventory test is not a speech - oracle. No new files, mechanism, queue, gate, dependency or provider allocation. +After the existing consented Start path connects either Live or Realtime, +Petrinaut renders one compact Voice dock while the existing conversation panel +continues to show the transcript and canonical Brunch output: + +1. The dock keeps microphone mute directly available. In Live, mute toggles the + one shared capture track that already feeds Live and the separate + transcription session; it does not mute playback or create another capture. + In Realtime, it preserves the existing microphone-gating behavior. +2. One audio popover contains session-local speaker mute and normalized volume + for both providers. The existing read-full-response, repeat-question and + interruption-by-speaking controls remain Realtime-only in that popover. +3. While canonical status is exactly `submitted` or `streaming`, both providers + show Stop and invoke the existing `onStop` path. Live consequently retains + the established `recordStopRequested()` → + `LiveBrunchBridge.stopResponse()` behavior: stop the current Brunch response + while leaving Live and transcription media connected. +4. End remains the separate Voice-session teardown. It does not stop canonical + work. The existing session-collapse control is relabelled Show conversation + or Hide conversation and changes only conversation visibility. +5. Status keeps the precedence connection/error → Speaking → Thinking → + microphone-muted → Listening. Speaker mute and volume zero do not make + Speaking false. +6. Speaker mute and volume start from their ordinary unmuted/full-volume + defaults for every new Voice session and are never persisted. + +The protected source is FE-1712 at the pinned parent above. Its browser capture +preferences, semantic VAD, patient-listening instruction and 500 ms +output-activity hold are unchanged. The production destinations and permitted +deltas are: + +- `libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/`: + keep the dock and existing conversation panel as the visible surface; thread + the canonical busy state and `onStop` to the dock; relabel the visibility + action; and compose the common audio popover from existing design-system + primitives. +- `libs/@hashintel/petrinaut/src/react/voice-session/` and + `libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.ts`: + extend the host/session contract only enough to report and change + session-local speaker mute and normalized volume. +- `apps/petrinaut-website/src/main/app/voice-interview/`: adapt the existing + Live and Realtime sessions to that contract, preserving shared capture, + Realtime gating, output ownership, admission, canonical Stop and teardown. + +Stop on an unlisted semantic delta. A local helper is warranted only when both +providers actually share the same contract; do not add a second control +surface, media owner or settings store. ### Owner decisions -- **2026-09-14:** Kostandin approves the separate constraints-only implementation - mission following the FE-1712 planning handoff. This permits its local branch, - separate authority commit, minimal implementation and provider-free checks. - Delegation-driven invocation and filtering remain deferred. -- **2026-09-14:** Kostandin authorizes pushing this branch and opening its draft PR - against FE-1664. This supersedes only the local-only publication restriction; - existing issues/PRs, parent branches, agent microphone/provider sessions, merge - and deployment remain outside scope. -- **2026-09-14:** Kostandin authorizes fixing this child's parent conflict by - rebasing, reconciling the mission, rerunning checks and pushing with an explicit - lease. Refresh this draft PR's proof record; leave other issues/PRs unchanged. -- **2026-09-14:** Kostandin accepts switching Live's separate transcription session - to semantic VAD after reporting single-word submissions. Use the discussed low - eagerness, keep Realtime unchanged, commit this authority separately and verify - locally without microphone/provider sessions. No new push or tracker write. -- **2026-09-14:** Kostandin authorizes one real `gpt-4o-transcribe` session with - at most three minutes of synthetic audio, no microphone access and no retries, - to test hesitation, short replies and correction retention headlessly. -- **2026-09-14:** After the low run, Kostandin accepts testing medium eagerness: - change only that transcription setting and its exact request assertion, then - repeat one session under the same 180-second/audio limit, without retries, - microphone access, other providers, Brunch inference or publication. -- **2026-09-14:** Kostandin authorizes one corrected medium session under the same - three-minute cap, with continuous synthetic silence, no microphone and no retries. -- **2026-09-14:** Kostandin authorizes pushing the prepared semantic-VAD change and - removing Amp thread-ID trailers from this child's commit messages. Preserve - authorship and parent commits; no merge, deployment or new provider allocation. -- **2026-09-14:** Kostandin accepts the patient-listening recut above with a 500 ms - indicator hold, not 800 ms, plus the prompt change. Test locally, commit without - Amp thread IDs, push and refresh this draft's proof. No restack, changed submission - timing, Realtime change, new provider run or other tracker write. -- **2026-09-14:** After CI Markdown lint and formatting failed on the inherited - `PR_DESCRIPTION.md`, Kostandin authorizes the cleanup, removal of Amp thread IDs - from this child's commits, and restacking onto the parent's current head with an - explicit lease, reconciling this mission and refreshing the draft PR. No Linear, - parent-branch, merge, deployment or microphone/provider change. +- **2026-09-15:** Kostandin accepts reduced Option B: compact dock, secondary + audio controls in one popover and the existing conversation panel for output. + The authority commit must remain separate from product code. ## Proof -### Patient-listening recut — provider-free proof passed, human witness pending - -The existing `live-conversation.test.ts` telemetry case owns the 500 ms boundary, -renewed activity and immediate Stop/late-sample behavior. Existing -`live-conversation-control.test.tsx` cases own Speaking/Thinking/Listening and -connection/error precedence. These tests do not establish conversational patience. -Verified 2026-09-14: the new 400 ms assertion failed on the old 300 ms hold, then -passed with 500 ms. The test also checks 499/500 ms, renewed activity, immediate -Stop at 499 ms during a pending stats read, and no late-state revival. All 386 tests -in 11 targeted Live/Realtime suites passed under OS network denial; after tightening -the Stop timing, all 40 transport tests passed again. A temporary jsdom render of -the real `VoiceDock` verified its accessible region and Speaking → Thinking → -Listening text transitions; the temporary probe was removed. No layout changed. -Website build, typecheck and lint passed all 16 Turbo tasks (10 cached); changed-file -formatting and whitespace checks passed. The full website suite was not rerun. -The Live session-creation test owns outgoing instruction carriage and unchanged -provider configuration; the prompt was inspected against OpenAI's -[pause-handling guidance](https://developers.openai.com/api/docs/guides/live-prompting). -Provider-free checks may establish timing and configuration only. Kostandin's next -fresh Live session remains the oracle for natural hesitation, short complete replies, -corrections, sparse acknowledgments and stopping speech when interrupted. Stop and -reorient if the indicator lingers misleadingly or the prompt worsens interruption -handling. The previous three synthetic transcription probes do not evaluate this -Live prompt, and their allocations remain consumed. - -### Corrected medium probe — synthetic retention verified, human latency pending - -One `gpt-4o-transcribe` session, semantic VAD / medium, at most 180 seconds of -synthetic audio and closed within 180 seconds after connection. No retries, -alternate model/setting, microphone, GPT-Live or Brunch inference. Use the same -actual endpoint, fixture bytes (compare hashes to the medium run), pause schedule -and 15-second final wait. Keep a zero-valued `ConstantSourceNode` connected and -active until teardown; inspect increasing RTP packet count and sample duration -through the final wait. No forced commits. Retain exact transcripts, timings, -session identity, usage and verified cleanup in the local-only native record at -`/tmp/fe1712-semantic-vad-medium-silence-T-01a09fe5/`. Remove temporary harness and -audio after inspection. Stop after this allocation for owner review. This can -adjudicate the three synthetic transcription cases, not human speech or echo. - -Observed 2026-09-14: session `sess_EO2X7U9GGGko3GvAOeQFU` confirmed semantic VAD -with medium eagerness. Input WAV hashes matched the inspected medium fixtures. -Exactly three committed items and three completed transcripts arrived in the -same predecessor order, with no extra inputs or errors: - -- “The inventory should contain twelve items, not twenty.” — 6.04 seconds after - speech ended; the one-second mid-sentence pause did not split the input. -- “Yes” — 5.42 seconds after speech ended, before the next case. -- “Set it to twenty. Actually, twelve.” — one item, 1.29 seconds after the - correction ended, retaining both values in order. - -All eight final-silence samples showed increasing packet count and source duration: -1698 → 2428 packets and 35.13 → 49.48 seconds. The corrected sender did not stall. -An independent read of `events.jsonl` asserted exact transcripts, item order, one -allocation, no provider errors, and closed peer/context/track with zero microphone -calls. The session lasted about 48 seconds; 5.49 seconds were synthesized speech. -Reported completed-item usage was 102 audio-input and 26 output tokens (128 total); -invoice cost is not established. `events.jsonl` and `attempt.json` in the directory -above are the local-only evidence. No product change, retry, push or deployment -was made during this probe. Keep medium locally for owner review; the 5–6 second -wait on the first two cases remains a usability concern, not an accepted latency. - -### Medium-eagerness comparison — completed, correction oracle invalid - -Reuse the actual panel endpoint and the low run's five locally generated clips, -one-second internal pauses, 12-second inter-case gaps and 15-second final wait. -Inspect fixture contents before dispatch. Record outbound RTP and audio-source -stats through the final silence to distinguish unfinished provider output from -a stopped synthetic sender. Do not force a commit or manufacture a final event. -The only new paid allocation is one `gpt-4o-transcribe` session, closed within -180 seconds after connection, with at most 180 seconds of synthetic input. -No retry or alternate setting in that session. Store safe native records under -`/tmp/fe1712-semantic-vad-medium-T-01a09fe5/`; remove temporary harness/audio after -inspection. The same boundary, retention, latency and cleanup oracles below apply. -Provider-free proof: the exact request-body assertion must fail on low and pass -on medium; rerun the five targeted suites, website typecheck and lint. -This single synthetic comparison cannot establish human speech or echo behavior. - -Observed 2026-09-14: session `sess_EO2SYiIYt07MDK84kntaf` confirmed semantic VAD -with medium eagerness. The inventory sentence stayed together and completed -5.52 seconds after its scheduled end (low: 9.50); “Yes.” completed in 4.44 seconds -(low: 6.48). The correction sequence started one item but never finalized. -RTP evidence explains why that last case cannot adjudicate VAD: after the final -clip, outbound packets remained at 1645 and source duration at 32.49 seconds -throughout the 15-second wait. The audio context still ran, but sent no silence. -The earlier cases had increasing packet counts during their pauses, so their -latencies remain observations, not controlled proof of improvement across runs. - -A local-only RTC pair reproduced the instrument defect and checked its repair: -after a completed clip, the old sender emitted zero additional packets over -three seconds; an active `ConstantSourceNode` with offset zero emitted 150 packets -and 3.01 additional audio seconds. No provider was called for this contrast. -Any future paid probe must retain that zero-valued source through the final wait -and inspect increasing outbound sample duration before judging finalization. -This repairs only the synthetic instrument, not product microphone behavior. - -Native `events.jsonl`, `attempt.json` and `local-silence-check.jsonl` in the medium -directory above retain the inspected evidence. One session, roughly 48 seconds, -5.49 seconds of synthetic speech; zero microphone calls and verified teardown. -Completed items reported 62 audio-input plus 16 output tokens (78 total); -unfinalized-item usage and invoice remain unknown. No retry or publication. -The medium request assertion failed on low then passed; all 127 targeted tests, -changed-file formatting and 15 website typecheck/lint tasks pass (10 cached). -Product code remains medium, unaccepted for full conversational quality. - -### Bounded headless transcription probe — completed, acceptance not established - -Use the actual `createOpenAITranscriptionSessionHandler` with the website's Vite -development environment loader (process values win). Drive its raw-SDP WebRTC -boundary from installed headless Chromium with a Web Audio synthetic track, not -`getUserMedia`. Generate only the three fixed witness phrases locally with macOS -speech synthesis; no TTS provider, Live session, Brunch inference or private data. -The only paid allocation is one `gpt-4o-transcribe` session, at most 180 seconds -of synthetic input, closed within 180 seconds after connection. No retries, -alternate models, second allocation or provider fallback on rejection/timeout. - -Inspect effective session configuration, provider item boundaries, exact completed -transcripts and timing relative to the scheduled one-second intra-phrase pauses. -Hesitation should remain one item; “Yes” must finalize without waiting for another -utterance; correction must retain both twenty and twelve in order. Report latency -rather than claiming a universal acceptable threshold. This is a synthetic -transcription-boundary probe, not proof of Brunch admission or physical echo. -Keep request/session IDs, returned usage and safe events in one local-only native -record under `/tmp/fe1712-semantic-vad-T-01a09fe5/`; unknown billing is not zero. -Stop after the single run or first rejection and return its evidence. Remove -temporary harness/audio after inspection; retain the safe native result. - -Observed 2026-09-14 via the already-running panel at `localhost:4915`, whose process -cwd is this checkout's website and whose API loader imports the current handler: -OpenAI session `sess_EO2KwbqCN3NRqcsrfVQ80` confirmed `gpt-4o-transcribe` and -`semantic_vad` / `low`. Safe native records are `events.jsonl` and `attempt.json` -in the local-only directory above. One attempt, roughly 48 seconds connected, -5.49 seconds of synthesized speech plus silence, zero microphone calls; the peer, -audio context and track all closed. No retry, GPT-Live or Brunch inference. - -- Hesitation: the one-second pause after “The inventory” stayed in one exact - completed sentence: “The inventory should contain twelve items, not twenty.” - Completion arrived 9.50 seconds after the scheduled end of that sentence. -- Short reply: “Yes.” finalized before the next input, but 6.48 seconds after - its scheduled end. Promptness is not established. -- Correction: “Set it to twenty.” finalized separately (1.34 seconds after its - end); the provider began another item for “Actually, twelve” but never emitted - its stop/commit/completion during the remaining 15 seconds. Both correction - words were verified in the exact input fixture after its internal pause. - **Correction oracle invalid:** the medium probe and local sender contrast above - exposed a shared harness defect: after the last clip it stops sending silence. - Retract the earlier retention/finalization-failure interpretation; this case - cannot establish provider loss or behavior with a real microphone. Brunch was - not invoked. No low-run RTP trace exists to adjudicate that session independently. - -Latencies use the browser's common monotonic clock for scheduled audio and event -receipt; they include provider/network delay, not just the VAD classifier. -Completed items reported 79 audio-input and 23 output tokens (102 total). -Unfinalized-item usage and invoice cost remain unknown, not zero. This evidence -motivates the separately accepted medium comparison above, not an acceptance claim. - -### Semantic turn-boundary recut — locally verified, owner witness pending - -Verified 2026-09-14: the exact outbound-body assertion failed on `server_vad` -before the change. After switching to semantic VAD with low eagerness, 127 tests -pass under OS network denial: `openai-transcription-session.test.ts`, -`openai-realtime-call.test.ts` and `openai-voice-policy.test.ts` under -`src/server/voice/`, plus `live-conversation.test.ts` and `live-brunch-bridge.test.ts` -under `src/main/app/voice-interview/`. Use the network-denied unit command below -with those five paths. These prove request configuration, unchanged Realtime policy -and existing failure/no-retry behavior, not provider acceptance or speech quality. -`turbo run lint:tsc lint:eslint --filter @apps/petrinaut-website ---output-logs=errors-only` passes all 15 tasks (10 cached). Changed TypeScript -formatting and `git diff --check` pass. Full website tests/build were not rerun -for this configuration-only recut. No provider session or UI change was made. - -Owner-held witness: in a fresh Live session, compare a hesitant phrase such as -“The inventory ... um ... purchase quantity is twelve, not twenty” against a -deliberately complete “Yes.” Check exact retained words, submission count/order, -and whether waiting feels excessive. Repeat with speakers and headphones. Do not -discard short legitimate answers to make the witness pass. Compatibility and -improved boundaries remain unproved until this actual product observation. - -### Provider-free configuration and regressions - -Baseline: `live-conversation.test.ts` passes 40 tests on the original comparison base under -OS network denial. The new capture assertion failed specifically because the old -call supplied `{ audio: true }`, then passed with the selected preferences. - -Run from the repository root with the pinned Node/Yarn toolchain: - -```sh -sandbox-exec -p '(version 1)(allow default)(deny network*)' yarn workspace @apps/petrinaut-website test:unit src/main/app/voice-interview/live-conversation.test.ts -``` - -Verified 2026-09-14 after conflict resolution: these seven files pass 317 tests: -`live-conversation.test.ts`, `live-brunch-bridge.test.ts`, -`live-conversation-control.test.tsx`, `openai-realtime-session.test.ts`, -`realtime-brunch-bridge.test.ts`, `voice-turn-controller.test.ts`, and -`voice-interview-control.test.tsx`, all under the cold-start directory above. -Existing late-permission, partial-failure and Stop tests retain media release and -stale-callback invalidation. These suites guard input admission, settlement, -interruption, consent, handoff and teardown; they do not establish acoustic correctness. -The full website suite was not rerun for this localized change. - -`yarn workspace @apps/petrinaut-website lint:tsc`, `lint:eslint` and `build` pass. -Lint reports zero warnings/errors. Build reports unchanged React Compiler -`try`/`finally` optimization and chunk-size warnings. Changed-file `oxfmt --check` -and `git diff --check` pass; Brunch Markdown is excluded by repository formatter -configuration and reviewed directly. No UI appearance or interaction controls change. - -Prior capture-only restack verification: `turbo run build lint:tsc lint:eslint --filter -@apps/petrinaut-website --output-logs=errors-only` passes all 16 tasks (9 cached). -The seven-suite run includes the parent's new consent and Thinking controller tests. -That production diff against the parent contained only the capture preferences; -consent and dock implementation are unchanged from that parent. - -### Manual speaker and headphone witness — pending, owner-held - -Allow about five minutes per output mode. Kostandin compares the pinned base and -this branch using the same browser, microphone, volume, prompt and disposable -document. Inspect effective capture settings using the browser's WebRTC diagnostics; -record unavailable settings as unknown, not confirmation. Do not start a second -capture just to inspect settings. Requested preferences may already be defaults. - -1. Connect both sessions and remain silent while a short no-tool Brunch answer - plays. Compare speakers and headphones; inspect completed input, canonical - admissions, delegation and commentary separately. Target zero unwanted admissions. -2. During playback give short novel replies and quantity/negation corrections; - hesitate, elaborate and deliberately quote the assistant. Check exact retained - words and admission order. Lost corrections invalidate apparent improvement. -3. Exit Voice and check silence/cleanup. If feedback persists, use headphones or - typed input. For the existing manual fallback, end Live, explicitly start - Realtime, disable “Interruption by speaking” and use “Your turn.” Never switch - or replay automatically. Unknown admission requires inspecting history first. - -This witness can support a limited mitigation claim, not an all-device guarantee, -native speech fidelity, tool-turn acceptance or migration readiness. Only the -separate bounded headless probe above grants an agent-run provider allocation. +### Authority cut + +Verified 2026-09-15: forced repository Markdown lint checked exactly this +mission, its future pointer and the website pointer with zero errors; +`git diff --check` also passed. These checks establish legible repository +authority only; they do not establish any product behavior. The pre-cut focused +baselines were reported as 51 Petrinaut tests and 218 website tests passing; +they contain no FE-1722 implementation. + +### Deterministic product proof — pending + +- `libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.test.tsx` + owns the compact dock, Show/Hide conversation as visibility only, canonical + Stop only for `submitted`/`streaming`, separate End, common audio controls and + provider-capability presentation. +- `libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.test.ts` + and + `libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx` + own the host contract and provider-optional action forwarding without making + controls mandatory for unrelated hosts. +- `apps/petrinaut-website/src/main/app/voice-interview/live-conversation.test.ts` + and `live-conversation-control.test.tsx` own shared-track microphone mute, + independent playback mute/volume, per-session reset and the unchanged + canonical Stop-to-`stopResponse()` path without media teardown. +- `apps/petrinaut-website/src/main/app/voice-interview/openai-realtime-session.test.ts`, + `voice-turn-controller.test.ts` and `voice-interview-control.test.tsx` own + unchanged Realtime microphone gating, common speaker settings, per-session + reset and Realtime-only controls. +- `apps/petrinaut-website/src/main/app/voice-interview/voice-session-state.test.ts` + and `live-conversation-control.test.tsx` own status precedence and prove that + speaker mute or volume zero does not rewrite Speaking. +- Run the affected Petrinaut and website suites, then each affected workspace's + typecheck, lint and build. Review user-facing Petrinaut Voice documentation + and add the required Petrinaut changeset with the implementation. These + checks can establish deterministic controls and regressions, not physical + audio, conversational quality or visual usability. + +### Product witness — pending, owner-held + +Kostandin starts a fresh Live session and a fresh Realtime session through the +real product door. In each, show and hide the conversation while speech and +canonical work continue; mute and unmute the microphone; mute the speaker and +move volume through zero during output; verify Speaking still reflects provider +output; and Stop one submitted or streaming Brunch response without ending +Voice. End Voice separately and confirm it does not stop canonical work. Start +a second session and confirm speaker mute and volume reset. In Realtime only, +also exercise read-full-response, repeat-question and +interruption-by-speaking. + +This witness may accept the interaction and audible effect on the tested +browser/device. It does not establish all-device media behavior, natural turn +boundaries, echo mitigation or FE-1712's remaining owner-held obligations. ## Constraints -- Preserve Realtime, default provider selection, consent, provider pinning, one - capture feeding both sessions, and teardown on failure/Stop. No extra capture, - session, dependency, telemetry store, retry or automatic fallback. -- Keep `gpt-4o-transcribe`, provider item ordering, no-delegation - admission, one waiting composer slot, frozen settled commentary and once-only - offering unchanged. No transcript suppression, fuzzy matching or new timers. -- Brunch remains canonical answer/tool authority. Live speech remains native and - best-effort; settlement gates supplied context, not every audible word. Append - acknowledgment is not consumption, speech, playback or execution completion. -- Preserve visible full text and no truncation/chunking/replay on commentary - rejection. Local Exit and canonical Stop remain distinct from acoustic interruption - and from canceling already-executed effects. -- No changes to parent branches, other issues/PRs, Brunch prompts, models, - services or infrastructure. Push and draft creation are authorized for this child - only; no merge, deployment or agent microphone access. The only provider exception - is the bounded transcription probe above. Prior FE-1664 - publication permissions do not transfer to this mission. -- Only the separate Live transcription session may switch VAD as specified above; - do not change Realtime. The only native Live behavior change is the accepted - patient-listening instruction; its effect is probabilistic, not enforced timing. +- Preserve explicit consent, one-capture ownership, teardown, transcript + admission, delegation policy, canonical Brunch authority, provider pinning + and Realtime response ownership. Do not start microphone or provider sessions + on an agent's behalf; real media evidence remains owner-held. +- Preserve FE-1712 browser capture preferences, semantic VAD, + patient-listening instruction and 500 ms output-activity hold. +- Live microphone mute disables the one shared capture track feeding Live and + transcription without muting playback, ending either session or changing + canonical work. Realtime keeps its existing gating semantics. +- Canonical Stop appears only for `submitted` or `streaming` and uses the + existing `onStop` path. Live Stop keeps media connected. End tears down Voice + and does not cancel canonical work. +- Connection/error, Speaking, Thinking, microphone-muted and Listening retain + that precedence. Audio settings describe local audibility, not provider + output activity. +- Show/Hide conversation changes visibility only. It must not change capture, + playback, work, session state, panel history or admission. +- Speaker mute and normalized volume are session-local for both providers and + reset for every new session. Do not persist them. +- A provider-finalized partial transcript admitted after mid-utterance mute is + allowed. Add no transcript suppression, fuzzy matching or timers. +- Existing read-full-response, repeat-question and interruption-by-speaking + behavior stays Realtime-only. ## Fog-line -The reported silent “No tengo.” admission demonstrates unwanted input, not whether -echo, background audio, routing or hallucination caused it. Browser defaults may -already apply these preferences; effective settings and the manual contrast decide -whether this change has acoustic value. A passing configuration test does not. - -OpenAI's [VAD guide](https://developers.openai.com/api/docs/guides/realtime-vad) -documents semantic VAD for supported transcription sessions and low eagerness for -larger chunks. Actual acceptance with this model/session is the headless probe's -first discriminator; natural human speech remains owner-witnessed. -Semantic VAD is probabilistic and may add latency; it does not guarantee a complete -thought or prevent echo. The earlier VAD rejection on a different model does not -establish incompatibility for `gpt-4o-transcribe`. - -Live-native transcripts and client delegation remain an alternative, not a selected -replacement. Transcript deltas lack authoritative finalization/item identity; -delegation has an opaque ID, target and offset, not task text. Separate transcription -finalizes audio items, not complete thoughts. Shared capture does not synchronize -the sessions' clocks. The cross-session range-selection oracle remains unresolved: -which immutable input range belongs to a delegation, and when is it complete? -Latest-item pairing, arrival order, silence timeouts or fuzzy matching cannot prove -it. This cut neither changes that policy nor claims to solve it. - -Mandatory half-duplex could enforce app playback/capture exclusion at the cost of -simultaneous listening. Live lacks Realtime's response-terminal/handoff lifecycle; -do not invent it from activity telemetry or append acknowledgments. +The exact compact spacing, icons, volume affordance and responsive fit remain +implementation details to validate against the existing Petrinaut design +system and accessibility semantics. They may not move microphone mute into the +popover, create another output surface or alter the control policy above. + +Muting a capture track cannot retract audio the provider already received; a +finalized partial transcript after mute is therefore neither automatically a +bug nor evidence of suppression. Speaker mute and zero volume change local +audibility, not whether output is active. Deterministic browser tests cannot +establish subjective volume feel, physical routing or whether the compact dock +is usable on Kostandin's device. + +FE-1712's acoustic benefit, natural turn-boundary quality, direct spoken-user +attribution, withheld-work recovery and comparative latency remain unresolved +at their existing parent or future-spine owners. This mission neither reruns +nor accepts them. ## Stop or reorient -Stop after the bounded headless probe for Kostandin's review. Do not add filtering -automatically. Return to the owner if preferences change neither settings nor -failure, headphones still produce silent admissions, or genuine corrections are -lost. Reclassify the observed failure before adding a mechanism. -For the turn-boundary recut, stop on provider rejection, continued fragmentation, -lost corrections or unacceptable delay. Preserve the existing visible connection -failure without silently reverting VAD; use typed input or explicitly ended Live -followed by Realtime. Return to the owner before selecting another setting. - -If deterministic prevention is required, select half-duplex/typed policy explicitly. -If delegation-driven invocation is required, resolve finalization/range selection -first. Any new queue, gate, prompt, model or handoff policy requires a new accepted -cut. Premature substantive speech, reordered/lost corrections, replay of uncertain -work or revived speech after Stop remain failures, not accepted side effects. +Stop and return to the owner if microphone mute silences output, creates a new +capture, changes admission, or fails to gate both Live consumers of the shared +track; if speaker controls alter microphone state or Speaking status; if Stop +tears down Voice or End stops canonical work; if Show/Hide changes anything +other than visibility; if settings survive a new session; or if Live gains +Realtime-only controls. + +Also stop on an inaccessible or unusable compact layout, a provider-specific +contract that cannot be represented without weakening the common invariants, +an unlisted persistence or timer, a new provider/media session, or a required +change to FE-1712's protected behavior. Do not select a broader redesign from +mechanism failure without a new owner decision. ## Deferred -[Voice feedback follow-up](MISSION.next.md#voice-feedback-follow-up) retains the -conditional filtering, native-delegation and half-duplex alternatives. The existing -[Voice recovery obligation](MISSION.next.md#voice-after-the-live-transport-cut), -parent integration witness and Mission 7c/7d obligations remain open under their -owners; this cut does not consume their waivers or acceptance. +[Voice control follow-up](MISSION.next.md#voice-control-follow-up) retains +device switching, voice and speed selection, helmet animation and settings +persistence. [Voice feedback follow-up](MISSION.next.md#voice-feedback-follow-up) +and [Voice after the live transport cut](MISSION.next.md#voice-after-the-live-transport-cut) +retain FE-1712's unfinished alternatives and owner-held obligations. None is +authorized by this cut. diff --git a/libs/@hashintel/brunch-agent/MISSION.next.md b/libs/@hashintel/brunch-agent/MISSION.next.md index 6a745eefb65..c89047ba6f9 100644 --- a/libs/@hashintel/brunch-agent/MISSION.next.md +++ b/libs/@hashintel/brunch-agent/MISSION.next.md @@ -2,16 +2,22 @@ > Future sequence and decision register only; not execution authority. [`MISSION.md`](MISSION.md) owns live scope and progress. Successor drafts become executable only after an owner-authorized cut; archives and git history retain prior contracts. -On this stacked voice branch, `MISSION.md` owns FE-1712. The inherited Mission 7c -map and its mission-section references below belong to the +On this stacked voice branch, `MISSION.md` owns FE-1722. Its pinned parent is +[FE-1712 at `377be52823`](https://github.com/hashintel/hash/blob/377be52823a65fcb3100d7b09e2ca471c374001e/libs/%40hashintel/brunch-agent/MISSION.md); +that parent's unfinished speech, acoustic and recovery obligations remain open +under their existing owners. The inherited Mission 7c map and its mission-section +references below belong to the [upstream contract](https://github.com/hashintel/hash/blob/dee90599e9a07d9fa3e55d0711c14491e9ce5c7c/libs/%40hashintel/brunch-agent/MISSION.md) on #9667, not to a second execution authority here. The stack does not close that mission or grant its paid-run permissions to voice work. ## Voice feedback follow-up -FE-1712 selects explicit capture preferences, semantic transcription turn detection -and owner-held speech/acoustic comparisons as specified in the live mission. The complete +FE-1712 selected explicit capture preferences, semantic transcription turn detection +and owner-held speech/acoustic comparisons as specified in its +[pinned mission](https://github.com/hashintel/hash/blob/377be52823a65fcb3100d7b09e2ca471c374001e/libs/%40hashintel/brunch-agent/MISSION.md). +FE-1722 preserves that behavior and those unfinished obligations while changing +only the controls admitted by its live mission. The complete [FE-1664 contract](https://github.com/hashintel/hash/blob/9499b9287bd69b751ebcdd61b0c6bf2586bc191e/libs/%40hashintel/brunch-agent/MISSION.md) is retained at the branch's pinned parent, not archived as accepted or replaced on that branch. Its input/delivery contracts, first no-tool exchange, later @@ -43,6 +49,13 @@ The broader alternatives, rejected shortcuts and discriminating test portfolio a planning context in [FE-1712](https://linear.app/hash/issue/FE-1712/stabilize-gpt-live-full-duplex-voice-feedback), not authority for these deferred changes. +## Voice control follow-up + +FE-1722's [live mission](MISSION.md) owns the current control cut. Device +switching, provider voice and speech-speed selection, helmet animation and +persistence of speaker settings remain future work and require a separate +owner-authorized cut. + ## How to use this spine Read this file to answer four questions: @@ -238,7 +251,8 @@ Immediate switching from a review or gap report into renewed elicitation remains ### Voice after the live transport cut -The FE-1664 integration is governed by this branch's [mission](MISSION.md). +The inherited FE-1664 integration is governed by the +[FE-1712 mission at this branch's pinned parent](https://github.com/hashintel/hash/blob/377be52823a65fcb3100d7b09e2ca471c374001e/libs/%40hashintel/brunch-agent/MISSION.md). Native Live delivery and canonical transcription do not waive the recovery obligation below or establish live provider compatibility. Historical waiver and attribution rationale remains in the From c0c9f25a0a0aeca133867497fa628e3b9239378e Mon Sep 17 00:00:00 2001 From: Kostandin Angjellari Date: Tue, 15 Sep 2026 15:51:37 +0200 Subject: [PATCH 02/14] Add shared Voice controls and audio options Co-authored-by: Cursor --- .../src/react/voice-session/store.ts | 2 + .../src/react/voice-session/types.ts | 4 + .../react/voice-session/use-voice-session.ts | 20 ++ .../ai-assistant-composer-control.test.ts | 22 ++ .../ui/types/ai-assistant-composer-control.ts | 4 + .../Editor/components/voice-session-labels.ts | 10 +- .../Editor/panels/ai-assistant-panel.test.tsx | 102 ++++--- .../Editor/panels/ai-assistant-panel.tsx | 12 + .../ai-assistant-contents.test.tsx | 280 +++++++++++++++--- .../ai-assistant-contents.tsx | 6 + .../ai-assistant-contents/voice-dock.tsx | 117 +++++--- .../voice-dock/audio-popover.tsx | 197 ++++++++++++ .../voice-dock/playback-menu.tsx | 61 ---- 13 files changed, 642 insertions(+), 195 deletions(-) create mode 100644 libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock/audio-popover.tsx delete mode 100644 libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock/playback-menu.tsx diff --git a/libs/@hashintel/petrinaut/src/react/voice-session/store.ts b/libs/@hashintel/petrinaut/src/react/voice-session/store.ts index 6bb3ad5230d..37ac1148765 100644 --- a/libs/@hashintel/petrinaut/src/react/voice-session/store.ts +++ b/libs/@hashintel/petrinaut/src/react/voice-session/store.ts @@ -14,6 +14,8 @@ export type VoiceSessionActions = { resume?: () => void; setInterruptionBySpeaking?: (enabled: boolean) => void; setMicrophoneMuted?: (muted: boolean) => void; + setSpeakerMuted?: (muted: boolean) => void; + setSpeakerVolume?: (volume: number) => void; takeTurn?: () => Promise | void; }; diff --git a/libs/@hashintel/petrinaut/src/react/voice-session/types.ts b/libs/@hashintel/petrinaut/src/react/voice-session/types.ts index be5e1b5fce5..a8f089f9217 100644 --- a/libs/@hashintel/petrinaut/src/react/voice-session/types.ts +++ b/libs/@hashintel/petrinaut/src/react/voice-session/types.ts @@ -35,6 +35,10 @@ export type PetrinautAiVoiceSessionState = { /** Temporary operational status shown in place of the current phase. */ notice?: string | null; phase: PetrinautAiVoiceSessionPhase; + /** Whether assistant audio is muted independently of its retained volume. */ + speakerMuted?: boolean; + /** Normalized 0–1 assistant audio volume. */ + speakerVolume?: number; /** Recoverable issue retained behind the Voice warning indicator. */ warningMessage?: string | null; }; diff --git a/libs/@hashintel/petrinaut/src/react/voice-session/use-voice-session.ts b/libs/@hashintel/petrinaut/src/react/voice-session/use-voice-session.ts index 3fa16ee72c7..2d1d9c1f7b9 100644 --- a/libs/@hashintel/petrinaut/src/react/voice-session/use-voice-session.ts +++ b/libs/@hashintel/petrinaut/src/react/voice-session/use-voice-session.ts @@ -43,6 +43,26 @@ export const useVoiceSessionMicrophoneMuted = (): boolean => { ); }; +export const useVoiceSessionSpeakerMuted = (): boolean => { + const store = use(VoiceSessionContext); + + return useSyncExternalStore( + store.subscribe, + () => store.getSnapshot().state?.speakerMuted ?? false, + () => false, + ); +}; + +export const useVoiceSessionSpeakerVolume = (): number => { + const store = use(VoiceSessionContext); + + return useSyncExternalStore( + store.subscribe, + () => store.getSnapshot().state?.speakerVolume ?? 1, + () => 1, + ); +}; + export const useVoiceSessionErrorMessage = (): string | null => { const store = use(VoiceSessionContext); diff --git a/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.test.ts b/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.test.ts index 37ec0af7bbc..84cd504b80f 100644 --- a/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.test.ts +++ b/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.test.ts @@ -5,6 +5,7 @@ import type { PetrinautAiVoiceModeControls, PetrinautAiVoiceModeSessionControls, } from "./ai-assistant-composer-control"; +import type { VoiceSessionActions } from "../../react/voice-session/store"; test("keeps legacy Voice controls required while sessions advertise capabilities", () => { expectTypeOf().toEqualTypeOf< @@ -43,3 +44,24 @@ test("accepts an existing context implementation with complete control registrat expectTypeOf(currentContext).toEqualTypeOf(); }); + +test("exposes optional provider-neutral speaker controls", () => { + expectTypeOf< + PetrinautAiVoiceModeControls["setSpeakerMuted"] + >().toEqualTypeOf<((muted: boolean) => void) | undefined>(); + expectTypeOf< + PetrinautAiVoiceModeControls["setSpeakerVolume"] + >().toEqualTypeOf<((volume: number) => void) | undefined>(); + expectTypeOf< + PetrinautAiVoiceModeSessionControls["setSpeakerMuted"] + >().toEqualTypeOf<((muted: boolean) => void) | undefined>(); + expectTypeOf< + PetrinautAiVoiceModeSessionControls["setSpeakerVolume"] + >().toEqualTypeOf<((volume: number) => void) | undefined>(); + expectTypeOf().toEqualTypeOf< + ((muted: boolean) => void) | undefined + >(); + expectTypeOf().toEqualTypeOf< + ((volume: number) => void) | undefined + >(); +}); diff --git a/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.ts b/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.ts index 1716a4ca3e2..aa621985f6e 100644 --- a/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.ts +++ b/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.ts @@ -79,6 +79,10 @@ export type PetrinautAiVoiceModeControls = { setMicrophoneMuted: (muted: boolean) => void; /** Allows speech to interrupt assistant playback without clearing input. */ setInterruptionBySpeaking?: (enabled: boolean) => void; + /** Mutes assistant audio without changing its retained volume. */ + setSpeakerMuted?: (muted: boolean) => void; + /** Sets normalized 0–1 assistant audio volume without changing mute state. */ + setSpeakerVolume?: (volume: number) => void; /** Cancels Voice output and hands the live microphone turn to the user. */ takeTurn?: () => Promise | void; }; diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/components/voice-session-labels.ts b/libs/@hashintel/petrinaut/src/ui/views/Editor/components/voice-session-labels.ts index b5b170792ca..984d5821d40 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/components/voice-session-labels.ts +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/components/voice-session-labels.ts @@ -28,20 +28,24 @@ export const voiceSessionStatusLabel = ( }; export const voiceSessionActionLabels = { - collapse: "Collapse voice session", + audioOptions: "Audio options", + collapse: "Hide conversation", end: "End voice mode", - expand: "Expand voice session", + expand: "Show conversation", interruptionBySpeaking: "Interruption by speaking", mute: "Mute microphone", + muteSpeaker: "Mute speaker", pause: "Pause voice mode", - playbackOptions: "Voice playback options", readFullResponse: "Read full response", reconnect: "Reconnect voice mode", repeatQuestion: "Repeat question", retryPlayback: "Play voice audio", resume: "Resume voice mode", + speakerVolume: "Speaker volume", + stop: "Stop AI response", takeTurn: "Your turn", unmute: "Unmute microphone", + unmuteSpeaker: "Unmute speaker", } as const; export const voiceSetupLabels = { diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx index 9198b5c6641..1fd711a13c3 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx @@ -2371,6 +2371,8 @@ describe("AiAssistantPanel composer submissions", () => { const takeTurn = vi.fn(); const repeatQuestion = vi.fn(); const readFullResponse = vi.fn(); + const setSpeakerMuted = vi.fn(); + const setSpeakerVolume = vi.fn(); const VoiceMode = ({ context, replayAllowed, @@ -2390,6 +2392,8 @@ describe("AiAssistantPanel composer submissions", () => { repeatQuestion, resume: vi.fn(), setMicrophoneMuted: vi.fn(), + setSpeakerMuted, + setSpeakerVolume, takeTurn, }), [registerVoiceModeControls], @@ -2403,6 +2407,8 @@ describe("AiAssistantPanel composer submissions", () => { microphoneLevel: 0, microphoneMuted: false, phase: "speaking", + speakerMuted: false, + speakerVolume: 0.25, }); return () => reportVoiceSessionState(null); }, [replayAllowed, reportVoiceSessionState]); @@ -2425,60 +2431,55 @@ describe("AiAssistantPanel composer submissions", () => { expect(takeTurn).toHaveBeenCalledOnce(); fireEvent.click( - screen.getByRole("button", { name: "Voice playback options" }), + screen.getByRole("button", { name: "Audio options" }), ); - const repeatQuestionItem = await screen.findByRole("menuitem", { + const repeatQuestionItem = await screen.findByRole("button", { name: "Repeat question", }); - expect(repeatQuestionItem.getAttribute("aria-disabled")).not.toBe("true"); - const repeatQuestionMenu = screen.getByRole("menu"); - fireEvent.keyDown(repeatQuestionMenu, { key: "ArrowDown" }); - await waitFor(() => - expect(repeatQuestionMenu.getAttribute("aria-activedescendant")).toBe( - repeatQuestionItem.id, - ), - ); - fireEvent.keyDown(repeatQuestionMenu, { key: "Enter" }); - await waitFor(() => expect(repeatQuestion).toHaveBeenCalledOnce()); + expect((repeatQuestionItem as HTMLButtonElement).disabled).toBe(false); + fireEvent.click(repeatQuestionItem); + expect(repeatQuestion).toHaveBeenCalledOnce(); - fireEvent.click( - screen.getByRole("button", { name: "Voice playback options" }), - ); - const readFullResponseItem = await screen.findByRole("menuitem", { + fireEvent.click(screen.getByRole("button", { name: "Mute speaker" })); + expect(setSpeakerMuted).toHaveBeenCalledWith(true); + const volume = screen.getByRole("slider", { name: "Speaker volume" }); + volume.focus(); + fireEvent.keyDown(volume, { key: "ArrowRight" }); + await waitFor(() => expect(setSpeakerVolume).toHaveBeenCalledWith(0.3)); + + const readFullResponseItem = screen.getByRole("button", { name: "Read full response", }); - expect(readFullResponseItem.getAttribute("aria-disabled")).not.toBe("true"); - const readFullResponseMenu = screen.getByRole("menu"); - fireEvent.keyDown(readFullResponseMenu, { key: "End" }); - await waitFor(() => - expect(readFullResponseMenu.getAttribute("aria-activedescendant")).toBe( - readFullResponseItem.id, - ), - ); - fireEvent.keyDown(readFullResponseMenu, { key: "Enter" }); - await waitFor(() => expect(readFullResponse).toHaveBeenCalledOnce()); + expect((readFullResponseItem as HTMLButtonElement).disabled).toBe(false); + fireEvent.click(readFullResponseItem); + expect(readFullResponse).toHaveBeenCalledOnce(); + fireEvent.keyDown(document.activeElement ?? document, { key: "Escape" }); rendered.rerenderPanel(aiAssistant(false), editorContextValue); fireEvent.click( - await screen.findByRole("button", { name: "Voice playback options" }), + await screen.findByRole("button", { name: "Audio options" }), ); expect( ( - await screen.findByRole("menuitem", { name: "Repeat question" }) - ).getAttribute("aria-disabled"), - ).toBe("true"); + await screen.findByRole("button", { name: "Repeat question" }) + ).hasAttribute("disabled"), + ).toBe(true); expect( - screen - .getByRole("menuitem", { name: "Read full response" }) - .getAttribute("aria-disabled"), - ).toBe("true"); + (screen.getByRole("button", { + name: "Read full response", + }) as HTMLButtonElement).disabled, + ).toBe(true); }); test("retires missing and unmounted optional host Voice actions", async () => { + const setSpeakerMuted = vi.fn(); + const setSpeakerVolume = vi.fn(); const VoiceMode = ({ context, + speakerControls, }: { context: PetrinautAiVoiceModeContext; + speakerControls: boolean; }) => { const { registerVoiceModeSessionControls, reportVoiceSessionState } = context; @@ -2488,8 +2489,9 @@ describe("AiAssistantPanel composer submissions", () => { return registerVoiceModeSessionControls({ end: async () => undefined, pause: vi.fn(), + ...(speakerControls ? { setSpeakerMuted, setSpeakerVolume } : {}), }); - }, [registerVoiceModeSessionControls]); + }, [registerVoiceModeSessionControls, speakerControls]); useEffect(() => { reportVoiceSessionState({ canReadFullResponse: true, @@ -2499,32 +2501,52 @@ describe("AiAssistantPanel composer submissions", () => { microphoneLevel: 0, microphoneMuted: false, phase: "speaking", + speakerMuted: false, + speakerVolume: 0.5, }); return () => reportVoiceSessionState(null); }, [reportVoiceSessionState]); return null; }; - const aiAssistant = (mounted: boolean): PetrinautAiAssistant => ({ + const aiAssistant = ( + mounted: boolean, + speakerControls: boolean, + ): PetrinautAiAssistant => ({ renderVoiceMode: (context) => - mounted ? : null, + mounted ? ( + + ) : null, transport: { reconnectToStream: () => Promise.resolve(null), sendMessages: vi.fn(), }, }); - const rendered = renderTestPanel({ aiAssistant: aiAssistant(true) }); + const rendered = renderTestPanel({ + aiAssistant: aiAssistant(true, true), + }); expect(screen.queryByRole("button", { name: "Your turn" })).toBeNull(); await screen.findByRole("region", { name: "Voice session" }); + fireEvent.click(screen.getByRole("button", { name: "Audio options" })); + fireEvent.click(await screen.findByRole("button", { name: "Mute speaker" })); + expect(setSpeakerMuted).toHaveBeenCalledWith(true); + fireEvent.keyDown(document.activeElement ?? document, { key: "Escape" }); + + rendered.rerenderPanel(aiAssistant(true, false), editorContextValue); + await waitFor(() => + expect( + screen.queryByRole("button", { name: "Audio options" }), + ).toBeNull(), + ); expect( - screen.queryByRole("button", { name: "Voice playback options" }), + screen.queryByRole("button", { name: "Audio options" }), ).toBeNull(); expect( screen.queryByRole("button", { name: "Mute microphone" }), ).toBeNull(); - rendered.rerenderPanel(aiAssistant(false), editorContextValue); + rendered.rerenderPanel(aiAssistant(false, false), editorContextValue); await waitFor(() => expect( diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.tsx index bc85f1fb6ac..ad17c65c424 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.tsx @@ -723,6 +723,18 @@ const ConversationAiAssistantPanel = ({ controls.setMicrophoneMuted?.(muted), } : {}), + ...(controls.setSpeakerMuted + ? { + setSpeakerMuted: (muted: boolean) => + controls.setSpeakerMuted?.(muted), + } + : {}), + ...(controls.setSpeakerVolume + ? { + setSpeakerVolume: (volume: number) => + controls.setSpeakerVolume?.(volume), + } + : {}), ...(controls.takeTurn ? { takeTurn: () => controls.takeTurn?.() } : {}), }); diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.test.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.test.tsx index 66176f457ef..cb80c5cc7b5 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.test.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.test.tsx @@ -81,32 +81,66 @@ const HostContent = ({ onMount }: { onMount: () => void }) => { return

Saved account

; }; -test("session-only dock shows Connected and End without unsupported controls", () => { +test("live-capability dock keeps microphone direct and Realtime controls absent", async () => { const end = vi.fn(); const collapse = vi.fn(); + const setMicrophoneMuted = vi.fn(); + const setSpeakerMuted = vi.fn(); + const setSpeakerVolume = vi.fn(); render( } microphoneMuted={false} + onStop={noop} onCollapsedToggle={collapse} phase="connected" + speakerMuted={false} + speakerVolume={1} />, ); expect(screen.getByText("Connected")).toBeTruthy(); - expect( - screen.queryByRole("button", { name: "Voice playback options" }), - ).toBeNull(); - expect(screen.queryByRole("button", { name: "Mute microphone" })).toBeNull(); + const microphone = screen.getByRole("button", { name: "Mute microphone" }); + expect(microphone).not.toBeNull(); expect(screen.queryByRole("button", { name: "Your turn" })).toBeNull(); fireEvent.click( - screen.getByRole("button", { name: "Collapse voice session" }), + screen.getByRole("button", { name: "Hide conversation" }), ); expect(collapse).toHaveBeenCalledOnce(); + fireEvent.click(microphone); + expect(setMicrophoneMuted).toHaveBeenCalledWith(true); + + fireEvent.click(screen.getByRole("button", { name: "Audio options" })); + expect(await screen.findByRole("button", { name: "Mute speaker" })).toBeTruthy(); + expect( + screen + .getByRole("slider", { name: "Speaker volume" }) + .getAttribute("aria-valuenow"), + ).toBe("1"); + expect( + screen.queryByRole("button", { name: "Repeat question" }), + ).toBeNull(); + expect( + screen.queryByRole("button", { name: "Read full response" }), + ).toBeNull(); + expect( + screen.queryByRole("button", { name: "Interruption by speaking" }), + ).toBeNull(); + expect( + microphone.closest('[data-scope="popover"][data-part="content"]'), + ).toBeNull(); + fireEvent.click(screen.getByRole("button", { name: "End voice mode" })); expect(end).toHaveBeenCalledOnce(); }); @@ -541,7 +575,7 @@ describe("AiAssistantContents", () => { expect(screen.getByText("Earlier answer")).not.toBeNull(); expect( within(dock) - .getByRole("button", { name: "Collapse voice session" }) + .getByRole("button", { name: "Hide conversation" }) .getAttribute("aria-expanded"), ).toBeNull(); @@ -558,9 +592,14 @@ describe("AiAssistantContents", () => { .closest("div")!; fireEvent.click( - within(dock).getByRole("button", { name: "Collapse voice session" }), + within(dock).getByRole("button", { name: "Hide conversation" }), ); + expect(actions.end).not.toHaveBeenCalled(); + expect(actions.pause).not.toHaveBeenCalled(); + expect(actions.reconnect).not.toHaveBeenCalled(); + expect(actions.resume).not.toHaveBeenCalled(); + expect(actions.setMicrophoneMuted).not.toHaveBeenCalled(); expect(screen.getByTestId("ai-transcript")).toBe(transcript); expect(screen.getByTestId("ai-voice-mode")).toBe(voiceMode); expect( @@ -580,13 +619,18 @@ describe("AiAssistantContents", () => { expect(onCollapsedVoiceEnd).toHaveBeenCalledOnce(); fireEvent.click( - within(dock).getByRole("button", { name: "Expand voice session" }), + within(dock).getByRole("button", { name: "Show conversation" }), ); expect(transcript.className).not.toContain("d_none"); expect(voiceMode.className).not.toContain("d_none"); expect(header.className).not.toContain("d_none"); expect(screen.getByText("Spoken request")).not.toBeNull(); + expect(actions.end).toHaveBeenCalledOnce(); + expect(actions.pause).not.toHaveBeenCalled(); + expect(actions.reconnect).not.toHaveBeenCalled(); + expect(actions.resume).not.toHaveBeenCalled(); + expect(actions.setMicrophoneMuted).not.toHaveBeenCalled(); fireEvent.click( within(dock).getByRole("button", { name: "End voice mode" }), @@ -657,7 +701,7 @@ describe("AiAssistantContents", () => { expect(onVoiceDockCollapsedChange).toHaveBeenCalledWith(false); }); - test("toggles interruption by speaking in the playback menu and reveals manual handover", async () => { + test("toggles interruption by speaking in audio options and reveals manual handover", async () => { const store = createVoiceSessionStore(); const state = { canTakeTurn: true, @@ -695,35 +739,28 @@ describe("AiAssistantContents", () => { ); expect(screen.queryByRole("button", { name: "Your turn" })).toBeNull(); fireEvent.click( - screen.getByRole("button", { name: "Voice playback options" }), + screen.getByRole("button", { name: "Audio options" }), ); - const preference = await screen.findByRole("menuitemcheckbox", { + const preference = await screen.findByRole("button", { name: "Interruption by speaking", }); - expect(preference.getAttribute("aria-checked")).toBe("true"); - expect(preference.hasAttribute("data-selected")).toBe(true); - const menu = screen.getByRole("menu"); - fireEvent.keyDown(menu, { key: "End" }); - await waitFor(() => - expect(menu.getAttribute("aria-activedescendant")).toBe(preference.id), - ); - fireEvent.keyDown(menu, { key: "Enter" }); + expect(preference.getAttribute("aria-pressed")).toBe("true"); + fireEvent.click(preference); await waitFor(() => expect(setInterruptionBySpeaking).toHaveBeenCalledWith(false), ); - expect(screen.getByRole("menu")).not.toBeNull(); - expect(preference.getAttribute("aria-checked")).toBe("false"); - expect(preference.hasAttribute("data-selected")).toBe(false); + expect(screen.getByText("Audio options")).not.toBeNull(); + expect(preference.getAttribute("aria-pressed")).toBe("false"); expect(screen.getByRole("button", { name: "Your turn" })).not.toBeNull(); - fireEvent.keyDown(menu, { key: "Enter" }); + fireEvent.click(preference); await waitFor(() => expect(setInterruptionBySpeaking).toHaveBeenLastCalledWith(true), ); - expect(preference.getAttribute("aria-checked")).toBe("true"); + expect(preference.getAttribute("aria-pressed")).toBe("true"); expect(screen.queryByRole("button", { name: "Your turn" })).toBeNull(); }); - test("keeps handoff and canonical playback controls in the Voice dock", async () => { + test("keeps Realtime playback and independent audio controls in the Voice dock", async () => { const store = createVoiceSessionStore(); const actions = { end: vi.fn(), @@ -732,7 +769,10 @@ describe("AiAssistantContents", () => { reconnect: vi.fn(), repeatQuestion: vi.fn(), resume: vi.fn(), + setInterruptionBySpeaking: vi.fn(), setMicrophoneMuted: vi.fn(), + setSpeakerMuted: vi.fn(), + setSpeakerVolume: vi.fn(), takeTurn: vi.fn(), }; store.setActions(actions); @@ -744,6 +784,8 @@ describe("AiAssistantContents", () => { microphoneLevel: 0.4, microphoneMuted: false, phase: "speaking", + speakerMuted: false, + speakerVolume: 0.4, }); render( @@ -775,38 +817,45 @@ describe("AiAssistantContents", () => { expect(actions.takeTurn).toHaveBeenCalledOnce(); fireEvent.click( - within(dock).getByRole("button", { name: "Voice playback options" }), + within(dock).getByRole("button", { name: "Audio options" }), ); - const repeatQuestion = await screen.findByRole("menuitem", { + const repeatQuestion = await screen.findByRole("button", { name: "Repeat question", }); - const repeatMenu = screen.getByRole("menu"); - fireEvent.keyDown(repeatMenu, { key: "ArrowDown" }); - await waitFor(() => - expect(repeatMenu.getAttribute("aria-activedescendant")).toBe( - repeatQuestion.id, - ), - ); - fireEvent.keyDown(repeatMenu, { key: "Enter" }); - await waitFor(() => expect(actions.repeatQuestion).toHaveBeenCalledOnce()); + fireEvent.click(repeatQuestion); + expect(actions.repeatQuestion).toHaveBeenCalledOnce(); - fireEvent.click( - within(dock).getByRole("button", { name: "Voice playback options" }), - ); - const readFullResponse = await screen.findByRole("menuitem", { + const readFullResponse = screen.getByRole("button", { name: "Read full response", }); - const fullResponseMenu = screen.getByRole("menu"); - fireEvent.keyDown(fullResponseMenu, { key: "End" }); + fireEvent.click(readFullResponse); + expect(actions.readFullResponse).toHaveBeenCalledOnce(); + expect( + screen.getByRole("button", { name: "Interruption by speaking" }), + ).not.toBeNull(); + + fireEvent.click(screen.getByRole("button", { name: "Mute speaker" })); + expect(actions.setSpeakerMuted).toHaveBeenCalledWith(true); + expect(actions.setSpeakerVolume).not.toHaveBeenCalled(); + + const volume = screen.getByRole("slider", { name: "Speaker volume" }); + volume.focus(); + fireEvent.keyDown(volume, { key: "ArrowRight" }); await waitFor(() => - expect(fullResponseMenu.getAttribute("aria-activedescendant")).toBe( - readFullResponse.id, - ), + expect(actions.setSpeakerVolume).toHaveBeenCalledWith(0.45), ); - fireEvent.keyDown(fullResponseMenu, { key: "Enter" }); + expect(actions.setSpeakerMuted).toHaveBeenCalledTimes(1); + actions.setSpeakerMuted.mockClear(); + fireEvent.keyDown(volume, { key: "Home" }); await waitFor(() => - expect(actions.readFullResponse).toHaveBeenCalledOnce(), + expect(actions.setSpeakerVolume).toHaveBeenCalledWith(0), ); + expect(actions.setSpeakerMuted).not.toHaveBeenCalled(); + expect( + within(dock) + .getByRole("button", { name: "Mute microphone" }) + .closest('[data-scope="popover"][data-part="content"]'), + ).toBeNull(); act(() => { store.setState({ @@ -814,10 +863,13 @@ describe("AiAssistantContents", () => { microphoneLevel: 0, microphoneMuted: true, phase: "speaking", + speakerMuted: true, + speakerVolume: 0.4, }); }); expect(within(dock).getByText("Speaking")).not.toBeNull(); + expect(screen.getByRole("button", { name: "Unmute speaker" })).not.toBeNull(); fireEvent.click( within(dock).getByRole("button", { name: "Unmute microphone" }), ); @@ -854,6 +906,136 @@ describe("AiAssistantContents", () => { expect(within(dock).getByText("Listening")).toBeTruthy(); }); + test("shows live Stop only while busy and keeps it independent from End", () => { + const store = createVoiceSessionStore(); + const end = vi.fn(); + const onStop = vi.fn(); + store.setActions({ end, pause: noop }); + store.setState({ + errorMessage: null, + microphoneLevel: 0, + microphoneMuted: false, + phase: "speaking", + }); + const props = { + input: "", + messages: [] as PetrinautAiMessage[], + onClose: noop, + onInputChange: noop, + onStop, + onSubmit: noop, + }; + const rendered = render( + + + , + ); + + expect( + screen.queryByRole("button", { name: "Stop AI response" }), + ).toBeNull(); + + rendered.rerender( + + + , + ); + fireEvent.click(screen.getByRole("button", { name: "Stop AI response" })); + expect(onStop).toHaveBeenCalledOnce(); + expect(end).not.toHaveBeenCalled(); + + rendered.rerender( + + + , + ); + fireEvent.click(screen.getByRole("button", { name: "Stop AI response" })); + expect(onStop).toHaveBeenCalledTimes(2); + expect(end).not.toHaveBeenCalled(); + + fireEvent.click(screen.getByRole("button", { name: "End voice mode" })); + expect(end).toHaveBeenCalledOnce(); + expect(onStop).toHaveBeenCalledTimes(2); + + rendered.rerender( + + + , + ); + expect( + screen.queryByRole("button", { name: "Stop AI response" }), + ).toBeNull(); + }); + + test("defaults speaker state safely and restores audio-trigger focus", async () => { + const store = createVoiceSessionStore(); + const setSpeakerMuted = vi.fn(); + const setSpeakerVolume = vi.fn(); + store.setActions({ + end: vi.fn(), + pause: noop, + setSpeakerMuted, + setSpeakerVolume, + }); + store.setState({ + errorMessage: null, + microphoneLevel: 0, + microphoneMuted: false, + phase: "connected", + }); + render( + + + , + ); + + const trigger = screen.getByRole("button", { name: "Audio options" }); + trigger.focus(); + fireEvent.click(trigger); + const speakerMute = await screen.findByRole("button", { + name: "Mute speaker", + }); + expect(speakerMute.getAttribute("aria-pressed")).toBe("false"); + const volume = screen.getByRole("slider", { name: "Speaker volume" }); + expect(volume.getAttribute("aria-valuenow")).toBe("1"); + + volume.focus(); + fireEvent.keyDown(volume, { key: "ArrowLeft" }); + await waitFor(() => + expect(setSpeakerVolume).toHaveBeenCalledWith(0.95), + ); + expect(setSpeakerMuted).not.toHaveBeenCalled(); + + fireEvent.keyDown(volume, { key: "Escape" }); + await waitFor(() => + expect( + screen.queryByRole("slider", { name: "Speaker volume" }), + ).toBeNull(), + ); + expect(document.activeElement).toBe(trigger); + + fireEvent.click(trigger); + const reopenedVolume = await screen.findByRole("slider", { + name: "Speaker volume", + }); + expect(reopenedVolume).not.toBeNull(); + fireEvent.pointerDown(document.body); + await waitFor(() => + expect( + screen.queryByRole("slider", { name: "Speaker volume" }), + ).toBeNull(), + ); + expect(document.activeElement).toBe(trigger); + }); + test.each([ "Voice admission could not be confirmed. Check canonical history before sending again; no automatic retry was made.", "That utterance was not retained. Wait for the pending input, then use the composer to send it.", diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.tsx index d4dcf4e6234..6465fc308a8 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.tsx @@ -976,12 +976,14 @@ export const AiAssistantContents = ({ className={panelContentStyle({ visible: isOpen })} > onVoiceDockCollapsedChange?.(!isVoiceDockCollapsed) } + onStop={onStop} /> ) : ( @@ -993,6 +995,7 @@ export const AiAssistantContents = ({ > } microphoneMuted={false} onCollapsedToggle={() => onVoiceDockCollapsedChange?.(false)} + onStop={onStop} phase="connecting" purpose="setup" + speakerMuted={false} + speakerVolume={1} /> )} diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock.tsx index 96696cd6a1c..b77089206c4 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock.tsx @@ -11,6 +11,8 @@ import { useVoiceSessionMicrophoneMuted, useVoiceSessionNotice, useVoiceSessionPhase, + useVoiceSessionSpeakerMuted, + useVoiceSessionSpeakerVolume, } from "../../../../../../react/voice-session/use-voice-session"; import { LiveVoiceSessionIndicator } from "../../../components/voice-session-indicator"; import { @@ -19,8 +21,8 @@ import { voiceSetupLabels, } from "../../../components/voice-session-labels"; import { aiFooterMinHeight } from "./footer-height"; +import { AudioPopover } from "./voice-dock/audio-popover"; import { MicrophoneIcon } from "./voice-dock/microphone-icon"; -import { VoicePlaybackMenu } from "./voice-dock/playback-menu"; import type { VoiceSessionActions } from "../../../../../../react/voice-session/store"; import type { PetrinautAiVoiceSessionPhase } from "../../../../../types/ai-assistant-composer-control"; @@ -119,6 +121,7 @@ const visuallyHiddenStyle = css({ export type VoiceDockProps = { actions: VoiceSessionActions | null; + assistantBusy: boolean; canReadFullResponse: boolean; canRepeatQuestion: boolean; canRetryPlayback?: boolean; @@ -132,8 +135,11 @@ export type VoiceDockProps = { notice?: string | null; onCollapsedEnd?: () => void; onCollapsedToggle: () => void; + onStop: () => void; phase: PetrinautAiVoiceSessionPhase; purpose?: "session" | "setup"; + speakerMuted: boolean; + speakerVolume: number; }; /** @@ -142,6 +148,7 @@ export type VoiceDockProps = { */ export const VoiceDock = ({ actions, + assistantBusy, canReadFullResponse, canRepeatQuestion, canRetryPlayback = false, @@ -154,8 +161,11 @@ export const VoiceDock = ({ notice, onCollapsedEnd, onCollapsedToggle, + onStop, phase, purpose = "session", + speakerMuted, + speakerVolume, }: VoiceDockProps) => { const collapseLabel = purpose === "setup" @@ -196,12 +206,16 @@ export const VoiceDock = ({ {actions !== null && (actions.readFullResponse || actions.repeatQuestion || - actions.setInterruptionBySpeaking) && ( - )} @@ -238,45 +252,54 @@ export const VoiceDock = ({ variant="ghost" /> )} - {phase === "error" - ? actions.reconnect && ( - + )} + {actions.setSpeakerVolume && ( + actions.setSpeakerVolume?.(volume)} + showValueText + step={0.05} + value={Math.min(1, Math.max(0, speakerVolume))} + variant="plain" + /> + )} + {actions.repeatQuestion && ( + + )} + {actions.readFullResponse && ( + + )} + {actions.setInterruptionBySpeaking && ( + + )} + + + + + )} + + ); +}; diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock/playback-menu.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock/playback-menu.tsx deleted file mode 100644 index b0674307019..00000000000 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock/playback-menu.tsx +++ /dev/null @@ -1,61 +0,0 @@ -import { Button, Menu, type MenuItem } from "@hashintel/ds-components"; - -import { voiceSessionActionLabels } from "../../../../components/voice-session-labels"; - -import type { VoiceSessionActions } from "../../../../../../../react/voice-session/store"; - -export const VoicePlaybackMenu = ({ - actions, - canReadFullResponse, - canRepeatQuestion, - interruptionBySpeaking, -}: { - actions: VoiceSessionActions; - canReadFullResponse: boolean; - canRepeatQuestion: boolean; - interruptionBySpeaking: boolean; -}) => { - const items: MenuItem[] = [ - { - disabled: !canRepeatQuestion || !actions.repeatQuestion, - id: "repeat-question", - onClick: () => actions.repeatQuestion?.(), - text: voiceSessionActionLabels.repeatQuestion, - }, - { - disabled: !canReadFullResponse || !actions.readFullResponse, - id: "read-full-response", - onClick: () => actions.readFullResponse?.(), - text: voiceSessionActionLabels.readFullResponse, - }, - ]; - - if (actions.setInterruptionBySpeaking) { - items.push({ - id: "interruption-by-speaking", - keepOpenOnSelect: true, - onClick: () => - actions.setInterruptionBySpeaking?.(!interruptionBySpeaking), - selected: interruptionBySpeaking, - selectedStyle: "checkbox", - text: voiceSessionActionLabels.interruptionBySpeaking, - }); - } - - return ( - - } - /> - ); -}; From 9a3064339b65bf7c88e3fb862df18d2e9db34f1f Mon Sep 17 00:00:00 2001 From: Kostandin Angjellari Date: Tue, 15 Sep 2026 16:09:34 +0200 Subject: [PATCH 03/14] Address shared Voice controls review findings Co-authored-by: Cursor --- .../ai-assistant-composer-control.test.ts | 8 +- .../Editor/components/voice-session-labels.ts | 1 + .../Editor/panels/ai-assistant-panel.test.tsx | 19 ++- .../ai-assistant-contents.test.tsx | 112 ++++++++++++++---- .../ai-assistant-contents.tsx | 2 - .../ai-assistant-contents/voice-dock.tsx | 26 ++-- .../voice-dock/audio-popover.tsx | 52 +------- 7 files changed, 121 insertions(+), 99 deletions(-) diff --git a/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.test.ts b/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.test.ts index 84cd504b80f..5b101be7640 100644 --- a/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.test.ts +++ b/libs/@hashintel/petrinaut/src/ui/types/ai-assistant-composer-control.test.ts @@ -1,11 +1,11 @@ import { expectTypeOf, test } from "vitest"; +import type { VoiceSessionActions } from "../../react/voice-session/store"; import type { PetrinautAiVoiceModeContext, PetrinautAiVoiceModeControls, PetrinautAiVoiceModeSessionControls, } from "./ai-assistant-composer-control"; -import type { VoiceSessionActions } from "../../react/voice-session/store"; test("keeps legacy Voice controls required while sessions advertise capabilities", () => { expectTypeOf().toEqualTypeOf< @@ -46,9 +46,9 @@ test("accepts an existing context implementation with complete control registrat }); test("exposes optional provider-neutral speaker controls", () => { - expectTypeOf< - PetrinautAiVoiceModeControls["setSpeakerMuted"] - >().toEqualTypeOf<((muted: boolean) => void) | undefined>(); + expectTypeOf().toEqualTypeOf< + ((muted: boolean) => void) | undefined + >(); expectTypeOf< PetrinautAiVoiceModeControls["setSpeakerVolume"] >().toEqualTypeOf<((volume: number) => void) | undefined>(); diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/components/voice-session-labels.ts b/libs/@hashintel/petrinaut/src/ui/views/Editor/components/voice-session-labels.ts index 984d5821d40..48969c791b3 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/components/voice-session-labels.ts +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/components/voice-session-labels.ts @@ -28,6 +28,7 @@ export const voiceSessionStatusLabel = ( }; export const voiceSessionActionLabels = { + audioControls: "Audio controls", audioOptions: "Audio options", collapse: "Hide conversation", end: "End voice mode", diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx index 1fd711a13c3..dc17353ba16 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel.test.tsx @@ -2430,9 +2430,7 @@ describe("AiAssistantPanel composer submissions", () => { fireEvent.click(await screen.findByRole("button", { name: "Your turn" })); expect(takeTurn).toHaveBeenCalledOnce(); - fireEvent.click( - screen.getByRole("button", { name: "Audio options" }), - ); + fireEvent.click(screen.getByRole("button", { name: "Audio options" })); const repeatQuestionItem = await screen.findByRole("button", { name: "Repeat question", }); @@ -2465,9 +2463,11 @@ describe("AiAssistantPanel composer submissions", () => { ).hasAttribute("disabled"), ).toBe(true); expect( - (screen.getByRole("button", { - name: "Read full response", - }) as HTMLButtonElement).disabled, + ( + screen.getByRole("button", { + name: "Read full response", + }) as HTMLButtonElement + ).disabled, ).toBe(true); }); @@ -2529,7 +2529,9 @@ describe("AiAssistantPanel composer submissions", () => { expect(screen.queryByRole("button", { name: "Your turn" })).toBeNull(); await screen.findByRole("region", { name: "Voice session" }); fireEvent.click(screen.getByRole("button", { name: "Audio options" })); - fireEvent.click(await screen.findByRole("button", { name: "Mute speaker" })); + fireEvent.click( + await screen.findByRole("button", { name: "Mute speaker" }), + ); expect(setSpeakerMuted).toHaveBeenCalledWith(true); fireEvent.keyDown(document.activeElement ?? document, { key: "Escape" }); @@ -2539,9 +2541,6 @@ describe("AiAssistantPanel composer submissions", () => { screen.queryByRole("button", { name: "Audio options" }), ).toBeNull(), ); - expect( - screen.queryByRole("button", { name: "Audio options" }), - ).toBeNull(); expect( screen.queryByRole("button", { name: "Mute microphone" }), ).toBeNull(); diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.test.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.test.tsx index cb80c5cc7b5..b5baa0f7954 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.test.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.test.tsx @@ -114,23 +114,21 @@ test("live-capability dock keeps microphone direct and Realtime controls absent" const microphone = screen.getByRole("button", { name: "Mute microphone" }); expect(microphone).not.toBeNull(); expect(screen.queryByRole("button", { name: "Your turn" })).toBeNull(); - fireEvent.click( - screen.getByRole("button", { name: "Hide conversation" }), - ); + fireEvent.click(screen.getByRole("button", { name: "Hide conversation" })); expect(collapse).toHaveBeenCalledOnce(); fireEvent.click(microphone); expect(setMicrophoneMuted).toHaveBeenCalledWith(true); fireEvent.click(screen.getByRole("button", { name: "Audio options" })); - expect(await screen.findByRole("button", { name: "Mute speaker" })).toBeTruthy(); + expect( + await screen.findByRole("button", { name: "Mute speaker" }), + ).toBeTruthy(); expect( screen .getByRole("slider", { name: "Speaker volume" }) .getAttribute("aria-valuenow"), ).toBe("1"); - expect( - screen.queryByRole("button", { name: "Repeat question" }), - ).toBeNull(); + expect(screen.queryByRole("button", { name: "Repeat question" })).toBeNull(); expect( screen.queryByRole("button", { name: "Read full response" }), ).toBeNull(); @@ -170,6 +168,52 @@ test("offers a user-gesture retry while session audio is blocked", () => { expect(screen.queryByRole("button", { name: "Play voice audio" })).toBeNull(); }); +test.each([ + { + absentAction: "Resume voice mode", + phase: "error" as const, + recoveryAction: "Reconnect voice mode", + }, + { + absentAction: "Reconnect voice mode", + phase: "paused" as const, + recoveryAction: "Resume voice mode", + }, +])( + "keeps direct microphone beside $phase recovery controls", + ({ absentAction, phase, recoveryAction }) => { + render( + } + microphoneMuted={false} + onCollapsedToggle={noop} + onStop={noop} + phase={phase} + speakerMuted={false} + speakerVolume={1} + />, + ); + + expect(screen.getByRole("button", { name: recoveryAction })).toBeTruthy(); + expect( + screen.getByRole("button", { name: "Mute microphone" }), + ).toBeTruthy(); + expect(screen.queryByRole("button", { name: absentAction })).toBeNull(); + }, +); + describe("AiAssistantContents", () => { test("switches to host content without unmounting chat or losing its draft and Stop control", () => { const onStop = vi.fn(); @@ -738,9 +782,7 @@ describe("AiAssistantContents", () => { , ); expect(screen.queryByRole("button", { name: "Your turn" })).toBeNull(); - fireEvent.click( - screen.getByRole("button", { name: "Audio options" }), - ); + fireEvent.click(screen.getByRole("button", { name: "Audio options" })); const preference = await screen.findByRole("button", { name: "Interruption by speaking", }); @@ -869,7 +911,9 @@ describe("AiAssistantContents", () => { }); expect(within(dock).getByText("Speaking")).not.toBeNull(); - expect(screen.getByRole("button", { name: "Unmute speaker" })).not.toBeNull(); + expect( + screen.getByRole("button", { name: "Unmute speaker" }), + ).not.toBeNull(); fireEvent.click( within(dock).getByRole("button", { name: "Unmute microphone" }), ); @@ -984,17 +1028,20 @@ describe("AiAssistantContents", () => { phase: "connected", }); render( - - - , + <> + + + + + , ); const trigger = screen.getByRole("button", { name: "Audio options" }); @@ -1009,11 +1056,12 @@ describe("AiAssistantContents", () => { volume.focus(); fireEvent.keyDown(volume, { key: "ArrowLeft" }); - await waitFor(() => - expect(setSpeakerVolume).toHaveBeenCalledWith(0.95), - ); + await waitFor(() => expect(setSpeakerVolume).toHaveBeenCalledWith(0.95)); expect(setSpeakerMuted).not.toHaveBeenCalled(); + await act(async () => { + await new Promise((resolve) => window.setTimeout(resolve, 50)); + }); fireEvent.keyDown(volume, { key: "Escape" }); await waitFor(() => expect( @@ -1027,7 +1075,19 @@ describe("AiAssistantContents", () => { name: "Speaker volume", }); expect(reopenedVolume).not.toBeNull(); - fireEvent.pointerDown(document.body); + await act(async () => { + await new Promise((resolve) => window.setTimeout(resolve, 50)); + }); + fireEvent.pointerDown( + screen.getByRole("button", { name: "Outside audio options" }), + { + button: 0, + clientX: 100, + clientY: 100, + isPrimary: true, + pointerType: "mouse", + }, + ); await waitFor(() => expect( screen.queryByRole("slider", { name: "Speaker volume" }), diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.tsx index 6465fc308a8..fcc53026553 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.tsx @@ -995,7 +995,6 @@ export const AiAssistantContents = ({ > } microphoneMuted={false} onCollapsedToggle={() => onVoiceDockCollapsedChange?.(false)} - onStop={onStop} phase="connecting" purpose="setup" speakerMuted={false} diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock.tsx index b77089206c4..202377e1f48 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents/voice-dock.tsx @@ -119,9 +119,7 @@ const visuallyHiddenStyle = css({ borderWidth: "[0]", }); -export type VoiceDockProps = { - actions: VoiceSessionActions | null; - assistantBusy: boolean; +type VoiceDockSharedProps = { canReadFullResponse: boolean; canRepeatQuestion: boolean; canRetryPlayback?: boolean; @@ -135,13 +133,27 @@ export type VoiceDockProps = { notice?: string | null; onCollapsedEnd?: () => void; onCollapsedToggle: () => void; - onStop: () => void; phase: PetrinautAiVoiceSessionPhase; - purpose?: "session" | "setup"; speakerMuted: boolean; speakerVolume: number; }; +export type VoiceDockProps = VoiceDockSharedProps & + ( + | { + actions: null; + assistantBusy?: never; + onStop?: never; + purpose: "setup"; + } + | { + actions: VoiceSessionActions | null; + assistantBusy: boolean; + onStop: () => void; + purpose?: "session"; + } + ); + /** * The compact Voice surface inside the assistant panel: one ribbon for setup * or live-session status and controls. @@ -278,9 +290,7 @@ export const VoiceDock = ({ - )} - {actions.setSpeakerVolume && ( - actions.setSpeakerVolume?.(volume)} - showValueText - step={0.05} - value={Math.min(1, Math.max(0, speakerVolume))} - variant="plain" - /> + {(actions.setSpeakerMuted || actions.setSpeakerVolume) && ( +
+ {actions.setSpeakerMuted && ( +
)} {actions.repeatQuestion && ( - )} - {actions.readFullResponse && ( - - )} - {actions.setInterruptionBySpeaking && ( - + {actions.repeatQuestion && ( + + )} + {actions.readFullResponse && ( + + )} + {actions.setInterruptionBySpeaking && ( + + )} + )} From 5746995852cab92453616e166854a259b32cbfe2 Mon Sep 17 00:00:00 2001 From: Kostandin Angjellari Date: Tue, 15 Sep 2026 19:45:13 +0200 Subject: [PATCH 14/14] Add keyboard coverage for voice controls Co-authored-by: Cursor --- libs/@hashintel/brunch-agent/MISSION.md | 37 +++++++++---------- .../ai-assistant-contents.stories.tsx | 14 ++++++- 2 files changed, 29 insertions(+), 22 deletions(-) diff --git a/libs/@hashintel/brunch-agent/MISSION.md b/libs/@hashintel/brunch-agent/MISSION.md index 2c6d94c14b7..f33fcbaee5d 100644 --- a/libs/@hashintel/brunch-agent/MISSION.md +++ b/libs/@hashintel/brunch-agent/MISSION.md @@ -4,27 +4,24 @@ Live execution authority for [FE-1722](https://linear.app/hash/issue/FE-1722/improve-brunch-voice-controls). -This branch is stacked on -[`origin/kostandin/fe-1664-land-voice-stack` at -`023a26b96b`](https://github.com/hashintel/hash/commit/023a26b96b51169da0acdb188697e159d001bcc0), -not `origin/main`. That FE-1664 squash base incorporates merged +This branch is based directly on `origin/main` after +[foundation PR #9745](https://github.com/hashintel/hash/pull/9745) merged. +That foundation incorporates [FE-1712 PR #9704](https://github.com/hashintel/hash/pull/9704). FE-1712's -implementation and evidence remain the protected behavior and evidence source -inherited through that base; its unfinished speech, acoustic and recovery -obligations are not accepted or replaced here. - -FE-1722 implementation exists on this branch through the shared-control, -provider-control, documentation and lifecycle commits from `49ffd6836a` through -`bdc699c923`, together with the final-review corrections recorded beside this -refresh. The deterministic product proof below is established. Real -microphone, speaker and headphone behavior remains unproven and owner-held; -Kostandin owns that browser witness, and no microphone or provider session is -authorized for an agent. - -The approved implementation plan and current owner direction authorize the -subsequent branch push and opening of a stacked draft PR after this committed -final gate. This correction task stops before either write. Merge, deployment -and tracker writes remain unauthorized. +implementation and evidence remain protected behavior; its unfinished speech, +acoustic and recovery obligations are not accepted or replaced here. + +FE-1722 implementation exists on this branch across the shared-control, +provider-control, documentation and lifecycle work reviewed in +[PR #9747](https://github.com/hashintel/hash/pull/9747). It is the bottom entry +of GitHub stack #9750, with follow-up +[PR #9748](https://github.com/hashintel/hash/pull/9748) above it. The +deterministic product proof below is established. Real microphone, speaker and +headphone behavior remains unproven and owner-held; Kostandin owns that browser +witness, and no microphone or provider session is authorized for an agent. + +The owner has authorized branch and PR maintenance for FE-1722. Merge, +deployment and tracker writes remain unauthorized unless separately requested. ## Imperative diff --git a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.stories.tsx b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.stories.tsx index 82f9e89f1f9..b86816136ef 100644 --- a/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.stories.tsx +++ b/libs/@hashintel/petrinaut/src/ui/views/Editor/panels/ai-assistant-panel/ai-assistant-contents.stories.tsx @@ -1,5 +1,5 @@ import { type ReactNode, useState } from "react"; -import { expect, userEvent, within } from "storybook/test"; +import { expect, userEvent, waitFor, within } from "storybook/test"; import { Button } from "@hashintel/ds-components"; import { css } from "@hashintel/ds-helpers/css"; @@ -481,11 +481,14 @@ export const LiveSessionAudioOptions: Story = { const audioOptions = within(dock).getByRole("button", { name: "Audio options", }); - await userEvent.click(audioOptions); + audioOptions.focus(); + await expect(audioOptions).toHaveFocus(); + await userEvent.keyboard("{Enter}"); const speakerMute = await canvas.findByRole("button", { name: "Mute speaker", }); + await waitFor(() => expect(speakerMute).toHaveFocus()); await expect(speakerMute.querySelector("svg")).not.toBeNull(); await expect(within(speakerMute).queryByText("Mute speaker")).toBeNull(); await expect( @@ -503,6 +506,13 @@ export const LiveSessionAudioOptions: Story = { await expect( canvas.queryByRole("button", { name: "Interruption by speaking" }), ).toBeNull(); + await userEvent.keyboard("{Escape}"); + await waitFor(() => + expect( + canvas.queryByRole("slider", { name: "Speaker volume" }), + ).toBeNull(), + ); + await expect(audioOptions).toHaveFocus(); }, };