From b0e23bb26351f25f1b294dd36116545f083dce6d Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Thu, 10 Sep 2026 14:22:54 -0400
Subject: [PATCH 01/21] feat(investigations): Add the hypothesis row for
agentic runs
Renders the hypotheses an agentic investigation is weighing as a row of
cards: the statement, where it landed, the confidence behind that, and the
checks the agent ran to get there.
The data already exists. Seer pushes the whole live state of a run as one
projection blob on every orchestration event, Sentry stores it on
InvestigationOrchestrationRun.projection, and
`/investigations/$id/orchestration/` serves the latest one. So the agent
decides what these cards say purely by what it writes into
`projection.hypotheses` -- there is no separate signal telling the frontend
to render a hypothesis, and no new block kind. Decisions travel back as
versioned commands fenced on workflowVersion, so one made against a stale
view is rejected rather than applied to a run that has moved on.
The row reflows on its container rather than the viewport: auto-fit over
minmax tracks drops columns whenever the available space stops fitting
another readable card. That lets the same component sit in a full-width
detail view and in a narrow drawer without a breakpoint prop, and it is
why this does not draw the connecting edges the flow-graph mock has.
Types and API signatures match the ones in the in-flight orchestration
branch (#122949) so the two collapse into one rather than conflicting.
Stories cover the row, its reflow, every status, and the connected
component running against the existing story fixture API, which now serves
the two orchestration routes and applies commands in memory so accept,
reject, and retry stay interactive on the stories page.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../investigationFixtureApi.spec.tsx | 90 ++++++
.../__stories__/investigationFixtureApi.tsx | 104 +++++++
static/app/views/investigations/api.ts | 94 ++++++
.../views/investigations/fixtures/index.ts | 193 ++++++++++++
.../hypotheses/hypotheses.stories.tsx | 210 +++++++++++++
.../hypotheses/hypothesisCard.spec.tsx | 199 +++++++++++++
.../hypotheses/hypothesisCard.tsx | 157 ++++++++++
.../hypotheses/hypothesisList.spec.tsx | 79 +++++
.../hypotheses/hypothesisList.tsx | 67 +++++
.../hypotheses/hypothesisStatus.tsx | 194 ++++++++++++
.../investigationHypotheses.spec.tsx | 180 +++++++++++
.../hypotheses/investigationHypotheses.tsx | 149 ++++++++++
static/app/views/investigations/types.ts | 281 ++++++++++++++++++
13 files changed, 1997 insertions(+)
create mode 100644 static/app/views/investigations/hypotheses/hypotheses.stories.tsx
create mode 100644 static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
create mode 100644 static/app/views/investigations/hypotheses/hypothesisCard.tsx
create mode 100644 static/app/views/investigations/hypotheses/hypothesisList.spec.tsx
create mode 100644 static/app/views/investigations/hypotheses/hypothesisList.tsx
create mode 100644 static/app/views/investigations/hypotheses/hypothesisStatus.tsx
create mode 100644 static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
create mode 100644 static/app/views/investigations/hypotheses/investigationHypotheses.tsx
diff --git a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
index 9fe7fbe9258b..f8ed03dbe307 100644
--- a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
+++ b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
@@ -10,7 +10,9 @@ import {
InvestigationBlockFixture,
InvestigationDetailFixture,
InvestigationListItemFixture,
+ InvestigationOrchestrationFixture,
} from 'sentry/views/investigations/fixtures';
+import {InvestigationHypotheses} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
const organization = OrganizationFixture({
features: ['investigations'],
@@ -153,4 +155,92 @@ describe('InvestigationFixtureApi', () => {
expect(firstDuplicate.id).toBe('fixture-id-collisions-copy');
expect(secondDuplicate.id).toBe('fixture-id-collisions-copy-2');
});
+
+ // These mirror the "Live, against a mocked orchestration API" story. The
+ // stories route is where this UI gets reviewed, so a fixture that no longer
+ // satisfies the component leaves a broken page rather than a failing build —
+ // rendering the story's contents here is what catches that.
+ describe('orchestration', () => {
+ function renderStoryHypotheses(
+ run = InvestigationOrchestrationFixture(),
+ investigationId = 'investigation-1'
+ ) {
+ return render(
+
+
+ ,
+ {organization}
+ );
+ }
+
+ it('serves the projection to the hypothesis row', async () => {
+ renderStoryHypotheses();
+
+ expect(await screen.findAllByTestId('investigation-hypothesis')).toHaveLength(3);
+ expect(
+ screen.getByRole('heading', {
+ name: 'Database or cache degradation delayed the response',
+ })
+ ).toBeInTheDocument();
+ expect(screen.getByText('Supported · 86% Confidence')).toBeInTheDocument();
+ expect(
+ screen.getByText('The delay begins before the document reaches the browser.')
+ ).toBeInTheDocument();
+ });
+
+ it('applies a disposition command and returns the new projection', async () => {
+ renderStoryHypotheses();
+
+ await userEvent.click(
+ await screen.findByRole('button', {
+ name: 'Actions for An external SSO provider slowed the response',
+ })
+ );
+ await userEvent.click(await screen.findByRole('menuitemradio', {name: 'Accept'}));
+
+ // The command response carries the updated projection, so the card
+ // changes without another read.
+ expect(
+ await screen.findByText('Accepted by you · 91% Confidence')
+ ).toBeInTheDocument();
+ });
+
+ it('clears a disposition back to the agent verdict', async () => {
+ renderStoryHypotheses();
+
+ const trigger = await screen.findByRole('button', {
+ name: 'Actions for An external SSO provider slowed the response',
+ });
+ await userEvent.click(trigger);
+ await userEvent.click(await screen.findByRole('menuitemradio', {name: 'Accept'}));
+ await screen.findByText('Accepted by you · 91% Confidence');
+
+ await userEvent.click(trigger);
+ await userEvent.click(
+ await screen.findByRole('menuitemradio', {name: 'Clear decision'})
+ );
+
+ expect(await screen.findByText('Refuted · 91% Confidence')).toBeInTheDocument();
+ });
+
+ it('puts a retried hypothesis back into investigation', async () => {
+ renderStoryHypotheses();
+
+ await userEvent.click(
+ await screen.findByRole('button', {
+ name: 'Actions for Session validation created a shared bottleneck',
+ })
+ );
+ await userEvent.click(
+ await screen.findByRole('menuitemradio', {name: 'Investigate again'})
+ );
+
+ expect(await screen.findByText('Investigating')).toBeInTheDocument();
+ expect(screen.getAllByText('Queued.').length).toBeGreaterThan(0);
+ });
+ });
});
diff --git a/static/app/views/investigations/__stories__/investigationFixtureApi.tsx b/static/app/views/investigations/__stories__/investigationFixtureApi.tsx
index b81c78b42ca0..0930d53ebc2e 100644
--- a/static/app/views/investigations/__stories__/investigationFixtureApi.tsx
+++ b/static/app/views/investigations/__stories__/investigationFixtureApi.tsx
@@ -18,7 +18,10 @@ import {
import type {
InvestigationDetail,
InvestigationExecutionDetail,
+ InvestigationHypothesis,
InvestigationListItem,
+ InvestigationOrchestration,
+ InvestigationOrchestrationCommand,
InvestigationTitleGeneration,
} from 'sentry/views/investigations/types';
@@ -33,6 +36,8 @@ type InvestigationFixtureApiProps = {
list?: InvestigationListItem[];
mode?: FixtureApiMode;
openMembership?: boolean;
+ /** Agentic run state, keyed by investigation id. */
+ orchestration?: Record;
pageLinks?: string;
titleGenerations?: Record;
};
@@ -69,6 +74,7 @@ export function InvestigationFixtureApi({
list = [],
mode = 'success',
openMembership = true,
+ orchestration = {},
pageLinks,
titleGenerations = {},
}: InvestigationFixtureApiProps) {
@@ -95,6 +101,7 @@ export function InvestigationFixtureApi({
list,
mode,
openMembership,
+ orchestration,
pageLinks,
titleGenerations,
})
@@ -213,6 +220,7 @@ type FixtureState = {
details: Map;
executions: Map;
list: InvestigationListItem[];
+ orchestration: Map;
titleGenerations: Map;
pageLinks?: string;
};
@@ -237,6 +245,12 @@ function createFixtureState(config: FixtureApiConfig): FixtureState {
]),
details,
list,
+ orchestration: new Map(
+ Object.entries(config.orchestration ?? {}).map(([key, run]) => [
+ key,
+ cloneFixture(run),
+ ])
+ ),
pageLinks: config.pageLinks,
executions: new Map(
Object.entries(config.executions ?? {}).map(([key, execution]) => [
@@ -350,6 +364,40 @@ function handleFixtureRequest(
return {body: cloneFixture(duplicate)};
}
+ if (parts[1] === 'orchestration' && parts.length === 2 && method === 'GET') {
+ return {body: cloneFixture(getFixtureOrchestration(state, investigationId))};
+ }
+
+ if (
+ parts[1] === 'orchestration' &&
+ parts[2] === 'commands' &&
+ parts.length === 3 &&
+ method === 'POST'
+ ) {
+ const run = getFixtureOrchestration(state, investigationId);
+ const command = data.command as InvestigationOrchestrationCommand;
+ // Seer would apply the command and push a new projection; the fixture
+ // applies it inline so a story stays interactive.
+ const updated: InvestigationOrchestration = {
+ ...run,
+ workflowVersion: run.workflowVersion + 1,
+ hypotheses: run.hypotheses.map(hypothesis =>
+ applyFixtureCommandToHypothesis(hypothesis, command)
+ ),
+ };
+ state.orchestration.set(investigationId, updated);
+ return {
+ body: {
+ accepted: true,
+ duplicate: false,
+ requestId: getDataString(data, 'requestId') ?? 'fixture-request',
+ runId: run.runId,
+ workflowVersion: updated.workflowVersion,
+ projection: cloneFixture(updated),
+ },
+ };
+ }
+
if (parts[1] === 'title-generation' && method === 'GET') {
const detail = getFixtureDetail(state, investigationId);
return {
@@ -552,6 +600,62 @@ function getFixtureDetail(state: FixtureState, investigationId: string) {
return generatedDetail;
}
+function getFixtureOrchestration(state: FixtureState, investigationId: string) {
+ const run = state.orchestration.get(investigationId);
+ if (!run) {
+ // The real endpoint 404s for an investigation with no agentic run behind
+ // it, so a story that forgot to supply one should say so loudly.
+ throw new Error(`No fixture orchestration run for investigation: ${investigationId}`);
+ }
+ return run;
+}
+
+function applyFixtureCommandToHypothesis(
+ hypothesis: InvestigationHypothesis,
+ command: InvestigationOrchestrationCommand
+): InvestigationHypothesis {
+ if (
+ command.type === 'set_hypothesis_disposition' &&
+ command.hypothesisId === hypothesis.id
+ ) {
+ if (command.disposition === null) {
+ // Clearing hands the hypothesis back to whatever the agent concluded.
+ return {
+ ...hypothesis,
+ decisionSource: 'agent',
+ effectiveStatus: hypothesis.agentVerdict?.verdict ?? 'inconclusive',
+ };
+ }
+ return {
+ ...hypothesis,
+ decisionSource: 'user',
+ effectiveStatus: command.disposition,
+ };
+ }
+
+ if (
+ command.type === 'retry' &&
+ command.target === 'hypothesis' &&
+ command.targetId === hypothesis.id
+ ) {
+ return {
+ ...hypothesis,
+ status: 'running',
+ effectiveStatus: 'investigating',
+ decisionSource: 'none',
+ confidence: null,
+ agentVerdict: null,
+ verificationSteps: hypothesis.verificationSteps.map(step => ({
+ ...step,
+ status: 'queued',
+ result: null,
+ })),
+ };
+ }
+
+ return hypothesis;
+}
+
function setFixtureDetail(state: FixtureState, detail: InvestigationDetail) {
state.details.set(detail.id, detail);
const listIndex = state.list.findIndex(item => item.id === detail.id);
diff --git a/static/app/views/investigations/api.ts b/static/app/views/investigations/api.ts
index 8428f2e7697f..46a870acbec0 100644
--- a/static/app/views/investigations/api.ts
+++ b/static/app/views/investigations/api.ts
@@ -15,6 +15,9 @@ import type {
InvestigationDetail,
InvestigationExecutionDetail,
InvestigationListItem,
+ InvestigationOrchestration,
+ InvestigationOrchestrationCommandResponse,
+ InvestigationOrchestrationCommandVariables,
InvestigationTitleGeneration,
MetricOpenPeriodInvestigationSource,
} from 'sentry/views/investigations/types';
@@ -97,6 +100,97 @@ export function investigationTitleGenerationQueryOptions(
);
}
+/**
+ * The live state of an agentic run: phase, broad scan, hypotheses, and report
+ * progress. Seer overwrites the whole projection on every orchestration event,
+ * so there is nothing to merge — the newest response wins outright.
+ *
+ * `staleTime: 0` because a running workflow changes constantly. Callers that
+ * render a run in progress should add a `refetchInterval`; use
+ * `isInvestigationRunSettled` to stop polling once it reaches a terminal state.
+ */
+export function investigationOrchestrationQueryOptions(
+ organizationSlug: string,
+ investigationId: string
+) {
+ return apiOptions.as()(
+ '/organizations/$organizationIdOrSlug/investigations/$investigationId/orchestration/',
+ {
+ path: {
+ organizationIdOrSlug: organizationSlug,
+ investigationId,
+ },
+ staleTime: 0,
+ }
+ );
+}
+
+/**
+ * Whether a rejected command lost a race rather than being malformed. The
+ * server answers 409 both when the workflow version has moved on and when an
+ * idempotency key is reused, and either way the fix is to re-read the
+ * projection rather than to show a hard failure.
+ */
+export function isInvestigationOrchestrationConflictError(error: unknown): boolean {
+ return (
+ typeof error === 'object' &&
+ error !== null &&
+ 'status' in error &&
+ (error as {status?: unknown}).status === 409
+ );
+}
+
+/**
+ * Send a viewer command — accept/reject a hypothesis, steer, retry, cancel — to
+ * a running workflow.
+ *
+ * The response carries the post-command projection, so it is written straight
+ * into the orchestration cache instead of triggering another fetch.
+ */
+export function useInvestigationOrchestrationCommandMutation(
+ organizationSlug: string,
+ investigationId: string,
+ options?: MutationOptions<
+ InvestigationOrchestrationCommandResponse,
+ InvestigationOrchestrationCommandVariables
+ >
+) {
+ const queryClient = useQueryClient();
+ const orchestrationOptions = investigationOrchestrationQueryOptions(
+ organizationSlug,
+ investigationId
+ );
+
+ return useMutation({
+ ...options,
+ mutationFn: ({command, expectedWorkflowVersion, requestId}) =>
+ fetchMutation({
+ url: getApiUrl(
+ '/organizations/$organizationIdOrSlug/investigations/$investigationId/orchestration/commands/',
+ {
+ path: {
+ organizationIdOrSlug: organizationSlug,
+ investigationId,
+ },
+ }
+ ),
+ method: 'POST',
+ data: {requestId, expectedWorkflowVersion, command},
+ }),
+ onSuccess: async (response, variables, onMutateResult, context) => {
+ queryClient.setQueryData(orchestrationOptions.queryKey, current =>
+ current ? {...current, json: response.projection} : current
+ );
+ await options?.onSuccess?.(response, variables, onMutateResult, context);
+ },
+ onError: async (error, variables, onMutateResult, context) => {
+ // A rejected command usually means the projection moved on beneath us.
+ await queryClient.invalidateQueries({queryKey: orchestrationOptions.queryKey});
+ await options?.onError?.(error, variables, onMutateResult, context);
+ },
+ });
+}
+
export function investigationCandidatesQueryOptions({
organizationSlug,
sources,
diff --git a/static/app/views/investigations/fixtures/index.ts b/static/app/views/investigations/fixtures/index.ts
index 2473526ec986..c8bf2c82ebb7 100644
--- a/static/app/views/investigations/fixtures/index.ts
+++ b/static/app/views/investigations/fixtures/index.ts
@@ -2,10 +2,13 @@ import type {
InvestigationBlock,
InvestigationDetail,
InvestigationExecutionDetail,
+ InvestigationHypothesis,
InvestigationListItem,
+ InvestigationOrchestration,
InvestigationQueryOutput,
InvestigationTitleGeneration,
InvestigationTranscriptBlock,
+ InvestigationVerificationStep,
} from 'sentry/views/investigations/types';
export function InvestigationListItemFixture(
@@ -393,3 +396,193 @@ export function InvestigationAwaitingInputExecutionFixture(
...overrides,
});
}
+
+export function InvestigationVerificationStepFixture(
+ overrides: Partial = {}
+): InvestigationVerificationStep {
+ return {
+ id: 'step-1',
+ order: 0,
+ title: 'Compare FCP with server response time',
+ objective: 'Establish whether the delay starts on the server or in the browser.',
+ method: 'Compare FCP and TTFB percentiles over the incident window.',
+ status: 'completed',
+ result: 'The delay begins before the document reaches the browser.',
+ evidence: [],
+ error: null,
+ ...overrides,
+ };
+}
+
+export function InvestigationHypothesisFixture(
+ overrides: Partial = {}
+): InvestigationHypothesis {
+ return {
+ id: 'hypothesis-1',
+ order: 0,
+ statement: 'Database or cache degradation delayed the response',
+ rationale:
+ 'FCP and TTFB rose together as cache misses exposed a much slower organization lookup.',
+ status: 'completed',
+ effectiveStatus: 'supported',
+ decisionSource: 'agent',
+ confidence: 0.86,
+ attempt: 0,
+ verificationSteps: [
+ InvestigationVerificationStepFixture(),
+ InvestigationVerificationStepFixture({
+ id: 'step-2',
+ order: 1,
+ title: 'Compare organization lookup spans',
+ objective: 'Isolate the slow span.',
+ method: 'Break lookup duration down by cache outcome.',
+ result: 'The lookup slowed sharply during the incident window.',
+ }),
+ InvestigationVerificationStepFixture({
+ id: 'step-3',
+ order: 2,
+ title: 'Inspect cache and Redis behavior',
+ objective: 'Confirm the cache is the source.',
+ method: 'Chart hit rate against response time.',
+ result: 'Cache misses increased at the same time as the slowdown.',
+ }),
+ ],
+ agentVerdict: {
+ verdict: 'supported',
+ confidence: 0.86,
+ rationale: 'Every check points at the same cache regression.',
+ supportingEvidenceIds: [],
+ refutingEvidenceIds: [],
+ remainingGaps: [],
+ },
+ evidence: [],
+ toolActivity: [],
+ error: null,
+ ...overrides,
+ };
+}
+
+/**
+ * The three-hypothesis shape the hypothesis row is designed around: one
+ * supported conclusion alongside a refuted and an inconclusive alternative.
+ */
+export function InvestigationHypothesesFixture(): InvestigationHypothesis[] {
+ return [
+ InvestigationHypothesisFixture(),
+ InvestigationHypothesisFixture({
+ id: 'hypothesis-2',
+ order: 1,
+ statement: 'An external SSO provider slowed the response',
+ rationale:
+ 'SSO and non-SSO organizations slowed together: provider spans stayed near baseline.',
+ effectiveStatus: 'refuted',
+ confidence: 0.91,
+ agentVerdict: {
+ verdict: 'refuted',
+ confidence: 0.91,
+ rationale: 'The shared delay contradicts an SSO-only explanation.',
+ supportingEvidenceIds: [],
+ refutingEvidenceIds: [],
+ remainingGaps: [],
+ },
+ verificationSteps: [
+ InvestigationVerificationStepFixture({
+ id: 'step-2-1',
+ order: 0,
+ title: 'Compare identity-provider spans',
+ objective: 'Check the provider call.',
+ method: 'Chart provider span duration over the window.',
+ result: 'No shared provider slowdown appears in the affected traces.',
+ }),
+ InvestigationVerificationStepFixture({
+ id: 'step-2-2',
+ order: 1,
+ title: 'Compare SSO and non-SSO organizations',
+ objective: 'Separate the two populations.',
+ method: 'Group response time by authentication method.',
+ result: 'Both groups show the same server-side delay.',
+ }),
+ ],
+ }),
+ InvestigationHypothesisFixture({
+ id: 'hypothesis-3',
+ order: 2,
+ statement: 'Session validation created a shared bottleneck',
+ rationale:
+ 'Available traces do not separate session-validation time from the cache and database delay.',
+ effectiveStatus: 'inconclusive',
+ confidence: 0.34,
+ agentVerdict: {
+ verdict: 'inconclusive',
+ confidence: 0.34,
+ rationale: 'Span coverage is too incomplete to isolate this contribution.',
+ supportingEvidenceIds: [],
+ refutingEvidenceIds: [],
+ remainingGaps: ['Session middleware spans are not instrumented.'],
+ },
+ verificationSteps: [
+ InvestigationVerificationStepFixture({
+ id: 'step-3-1',
+ order: 0,
+ title: 'Inspect session and middleware spans',
+ objective: 'Measure validation time.',
+ method: 'Break the request down by middleware span.',
+ result: 'Span coverage is incomplete in the affected trace sample.',
+ }),
+ InvestigationVerificationStepFixture({
+ id: 'step-3-2',
+ order: 1,
+ title: 'Check shared Redis pressure',
+ objective: 'Separate session load from cache load.',
+ method: 'Compare Redis command latency by key prefix.',
+ result:
+ 'Redis contention overlaps the slowdown but does not isolate session validation.',
+ }),
+ ],
+ }),
+ ];
+}
+
+export function InvestigationOrchestrationFixture(
+ overrides: Partial = {}
+): InvestigationOrchestration {
+ return {
+ runId: '9001',
+ investigationId: 'investigation-1',
+ workflowVersion: 4,
+ generation: 1,
+ notebookRevision: 5,
+ phase: 'reporting',
+ status: 'processing',
+ sourceType: 'breached_metric',
+ broadScan: {
+ status: 'completed',
+ summary: 'FCP regressed on organization login pages across every active release.',
+ toolActivity: [],
+ error: null,
+ },
+ hypotheses: InvestigationHypothesesFixture(),
+ report: {
+ status: 'composing',
+ revision: 2,
+ notebookRevision: 5,
+ currentBlockKey: null,
+ includedHypothesisIds: ['hypothesis-1', 'hypothesis-3'],
+ primaryHypothesisId: 'hypothesis-1',
+ error: null,
+ metadata: {
+ status: 'completed',
+ title: 'Why did FCP spike on organization login pages?',
+ summary: 'A cache regression slowed organization lookups',
+ summaryDescription:
+ 'Cache misses exposed a much slower organization lookup, delaying the server response.',
+ error: null,
+ },
+ },
+ pendingInput: null,
+ errors: [],
+ heartbeatAt: '2026-08-27T11:06:30Z',
+ updatedAt: '2026-08-27T11:06:30Z',
+ ...overrides,
+ };
+}
diff --git a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
new file mode 100644
index 000000000000..74f2f665a283
--- /dev/null
+++ b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
@@ -0,0 +1,210 @@
+import {Fragment} from 'react';
+
+import {Container, Stack} from '@sentry/scraps/layout';
+import {Text} from '@sentry/scraps/text';
+
+import * as Storybook from 'sentry/stories';
+import {InvestigationFixtureApi} from 'sentry/views/investigations/__stories__/investigationFixtureApi';
+import {
+ InvestigationDetailFixture,
+ InvestigationHypothesesFixture,
+ InvestigationHypothesisFixture,
+ InvestigationOrchestrationFixture,
+ InvestigationVerificationStepFixture,
+} from 'sentry/views/investigations/fixtures';
+import {HypothesisList} from 'sentry/views/investigations/hypotheses/hypothesisList';
+import {InvestigationHypotheses} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
+
+export default Storybook.story('Investigations — Hypotheses', story => {
+ story('The hypothesis row', () => (
+
+
+ An agentic investigation proposes several explanations, tests each one, and lands
+ on a verdict. The row shows them side by side so the alternatives that were ruled
+ out stay visible next to the one that survived.
+
+
+ The data comes from projection.hypotheses on the orchestration
+ endpoint, and the highlighted card is report.primaryHypothesisId.
+
+
+
+ ));
+
+ story('Reflows on its container, not the viewport', () => (
+
+
+ The row is a grid of minmax(260px, 1fr) tracks with{' '}
+ auto-fit, so it drops columns whenever its own box stops fitting
+ another readable card. Nothing about the viewport is consulted, which is what lets
+ the same component sit in a full-width detail view and in a narrow drawer.
+
+ A card renders effectiveStatus, which already folds the agent verdict
+ and any user disposition into the run status. Confidence only appears once the
+ agent has settled on a verdict, so work in progress shows a bare label and a
+ pulsing dot.
+
+
+
+ ));
+
+ story('Live, against a mocked orchestration API', () => (
+
+
+ InvestigationHypotheses reads the projection from{' '}
+ /investigations/$id/orchestration/ and posts decisions back to{' '}
+ /orchestration/commands/. Here both are served by{' '}
+ InvestigationFixtureApi, the in-memory fake the other investigations
+ stories use, so this is the real component and the real data flow with only the
+ network swapped out.
+
+
+ Accept or reject a hypothesis from its overflow menu: the fixture applies the
+ command, bumps workflowVersion, and returns the new projection, which
+ the mutation writes straight into the query cache. Choosing the same decision
+ twice clears it and hands the hypothesis back to the agent's verdict.
+
+ Cards do not own commands. The surface rendering them decides which of accept,
+ reject, steer, and retry apply, and posts the chosen one to{' '}
+ /orchestration/commands/. Leave getActions off for a
+ read-only surface.
+
+ [
+ {
+ key: 'accept',
+ label: 'Accept',
+ onAction: () => {},
+ },
+ {
+ key: 'reject',
+ label: 'Reject',
+ onAction: () => {},
+ },
+ {
+ key: 'retry',
+ label: 'Investigate again',
+ disabled: hypothesis.effectiveStatus === 'investigating',
+ onAction: () => {},
+ },
+ ]}
+ />
+
+ ));
+});
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
new file mode 100644
index 000000000000..ac1343e92fd2
--- /dev/null
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
@@ -0,0 +1,199 @@
+import {render, screen, userEvent, within} from 'sentry-test/reactTestingLibrary';
+
+import {
+ InvestigationHypothesisFixture,
+ InvestigationVerificationStepFixture,
+} from 'sentry/views/investigations/fixtures';
+import {HypothesisCard} from 'sentry/views/investigations/hypotheses/hypothesisCard';
+
+describe('HypothesisCard', () => {
+ it('renders the statement, rationale, and one-based ordinal', () => {
+ render(
+
+ );
+
+ expect(
+ screen.getByRole('heading', {name: 'An external SSO provider slowed the response'})
+ ).toBeInTheDocument();
+ expect(
+ screen.getByText('SSO and non-SSO organizations slowed together.')
+ ).toBeInTheDocument();
+ // `order` is zero-based on the wire, so the second hypothesis reads as 2.
+ expect(screen.getByText('Hypothesis 2')).toBeInTheDocument();
+ });
+
+ it('shows confidence once the agent has reached a verdict', () => {
+ render(
+
+ );
+
+ expect(screen.getByText('Supported · 86% Confidence')).toBeInTheDocument();
+ });
+
+ it('falls back to the verdict confidence when the hypothesis omits it', () => {
+ render(
+
+ );
+
+ expect(screen.getByText('Inconclusive · 34% Confidence')).toBeInTheDocument();
+ });
+
+ it('omits confidence while the hypothesis is still being investigated', () => {
+ render(
+
+ );
+
+ expect(screen.getByText('Investigating')).toBeInTheDocument();
+ expect(screen.queryByText(/Confidence/)).not.toBeInTheDocument();
+ });
+
+ it('lists verification steps in order with their results', () => {
+ render(
+
+ );
+
+ expect(screen.getByText('Evidence checked')).toBeInTheDocument();
+ // The only list inside a card is the evidence list; the card itself is an
+ // `li` belonging to the surrounding hypothesis row.
+ const steps = within(screen.getByRole('list')).getAllByRole('listitem');
+ expect(steps[0]).toHaveTextContent('First check');
+ expect(steps[1]).toHaveTextContent('Second check');
+ });
+
+ it('describes a step that has not produced a result yet', () => {
+ render(
+
+ );
+
+ expect(screen.getByText('Checking…')).toBeInTheDocument();
+ });
+
+ it("prefers a failed step's error message over the generic failure label", () => {
+ render(
+
+ );
+
+ expect(screen.getByText('The query timed out.')).toBeInTheDocument();
+ expect(screen.queryByText('This check failed.')).not.toBeInTheDocument();
+ });
+
+ it('surfaces a hypothesis-level error', () => {
+ render(
+
+ );
+
+ expect(
+ screen.getByText('No traces covered the incident window.')
+ ).toBeInTheDocument();
+ });
+
+ it('hides the evidence section when there are no steps', () => {
+ render(
+
+ );
+
+ expect(screen.queryByText('Evidence checked')).not.toBeInTheDocument();
+ });
+
+ it('renders no overflow menu without actions', () => {
+ render();
+
+ expect(screen.queryByRole('button', {name: /Actions for/})).not.toBeInTheDocument();
+ });
+
+ it('opens the overflow menu and runs an action', async () => {
+ const onAction = jest.fn();
+ const hypothesis = InvestigationHypothesisFixture();
+ render(
+
+ );
+
+ await userEvent.click(
+ screen.getByRole('button', {name: `Actions for ${hypothesis.statement}`})
+ );
+ await userEvent.click(await screen.findByRole('menuitemradio', {name: 'Accept'}));
+
+ expect(onAction).toHaveBeenCalled();
+ });
+});
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
new file mode 100644
index 000000000000..f4295f058751
--- /dev/null
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
@@ -0,0 +1,157 @@
+import styled from '@emotion/styled';
+
+import {Container, Flex, Stack} from '@sentry/scraps/layout';
+import {Heading, Text} from '@sentry/scraps/text';
+
+import {DropdownMenu, type MenuItemProps} from 'sentry/components/dropdownMenu';
+import {IconEllipsis} from 'sentry/icons';
+import {t} from 'sentry/locale';
+import {
+ getVerificationStepStatusLabel,
+ HypothesisStatus,
+} from 'sentry/views/investigations/hypotheses/hypothesisStatus';
+import type {
+ InvestigationHypothesis,
+ InvestigationVerificationStep,
+} from 'sentry/views/investigations/types';
+
+type HypothesisCardProps = {
+ hypothesis: InvestigationHypothesis;
+ /**
+ * Menu items for the card's overflow menu. The card does not own commands —
+ * the surface rendering it decides which of accept, reject, steer, and retry
+ * apply, and supplies them here. No menu renders when this is empty.
+ */
+ actions?: MenuItemProps[];
+ className?: string;
+ /**
+ * Whether this is the hypothesis the report leads with
+ * (`report.primaryHypothesisId`). It gets an accent border so the conclusion
+ * is findable without reading every card.
+ */
+ isPrimary?: boolean;
+};
+
+/**
+ * One hypothesis in an agentic investigation: what the agent proposed, where it
+ * landed, and the checks it ran to get there.
+ *
+ * The card is presentational and self-contained so it can appear in the
+ * investigation detail view, in a monitor alert drawer, or anywhere else a run
+ * is summarized. It sizes to its container rather than to the viewport.
+ */
+export function HypothesisCard({
+ actions,
+ className,
+ hypothesis,
+ isPrimary = false,
+}: HypothesisCardProps) {
+ const steps = [...(hypothesis.verificationSteps ?? [])].sort(
+ (a, b) => a.order - b.order
+ );
+
+ return (
+
+
+
+
+ {/* `order` is zero-based in the projection; people count from one. */}
+ {t('Hypothesis %s', hypothesis.order + 1)}
+
+
+
+ {actions?.length ? (
+ ,
+ 'aria-label': t('Actions for %s', hypothesis.statement),
+ }}
+ items={actions}
+ />
+ ) : null}
+
+
+
+
+ {hypothesis.statement}
+
+ {hypothesis.rationale ? (
+
+ {hypothesis.rationale}
+
+ ) : null}
+
+
+ {hypothesis.error ? (
+
+ {hypothesis.error.message}
+
+ ) : null}
+
+ {steps.length > 0 ? (
+
+
+ {t('Evidence checked')}
+
+
+ {steps.map(step => (
+
+ ))}
+
+
+ ) : null}
+
+ );
+}
+
+function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
+ const failed = step.status === 'failed';
+ // A step's own error is more specific than the generic failure label, so it
+ // wins when both are present.
+ const detail =
+ step.result || step.error?.message || getVerificationStepStatusLabel(step.status);
+
+ return (
+
+
+ {step.title}
+
+ {detail}
+
+
+
+ );
+}
+
+// The accent border alone is easy to miss against a wall of cards, so the
+// primary hypothesis also lifts off the page. Driven by a data attribute rather
+// than a styled prop: `Stack` forwards every prop it does not recognize to the
+// DOM, and a bare `isPrimary` would land there as an unknown attribute.
+const Card = styled(Stack)`
+ list-style: none;
+
+ &[data-primary='true'] {
+ box-shadow: ${p => p.theme.shadow.low};
+ }
+`;
diff --git a/static/app/views/investigations/hypotheses/hypothesisList.spec.tsx b/static/app/views/investigations/hypotheses/hypothesisList.spec.tsx
new file mode 100644
index 000000000000..7da9aceff0a3
--- /dev/null
+++ b/static/app/views/investigations/hypotheses/hypothesisList.spec.tsx
@@ -0,0 +1,79 @@
+import {render, screen, within} from 'sentry-test/reactTestingLibrary';
+
+import {
+ InvestigationHypothesesFixture,
+ InvestigationHypothesisFixture,
+} from 'sentry/views/investigations/fixtures';
+import {HypothesisList} from 'sentry/views/investigations/hypotheses/hypothesisList';
+
+describe('HypothesisList', () => {
+ it('renders one card per hypothesis', () => {
+ render();
+
+ expect(screen.getAllByTestId('investigation-hypothesis')).toHaveLength(3);
+ expect(
+ screen.getByRole('heading', {
+ name: 'Database or cache degradation delayed the response',
+ })
+ ).toBeInTheDocument();
+ });
+
+ it('orders cards by the projection order, not array position', () => {
+ render(
+
+ );
+
+ const cards = within(screen.getByTestId('investigation-hypotheses')).getAllByTestId(
+ 'investigation-hypothesis'
+ );
+ expect(cards[0]).toHaveTextContent('First idea');
+ expect(cards[1]).toHaveTextContent('Third idea');
+ });
+
+ it('renders nothing when there are no hypotheses', () => {
+ render();
+
+ expect(screen.queryByTestId('investigation-hypotheses')).not.toBeInTheDocument();
+ });
+
+ it('marks the report primary hypothesis', () => {
+ render(
+
+ );
+
+ const cards = screen.getAllByTestId('investigation-hypothesis');
+ expect(cards[0]).toHaveAttribute('data-primary', 'false');
+ expect(cards[1]).toHaveAttribute('data-primary', 'true');
+ });
+
+ it('passes per-hypothesis actions to each card', async () => {
+ render(
+ [
+ {key: 'retry', label: `Retry ${hypothesis.id}`, onAction: jest.fn()},
+ ]}
+ />
+ );
+
+ expect(await screen.findAllByRole('button', {name: /Actions for/})).toHaveLength(3);
+ });
+});
diff --git a/static/app/views/investigations/hypotheses/hypothesisList.tsx b/static/app/views/investigations/hypotheses/hypothesisList.tsx
new file mode 100644
index 000000000000..8cbddf802f8d
--- /dev/null
+++ b/static/app/views/investigations/hypotheses/hypothesisList.tsx
@@ -0,0 +1,67 @@
+import {Grid} from '@sentry/scraps/layout';
+
+import type {MenuItemProps} from 'sentry/components/dropdownMenu';
+import {HypothesisCard} from 'sentry/views/investigations/hypotheses/hypothesisCard';
+import type {InvestigationHypothesis} from 'sentry/views/investigations/types';
+
+/**
+ * The narrowest a hypothesis card may get before the row drops to fewer
+ * columns. Below roughly this width the statement and its evidence rows stop
+ * being scannable.
+ */
+const MIN_CARD_WIDTH = '260px';
+
+type HypothesisListProps = {
+ hypotheses: InvestigationHypothesis[];
+ className?: string;
+ /**
+ * Builds the overflow menu for one hypothesis. Left out, cards render without
+ * a menu — which is what a read-only surface wants.
+ */
+ getActions?: (hypothesis: InvestigationHypothesis) => MenuItemProps[];
+ /** `report.primaryHypothesisId` from the projection, if the report has one. */
+ primaryHypothesisId?: string | null;
+};
+
+/**
+ * The hypotheses an agentic investigation is weighing, side by side.
+ *
+ * The row reflows on the *container's* width rather than the viewport's:
+ * `auto-fit` + `minmax` drops to fewer columns whenever the available space
+ * stops fitting another readable card. That is what lets the same component sit
+ * in a full-width detail view and in a narrow drawer without a breakpoint prop
+ * or a `containerType` on the parent.
+ */
+export function HypothesisList({
+ className,
+ getActions,
+ hypotheses,
+ primaryHypothesisId,
+}: HypothesisListProps) {
+ if (hypotheses.length === 0) {
+ return null;
+ }
+
+ const ordered = [...hypotheses].sort((a, b) => a.order - b.order);
+
+ return (
+
+ {ordered.map(hypothesis => (
+
+ ))}
+
+ );
+}
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
new file mode 100644
index 000000000000..4ae214cdd497
--- /dev/null
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -0,0 +1,194 @@
+import {Flex} from '@sentry/scraps/layout';
+import {StatusIndicator} from '@sentry/scraps/statusIndicator';
+import {Text} from '@sentry/scraps/text';
+
+import {t} from 'sentry/locale';
+import type {
+ InvestigationHypothesis,
+ InvestigationHypothesisStatus,
+ InvestigationOrchestrationWorkStatus,
+} from 'sentry/views/investigations/types';
+
+type StatusVariant = 'success' | 'warning' | 'danger' | 'accent' | 'muted';
+
+/**
+ * Statuses where the agent has reached a verdict. Confidence is only meaningful
+ * once it has, so an in-flight hypothesis shows a bare label.
+ */
+const SETTLED_STATUSES = new Set([
+ 'supported',
+ 'refuted',
+ 'inconclusive',
+ 'accepted',
+ 'rejected',
+]);
+
+/** Statuses where the agent is still working, so the dot keeps pulsing. */
+const IN_FLIGHT_STATUSES = new Set(['pending', 'investigating']);
+
+export function isHypothesisSettled(status: InvestigationHypothesisStatus): boolean {
+ return SETTLED_STATUSES.has(status);
+}
+
+export function isHypothesisInFlight(status: InvestigationHypothesisStatus): boolean {
+ return IN_FLIGHT_STATUSES.has(status);
+}
+
+/**
+ * Turn an unrecognized wire value into something readable rather than dropping
+ * it. Statuses are an open set — Seer can introduce one before Sentry knows the
+ * name — so every lookup here needs a fallback.
+ */
+function humanize(status: string): string {
+ return status.replaceAll('_', ' ').replace(/^./, character => character.toUpperCase());
+}
+
+export function getHypothesisStatusLabel(hypothesis: InvestigationHypothesis): string {
+ // A decision the viewer made themselves reads differently from one the agent
+ // reached, even though both land in `effectiveStatus`.
+ if (hypothesis.decisionSource === 'user') {
+ if (hypothesis.effectiveStatus === 'accepted') {
+ return t('Accepted by you');
+ }
+ if (hypothesis.effectiveStatus === 'rejected') {
+ return t('Rejected by you');
+ }
+ }
+
+ switch (hypothesis.effectiveStatus) {
+ case 'pending':
+ return t('Pending');
+ case 'investigating':
+ return t('Investigating');
+ case 'supported':
+ return t('Supported');
+ case 'refuted':
+ return t('Refuted');
+ case 'inconclusive':
+ return t('Inconclusive');
+ case 'accepted':
+ return t('Accepted');
+ case 'rejected':
+ return t('Rejected');
+ case 'failed':
+ return t('Failed');
+ case 'cancelled':
+ return t('Cancelled');
+ default:
+ return humanize(hypothesis.effectiveStatus);
+ }
+}
+
+export function getHypothesisStatusVariant(
+ status: InvestigationHypothesisStatus
+): StatusVariant {
+ switch (status) {
+ // A hypothesis the evidence backs, or one a person has endorsed.
+ case 'supported':
+ case 'accepted':
+ return 'success';
+ // Ruled out cleanly. This is a useful outcome rather than an error, so it
+ // reads as neutral instead of dangerous.
+ case 'refuted':
+ case 'rejected':
+ return 'muted';
+ // Checked, but the evidence did not settle it either way.
+ case 'inconclusive':
+ return 'warning';
+ case 'investigating':
+ return 'accent';
+ case 'failed':
+ return 'danger';
+ case 'pending':
+ case 'cancelled':
+ return 'muted';
+ default:
+ return 'muted';
+ }
+}
+
+/**
+ * What a verification step says about itself while it has no result yet. A step
+ * only carries a `result` once it has finished, so everything short of that
+ * needs a stand-in line rather than an empty row.
+ */
+export function getVerificationStepStatusLabel(
+ status: InvestigationOrchestrationWorkStatus
+): string {
+ switch (status) {
+ case 'not_started':
+ case 'queued':
+ return t('Queued.');
+ case 'running':
+ return t('Checking…');
+ case 'blocked':
+ return t('Blocked on an earlier step.');
+ case 'reauth_required':
+ return t('Waiting on reauthentication.');
+ case 'stalled':
+ return t('Stalled.');
+ case 'cancelled':
+ return t('Cancelled before it finished.');
+ case 'failed':
+ return t('This check failed.');
+ case 'completed':
+ // A completed step with no result is a gap in the projection, not a state
+ // worth naming in the UI.
+ return t('No result was recorded.');
+ default:
+ return humanize(status);
+ }
+}
+
+/**
+ * The confidence the card should show, as a whole percentage, or null when
+ * there is nothing meaningful to show yet.
+ *
+ * The projection carries confidence in two places: denormalized onto the
+ * hypothesis, and on the agent's verdict. The hypothesis-level value is the one
+ * kept in step with `effectiveStatus`, so it wins; the verdict is the fallback
+ * for a projection that has only filled the latter in.
+ */
+export function getHypothesisConfidencePercent(
+ hypothesis: InvestigationHypothesis
+): number | null {
+ if (!isHypothesisSettled(hypothesis.effectiveStatus)) {
+ return null;
+ }
+ const confidence = hypothesis.confidence ?? hypothesis.agentVerdict?.confidence;
+ if (typeof confidence !== 'number' || Number.isNaN(confidence)) {
+ return null;
+ }
+ return Math.round(confidence * 100);
+}
+
+type HypothesisStatusProps = {
+ hypothesis: InvestigationHypothesis;
+};
+
+/**
+ * The dot-and-label line above a hypothesis statement, e.g.
+ * "● Supported · 86% Confidence".
+ */
+export function HypothesisStatus({hypothesis}: HypothesisStatusProps) {
+ const status = hypothesis.effectiveStatus;
+ const variant = getHypothesisStatusVariant(status);
+ const label = getHypothesisStatusLabel(hypothesis);
+ const confidence = getHypothesisConfidencePercent(hypothesis);
+
+ return (
+
+
+
+ {confidence === null
+ ? label
+ : // Translators: e.g. "Supported · 86% Confidence"
+ t('%s · %s%% Confidence', label, confidence)}
+
+
+ );
+}
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
new file mode 100644
index 000000000000..d9800fad6fe5
--- /dev/null
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
@@ -0,0 +1,180 @@
+import {QueryClientProvider} from '@tanstack/react-query';
+import {OrganizationFixture} from 'sentry-fixture/organization';
+
+import {makeTestQueryClient} from 'sentry-test/queryClient';
+import {render, screen, userEvent, waitFor} from 'sentry-test/reactTestingLibrary';
+
+import {InvestigationOrchestrationFixture} from 'sentry/views/investigations/fixtures';
+import {InvestigationHypotheses} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
+import type {InvestigationOrchestration} from 'sentry/views/investigations/types';
+
+const organization = OrganizationFixture({features: ['investigations']});
+const orchestrationUrl =
+ '/organizations/org-slug/investigations/investigation-1/orchestration/';
+const commandsUrl = `${orchestrationUrl}commands/`;
+
+function renderHypotheses() {
+ return render(, {
+ additionalWrapper: ({children}) => (
+ {children}
+ ),
+ organization,
+ });
+}
+
+describe('InvestigationHypotheses', () => {
+ it('renders the hypotheses carried on the projection', async () => {
+ MockApiClient.addMockResponse({
+ url: orchestrationUrl,
+ body: InvestigationOrchestrationFixture(),
+ });
+
+ renderHypotheses();
+
+ expect(await screen.findAllByTestId('investigation-hypothesis')).toHaveLength(3);
+ expect(
+ screen.getByRole('heading', {
+ name: 'Database or cache degradation delayed the response',
+ })
+ ).toBeInTheDocument();
+ expect(screen.getByText('Supported · 86% Confidence')).toBeInTheDocument();
+ });
+
+ it('highlights the report primary hypothesis', async () => {
+ MockApiClient.addMockResponse({
+ url: orchestrationUrl,
+ body: InvestigationOrchestrationFixture(),
+ });
+
+ renderHypotheses();
+
+ const cards = await screen.findAllByTestId('investigation-hypothesis');
+ expect(cards[0]).toHaveAttribute('data-primary', 'true');
+ expect(cards[1]).toHaveAttribute('data-primary', 'false');
+ });
+
+ it('renders nothing when the projection carries no hypotheses', async () => {
+ const request = MockApiClient.addMockResponse({
+ url: orchestrationUrl,
+ body: InvestigationOrchestrationFixture({hypotheses: []}),
+ });
+
+ renderHypotheses();
+
+ await waitFor(() => expect(request).toHaveBeenCalled());
+ expect(screen.queryByTestId('investigation-hypotheses')).not.toBeInTheDocument();
+ });
+
+ it('does not fetch when there is no agentic run behind the investigation', () => {
+ const request = MockApiClient.addMockResponse({
+ url: orchestrationUrl,
+ body: InvestigationOrchestrationFixture(),
+ });
+
+ render(
+ ,
+ {
+ additionalWrapper: ({children}) => (
+
+ {children}
+
+ ),
+ organization,
+ }
+ );
+
+ expect(request).not.toHaveBeenCalled();
+ });
+
+ it('sends a disposition command fenced on the workflow version', async () => {
+ MockApiClient.addMockResponse({
+ url: orchestrationUrl,
+ body: InvestigationOrchestrationFixture({workflowVersion: 7}),
+ });
+ const commandRequest = MockApiClient.addMockResponse({
+ url: commandsUrl,
+ method: 'POST',
+ body: {
+ accepted: true,
+ duplicate: false,
+ requestId: 'request-1',
+ workflowVersion: 8,
+ commandStatus: 'accepted',
+ commandError: null,
+ runId: '9001',
+ projection: InvestigationOrchestrationFixture({workflowVersion: 8}),
+ },
+ });
+
+ renderHypotheses();
+
+ await userEvent.click(
+ await screen.findByRole('button', {
+ name: 'Actions for Database or cache degradation delayed the response',
+ })
+ );
+ await userEvent.click(await screen.findByRole('menuitemradio', {name: 'Accept'}));
+
+ await waitFor(() =>
+ expect(commandRequest).toHaveBeenCalledWith(
+ commandsUrl,
+ expect.objectContaining({
+ method: 'POST',
+ data: expect.objectContaining({
+ expectedWorkflowVersion: 7,
+ command: {
+ type: 'set_hypothesis_disposition',
+ hypothesisId: 'hypothesis-1',
+ disposition: 'accepted',
+ },
+ }),
+ })
+ )
+ );
+ });
+
+ it('writes the returned projection straight into the cache', async () => {
+ MockApiClient.addMockResponse({
+ url: orchestrationUrl,
+ body: InvestigationOrchestrationFixture({workflowVersion: 7}),
+ });
+ const updated: InvestigationOrchestration = InvestigationOrchestrationFixture({
+ workflowVersion: 8,
+ hypotheses: InvestigationOrchestrationFixture().hypotheses!.map(hypothesis =>
+ hypothesis.id === 'hypothesis-1'
+ ? {
+ ...hypothesis,
+ effectiveStatus: 'accepted' as const,
+ userDisposition: {disposition: 'accepted' as const, userId: 1},
+ }
+ : hypothesis
+ ),
+ });
+ MockApiClient.addMockResponse({
+ url: commandsUrl,
+ method: 'POST',
+ body: {
+ accepted: true,
+ duplicate: false,
+ requestId: 'request-1',
+ workflowVersion: 8,
+ commandStatus: 'accepted',
+ commandError: null,
+ runId: '9001',
+ projection: updated,
+ },
+ });
+
+ renderHypotheses();
+
+ await userEvent.click(
+ await screen.findByRole('button', {
+ name: 'Actions for Database or cache degradation delayed the response',
+ })
+ );
+ await userEvent.click(await screen.findByRole('menuitemradio', {name: 'Accept'}));
+
+ // No refetch is needed: the command response carries the new projection.
+ expect(await screen.findByText('Accepted · 86% Confidence')).toBeInTheDocument();
+ });
+});
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
new file mode 100644
index 000000000000..9c823504ef7e
--- /dev/null
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
@@ -0,0 +1,149 @@
+import {uuid4} from '@sentry/core';
+import {useQuery} from '@tanstack/react-query';
+
+import type {MenuItemProps} from 'sentry/components/dropdownMenu';
+import {t} from 'sentry/locale';
+import {useOrganization} from 'sentry/utils/useOrganization';
+import {
+ investigationOrchestrationQueryOptions,
+ useInvestigationOrchestrationCommandMutation,
+} from 'sentry/views/investigations/api';
+import {HypothesisList} from 'sentry/views/investigations/hypotheses/hypothesisList';
+import type {
+ InvestigationHypothesis,
+ InvestigationOrchestration,
+} from 'sentry/views/investigations/types';
+
+/** How often to re-read the projection while a workflow is still moving. */
+const POLL_INTERVAL_MS = 2000;
+
+/**
+ * Whether the workflow has stopped moving on its own. `awaiting_input` is
+ * deliberately not settled: the run resumes as soon as input arrives, which may
+ * happen from another surface, so polling has to continue.
+ */
+export function isInvestigationRunSettled(
+ projection: InvestigationOrchestration | undefined
+): boolean {
+ if (!projection) {
+ return false;
+ }
+ return (
+ projection.status === 'completed' ||
+ projection.status === 'failed' ||
+ projection.status === 'cancelled'
+ );
+}
+
+type InvestigationHypothesesProps = {
+ investigationId: string;
+ /**
+ * Set false for an investigation with no agentic run behind it — the
+ * orchestration endpoint 404s for those, and there is nothing to poll.
+ */
+ enabled?: boolean;
+};
+
+/**
+ * The hypothesis row for an agentic investigation, wired to the live run.
+ *
+ * This is the whole agent-to-frontend path in one place. Seer overwrites the
+ * projection on every orchestration event; Sentry stores it on
+ * `InvestigationOrchestrationRun.projection` and serves the latest one here. So
+ * the agent decides what these cards say purely by what it writes into
+ * `projection.hypotheses` — there is no separate signal telling the frontend to
+ * render a hypothesis, and no block kind to add.
+ *
+ * Actions travel back the other way as versioned commands, fenced on
+ * `workflowVersion` so a decision made against a stale view is rejected rather
+ * than applied to a run that has moved on.
+ */
+export function InvestigationHypotheses({
+ enabled = true,
+ investigationId,
+}: InvestigationHypothesesProps) {
+ const organization = useOrganization();
+ const {data: projection} = useQuery({
+ ...investigationOrchestrationQueryOptions(organization.slug, investigationId),
+ enabled,
+ refetchInterval: query =>
+ isInvestigationRunSettled(query.state.data?.json) ? false : POLL_INTERVAL_MS,
+ });
+
+ const commandMutation = useInvestigationOrchestrationCommandMutation(
+ organization.slug,
+ investigationId
+ );
+
+ if (!projection?.hypotheses?.length) {
+ return null;
+ }
+
+ const {workflowVersion} = projection;
+ const commandPending = commandMutation.isPending;
+
+ function setDisposition(
+ hypothesis: InvestigationHypothesis,
+ disposition: 'accepted' | 'rejected'
+ ) {
+ commandMutation.mutate({
+ requestId: uuid4(),
+ expectedWorkflowVersion: workflowVersion,
+ command: {
+ type: 'set_hypothesis_disposition',
+ hypothesisId: hypothesis.id,
+ // Choosing the decision the viewer already made clears it, so the same
+ // menu entry toggles rather than needing a separate "undo".
+ disposition:
+ hypothesis.decisionSource === 'user' &&
+ hypothesis.effectiveStatus === disposition
+ ? null
+ : disposition,
+ },
+ });
+ }
+
+ function getActions(hypothesis: InvestigationHypothesis): MenuItemProps[] {
+ const decidedByUser = hypothesis.decisionSource === 'user';
+
+ return [
+ {
+ key: 'accept',
+ label:
+ decidedByUser && hypothesis.effectiveStatus === 'accepted'
+ ? t('Clear decision')
+ : t('Accept'),
+ disabled: commandPending,
+ onAction: () => setDisposition(hypothesis, 'accepted'),
+ },
+ {
+ key: 'reject',
+ label:
+ decidedByUser && hypothesis.effectiveStatus === 'rejected'
+ ? t('Clear decision')
+ : t('Reject'),
+ disabled: commandPending,
+ onAction: () => setDisposition(hypothesis, 'rejected'),
+ },
+ {
+ key: 'retry',
+ label: t('Investigate again'),
+ disabled: commandPending || hypothesis.effectiveStatus === 'investigating',
+ onAction: () =>
+ commandMutation.mutate({
+ requestId: uuid4(),
+ expectedWorkflowVersion: workflowVersion,
+ command: {type: 'retry', target: 'hypothesis', targetId: hypothesis.id},
+ }),
+ },
+ ];
+ }
+
+ return (
+
+ );
+}
diff --git a/static/app/views/investigations/types.ts b/static/app/views/investigations/types.ts
index cae66a822266..c34c0262cb99 100644
--- a/static/app/views/investigations/types.ts
+++ b/static/app/views/investigations/types.ts
@@ -154,3 +154,284 @@ export type InvestigationCandidate =
| {status: 'investigate'}
| {status: 'unavailable'}
| {investigationId: string; status: 'view'};
+
+// The orchestration projection: the whole live state of an agentic run, which
+// Seer pushes as one JSON blob on every orchestration event and Sentry serves
+// from `/investigations/$investigationId/orchestration/`. The server contract
+// lives in `src/sentry/investigations/contracts.py` — keep these types in step
+// with it.
+//
+// Those serializers are deliberately relaxed: fields Sentry does not know about
+// pass straight through instead of failing validation, so Seer can ship a new
+// field ahead of a Sentry deploy. `InvestigationOrchestrationOpenString` encodes
+// the same tolerance for enum values — a status Sentry has never heard of still
+// typechecks, so a `switch` over one has to keep its default branch.
+
+/** A known set of string values that still accepts one Seer added later. */
+type InvestigationOrchestrationOpenString = T | (string & {});
+
+export type InvestigationOrchestrationPhase = InvestigationOrchestrationOpenString<
+ | 'intake'
+ | 'broad_scan'
+ | 'planning'
+ | 'investigating'
+ | 'judging'
+ | 'reporting'
+ | 'metadata'
+ | 'completed'
+ | 'failed'
+ | 'cancelled'
+>;
+
+export type InvestigationOrchestrationStatus = InvestigationOrchestrationOpenString<
+ 'pending' | 'processing' | 'awaiting_input' | 'completed' | 'failed' | 'cancelled'
+>;
+
+/** Lifecycle of one unit of agent work. Mirrors `WORK_STATUSES`. */
+export type InvestigationOrchestrationWorkStatus = InvestigationOrchestrationOpenString<
+ | 'not_started'
+ | 'queued'
+ | 'running'
+ | 'blocked'
+ | 'reauth_required'
+ | 'stalled'
+ | 'completed'
+ | 'failed'
+ | 'cancelled'
+>;
+
+/**
+ * What the UI shows for a hypothesis. The server folds the agent's verdict and
+ * the user's disposition into the run status to produce this, so it — not
+ * `status` — is what a hypothesis card renders.
+ * Mirrors `EFFECTIVE_HYPOTHESIS_STATUSES`.
+ */
+export type InvestigationHypothesisStatus = InvestigationOrchestrationOpenString<
+ | 'pending'
+ | 'investigating'
+ | 'supported'
+ | 'refuted'
+ | 'inconclusive'
+ | 'accepted'
+ | 'rejected'
+ | 'failed'
+ | 'cancelled'
+>;
+
+export type InvestigationOrchestrationError = {
+ code: string;
+ message: string;
+ retryable: boolean;
+ occurredAt?: string;
+ requestId?: string;
+ source?: string | null;
+};
+
+export type InvestigationToolActivity = {
+ id: string;
+ kind: InvestigationOrchestrationOpenString<'api' | 'library' | 'step' | 'tool'>;
+ status: InvestigationOrchestrationOpenString<
+ 'queued' | 'running' | 'completed' | 'failed'
+ >;
+ title: string;
+};
+
+export type InvestigationOrchestrationEvidence = {
+ data: Record;
+ id: string;
+ kind: InvestigationOrchestrationOpenString<
+ | 'issue'
+ | 'event'
+ | 'trace'
+ | 'profile'
+ | 'replay'
+ | 'query'
+ | 'chart'
+ | 'release'
+ | 'monitor'
+ | 'external'
+ | 'other'
+ >;
+ title: string;
+ reference?: string | null;
+ summary?: string | null;
+ url?: string | null;
+};
+
+/** One check the agent ran against a hypothesis — an "Evidence checked" row. */
+export type InvestigationVerificationStep = {
+ error: InvestigationOrchestrationError | null;
+ evidence: InvestigationOrchestrationEvidence[];
+ id: string;
+ method: string;
+ objective: string;
+ order: number;
+ result: string | null;
+ status: InvestigationOrchestrationWorkStatus;
+ title: string;
+};
+
+export type InvestigationAgentVerdict = {
+ confidence: number;
+ rationale: string;
+ refutingEvidenceIds: string[];
+ remainingGaps: string[];
+ supportingEvidenceIds: string[];
+ verdict: InvestigationOrchestrationOpenString<'supported' | 'refuted' | 'inconclusive'>;
+};
+
+export type InvestigationHypothesis = {
+ confidence: number | null;
+ decisionSource: InvestigationOrchestrationOpenString<'none' | 'agent' | 'user'>;
+ effectiveStatus: InvestigationHypothesisStatus;
+ error: InvestigationOrchestrationError | null;
+ evidence: InvestigationOrchestrationEvidence[];
+ id: string;
+ order: number;
+ rationale: string;
+ statement: string;
+ status: InvestigationOrchestrationWorkStatus;
+ verificationSteps: InvestigationVerificationStep[];
+ agentVerdict?: InvestigationAgentVerdict | null;
+ attempt?: number;
+ automaticRetryCount?: number;
+ heartbeatAt?: string | null;
+ investigatorRunId?: number | null;
+ toolActivity?: InvestigationToolActivity[];
+};
+
+export type InvestigationOrchestrationReport = {
+ currentBlockKey: string | null;
+ error: InvestigationOrchestrationError | null;
+ includedHypothesisIds: string[];
+ metadata: {
+ error: InvestigationOrchestrationError | null;
+ status: InvestigationOrchestrationOpenString<
+ 'not_started' | 'generating' | 'completed' | 'failed'
+ >;
+ summary: string | null;
+ summaryDescription: string | null;
+ title: string | null;
+ };
+ notebookRevision: number;
+ primaryHypothesisId: string | null;
+ revision: number;
+ status: InvestigationOrchestrationOpenString<
+ | 'not_started'
+ | 'waiting'
+ | 'composing'
+ | 'completed'
+ | 'partial_failed'
+ | 'failed'
+ | 'cancelled'
+ >;
+ automaticRetryCount?: number;
+ currentBlockStatus?: InvestigationOrchestrationWorkStatus | null;
+ currentBlockToolActivity?: InvestigationToolActivity[];
+ heartbeatAt?: string | null;
+ suggestedHypotheses?: Array<{
+ statement: string;
+ rationale?: string | null;
+ }>;
+};
+
+export type InvestigationOrchestration = {
+ broadScan: {
+ error: InvestigationOrchestrationError | null;
+ status: InvestigationOrchestrationWorkStatus;
+ summary: string | null;
+ attempt?: number;
+ automaticRetryCount?: number;
+ heartbeatAt?: string | null;
+ runId?: number | string | null;
+ toolActivity?: InvestigationToolActivity[];
+ };
+ errors: InvestigationOrchestrationError[];
+ generation: number;
+ heartbeatAt: string | null;
+ hypotheses: InvestigationHypothesis[];
+ investigationId: string;
+ notebookRevision: number;
+ phase: InvestigationOrchestrationPhase;
+ report: InvestigationOrchestrationReport;
+ runId: string | null;
+ sourceType: InvestigationOrchestrationOpenString<'manual' | 'breached_metric'>;
+ status: InvestigationOrchestrationStatus;
+ updatedAt: string;
+ workflowVersion: number;
+ pendingInput?: {
+ missingFields: Array<'prompt' | 'time_range'>;
+ prompt: string;
+ } | null;
+ steeringIntents?: Array<{
+ createdAt: string;
+ id: string;
+ instruction: string;
+ requestId: string;
+ target: InvestigationOrchestrationOpenString<
+ 'workflow' | 'hypothesis' | 'report' | 'block'
+ >;
+ targetId: string | null;
+ }>;
+};
+
+/**
+ * A viewer-issued instruction to a running workflow. Mirrors
+ * `COMMAND_VALIDATORS`. The inner payload stays camelCase because the server
+ * only case-converts the envelope keys.
+ */
+export type InvestigationOrchestrationCommand =
+ | {
+ type: 'provide_input';
+ prompt?: string;
+ timeRange?: {end: string; start: string};
+ }
+ | {
+ statement: string;
+ type: 'add_hypothesis';
+ rationale?: string | null;
+ }
+ | {
+ disposition: 'accepted' | 'rejected' | null;
+ hypothesisId: string;
+ type: 'set_hypothesis_disposition';
+ }
+ | {
+ instruction: string;
+ target: 'workflow' | 'hypothesis' | 'report' | 'block';
+ type: 'steer';
+ targetId?: string | null;
+ }
+ | {
+ target: 'run' | 'hypothesis' | 'report';
+ type: 'retry';
+ targetId?: string | null;
+ }
+ | {
+ type: 'cancel';
+ reason?: string | null;
+ };
+
+export type InvestigationOrchestrationCommandVariables = {
+ command: InvestigationOrchestrationCommand;
+ /**
+ * The version the caller believes it is acting on. The server rejects the
+ * command with a 409 if the workflow has moved on, so pass the version from
+ * the projection this command was composed against.
+ */
+ expectedWorkflowVersion: number;
+ /**
+ * Idempotency key. A repeat of the same id is accepted and reported back as a
+ * duplicate rather than applied twice.
+ */
+ requestId: string;
+};
+
+export type InvestigationOrchestrationCommandResponse = {
+ accepted: boolean;
+ duplicate: boolean;
+ projection: InvestigationOrchestration;
+ requestId: string;
+ runId: string | null;
+ workflowVersion: number;
+};
From ab819b767947c88d6bea6beab1b6bce714e5e759 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Thu, 10 Sep 2026 14:47:11 -0400
Subject: [PATCH 02/21] fix(investigations): Give hypothesis cards
verdict-driven borders
Four fixes from design review of the hypothesis row.
The evidence list is a `ul` nested inside the row's own `ul`, so browsers
gave it `list-style-type: circle` and drew a marker beside every step,
sitting in the card's padding. Both lists now clear their markers.
The card border carries the verdict rather than only marking the report's
primary hypothesis: accent for supported or user-accepted, dotted while
inconclusive, ordinary otherwise -- refuted included, since ruling
something out is a result rather than a fault. `getBorder` only ever emits
`1px solid`, so the dotted case is CSS; the colors are still border tokens.
It is driven by a data attribute, which also makes it assertable, unlike
emotion styles, which this repo's stubbed getComputedStyle cannot see.
Stories now sit inside `Storybook.Demo`. That excludes their headings from
the page's table of contents, which was listing every hypothesis statement
and, because the same three statements repeat across stories, giving
several entries the same id and marking them all active at once. The demo
also supplies the container query context these cards size against.
The three hardcoded-width reflow stories are gone; `Demo resizable` gives a
drag handle and a live breakpoint readout instead.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../investigationFixtureApi.spec.tsx | 5 +
.../hypotheses/hypotheses.stories.tsx | 265 ++++++++----------
.../hypotheses/hypothesisCard.spec.tsx | 21 ++
.../hypotheses/hypothesisCard.tsx | 41 ++-
.../hypotheses/hypothesisStatus.tsx | 22 ++
5 files changed, 203 insertions(+), 151 deletions(-)
diff --git a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
index f8ed03dbe307..3f80f19b3e82 100644
--- a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
+++ b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
@@ -207,6 +207,11 @@ describe('InvestigationFixtureApi', () => {
expect(
await screen.findByText('Accepted by you · 91% Confidence')
).toBeInTheDocument();
+ // Accepting settles the hypothesis, so its edge picks up the accent.
+ expect(screen.getAllByTestId('investigation-hypothesis')[1]).toHaveAttribute(
+ 'data-border',
+ 'accent'
+ );
});
it('clears a disposition back to the agent verdict', async () => {
diff --git a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
index 74f2f665a283..62d758f301f1 100644
--- a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
+++ b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
@@ -1,8 +1,5 @@
import {Fragment} from 'react';
-import {Container, Stack} from '@sentry/scraps/layout';
-import {Text} from '@sentry/scraps/text';
-
import * as Storybook from 'sentry/stories';
import {InvestigationFixtureApi} from 'sentry/views/investigations/__stories__/investigationFixtureApi';
import {
@@ -25,49 +22,21 @@ export default Storybook.story('Investigations — Hypotheses', story => {
The data comes from projection.hypotheses on the orchestration
- endpoint, and the highlighted card is report.primaryHypothesisId.
+ endpoint, and the lifted card is report.primaryHypothesisId.
-
-
- ));
-
- story('Reflows on its container, not the viewport', () => (
-
The row is a grid of minmax(260px, 1fr) tracks with{' '}
auto-fit, so it drops columns whenever its own box stops fitting
- another readable card. Nothing about the viewport is consulted, which is what lets
- the same component sit in a full-width detail view and in a narrow drawer.
+ another readable card — the viewport is never consulted. Drag the demo's edge to
+ watch it reflow.
- {(
- [
- {label: 'Wide — three columns', width: '900px'},
- {label: 'Medium — two columns', width: '600px'},
- {label: 'Narrow — one column', width: '340px'},
- ] as const
- ).map(({label, width}) => (
-
-
- {label}
-
-
-
-
-
- ))}
-
+
+
+
+
));
story('Statuses', () => (
@@ -78,74 +47,88 @@ export default Storybook.story('Investigations — Hypotheses', story => {
agent has settled on a verdict, so work in progress shows a bare label and a
pulsing dot.
-
+
+ The border carries the verdict: accent for a supported hypothesis, dotted while a
+ hypothesis is inconclusive, and an ordinary border everywhere else — refuted
+ included, since ruling something out is a result rather than a fault.
+
+
+
+
+
+ Nothing has settled yet in these, so none of them carry confidence and the failed
+ hypothesis shows why it stopped.
+
Accept or reject a hypothesis from its overflow menu: the fixture applies the
command, bumps workflowVersion, and returns the new projection, which
- the mutation writes straight into the query cache. Choosing the same decision
- twice clears it and hands the hypothesis back to the agent's verdict.
+ the mutation writes straight into the query cache. Accepting turns the border
+ accent; choosing the same decision twice clears it and hands the hypothesis back
+ to the agent's verdict.
Cards do not own commands. The surface rendering them decides which of accept,
reject, steer, and retry apply, and posts the chosen one to{' '}
- /orchestration/commands/. Leave getActions off for a
- read-only surface.
+ /orchestration/commands/. Leave getActions off — as the
+ rows above do — and the overflow menu disappears, which is what a read-only
+ surface wants.
- [
- {
- key: 'accept',
- label: 'Accept',
- onAction: () => {},
- },
- {
- key: 'reject',
- label: 'Reject',
- onAction: () => {},
- },
- {
- key: 'retry',
- label: 'Investigate again',
- disabled: hypothesis.effectiveStatus === 'investigating',
- onAction: () => {},
- },
- ]}
- />
+
+ [
+ {key: 'accept', label: 'Accept', onAction: () => {}},
+ {key: 'reject', label: 'Reject', onAction: () => {}},
+ {
+ key: 'retry',
+ label: 'Investigate again',
+ disabled: hypothesis.effectiveStatus === 'investigating',
+ onAction: () => {},
+ },
+ ]}
+ />
+
));
});
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
index ac1343e92fd2..2fade1b975fc 100644
--- a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
@@ -163,6 +163,27 @@ describe('HypothesisCard', () => {
).toBeInTheDocument();
});
+ it.each([
+ ['supported', 'accent'],
+ ['accepted', 'accent'],
+ ['inconclusive', 'dotted'],
+ // Ruling a hypothesis out is a real outcome, so it gets an ordinary border
+ // rather than one that reads as a fault.
+ ['refuted', 'default'],
+ ['rejected', 'default'],
+ ['investigating', 'default'],
+ ['failed', 'default'],
+ ] as const)('draws a %s hypothesis with a %s border', (effectiveStatus, border) => {
+ render(
+
+ );
+
+ expect(screen.getByTestId('investigation-hypothesis')).toHaveAttribute(
+ 'data-border',
+ border
+ );
+ });
+
it('hides the evidence section when there are no steps', () => {
render(
@@ -108,11 +110,11 @@ export function HypothesisCard({
{t('Evidence checked')}
-
+
{steps.map(step => (
))}
-
+
) : null}
@@ -144,14 +146,35 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
);
}
-// The accent border alone is easy to miss against a wall of cards, so the
-// primary hypothesis also lifts off the page. Driven by a data attribute rather
-// than a styled prop: `Stack` forwards every prop it does not recognize to the
-// DOM, and a bare `isPrimary` would land there as an unknown attribute.
+/**
+ * The card border carries the verdict, which is why it is CSS rather than the
+ * `border` prop: `getBorder` only ever emits `1px solid`, and an unsettled
+ * hypothesis needs a dotted edge. The colors still come from border tokens.
+ *
+ * Both variants are driven by data attributes because `Stack` forwards props it
+ * does not recognize to the DOM, where a bare `isPrimary` would land as an
+ * unknown attribute.
+ */
const Card = styled(Stack)`
list-style: none;
+ border: 1px solid ${p => p.theme.tokens.border.primary};
+
+ /* Supported, or endorsed by a person: the explanation the evidence backs. */
+ &[data-border='accent'] {
+ border-color: ${p => p.theme.tokens.border.accent.vibrant};
+ }
+
+ /* Checked, but not settled either way. The broken edge reads as unfinished. */
+ &[data-border='dotted'] {
+ border-style: dotted;
+ }
&[data-primary='true'] {
box-shadow: ${p => p.theme.shadow.low};
}
`;
+
+// `ul` markers would otherwise sit in the card's padding next to each step.
+const EvidenceList = styled(Stack)`
+ list-style: none;
+`;
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
index 4ae214cdd497..5d59fff297ba 100644
--- a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -107,6 +107,28 @@ export function getHypothesisStatusVariant(
}
}
+/**
+ * How a card's edge should be drawn for a given verdict.
+ *
+ * - `accent` — supported, or endorsed by a person. The purple border marks the
+ * explanation the evidence backs.
+ * - `dotted` — inconclusive. Checked, but not settled either way, so the edge
+ * reads as unfinished rather than as a result.
+ * - `default` — everything else, including refuted. Ruling a hypothesis out is
+ * a real outcome, so it gets an ordinary border rather than a warning color.
+ */
+export function getHypothesisCardBorder(
+ status: InvestigationHypothesisStatus
+): 'accent' | 'dotted' | 'default' {
+ if (status === 'supported' || status === 'accepted') {
+ return 'accent';
+ }
+ if (status === 'inconclusive') {
+ return 'dotted';
+ }
+ return 'default';
+}
+
/**
* What a verification step says about itself while it has no result yet. A step
* only carries a `result` once it has finished, so everything short of that
From 6eac7f8440b28a5acd8cc935e96d947ffcea8643 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Thu, 10 Sep 2026 15:09:58 -0400
Subject: [PATCH 03/21] fix(investigations): Dash every hypothesis that is not
the answer
A dotted hairline is barely visible against the default border color, so
the broken edge is dashed now.
It also applies more widely. The border had three states; it has two. A
solid accent edge marks the explanation that stands -- supported by the
evidence, or accepted by a person -- and every other card is dashed.
Investigating, refuted, inconclusive, cancelled and failed all mean the
same thing to someone scanning the row: not the answer. The status line
already carries which of them it is, so the border does not need to.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../investigationFixtureApi.spec.tsx | 5 +++++
.../hypotheses/hypotheses.stories.tsx | 8 ++++---
.../hypotheses/hypothesisCard.spec.tsx | 17 ++++++++------
.../hypotheses/hypothesisCard.tsx | 15 ++++++++-----
.../hypotheses/hypothesisStatus.tsx | 22 +++++++------------
5 files changed, 37 insertions(+), 30 deletions(-)
diff --git a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
index 3f80f19b3e82..6d24fcf01b4a 100644
--- a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
+++ b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
@@ -230,6 +230,11 @@ describe('InvestigationFixtureApi', () => {
);
expect(await screen.findByText('Refuted · 91% Confidence')).toBeInTheDocument();
+ // Back to the agent's verdict, so the edge breaks again.
+ expect(screen.getAllByTestId('investigation-hypothesis')[1]).toHaveAttribute(
+ 'data-border',
+ 'dashed'
+ );
});
it('puts a retried hypothesis back into investigation', async () => {
diff --git a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
index 62d758f301f1..9fbe8e353c79 100644
--- a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
+++ b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
@@ -48,9 +48,11 @@ export default Storybook.story('Investigations — Hypotheses', story => {
pulsing dot.
- The border carries the verdict: accent for a supported hypothesis, dotted while a
- hypothesis is inconclusive, and an ordinary border everywhere else — refuted
- included, since ruling something out is a result rather than a fault.
+ The border carries the verdict, and only two ways: a solid accent edge on the
+ explanation that stands — supported by the evidence, or accepted by a person — and
+ a dashed edge on every other card. Still running, ruled out, inconclusive and
+ failed all read the same way to someone scanning the row, so the status line
+ carries the distinction rather than the border.
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
index 2fade1b975fc..c43fed9c1b82 100644
--- a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
@@ -164,15 +164,18 @@ describe('HypothesisCard', () => {
});
it.each([
+ // Only an explanation that stands gets the solid accent edge.
['supported', 'accent'],
['accepted', 'accent'],
- ['inconclusive', 'dotted'],
- // Ruling a hypothesis out is a real outcome, so it gets an ordinary border
- // rather than one that reads as a fault.
- ['refuted', 'default'],
- ['rejected', 'default'],
- ['investigating', 'default'],
- ['failed', 'default'],
+ // Everything else reads the same to someone scanning the row: not the
+ // answer, whether that is because it is unfinished or because it lost.
+ ['inconclusive', 'dashed'],
+ ['refuted', 'dashed'],
+ ['rejected', 'dashed'],
+ ['investigating', 'dashed'],
+ ['pending', 'dashed'],
+ ['failed', 'dashed'],
+ ['cancelled', 'dashed'],
] as const)('draws a %s hypothesis with a %s border', (effectiveStatus, border) => {
render(
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
index d6c01ce9f43c..45462093b05c 100644
--- a/static/app/views/investigations/hypotheses/hypothesisCard.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
@@ -148,8 +148,9 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
/**
* The card border carries the verdict, which is why it is CSS rather than the
- * `border` prop: `getBorder` only ever emits `1px solid`, and an unsettled
- * hypothesis needs a dotted edge. The colors still come from border tokens.
+ * `border` prop: `getBorder` only ever emits `1px solid`, and a hypothesis that
+ * has not been established needs a broken edge. The colors still come from
+ * border tokens.
*
* Both variants are driven by data attributes because `Stack` forwards props it
* does not recognize to the DOM, where a bare `isPrimary` would land as an
@@ -159,14 +160,16 @@ const Card = styled(Stack)`
list-style: none;
border: 1px solid ${p => p.theme.tokens.border.primary};
- /* Supported, or endorsed by a person: the explanation the evidence backs. */
+ /* The explanation that stands. */
&[data-border='accent'] {
border-color: ${p => p.theme.tokens.border.accent.vibrant};
}
- /* Checked, but not settled either way. The broken edge reads as unfinished. */
- &[data-border='dotted'] {
- border-style: dotted;
+ /* Not the answer: still running, ruled out, inconclusive, or failed. Dashed
+ * rather than dotted because a dotted hairline all but disappears at this
+ * border color. */
+ &[data-border='dashed'] {
+ border-style: dashed;
}
&[data-primary='true'] {
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
index 5d59fff297ba..871b5c212dbe 100644
--- a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -110,23 +110,17 @@ export function getHypothesisStatusVariant(
/**
* How a card's edge should be drawn for a given verdict.
*
- * - `accent` — supported, or endorsed by a person. The purple border marks the
- * explanation the evidence backs.
- * - `dotted` — inconclusive. Checked, but not settled either way, so the edge
- * reads as unfinished rather than as a result.
- * - `default` — everything else, including refuted. Ruling a hypothesis out is
- * a real outcome, so it gets an ordinary border rather than a warning color.
+ * - `accent` — the explanation that stands: supported by the evidence, or
+ * endorsed by a person. A solid purple edge means "this is the answer".
+ * - `dashed` — everything else. A hypothesis still being investigated, ruled
+ * out, inconclusive, or failed is all the same thing to a reader scanning the
+ * row: not the answer. One broken edge says that without needing a colour per
+ * status, which the status line already carries.
*/
export function getHypothesisCardBorder(
status: InvestigationHypothesisStatus
-): 'accent' | 'dotted' | 'default' {
- if (status === 'supported' || status === 'accepted') {
- return 'accent';
- }
- if (status === 'inconclusive') {
- return 'dotted';
- }
- return 'default';
+): 'accent' | 'dashed' {
+ return status === 'supported' || status === 'accepted' ? 'accent' : 'dashed';
}
/**
From 605046de7401690d7b675940f9eabd74282012e4 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Thu, 10 Sep 2026 15:15:03 -0400
Subject: [PATCH 04/21] ref(investigations): Stop exporting hypothesis
internals
Knip flagged fourteen exports nothing outside their own module reads.
Thirteen were module internals -- the status vocabulary behind
`HypothesisStatus`, the polling predicate behind the orchestration query,
and the projection types composed into `InvestigationOrchestration` -- so
they lose the `export` keyword and keep working. Anything that needs one
later can export it then.
`isInvestigationOrchestrationConflictError` was speculative: added to match
the in-flight orchestration branch, called by nothing here. Deleted; it
arrives with the hook that uses it.
`knip --production` separately reports the hypothesis components as
unreachable, which is accurate -- no investigation surface renders the row
yet, only stories and tests do. That is the same situation the config
already records for `autofixChatContext` and the chat blocks, so it gets
the same treatment: one entry point with a TODO. Wiring the row into the
detail view belongs with the surface work, not here, and that file is
being rewritten on #122949.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
knip.config.ts | 3 +++
static/app/views/investigations/api.ts | 19 ++-----------------
.../hypotheses/hypothesisStatus.tsx | 10 +++++-----
.../hypotheses/investigationHypotheses.tsx | 2 +-
static/app/views/investigations/types.ts | 14 +++++++-------
5 files changed, 18 insertions(+), 30 deletions(-)
diff --git a/knip.config.ts b/knip.config.ts
index 8bad3af97f9f..e549a2705122 100644
--- a/knip.config.ts
+++ b/knip.config.ts
@@ -24,6 +24,9 @@ const productionEntryPoints = [
'static/app/chartcuterie/**/*.{js,ts,tsx}',
// TODO: Remove when the autofixRef embed consumes it (#122099)
'static/app/components/seer/autofixChatContext.tsx',
+ // Pulls in the whole hypotheses/ directory.
+ // TODO: Remove when an investigation surface renders it (#124086)
+ 'static/app/views/investigations/hypotheses/investigationHypotheses.tsx',
'static/app/components/brandPageLayout/**/*.{ts,tsx}',
// React authentication routes are discovered dynamically by the frontend route registry
'static/app/views/authV2/authLogin/**/*.{ts,tsx}',
diff --git a/static/app/views/investigations/api.ts b/static/app/views/investigations/api.ts
index 46a870acbec0..016d617510de 100644
--- a/static/app/views/investigations/api.ts
+++ b/static/app/views/investigations/api.ts
@@ -106,8 +106,8 @@ export function investigationTitleGenerationQueryOptions(
* so there is nothing to merge — the newest response wins outright.
*
* `staleTime: 0` because a running workflow changes constantly. Callers that
- * render a run in progress should add a `refetchInterval`; use
- * `isInvestigationRunSettled` to stop polling once it reaches a terminal state.
+ * render a run in progress should add a `refetchInterval` and drop it once
+ * `status` reaches a terminal value, as `InvestigationHypotheses` does.
*/
export function investigationOrchestrationQueryOptions(
organizationSlug: string,
@@ -125,21 +125,6 @@ export function investigationOrchestrationQueryOptions(
);
}
-/**
- * Whether a rejected command lost a race rather than being malformed. The
- * server answers 409 both when the workflow version has moved on and when an
- * idempotency key is reused, and either way the fix is to re-read the
- * projection rather than to show a hard failure.
- */
-export function isInvestigationOrchestrationConflictError(error: unknown): boolean {
- return (
- typeof error === 'object' &&
- error !== null &&
- 'status' in error &&
- (error as {status?: unknown}).status === 409
- );
-}
-
/**
* Send a viewer command — accept/reject a hypothesis, steer, retry, cancel — to
* a running workflow.
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
index 871b5c212dbe..5574044cb758 100644
--- a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -26,11 +26,11 @@ const SETTLED_STATUSES = new Set([
/** Statuses where the agent is still working, so the dot keeps pulsing. */
const IN_FLIGHT_STATUSES = new Set(['pending', 'investigating']);
-export function isHypothesisSettled(status: InvestigationHypothesisStatus): boolean {
+function isHypothesisSettled(status: InvestigationHypothesisStatus): boolean {
return SETTLED_STATUSES.has(status);
}
-export function isHypothesisInFlight(status: InvestigationHypothesisStatus): boolean {
+function isHypothesisInFlight(status: InvestigationHypothesisStatus): boolean {
return IN_FLIGHT_STATUSES.has(status);
}
@@ -43,7 +43,7 @@ function humanize(status: string): string {
return status.replaceAll('_', ' ').replace(/^./, character => character.toUpperCase());
}
-export function getHypothesisStatusLabel(hypothesis: InvestigationHypothesis): string {
+function getHypothesisStatusLabel(hypothesis: InvestigationHypothesis): string {
// A decision the viewer made themselves reads differently from one the agent
// reached, even though both land in `effectiveStatus`.
if (hypothesis.decisionSource === 'user') {
@@ -79,7 +79,7 @@ export function getHypothesisStatusLabel(hypothesis: InvestigationHypothesis): s
}
}
-export function getHypothesisStatusVariant(
+function getHypothesisStatusVariant(
status: InvestigationHypothesisStatus
): StatusVariant {
switch (status) {
@@ -165,7 +165,7 @@ export function getVerificationStepStatusLabel(
* kept in step with `effectiveStatus`, so it wins; the verdict is the fallback
* for a projection that has only filled the latter in.
*/
-export function getHypothesisConfidencePercent(
+function getHypothesisConfidencePercent(
hypothesis: InvestigationHypothesis
): number | null {
if (!isHypothesisSettled(hypothesis.effectiveStatus)) {
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
index 9c823504ef7e..2e3169c88928 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
@@ -22,7 +22,7 @@ const POLL_INTERVAL_MS = 2000;
* deliberately not settled: the run resumes as soon as input arrives, which may
* happen from another surface, so polling has to continue.
*/
-export function isInvestigationRunSettled(
+function isInvestigationRunSettled(
projection: InvestigationOrchestration | undefined
): boolean {
if (!projection) {
diff --git a/static/app/views/investigations/types.ts b/static/app/views/investigations/types.ts
index c34c0262cb99..6455ac59b1b8 100644
--- a/static/app/views/investigations/types.ts
+++ b/static/app/views/investigations/types.ts
@@ -170,7 +170,7 @@ export type InvestigationCandidate =
/** A known set of string values that still accepts one Seer added later. */
type InvestigationOrchestrationOpenString = T | (string & {});
-export type InvestigationOrchestrationPhase = InvestigationOrchestrationOpenString<
+type InvestigationOrchestrationPhase = InvestigationOrchestrationOpenString<
| 'intake'
| 'broad_scan'
| 'planning'
@@ -183,7 +183,7 @@ export type InvestigationOrchestrationPhase = InvestigationOrchestrationOpenStri
| 'cancelled'
>;
-export type InvestigationOrchestrationStatus = InvestigationOrchestrationOpenString<
+type InvestigationOrchestrationStatus = InvestigationOrchestrationOpenString<
'pending' | 'processing' | 'awaiting_input' | 'completed' | 'failed' | 'cancelled'
>;
@@ -218,7 +218,7 @@ export type InvestigationHypothesisStatus = InvestigationOrchestrationOpenString
| 'cancelled'
>;
-export type InvestigationOrchestrationError = {
+type InvestigationOrchestrationError = {
code: string;
message: string;
retryable: boolean;
@@ -227,7 +227,7 @@ export type InvestigationOrchestrationError = {
source?: string | null;
};
-export type InvestigationToolActivity = {
+type InvestigationToolActivity = {
id: string;
kind: InvestigationOrchestrationOpenString<'api' | 'library' | 'step' | 'tool'>;
status: InvestigationOrchestrationOpenString<
@@ -236,7 +236,7 @@ export type InvestigationToolActivity = {
title: string;
};
-export type InvestigationOrchestrationEvidence = {
+type InvestigationOrchestrationEvidence = {
data: Record;
id: string;
kind: InvestigationOrchestrationOpenString<
@@ -271,7 +271,7 @@ export type InvestigationVerificationStep = {
title: string;
};
-export type InvestigationAgentVerdict = {
+type InvestigationAgentVerdict = {
confidence: number;
rationale: string;
refutingEvidenceIds: string[];
@@ -300,7 +300,7 @@ export type InvestigationHypothesis = {
toolActivity?: InvestigationToolActivity[];
};
-export type InvestigationOrchestrationReport = {
+type InvestigationOrchestrationReport = {
currentBlockKey: string | null;
error: InvestigationOrchestrationError | null;
includedHypothesisIds: string[];
From 62da57ddd38ffc732f44a77a248636aacdf2a8de Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Thu, 10 Sep 2026 15:18:25 -0400
Subject: [PATCH 05/21] style(investigations): Match the hypothesis card spec
Measured against the mock, every value from a token.
The evidence rows sat on a grey surface; in the spec they sit on the card
and are separated by their border alone, so they move to the primary
background. The rationale was muted, which put it in the same register as
the evidence results below it; the spec reads it as body copy, so it takes
the primary content colour and only the results stay muted.
Spacing was uniformly tighter than the spec: card padding lg -> xl, card
gap md -> lg, evidence rows sm/md -> md/lg with a sm gap between them, and
the gap between cards md -> xl, which now matches the card's own padding.
Type sizes and weights were already right and are unchanged. Worth noting
the scale tops out at 500 -- there is no bolder weight token -- so the
statement is as heavy as the system goes.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../investigations/hypotheses/hypothesisCard.tsx | 12 ++++++------
.../investigations/hypotheses/hypothesisList.tsx | 2 +-
2 files changed, 7 insertions(+), 7 deletions(-)
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
index 45462093b05c..b0ff155d3308 100644
--- a/static/app/views/investigations/hypotheses/hypothesisCard.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
@@ -56,8 +56,8 @@ export function HypothesisCard({
{hypothesis.rationale ? (
-
+
{hypothesis.rationale}
) : null}
@@ -110,7 +110,7 @@ export function HypothesisCard({
{t('Evidence checked')}
-
+
{steps.map(step => (
))}
@@ -133,8 +133,8 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
as="li"
border={failed ? 'danger' : 'primary'}
radius="sm"
- padding="sm md"
- background="secondary"
+ padding="md lg"
+ background="primary"
>
{step.title}
diff --git a/static/app/views/investigations/hypotheses/hypothesisList.tsx b/static/app/views/investigations/hypotheses/hypothesisList.tsx
index 8cbddf802f8d..f9bafa7b6ffd 100644
--- a/static/app/views/investigations/hypotheses/hypothesisList.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisList.tsx
@@ -49,7 +49,7 @@ export function HypothesisList({
as="ul"
className={className}
columns={`repeat(auto-fit, minmax(${MIN_CARD_WIDTH}, 1fr))`}
- gap="md"
+ gap="xl"
align="start"
padding="0"
data-test-id="investigation-hypotheses"
From c720365a4eeb84ea52bd4a6eee9d90256c86cf48 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Fri, 11 Sep 2026 15:18:36 -0400
Subject: [PATCH 06/21] feat(investigations): Name the states a hypothesis
passes through
Taken from the prototype recording, which shows the row moving through
states the card had been collapsing into one.
A hypothesis in flight is a single effectiveStatus, but the recording names
four moments inside it: Formed, Preparing checks, Checking, and Evidence
checked. The distinction only exists in the verification steps -- whether
any are planned, and whether they have produced anything -- so that is
where it is read from. Only Checking is coloured and keeps its dot moving;
the others are staging posts, not outcomes, and the old code lit all of
them accent. The heading above the steps moves with them, from "Evidence to
check" to "Evidence checked", and a step with nothing yet reads "Awaiting
evidence" rather than naming its queue position.
A step that has produced something now opens, revealing the objective and
method behind it -- fields the projection has always carried and the card
never showed. A step with no result stays a bare row, because a chevron
there would promise a finding that does not exist yet.
Accepted and rejected keep their own labels rather than folding into
supported and refuted, so a decision is never misread as a verdict.
Confidence is lowercase, matching the recording.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../investigationFixtureApi.spec.tsx | 12 +-
.../hypotheses/hypotheses.stories.tsx | 135 ++++++++++++-----
.../hypotheses/hypothesisCard.spec.tsx | 55 ++++++-
.../hypotheses/hypothesisCard.tsx | 50 ++++++-
.../hypotheses/hypothesisStatus.tsx | 140 ++++++++++--------
.../investigationHypotheses.spec.tsx | 9 +-
6 files changed, 273 insertions(+), 128 deletions(-)
diff --git a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
index 6d24fcf01b4a..2a5a8658e29b 100644
--- a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
+++ b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
@@ -186,7 +186,7 @@ describe('InvestigationFixtureApi', () => {
name: 'Database or cache degradation delayed the response',
})
).toBeInTheDocument();
- expect(screen.getByText('Supported · 86% Confidence')).toBeInTheDocument();
+ expect(screen.getByText('Supported · 86% confidence')).toBeInTheDocument();
expect(
screen.getByText('The delay begins before the document reaches the browser.')
).toBeInTheDocument();
@@ -205,7 +205,7 @@ describe('InvestigationFixtureApi', () => {
// The command response carries the updated projection, so the card
// changes without another read.
expect(
- await screen.findByText('Accepted by you · 91% Confidence')
+ await screen.findByText('Accepted by you · 91% confidence')
).toBeInTheDocument();
// Accepting settles the hypothesis, so its edge picks up the accent.
expect(screen.getAllByTestId('investigation-hypothesis')[1]).toHaveAttribute(
@@ -222,14 +222,14 @@ describe('InvestigationFixtureApi', () => {
});
await userEvent.click(trigger);
await userEvent.click(await screen.findByRole('menuitemradio', {name: 'Accept'}));
- await screen.findByText('Accepted by you · 91% Confidence');
+ await screen.findByText('Accepted by you · 91% confidence');
await userEvent.click(trigger);
await userEvent.click(
await screen.findByRole('menuitemradio', {name: 'Clear decision'})
);
- expect(await screen.findByText('Refuted · 91% Confidence')).toBeInTheDocument();
+ expect(await screen.findByText('Refuted · 91% confidence')).toBeInTheDocument();
// Back to the agent's verdict, so the edge breaks again.
expect(screen.getAllByTestId('investigation-hypothesis')[1]).toHaveAttribute(
'data-border',
@@ -249,8 +249,8 @@ describe('InvestigationFixtureApi', () => {
await screen.findByRole('menuitemradio', {name: 'Investigate again'})
);
- expect(await screen.findByText('Investigating')).toBeInTheDocument();
- expect(screen.getAllByText('Queued.').length).toBeGreaterThan(0);
+ expect(await screen.findByText('Checking')).toBeInTheDocument();
+ expect(screen.getAllByText('Awaiting evidence').length).toBeGreaterThan(0);
});
});
});
diff --git a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
index 9fbe8e353c79..fbd711baa07b 100644
--- a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
+++ b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
@@ -44,8 +44,7 @@ export default Storybook.story('Investigations — Hypotheses', story => {
A card renders effectiveStatus, which already folds the agent verdict
and any user disposition into the run status. Confidence only appears once the
- agent has settled on a verdict, so work in progress shows a bare label and a
- pulsing dot.
+ agent has settled on a verdict.
The border carries the verdict, and only two ways: a solid accent edge on the
@@ -58,51 +57,27 @@ export default Storybook.story('Investigations — Hypotheses', story => {
- Nothing has settled yet in these, so none of them carry confidence and the failed
- hypothesis shows why it stopped.
+ A hypothesis in flight is all one effectiveStatus, but it passes
+ through several states worth naming: formed, having its checks planned, running
+ them, and done checking but not yet judged. Those are read off the verification
+ steps, since that is the only place the distinction exists. Only the running state
+ is coloured and keeps its dot moving — the rest are staging posts, not outcomes.
+ The heading over the steps moves with them, from "Evidence to check" to "Evidence
+ checked".
+
+
+
+
+
+ A failure is the one in-flight state that gets a colour, because it is the only
+ one that has stopped. The hypothesis says why, and so does each check that broke.
{
));
});
+
+/** The four states a hypothesis passes through before it is judged. */
+function inFlightHypotheses() {
+ return [
+ InvestigationHypothesisFixture({
+ id: 'formed',
+ order: 0,
+ statement: 'A slow dependency upgrade changed request timing',
+ rationale: 'Nothing has been planned to test this yet.',
+ status: 'queued',
+ effectiveStatus: 'pending',
+ confidence: null,
+ agentVerdict: null,
+ verificationSteps: [],
+ }),
+ InvestigationHypothesisFixture({
+ id: 'preparing',
+ order: 1,
+ statement: 'A cache warm-up left the first requests cold',
+ rationale: 'The checks are planned but none has started.',
+ status: 'queued',
+ effectiveStatus: 'investigating',
+ confidence: null,
+ agentVerdict: null,
+ verificationSteps: [
+ InvestigationVerificationStepFixture({
+ id: 'preparing-step',
+ title: 'Compare cold and warm cache windows',
+ status: 'queued',
+ result: null,
+ }),
+ ],
+ }),
+ InvestigationHypothesisFixture({
+ id: 'checking',
+ order: 2,
+ statement: 'A noisy neighbour saturated the shared pool',
+ rationale: 'One check is running; the rest are queued behind it.',
+ status: 'running',
+ effectiveStatus: 'investigating',
+ confidence: null,
+ agentVerdict: null,
+ verificationSteps: [
+ InvestigationVerificationStepFixture({
+ id: 'checking-step',
+ title: 'Compare pool saturation across tenants',
+ status: 'running',
+ result: null,
+ }),
+ InvestigationVerificationStepFixture({
+ id: 'checking-queued',
+ order: 1,
+ title: 'Inspect connection wait time',
+ status: 'queued',
+ result: null,
+ }),
+ ],
+ }),
+ InvestigationHypothesisFixture({
+ id: 'checked',
+ order: 3,
+ statement: 'A retry storm amplified the original delay',
+ rationale: 'Every check has reported; the verdict has not landed yet.',
+ status: 'running',
+ effectiveStatus: 'investigating',
+ confidence: null,
+ agentVerdict: null,
+ verificationSteps: [
+ InvestigationVerificationStepFixture({
+ id: 'checked-step',
+ title: 'Compare retry volume with latency',
+ status: 'completed',
+ result: 'Retries tripled while the p95 climbed.',
+ }),
+ ],
+ }),
+ ];
+}
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
index c43fed9c1b82..1a41730fa007 100644
--- a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
@@ -38,7 +38,7 @@ describe('HypothesisCard', () => {
/>
);
- expect(screen.getByText('Supported · 86% Confidence')).toBeInTheDocument();
+ expect(screen.getByText('Supported · 86% confidence')).toBeInTheDocument();
});
it('falls back to the verdict confidence when the hypothesis omits it', () => {
@@ -59,22 +59,65 @@ describe('HypothesisCard', () => {
/>
);
- expect(screen.getByText('Inconclusive · 34% Confidence')).toBeInTheDocument();
+ expect(screen.getByText('Inconclusive · 34% confidence')).toBeInTheDocument();
});
- it('omits confidence while the hypothesis is still being investigated', () => {
+ it('omits confidence while the hypothesis is still in flight', () => {
render(
+ );
+
+ expect(screen.getByText('Checking')).toBeInTheDocument();
+ expect(screen.queryByText(/confidence/i)).not.toBeInTheDocument();
+ });
+
+ // A hypothesis in flight is one `effectiveStatus`, but it passes through
+ // several states worth naming. They are read off the verification steps,
+ // since that is the only place the distinction exists.
+ it.each([
+ ['no steps planned yet', 'Formed', [], 'queued'],
+ [
+ 'steps planned but not started',
+ 'Preparing checks',
+ [InvestigationVerificationStepFixture({status: 'queued', result: null})],
+ 'queued',
+ ],
+ [
+ 'steps running',
+ 'Checking',
+ [InvestigationVerificationStepFixture({status: 'running', result: null})],
+ 'running',
+ ],
+ [
+ 'every step finished, no verdict',
+ 'Evidence checked',
+ [InvestigationVerificationStepFixture({status: 'completed', result: 'Done.'})],
+ 'running',
+ ],
+ ] as const)('reads %s as "%s"', (_name, label, verificationSteps, status) => {
+ render(
+
);
- expect(screen.getByText('Investigating')).toBeInTheDocument();
- expect(screen.queryByText(/Confidence/)).not.toBeInTheDocument();
+ // Scoped, because "Evidence checked" is also the heading over the steps.
+ expect(
+ within(screen.getByTestId('hypothesis-status')).getByText(label)
+ ).toBeInTheDocument();
});
it('lists verification steps in order with their results', () => {
@@ -118,7 +161,7 @@ describe('HypothesisCard', () => {
/>
);
- expect(screen.getByText('Checking…')).toBeInTheDocument();
+ expect(screen.getByText('Awaiting evidence')).toBeInTheDocument();
});
it("prefers a failed step's error message over the generic failure label", () => {
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
index b0ff155d3308..f59df500af87 100644
--- a/static/app/views/investigations/hypotheses/hypothesisCard.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
@@ -1,5 +1,6 @@
import styled from '@emotion/styled';
+import {Disclosure} from '@sentry/scraps/disclosure';
import {Container, Flex, Stack} from '@sentry/scraps/layout';
import {Heading, Text} from '@sentry/scraps/text';
@@ -7,6 +8,7 @@ import {DropdownMenu, type MenuItemProps} from 'sentry/components/dropdownMenu';
import {IconEllipsis} from 'sentry/icons';
import {t} from 'sentry/locale';
import {
+ getEvidenceSectionLabel,
getHypothesisCardBorder,
getVerificationStepStatusLabel,
HypothesisStatus,
@@ -108,7 +110,7 @@ export function HypothesisCard({
{steps.length > 0 ? (
- {t('Evidence checked')}
+ {getEvidenceSectionLabel(steps)}
{steps.map(step => (
@@ -127,6 +129,19 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
// wins when both are present.
const detail =
step.result || step.error?.message || getVerificationStepStatusLabel(step.status);
+ const summary = (
+
+ {step.title}
+
+ {detail}
+
+
+ );
+
+ // A step that has produced something can be opened for how the agent got
+ // there. One that has not is a bare row — there is no finding to unpack yet,
+ // and a chevron would promise one.
+ const hasRun = Boolean(step.result) || Boolean(step.error);
return (
-
- {step.title}
-
- {detail}
-
-
+ {hasRun ? (
+
+ {summary}
+
+
+
+
+ {t('Objective')}
+
+
+ {step.objective}
+
+
+
+
+ {t('Method')}
+
+
+ {step.method}
+
+
+
+
+
+ ) : (
+ summary
+ )}
);
}
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
index 5574044cb758..18f152803315 100644
--- a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -7,13 +7,14 @@ import type {
InvestigationHypothesis,
InvestigationHypothesisStatus,
InvestigationOrchestrationWorkStatus,
+ InvestigationVerificationStep,
} from 'sentry/views/investigations/types';
type StatusVariant = 'success' | 'warning' | 'danger' | 'accent' | 'muted';
/**
* Statuses where the agent has reached a verdict. Confidence is only meaningful
- * once it has, so an in-flight hypothesis shows a bare label.
+ * once it has, so a hypothesis still in flight shows a bare label.
*/
const SETTLED_STATUSES = new Set([
'supported',
@@ -23,17 +24,10 @@ const SETTLED_STATUSES = new Set([
'rejected',
]);
-/** Statuses where the agent is still working, so the dot keeps pulsing. */
-const IN_FLIGHT_STATUSES = new Set(['pending', 'investigating']);
-
function isHypothesisSettled(status: InvestigationHypothesisStatus): boolean {
return SETTLED_STATUSES.has(status);
}
-function isHypothesisInFlight(status: InvestigationHypothesisStatus): boolean {
- return IN_FLIGHT_STATUSES.has(status);
-}
-
/**
* Turn an unrecognized wire value into something readable rather than dropping
* it. Statuses are an open set — Seer can introduce one before Sentry knows the
@@ -43,68 +37,78 @@ function humanize(status: string): string {
return status.replaceAll('_', ' ').replace(/^./, character => character.toUpperCase());
}
-function getHypothesisStatusLabel(hypothesis: InvestigationHypothesis): string {
+function hasRun(step: InvestigationVerificationStep): boolean {
+ return Boolean(step.result) || Boolean(step.error);
+}
+
+type HypothesisStatusDisplay = {
+ /** Whether the agent is actively working, which is what keeps the dot moving. */
+ inFlight: boolean;
+ label: string;
+ variant: StatusVariant;
+};
+
+/**
+ * What the status line says, and in what colour.
+ *
+ * A hypothesis in flight is all one `effectiveStatus`, but it passes through
+ * several states worth naming: formed, having its checks planned, running them,
+ * and done checking but not yet judged. Those are read off the verification
+ * steps, since that is the only place the distinction exists. Only the running
+ * state is coloured — the rest are staging posts, not outcomes.
+ */
+export function getHypothesisStatusDisplay(
+ hypothesis: InvestigationHypothesis
+): HypothesisStatusDisplay {
+ const status = hypothesis.effectiveStatus;
+
// A decision the viewer made themselves reads differently from one the agent
// reached, even though both land in `effectiveStatus`.
- if (hypothesis.decisionSource === 'user') {
- if (hypothesis.effectiveStatus === 'accepted') {
- return t('Accepted by you');
- }
- if (hypothesis.effectiveStatus === 'rejected') {
- return t('Rejected by you');
- }
+ if (hypothesis.decisionSource === 'user' && status === 'accepted') {
+ return {label: t('Accepted by you'), variant: 'success', inFlight: false};
}
-
- switch (hypothesis.effectiveStatus) {
- case 'pending':
- return t('Pending');
- case 'investigating':
- return t('Investigating');
- case 'supported':
- return t('Supported');
- case 'refuted':
- return t('Refuted');
- case 'inconclusive':
- return t('Inconclusive');
- case 'accepted':
- return t('Accepted');
- case 'rejected':
- return t('Rejected');
- case 'failed':
- return t('Failed');
- case 'cancelled':
- return t('Cancelled');
- default:
- return humanize(hypothesis.effectiveStatus);
+ if (hypothesis.decisionSource === 'user' && status === 'rejected') {
+ return {label: t('Rejected by you'), variant: 'muted', inFlight: false};
}
-}
-function getHypothesisStatusVariant(
- status: InvestigationHypothesisStatus
-): StatusVariant {
switch (status) {
- // A hypothesis the evidence backs, or one a person has endorsed.
case 'supported':
+ return {label: t('Supported'), variant: 'success', inFlight: false};
case 'accepted':
- return 'success';
- // Ruled out cleanly. This is a useful outcome rather than an error, so it
- // reads as neutral instead of dangerous.
+ return {label: t('Accepted'), variant: 'success', inFlight: false};
+ // Ruled out cleanly. A useful outcome rather than an error, so it reads
+ // neutral instead of dangerous.
case 'refuted':
+ return {label: t('Refuted'), variant: 'muted', inFlight: false};
case 'rejected':
- return 'muted';
+ return {label: t('Rejected'), variant: 'muted', inFlight: false};
// Checked, but the evidence did not settle it either way.
case 'inconclusive':
- return 'warning';
- case 'investigating':
- return 'accent';
+ return {label: t('Inconclusive'), variant: 'warning', inFlight: false};
case 'failed':
- return 'danger';
- case 'pending':
+ return {label: t('Failed'), variant: 'danger', inFlight: false};
case 'cancelled':
- return 'muted';
+ return {label: t('Cancelled'), variant: 'muted', inFlight: false};
+ case 'pending':
+ case 'investigating':
+ break;
default:
- return 'muted';
+ return {label: humanize(status), variant: 'muted', inFlight: false};
+ }
+
+ const steps = hypothesis.verificationSteps;
+ if (steps.length === 0) {
+ // Proposed, with nothing planned to test it yet.
+ return {label: t('Formed'), variant: 'muted', inFlight: false};
+ }
+ if (steps.every(hasRun)) {
+ // Every check has produced something; the verdict is what is missing.
+ return {label: t('Evidence checked'), variant: 'muted', inFlight: false};
}
+ if (hypothesis.status === 'running') {
+ return {label: t('Checking'), variant: 'accent', inFlight: true};
+ }
+ return {label: t('Preparing checks'), variant: 'muted', inFlight: false};
}
/**
@@ -123,6 +127,11 @@ export function getHypothesisCardBorder(
return status === 'supported' || status === 'accepted' ? 'accent' : 'dashed';
}
+/** The heading above the steps, which depends on whether any have run yet. */
+export function getEvidenceSectionLabel(steps: InvestigationVerificationStep[]): string {
+ return steps.some(hasRun) ? t('Evidence checked') : t('Evidence to check');
+}
+
/**
* What a verification step says about itself while it has no result yet. A step
* only carries a `result` once it has finished, so everything short of that
@@ -132,11 +141,12 @@ export function getVerificationStepStatusLabel(
status: InvestigationOrchestrationWorkStatus
): string {
switch (status) {
+ // Queued and running read the same from outside: the answer is not here
+ // yet. Only the states that need someone to act get their own line.
case 'not_started':
case 'queued':
- return t('Queued.');
case 'running':
- return t('Checking…');
+ return t('Awaiting evidence');
case 'blocked':
return t('Blocked on an earlier step.');
case 'reauth_required':
@@ -184,26 +194,26 @@ type HypothesisStatusProps = {
/**
* The dot-and-label line above a hypothesis statement, e.g.
- * "● Supported · 86% Confidence".
+ * "● Supported · 86% confidence".
*/
export function HypothesisStatus({hypothesis}: HypothesisStatusProps) {
- const status = hypothesis.effectiveStatus;
- const variant = getHypothesisStatusVariant(status);
- const label = getHypothesisStatusLabel(hypothesis);
+ const {inFlight, label, variant} = getHypothesisStatusDisplay(hypothesis);
const confidence = getHypothesisConfidencePercent(hypothesis);
return (
-
+ // "Evidence checked" is both a status and the heading over the steps, so
+ // this needs to be addressable on its own.
+
{confidence === null
? label
- : // Translators: e.g. "Supported · 86% Confidence"
- t('%s · %s%% Confidence', label, confidence)}
+ : // Translators: e.g. "Supported · 86% confidence"
+ t('%s · %s%% confidence', label, confidence)}
);
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
index d9800fad6fe5..dbd02fdd3794 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
@@ -37,7 +37,7 @@ describe('InvestigationHypotheses', () => {
name: 'Database or cache degradation delayed the response',
})
).toBeInTheDocument();
- expect(screen.getByText('Supported · 86% Confidence')).toBeInTheDocument();
+ expect(screen.getByText('Supported · 86% confidence')).toBeInTheDocument();
});
it('highlights the report primary hypothesis', async () => {
@@ -145,7 +145,8 @@ describe('InvestigationHypotheses', () => {
? {
...hypothesis,
effectiveStatus: 'accepted' as const,
- userDisposition: {disposition: 'accepted' as const, userId: 1},
+ // A viewer decision, which the card credits to them by name.
+ decisionSource: 'user' as const,
}
: hypothesis
),
@@ -175,6 +176,8 @@ describe('InvestigationHypotheses', () => {
await userEvent.click(await screen.findByRole('menuitemradio', {name: 'Accept'}));
// No refetch is needed: the command response carries the new projection.
- expect(await screen.findByText('Accepted · 86% Confidence')).toBeInTheDocument();
+ expect(
+ await screen.findByText('Accepted by you · 86% confidence')
+ ).toBeInTheDocument();
});
});
From 876d22ff3f56666520145316279fc46ce8be9388 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Fri, 11 Sep 2026 15:24:20 -0400
Subject: [PATCH 07/21] fix(investigations): Lay out evidence rows against the
prototype
Verified in a browser this time, which caught four things reading the code
did not.
The evidence row's summary was passed as the Disclosure title's children,
so it landed inside a Button: centred, held to one line, and spilling past
the row's right edge. Moving it to `leadingItems` puts it outside the
button, where it lays out as ordinary content and, taking the row's spare
width, pushes the chevron to the edge the prototype has it on. That also
removed the doubled padding, since the Disclosure supplies its own.
The toggle had no accessible name once the summary moved out of it -- a
screen reader heard "button, collapsed" and nothing else. It now says which
check it opens.
The statement and its rationale were running together as one block, and the
story demos were clipping at Storybook.Demo's 512px maximum, so most of
each row was only reachable by scrolling inside the frame.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../hypotheses/hypotheses.stories.tsx | 10 +++---
.../hypotheses/hypothesisCard.tsx | 32 +++++++++++++------
.../hypotheses/hypothesisStatus.tsx | 2 +-
3 files changed, 29 insertions(+), 15 deletions(-)
diff --git a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
index fbd711baa07b..a20a2a524b8d 100644
--- a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
+++ b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
@@ -53,7 +53,7 @@ export default Storybook.story('Investigations — Hypotheses', story => {
failed all read the same way to someone scanning the row, so the status line
carries the distinction rather than the border.
-
+
@@ -65,14 +65,14 @@ export default Storybook.story('Investigations — Hypotheses', story => {
The heading over the steps moves with them, from "Evidence to check" to "Evidence
checked".
-
+
A failure is the one in-flight state that gets a colour, because it is the only
one that has stopped. The hypothesis says why, and so does each check that broke.
-
+ {
accent; choosing the same decision twice clears it and hands the hypothesis back
to the agent's verdict.
-
+ {
rows above do — and the overflow menu disappears, which is what a read-only
surface wants.
-
+
-
+
{hypothesis.statement}
@@ -129,8 +129,14 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
// wins when both are present.
const detail =
step.result || step.error?.message || getVerificationStepStatusLabel(step.status);
+ // A step that has produced something can be opened for how the agent got
+ // there. One that has not is a bare row — there is no finding to unpack yet,
+ // and a chevron would promise one.
+ const hasRun = Boolean(step.result) || Boolean(step.error);
const summary = (
-
+ // A full flex-basis, because the chevron's button grows too — without this
+ // the two split the row and the text wraps in half the width it has.
+ {step.title}
{detail}
@@ -138,22 +144,30 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
);
- // A step that has produced something can be opened for how the agent got
- // there. One that has not is a bare row — there is no finding to unpack yet,
- // and a chevron would promise one.
- const hasRun = Boolean(step.result) || Boolean(step.error);
-
return (
{hasRun ? (
- {summary}
+ {/*
+ * The summary goes in `leadingItems`, not as the title's children:
+ * children land inside a Button, which is one line tall and centres
+ * what it holds. From the leading slot the summary lays out normally
+ * and, taking the row's spare width, pushes the chevron to the edge.
+ */}
+
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
index 18f152803315..1a14ca71a5cc 100644
--- a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -57,7 +57,7 @@ type HypothesisStatusDisplay = {
* steps, since that is the only place the distinction exists. Only the running
* state is coloured — the rest are staging posts, not outcomes.
*/
-export function getHypothesisStatusDisplay(
+function getHypothesisStatusDisplay(
hypothesis: InvestigationHypothesis
): HypothesisStatusDisplay {
const status = hypothesis.effectiveStatus;
From a3ca10cabe97c6fadae172bc9ec4789cf40081f0 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Sun, 13 Sep 2026 09:35:41 -0400
Subject: [PATCH 08/21] feat(investigations): Render the hypothesis row on the
detail view
The row talked to the real endpoints already, but nothing mounted it, so it
never appeared in the product. `knip --production` had been reporting
exactly that and the config carried a TODO entry point to keep it quiet;
both are gone now because the detail view reaches it.
`orchestration` on the investigation is what gates it. The field is already
served inline on the list and detail responses and is null for manual and
template investigations, whose orchestration endpoint 404s. The proof of
concept gated on `investigation.mode === 'agentic'` instead, but no `mode`
field exists -- the summary is the only marker an investigation carries.
The detail query now also polls while a run is live. It previously stopped
as soon as no block was executing, which would freeze the gate: a run that
started after load would never show its row, and one that finished would
keep claiming to be running.
---
knip.config.ts | 3 --
.../investigations/detail/index.spec.tsx | 51 ++++++++++++++++++-
.../app/views/investigations/detail/index.tsx | 25 ++++++++-
.../views/investigations/fixtures/index.ts | 25 +++++++++
.../hypotheses/investigationHypotheses.tsx | 27 +++++-----
static/app/views/investigations/types.ts | 18 ++++++-
6 files changed, 128 insertions(+), 21 deletions(-)
diff --git a/knip.config.ts b/knip.config.ts
index e549a2705122..8bad3af97f9f 100644
--- a/knip.config.ts
+++ b/knip.config.ts
@@ -24,9 +24,6 @@ const productionEntryPoints = [
'static/app/chartcuterie/**/*.{js,ts,tsx}',
// TODO: Remove when the autofixRef embed consumes it (#122099)
'static/app/components/seer/autofixChatContext.tsx',
- // Pulls in the whole hypotheses/ directory.
- // TODO: Remove when an investigation surface renders it (#124086)
- 'static/app/views/investigations/hypotheses/investigationHypotheses.tsx',
'static/app/components/brandPageLayout/**/*.{ts,tsx}',
// React authentication routes are discovered dynamically by the frontend route registry
'static/app/views/authV2/authLogin/**/*.{ts,tsx}',
diff --git a/static/app/views/investigations/detail/index.spec.tsx b/static/app/views/investigations/detail/index.spec.tsx
index 96548c469809..77029714cc81 100644
--- a/static/app/views/investigations/detail/index.spec.tsx
+++ b/static/app/views/investigations/detail/index.spec.tsx
@@ -28,7 +28,11 @@ import {
investigationListQueryOptions,
} from 'sentry/views/investigations/api';
import InvestigationDetailView from 'sentry/views/investigations/detail';
-import {InvestigationDetailFixture} from 'sentry/views/investigations/fixtures';
+import {
+ InvestigationAgenticDetailFixture,
+ InvestigationDetailFixture,
+ InvestigationOrchestrationFixture,
+} from 'sentry/views/investigations/fixtures';
jest.unmock('@tanstack/react-pacer');
@@ -39,6 +43,8 @@ const organization = OrganizationFixture({
const detailUrl = '/organizations/org-slug/investigations/investigation-1/';
const titleGenerationUrl =
'/organizations/org-slug/investigations/investigation-1/title-generation/';
+const orchestrationUrl =
+ '/organizations/org-slug/investigations/investigation-1/orchestration/';
const feedbackForm = {
appendToDom: jest.fn(),
@@ -1854,4 +1860,47 @@ describe('Investigation detail', () => {
).toBeInTheDocument();
expect(request).not.toHaveBeenCalled();
});
+
+ // `orchestration` being present is the only thing that marks an investigation
+ // as agentic, and the orchestration endpoint 404s without a run, so the gate
+ // has to hold in both directions.
+ it('renders the hypothesis row for an agentic investigation', async () => {
+ MockApiClient.addMockResponse({
+ url: detailUrl,
+ body: InvestigationAgenticDetailFixture(),
+ });
+ const orchestrationRequest = MockApiClient.addMockResponse({
+ url: orchestrationUrl,
+ body: InvestigationOrchestrationFixture(),
+ });
+
+ renderView();
+
+ expect(await screen.findAllByTestId('investigation-hypothesis')).toHaveLength(3);
+ expect(
+ screen.getByRole('heading', {
+ name: 'Database or cache degradation delayed the response',
+ })
+ ).toBeInTheDocument();
+ expect(orchestrationRequest).toHaveBeenCalled();
+ });
+
+ it('does not reach for orchestration on a manual investigation', async () => {
+ MockApiClient.addMockResponse({
+ url: detailUrl,
+ body: InvestigationDetailFixture(),
+ });
+ const orchestrationRequest = MockApiClient.addMockResponse({
+ url: orchestrationUrl,
+ body: InvestigationOrchestrationFixture(),
+ });
+
+ renderView();
+
+ expect(
+ await screen.findByRole('textbox', {name: 'Investigation title'})
+ ).toBeInTheDocument();
+ expect(screen.queryByTestId('investigation-hypotheses')).not.toBeInTheDocument();
+ expect(orchestrationRequest).not.toHaveBeenCalled();
+ });
});
diff --git a/static/app/views/investigations/detail/index.tsx b/static/app/views/investigations/detail/index.tsx
index 0bfb5101841e..5bb08275242c 100644
--- a/static/app/views/investigations/detail/index.tsx
+++ b/static/app/views/investigations/detail/index.tsx
@@ -44,6 +44,10 @@ import {
shouldDisplayInvestigationBlock,
shouldPollInvestigationBlocks,
} from 'sentry/views/investigations/detail/cell';
+import {
+ InvestigationHypotheses,
+ isInvestigationRunSettled,
+} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
import {updateInvestigationCache} from 'sentry/views/investigations/investigationCache';
import {InvestigationSummaryCard} from 'sentry/views/investigations/investigationSummaryCard';
import type {
@@ -92,7 +96,14 @@ export function InvestigationBootstrapPage({investigationId}: {investigationId:
...detailOptions,
refetchInterval: query => {
const data = query.state.data?.json;
- return shouldPollInvestigationBlocks(data?.blocks ?? []) ||
+ // A live agentic run keeps this polling too: the notebook fills in as the
+ // agent writes blocks, and `orchestration` is what gates the hypothesis
+ // row, so a stale copy would leave the row hidden or showing a run that
+ // has since finished.
+ const orchestrationActive =
+ data?.orchestration && !isInvestigationRunSettled(data.orchestration.status);
+ return orchestrationActive ||
+ shouldPollInvestigationBlocks(data?.blocks ?? []) ||
isTitleGenerationActive(data?.titleGeneration?.status)
? 2000
: false;
@@ -408,6 +419,18 @@ function InvestigationPageContent({investigation}: {investigation: Investigation
summaryDescription={investigation.summaryDescription}
/>
+ {/*
+ * Only an agentic investigation has hypotheses, and `orchestration`
+ * being present is the only thing that says one is: it is null for
+ * manual and template investigations, whose orchestration endpoint
+ * 404s.
+ */}
+ {investigation.orchestration ? (
+
+
+
+ ) : null}
+
{visibleSummaryBlock ? (
= {}
+): InvestigationDetail & {blocks: InvestigationBlock[]} {
+ return InvestigationDetailFixture({
+ orchestration: {
+ phase: 'investigating',
+ status: 'processing',
+ heartbeatAt: '2026-08-27T11:06:30Z',
+ notebookRevision: 5,
+ },
+ ...overrides,
+ });
+}
+
export function InvestigationTranscriptBlockFixture(
overrides: Partial = {}
): InvestigationTranscriptBlock {
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
index 2e3169c88928..921ebda159f1 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
@@ -11,28 +11,25 @@ import {
import {HypothesisList} from 'sentry/views/investigations/hypotheses/hypothesisList';
import type {
InvestigationHypothesis,
- InvestigationOrchestration,
+ InvestigationOrchestrationStatus,
} from 'sentry/views/investigations/types';
/** How often to re-read the projection while a workflow is still moving. */
const POLL_INTERVAL_MS = 2000;
/**
- * Whether the workflow has stopped moving on its own. `awaiting_input` is
- * deliberately not settled: the run resumes as soon as input arrives, which may
- * happen from another surface, so polling has to continue.
+ * Whether a workflow has stopped moving on its own.
+ *
+ * `awaiting_input` is deliberately not terminal: the run resumes as soon as
+ * input arrives, which may happen from another surface, so polling has to
+ * continue. Exported because the detail view decides from the summary served
+ * alongside the investigation, and this component from the full projection —
+ * the same three statuses either way.
*/
-function isInvestigationRunSettled(
- projection: InvestigationOrchestration | undefined
+export function isInvestigationRunSettled(
+ status: InvestigationOrchestrationStatus | undefined
): boolean {
- if (!projection) {
- return false;
- }
- return (
- projection.status === 'completed' ||
- projection.status === 'failed' ||
- projection.status === 'cancelled'
- );
+ return status === 'completed' || status === 'failed' || status === 'cancelled';
}
type InvestigationHypothesesProps = {
@@ -67,7 +64,7 @@ export function InvestigationHypotheses({
...investigationOrchestrationQueryOptions(organization.slug, investigationId),
enabled,
refetchInterval: query =>
- isInvestigationRunSettled(query.state.data?.json) ? false : POLL_INTERVAL_MS,
+ isInvestigationRunSettled(query.state.data?.json.status) ? false : POLL_INTERVAL_MS,
});
const commandMutation = useInvestigationOrchestrationCommandMutation(
diff --git a/static/app/views/investigations/types.ts b/static/app/views/investigations/types.ts
index 6455ac59b1b8..647cc9c92e75 100644
--- a/static/app/views/investigations/types.ts
+++ b/static/app/views/investigations/types.ts
@@ -1,5 +1,20 @@
import type {ToolResult} from 'sentry/views/seerExplorer/types';
+/**
+ * The scalar head of an agentic run, served inline with an investigation so a
+ * caller can tell one apart without a second request.
+ *
+ * This is the only marker an investigation carries for being agentic: it is
+ * `null` on manual and template investigations, whose orchestration endpoint
+ * 404s. Anything that needs the full run state fetches the projection.
+ */
+type InvestigationOrchestrationSummary = {
+ heartbeatAt: string | null;
+ notebookRevision: number;
+ phase: InvestigationOrchestrationPhase;
+ status: InvestigationOrchestrationStatus;
+};
+
export type InvestigationListItem = {
blockCount: number;
createdBy: string | null;
@@ -13,6 +28,7 @@ export type InvestigationListItem = {
summaryDescription: string | null;
title: string;
version: number;
+ orchestration?: InvestigationOrchestrationSummary | null;
titleGeneration?: {
status: 'pending' | 'running' | 'completed' | 'failed' | null;
};
@@ -183,7 +199,7 @@ type InvestigationOrchestrationPhase = InvestigationOrchestrationOpenString<
| 'cancelled'
>;
-type InvestigationOrchestrationStatus = InvestigationOrchestrationOpenString<
+export type InvestigationOrchestrationStatus = InvestigationOrchestrationOpenString<
'pending' | 'processing' | 'awaiting_input' | 'completed' | 'failed' | 'cancelled'
>;
From bd8cda0869e080e25ca8df8325587f80f376c5a2 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Mon, 14 Sep 2026 10:51:37 -0400
Subject: [PATCH 09/21] fix(investigations): Say why launching an investigation
is unavailable
The button went grey with nothing to explain it, and the two reasons it
does so are not guessable from the page.
A missing open period is named outright: the page already lists open
periods, so saying there is none gives nothing away.
The other reason cannot be as specific. `unavailable` from the candidates
endpoint covers an issue that cannot be investigated at all, an existing
investigation in a project the viewer cannot see, and a viewer who may not
create one -- collapsed on purpose, since the resolver it comes from notes
that a caller "should not reveal whether an inaccessible or invalid issue
exists". Distinguishing those in a tooltip would leak precisely that, so
the wording covers them together and points at the cause someone can
actually act on.
Neither reason is ever "still loading": a pending query renders a
placeholder in place of the button, and a failed one renders an alert.
Claude-Session: https://claude.ai/code/session_012CtaiBZtJdz8uMRtUgvbmv
---
.../metricDetectorTriggeredSection.spec.tsx | 53 +++++++++++++++++++
.../metricDetectorTriggeredSection.tsx | 40 +++++++++++++-
2 files changed, 91 insertions(+), 2 deletions(-)
diff --git a/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx b/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx
index fa90c0ff5ba7..b1f4f94edcd5 100644
--- a/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx
+++ b/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx
@@ -307,6 +307,59 @@ describe('MetricDetectorTriggeredSection', () => {
expect(router.location.pathname).toBe('/explore/investigations/4567/');
});
+ it('explains why launching is unavailable', async () => {
+ const organization = OrganizationFixture({
+ slug: 'org-slug',
+ features: ['investigations'],
+ });
+ MockApiClient.addMockResponse({
+ url: '/organizations/org-slug/investigations/candidates/',
+ method: 'POST',
+ body: {items: [{status: 'unavailable'}]},
+ });
+
+ render(, {organization});
+
+ const button = await screen.findByRole('button', {name: 'Launch Investigation'});
+ expect(button).toBeDisabled();
+ await userEvent.hover(button);
+ // Deliberately vague: naming the cause would reveal whether an issue the
+ // viewer cannot access exists.
+ expect(
+ await screen.findByText(
+ 'Seer cannot investigate this issue. It may not be linked to an active monitor, or you may not have access.'
+ )
+ ).toBeInTheDocument();
+ });
+
+ it('explains that an issue with no open period cannot be investigated', async () => {
+ const organization = OrganizationFixture({
+ slug: 'org-slug',
+ features: ['investigations'],
+ });
+ const candidatesMock = MockApiClient.addMockResponse({
+ url: '/organizations/org-slug/investigations/candidates/',
+ method: 'POST',
+ body: {items: [{status: 'investigate'}]},
+ });
+ MockApiClient.addMockResponse({
+ url: '/organizations/org-slug/open-periods/',
+ body: [],
+ });
+
+ render(, {organization});
+
+ const button = await screen.findByRole('button', {name: 'Launch Investigation'});
+ expect(button).toBeDisabled();
+ // Open periods are already on the page, so naming this one gives nothing away.
+ await userEvent.hover(button);
+ expect(
+ await screen.findByText('This issue has no open period to investigate.')
+ ).toBeInTheDocument();
+ // Without a source there is nothing to ask about.
+ expect(candidatesMock).not.toHaveBeenCalled();
+ });
+
it('uses the latest open period when the displayed event is not linked to one', async () => {
const organization = OrganizationFixture({
slug: 'org-slug',
diff --git a/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.tsx b/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.tsx
index 9e120746eefc..cf41d36dcd40 100644
--- a/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.tsx
+++ b/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.tsx
@@ -61,7 +61,10 @@ import {
} from 'sentry/views/investigations/api';
import {shouldPollInvestigationBlocks} from 'sentry/views/investigations/detail/cell';
import {InvestigationSummaryCard} from 'sentry/views/investigations/investigationSummaryCard';
-import type {MetricOpenPeriodInvestigationSource} from 'sentry/views/investigations/types';
+import type {
+ InvestigationCandidate,
+ MetricOpenPeriodInvestigationSource,
+} from 'sentry/views/investigations/types';
import {FoldSection} from 'sentry/views/issueDetails/foldSection';
import {AttributeComparisonSection} from './attributeComparisonSection';
@@ -566,6 +569,34 @@ const GroupListWrapper = styled('div')`
margin-top: ${p => p.theme.space.md};
`;
+/**
+ * Why the launch button is off, or undefined when it is available.
+ *
+ * Both queries have settled by the time the button renders — a pending one
+ * shows a placeholder instead — so this is never "not yet".
+ *
+ * `unavailable` covers several situations the server deliberately does not
+ * separate: an issue that cannot be investigated at all, an existing
+ * investigation in a project the viewer cannot see, and a viewer who may not
+ * create one. Saying which would reveal whether an issue the viewer has no
+ * access to exists, so that wording stays vague on purpose. A missing open
+ * period is safe to name: the page already lists them.
+ */
+function getLaunchDisabledReason(
+ source: MetricOpenPeriodInvestigationSource | null,
+ candidateStatus: InvestigationCandidate['status'] | undefined
+): string | undefined {
+ if (source === null) {
+ return t('This issue has no open period to investigate.');
+ }
+ if (candidateStatus === 'unavailable') {
+ return t(
+ 'Seer cannot investigate this issue. It may not be linked to an active monitor, or you may not have access.'
+ );
+ }
+ return undefined;
+}
+
function SeerInvestigationSection({
eventId,
groupId,
@@ -687,6 +718,8 @@ function SeerInvestigationSection({
)
: null;
+ const launchDisabledReason = getLaunchDisabledReason(source, candidate?.status);
+
return (
source && launchMutation.mutate(source)}
>
{t('Launch Investigation')}
From 51f43806180055c082961c3bf96844121c1ff6ab Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Mon, 14 Sep 2026 11:39:30 -0400
Subject: [PATCH 10/21] feat(investigations): Show what the Seer run is doing
above the hypotheses
Frames 4705-4707 of the Seer Investigation design put a status block above
the hypothesis row: one line saying what the agent is doing, with a chip and
an elapsed counter, and the row beneath it inside a shared panel.
The block is one component for the whole lifecycle, because the shape never
changes -- icon, sentence, chip, time -- only the words and the colour do. It
sits in a fixed spot for the life of a run, so a reader who has learned where
to look for "what is happening" never has to relearn it. `status` picks the
variant and `phase` picks the words: every in-flight phase shares one
`processing` status, so the phase is the only thing separating "gathering
context" from "finalizing".
The design's chip turned out to be the existing `Tag`: `variant="info"`
resolves to `content.accent` on `background.transparent.accent.muted`, which
is exactly what the Figma variables name. No new styled component needed.
Two states are not from the frames and are marked as such. `cancelled` is
here because the projection can report it and falling through to `failed`
would paint a decision someone deliberately made bright red. `elapsed` is an
optional prop the wired block leaves empty -- the projection carries no
run-level start time, only per-block `startedAt` -- so it renders in the
story and nowhere else rather than counting from an invented origin.
The cards move with it. A hypothesis being verified now keeps a solid border
instead of a dashed one, because dashing it announces a verdict the agent has
not reached; dashed is reserved for a card that was checked and is not the
answer. "Checking" becomes "Verifying..." and is drawn as a spinning ring
rather than a pulsing dot, and `refuted` reads amber rather than muted -- a
hypothesis the agent tested and closed is not an error, but it is a result
worth registering as you scan the row.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../investigationFixtureApi.spec.tsx | 10 +-
.../hypotheses/hypotheses.stories.tsx | 19 +-
.../hypotheses/hypothesisCard.spec.tsx | 18 +-
.../hypotheses/hypothesisCard.tsx | 7 +-
.../hypotheses/hypothesisStatus.stories.tsx | 265 ++++++++++++++++++
.../hypotheses/hypothesisStatus.tsx | 46 +--
.../investigationHypotheses.spec.tsx | 4 +-
.../hypotheses/investigationHypotheses.tsx | 35 ++-
.../statusBlock/getSeerStatusBlock.tsx | 191 +++++++++++++
.../statusBlock/seerStatusBlock.spec.tsx | 169 +++++++++++
.../statusBlock/seerStatusBlock.stories.tsx | 201 +++++++++++++
.../statusBlock/seerStatusBlock.tsx | 204 ++++++++++++++
12 files changed, 1120 insertions(+), 49 deletions(-)
create mode 100644 static/app/views/investigations/hypotheses/hypothesisStatus.stories.tsx
create mode 100644 static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx
create mode 100644 static/app/views/investigations/statusBlock/seerStatusBlock.spec.tsx
create mode 100644 static/app/views/investigations/statusBlock/seerStatusBlock.stories.tsx
create mode 100644 static/app/views/investigations/statusBlock/seerStatusBlock.tsx
diff --git a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
index 2a5a8658e29b..630200d99581 100644
--- a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
+++ b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
@@ -186,7 +186,7 @@ describe('InvestigationFixtureApi', () => {
name: 'Database or cache degradation delayed the response',
})
).toBeInTheDocument();
- expect(screen.getByText('Supported · 86% confidence')).toBeInTheDocument();
+ expect(screen.getByText('Supported · 86% Confidence')).toBeInTheDocument();
expect(
screen.getByText('The delay begins before the document reaches the browser.')
).toBeInTheDocument();
@@ -205,7 +205,7 @@ describe('InvestigationFixtureApi', () => {
// The command response carries the updated projection, so the card
// changes without another read.
expect(
- await screen.findByText('Accepted by you · 91% confidence')
+ await screen.findByText('Accepted by you · 91% Confidence')
).toBeInTheDocument();
// Accepting settles the hypothesis, so its edge picks up the accent.
expect(screen.getAllByTestId('investigation-hypothesis')[1]).toHaveAttribute(
@@ -222,14 +222,14 @@ describe('InvestigationFixtureApi', () => {
});
await userEvent.click(trigger);
await userEvent.click(await screen.findByRole('menuitemradio', {name: 'Accept'}));
- await screen.findByText('Accepted by you · 91% confidence');
+ await screen.findByText('Accepted by you · 91% Confidence');
await userEvent.click(trigger);
await userEvent.click(
await screen.findByRole('menuitemradio', {name: 'Clear decision'})
);
- expect(await screen.findByText('Refuted · 91% confidence')).toBeInTheDocument();
+ expect(await screen.findByText('Refuted · 91% Confidence')).toBeInTheDocument();
// Back to the agent's verdict, so the edge breaks again.
expect(screen.getAllByTestId('investigation-hypothesis')[1]).toHaveAttribute(
'data-border',
@@ -249,7 +249,7 @@ describe('InvestigationFixtureApi', () => {
await screen.findByRole('menuitemradio', {name: 'Investigate again'})
);
- expect(await screen.findByText('Checking')).toBeInTheDocument();
+ expect(await screen.findByText('Verifying…')).toBeInTheDocument();
expect(screen.getAllByText('Awaiting evidence').length).toBeGreaterThan(0);
});
});
diff --git a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
index a20a2a524b8d..9aa6621ab5fc 100644
--- a/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
+++ b/static/app/views/investigations/hypotheses/hypotheses.stories.tsx
@@ -47,11 +47,13 @@ export default Storybook.story('Investigations — Hypotheses', story => {
agent has settled on a verdict.
- The border carries the verdict, and only two ways: a solid accent edge on the
- explanation that stands — supported by the evidence, or accepted by a person — and
- a dashed edge on every other card. Still running, ruled out, inconclusive and
- failed all read the same way to someone scanning the row, so the status line
- carries the distinction rather than the border.
+ The border carries the verdict three ways. A solid accent edge marks the
+ explanation that stands — supported by the evidence, or accepted by a person. A
+ dashed edge marks a card that was checked and is not the answer: ruled out,
+ inconclusive, failed and cancelled all read the same way to someone scanning the
+ row, so the status line carries that distinction rather than the border. A
+ hypothesis still being investigated keeps an ordinary solid edge, because dashing
+ it would announce a verdict the agent has not reached.
@@ -61,9 +63,10 @@ export default Storybook.story('Investigations — Hypotheses', story => {
through several states worth naming: formed, having its checks planned, running
them, and done checking but not yet judged. Those are read off the verification
steps, since that is the only place the distinction exists. Only the running state
- is coloured and keeps its dot moving — the rest are staging posts, not outcomes.
- The heading over the steps moves with them, from "Evidence to check" to "Evidence
- checked".
+ is coloured, and it is the only one drawn as a spinning ring rather than a dot —
+ the rest are staging posts, not outcomes. All four keep a solid border: dashing
+ one would announce a verdict the agent has not reached. The heading over the steps
+ moves with them, from "Evidence to check" to "Evidence checked".
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
index 1a41730fa007..d3a4b539d279 100644
--- a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
@@ -38,7 +38,7 @@ describe('HypothesisCard', () => {
/>
);
- expect(screen.getByText('Supported · 86% confidence')).toBeInTheDocument();
+ expect(screen.getByText('Supported · 86% Confidence')).toBeInTheDocument();
});
it('falls back to the verdict confidence when the hypothesis omits it', () => {
@@ -59,7 +59,7 @@ describe('HypothesisCard', () => {
/>
);
- expect(screen.getByText('Inconclusive · 34% confidence')).toBeInTheDocument();
+ expect(screen.getByText('Inconclusive · 34% Confidence')).toBeInTheDocument();
});
it('omits confidence while the hypothesis is still in flight', () => {
@@ -76,7 +76,7 @@ describe('HypothesisCard', () => {
/>
);
- expect(screen.getByText('Checking')).toBeInTheDocument();
+ expect(screen.getByText('Verifying…')).toBeInTheDocument();
expect(screen.queryByText(/confidence/i)).not.toBeInTheDocument();
});
@@ -93,7 +93,7 @@ describe('HypothesisCard', () => {
],
[
'steps running',
- 'Checking',
+ 'Verifying…',
[InvestigationVerificationStepFixture({status: 'running', result: null})],
'running',
],
@@ -210,15 +210,17 @@ describe('HypothesisCard', () => {
// Only an explanation that stands gets the solid accent edge.
['supported', 'accent'],
['accepted', 'accent'],
- // Everything else reads the same to someone scanning the row: not the
- // answer, whether that is because it is unfinished or because it lost.
+ // Checked, and not the answer. These read the same to someone scanning the
+ // row, so one broken edge covers all of them.
['inconclusive', 'dashed'],
['refuted', 'dashed'],
['rejected', 'dashed'],
- ['investigating', 'dashed'],
- ['pending', 'dashed'],
['failed', 'dashed'],
['cancelled', 'dashed'],
+ // Still being investigated. An ordinary edge, because dashing it would
+ // announce a verdict the agent has not reached.
+ ['investigating', 'solid'],
+ ['pending', 'solid'],
] as const)('draws a %s hypothesis with a %s border', (effectiveStatus, border) => {
render(
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
index 15874cce5fb5..aef444dccd4a 100644
--- a/static/app/views/investigations/hypotheses/hypothesisCard.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
@@ -215,9 +215,10 @@ const Card = styled(Stack)`
border-color: ${p => p.theme.tokens.border.accent.vibrant};
}
- /* Not the answer: still running, ruled out, inconclusive, or failed. Dashed
- * rather than dotted because a dotted hairline all but disappears at this
- * border color. */
+ /* Checked, and not the answer: ruled out, inconclusive, failed or cancelled.
+ * Dashed rather than dotted because a dotted hairline all but disappears at
+ * this border color. A hypothesis still being investigated keeps the solid
+ * default above — dashing it would announce a verdict nobody has reached. */
&[data-border='dashed'] {
border-style: dashed;
}
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.stories.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.stories.tsx
new file mode 100644
index 000000000000..37c79011689d
--- /dev/null
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.stories.tsx
@@ -0,0 +1,265 @@
+import {Fragment} from 'react';
+
+import {Grid} from '@sentry/scraps/layout';
+import {Text} from '@sentry/scraps/text';
+
+import * as Storybook from 'sentry/stories';
+import {
+ InvestigationHypothesisFixture,
+ InvestigationVerificationStepFixture,
+} from 'sentry/views/investigations/fixtures';
+import {HypothesisStatus} from 'sentry/views/investigations/hypotheses/hypothesisStatus';
+import type {InvestigationHypothesis} from 'sentry/views/investigations/types';
+
+export default Storybook.story('Investigations — Hypothesis status', story => {
+ story('A verdict the agent reached', () => (
+
+
+ The status box is the dot-and-label line a hypothesis card leads with. It is not a
+ render of one field: the label comes from effectiveStatus,{' '}
+ decisionSource and the shape of verificationSteps{' '}
+ together. Every row below is the projection shape that actually produces that box.
+
+
+ Confidence is only meaningful once the agent has settled, so it is appended only
+ for those statuses. It is read from hypothesis.confidence, falling
+ back to agentVerdict.confidence for a projection that has filled in
+ only the latter.
+
+
+ Colour is deliberately sparing. supported and accepted{' '}
+ are the only greens. refuted and inconclusive are amber:
+ a hypothesis the agent tested and closed is not an error, but it is still a result
+ worth registering as you scan the row. rejected and{' '}
+ cancelled stay muted — nobody tested those — and of the settled
+ statuses only failed is dangerous.
+
+ A person accepting or rejecting a hypothesis lands in the same{' '}
+ effectiveStatus as the agent doing it, so{' '}
+ decisionSource: 'user' is the only thing separating them. The box
+ says which it was out loud, because whose call it was changes how the rest of the
+ report should be read.
+
+
+
+ ));
+
+ story('While the agent is still working', () => (
+
+
+ These four are all effectiveStatus: 'pending' or{' '}
+ 'investigating'. The distinction between them exists nowhere but the
+ verification steps, so the box reads it off them: no steps at all means nothing
+ has been planned yet, and steps that have every one produced something means only
+ the verdict is missing.
+
+
+ Only Verifying… is coloured, and it is the only one drawn as a spinning
+ ring rather than a dot — the agent is doing something, where the others are places
+ the hypothesis has come to a stop, however briefly. A column of moving indicators
+ would claim everything is live when nothing is.
+
+
+
+ ));
+
+ story('A status Sentry does not know yet', () => (
+
+
+ Hypothesis statuses are an open set: Seer can introduce one before this code knows
+ its name. Rather than drop it, an unrecognized value is humanized and rendered
+ muted, so a new status degrades to a readable label instead of an empty box.
+
+
+
+ ));
+});
+
+type StatusBoxRow = {
+ /** What in the projection puts the hypothesis into this state. */
+ caption: string;
+ hypothesis: InvestigationHypothesis;
+ key: string;
+};
+
+/** Each box beside the projection shape that produces it. */
+function StatusBoxes({rows}: {rows: StatusBoxRow[]}) {
+ return (
+
+
+ {rows.map(row => (
+
+
+
+ {row.caption}
+
+
+ ))}
+
+
+ );
+}
+
+const SETTLED: StatusBoxRow[] = [
+ {
+ key: 'supported',
+ caption: "effectiveStatus: 'supported' — the explanation that stands",
+ hypothesis: InvestigationHypothesisFixture(),
+ },
+ {
+ key: 'accepted',
+ caption: "effectiveStatus: 'accepted', decisionSource: 'agent'",
+ hypothesis: InvestigationHypothesisFixture({
+ effectiveStatus: 'accepted',
+ decisionSource: 'agent',
+ confidence: 0.92,
+ }),
+ },
+ {
+ key: 'refuted',
+ caption: "effectiveStatus: 'refuted' — tested and closed, so it reads amber",
+ hypothesis: InvestigationHypothesisFixture({
+ effectiveStatus: 'refuted',
+ confidence: 0.91,
+ }),
+ },
+ {
+ key: 'rejected',
+ caption: "effectiveStatus: 'rejected', decisionSource: 'agent'",
+ hypothesis: InvestigationHypothesisFixture({
+ effectiveStatus: 'rejected',
+ decisionSource: 'agent',
+ confidence: 0.77,
+ }),
+ },
+ {
+ key: 'inconclusive',
+ caption: "effectiveStatus: 'inconclusive' — checked, but nothing settled it",
+ hypothesis: InvestigationHypothesisFixture({
+ effectiveStatus: 'inconclusive',
+ confidence: 0.34,
+ }),
+ },
+ {
+ key: 'failed',
+ caption: "effectiveStatus: 'failed' — no confidence, because there is no verdict",
+ hypothesis: InvestigationHypothesisFixture({
+ status: 'failed',
+ effectiveStatus: 'failed',
+ confidence: null,
+ agentVerdict: null,
+ }),
+ },
+ {
+ key: 'cancelled',
+ caption: "effectiveStatus: 'cancelled' — stopped before it reached a verdict",
+ hypothesis: InvestigationHypothesisFixture({
+ status: 'cancelled',
+ effectiveStatus: 'cancelled',
+ confidence: null,
+ agentVerdict: null,
+ }),
+ },
+];
+
+const USER_DECISIONS: StatusBoxRow[] = [
+ {
+ key: 'accepted-by-you',
+ caption: "effectiveStatus: 'accepted', decisionSource: 'user'",
+ hypothesis: InvestigationHypothesisFixture({
+ effectiveStatus: 'accepted',
+ decisionSource: 'user',
+ confidence: null,
+ }),
+ },
+ {
+ key: 'rejected-by-you',
+ caption: "effectiveStatus: 'rejected', decisionSource: 'user'",
+ hypothesis: InvestigationHypothesisFixture({
+ effectiveStatus: 'rejected',
+ decisionSource: 'user',
+ confidence: null,
+ }),
+ },
+];
+
+const IN_FLIGHT: StatusBoxRow[] = [
+ {
+ key: 'formed',
+ caption: "effectiveStatus: 'pending', verificationSteps: [] — nothing planned yet",
+ hypothesis: InvestigationHypothesisFixture({
+ status: 'queued',
+ effectiveStatus: 'pending',
+ confidence: null,
+ agentVerdict: null,
+ verificationSteps: [],
+ }),
+ },
+ {
+ key: 'preparing',
+ caption: "status: 'queued' — the checks are planned, but none has started",
+ hypothesis: InvestigationHypothesisFixture({
+ status: 'queued',
+ effectiveStatus: 'investigating',
+ confidence: null,
+ agentVerdict: null,
+ verificationSteps: [
+ InvestigationVerificationStepFixture({status: 'queued', result: null}),
+ ],
+ }),
+ },
+ {
+ key: 'checking',
+ caption: "status: 'running' — live work, and the only box that spins",
+ hypothesis: InvestigationHypothesisFixture({
+ status: 'running',
+ effectiveStatus: 'investigating',
+ confidence: null,
+ agentVerdict: null,
+ verificationSteps: [
+ InvestigationVerificationStepFixture({status: 'running', result: null}),
+ InvestigationVerificationStepFixture({
+ id: 'step-2',
+ order: 1,
+ status: 'queued',
+ result: null,
+ }),
+ ],
+ }),
+ },
+ {
+ key: 'evidence-checked',
+ caption: 'every step has produced a result — only the verdict is missing',
+ hypothesis: InvestigationHypothesisFixture({
+ status: 'running',
+ effectiveStatus: 'investigating',
+ confidence: null,
+ agentVerdict: null,
+ verificationSteps: [
+ InvestigationVerificationStepFixture({
+ status: 'completed',
+ result: 'Retries tripled while the p95 climbed.',
+ }),
+ ],
+ }),
+ },
+];
+
+const UNKNOWN: StatusBoxRow[] = [
+ {
+ key: 'unknown',
+ caption: "effectiveStatus: 'needs_more_data' — humanized rather than dropped",
+ hypothesis: InvestigationHypothesisFixture({
+ effectiveStatus: 'needs_more_data',
+ confidence: null,
+ agentVerdict: null,
+ }),
+ },
+];
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
index 1a14ca71a5cc..009170e316cd 100644
--- a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -2,6 +2,7 @@ import {Flex} from '@sentry/scraps/layout';
import {StatusIndicator} from '@sentry/scraps/statusIndicator';
import {Text} from '@sentry/scraps/text';
+import {LoadingIndicator} from 'sentry/components/loadingIndicator';
import {t} from 'sentry/locale';
import type {
InvestigationHypothesis,
@@ -76,10 +77,11 @@ function getHypothesisStatusDisplay(
return {label: t('Supported'), variant: 'success', inFlight: false};
case 'accepted':
return {label: t('Accepted'), variant: 'success', inFlight: false};
- // Ruled out cleanly. A useful outcome rather than an error, so it reads
- // neutral instead of dangerous.
+ // Ruled out by the evidence. Not an error — a hypothesis the agent tested
+ // and closed — but still a result worth registering as you scan the row,
+ // which is why it is warning rather than muted.
case 'refuted':
- return {label: t('Refuted'), variant: 'muted', inFlight: false};
+ return {label: t('Refuted'), variant: 'warning', inFlight: false};
case 'rejected':
return {label: t('Rejected'), variant: 'muted', inFlight: false};
// Checked, but the evidence did not settle it either way.
@@ -106,7 +108,7 @@ function getHypothesisStatusDisplay(
return {label: t('Evidence checked'), variant: 'muted', inFlight: false};
}
if (hypothesis.status === 'running') {
- return {label: t('Checking'), variant: 'accent', inFlight: true};
+ return {label: t('Verifying…'), variant: 'accent', inFlight: true};
}
return {label: t('Preparing checks'), variant: 'muted', inFlight: false};
}
@@ -116,15 +118,20 @@ function getHypothesisStatusDisplay(
*
* - `accent` — the explanation that stands: supported by the evidence, or
* endorsed by a person. A solid purple edge means "this is the answer".
- * - `dashed` — everything else. A hypothesis still being investigated, ruled
- * out, inconclusive, or failed is all the same thing to a reader scanning the
- * row: not the answer. One broken edge says that without needing a colour per
- * status, which the status line already carries.
+ * - `solid` — still being investigated. Nothing has been ruled out yet, so the
+ * card gets an ordinary edge; dashing it would announce a verdict the agent
+ * has not reached.
+ * - `dashed` — checked, and not the answer. Ruled out, inconclusive, failed and
+ * cancelled all read the same way to someone scanning the row, so one broken
+ * edge covers them and the status line carries the distinction.
*/
export function getHypothesisCardBorder(
status: InvestigationHypothesisStatus
-): 'accent' | 'dashed' {
- return status === 'supported' || status === 'accepted' ? 'accent' : 'dashed';
+): 'accent' | 'solid' | 'dashed' {
+ if (status === 'supported' || status === 'accepted') {
+ return 'accent';
+ }
+ return status === 'pending' || status === 'investigating' ? 'solid' : 'dashed';
}
/** The heading above the steps, which depends on whether any have run yet. */
@@ -204,16 +211,21 @@ export function HypothesisStatus({hypothesis}: HypothesisStatusProps) {
// "Evidence checked" is both a status and the heading over the steps, so
// this needs to be addressable on its own.
-
+ {inFlight ? (
+ // Live work gets a ring rather than a dot: the agent is doing
+ // something, not resting in a state. Every other status is a place the
+ // hypothesis has come to a stop, however briefly.
+
+
+
+ ) : (
+
+ )}
{confidence === null
? label
- : // Translators: e.g. "Supported · 86% confidence"
- t('%s · %s%% confidence', label, confidence)}
+ : // Translators: e.g. "Supported · 86% Confidence"
+ t('%s · %s%% Confidence', label, confidence)}
);
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
index dbd02fdd3794..b660ee86b5b5 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
@@ -37,7 +37,7 @@ describe('InvestigationHypotheses', () => {
name: 'Database or cache degradation delayed the response',
})
).toBeInTheDocument();
- expect(screen.getByText('Supported · 86% confidence')).toBeInTheDocument();
+ expect(screen.getByText('Supported · 86% Confidence')).toBeInTheDocument();
});
it('highlights the report primary hypothesis', async () => {
@@ -177,7 +177,7 @@ describe('InvestigationHypotheses', () => {
// No refetch is needed: the command response carries the new projection.
expect(
- await screen.findByText('Accepted by you · 86% confidence')
+ await screen.findByText('Accepted by you · 86% Confidence')
).toBeInTheDocument();
});
});
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
index 921ebda159f1..e1f58eaee57c 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
@@ -1,6 +1,8 @@
import {uuid4} from '@sentry/core';
import {useQuery} from '@tanstack/react-query';
+import {Container, Stack} from '@sentry/scraps/layout';
+
import type {MenuItemProps} from 'sentry/components/dropdownMenu';
import {t} from 'sentry/locale';
import {useOrganization} from 'sentry/utils/useOrganization';
@@ -9,6 +11,8 @@ import {
useInvestigationOrchestrationCommandMutation,
} from 'sentry/views/investigations/api';
import {HypothesisList} from 'sentry/views/investigations/hypotheses/hypothesisList';
+import {getSeerStatusBlock} from 'sentry/views/investigations/statusBlock/getSeerStatusBlock';
+import {SeerStatusBlock} from 'sentry/views/investigations/statusBlock/seerStatusBlock';
import type {
InvestigationHypothesis,
InvestigationOrchestrationStatus,
@@ -72,10 +76,14 @@ export function InvestigationHypotheses({
investigationId
);
- if (!projection?.hypotheses?.length) {
+ // The status block is the run talking, so it appears as soon as there is a
+ // run — before the first hypothesis exists, which is exactly when a viewer
+ // most needs to be told that something is happening.
+ if (!projection) {
return null;
}
+ const statusBlock = getSeerStatusBlock(projection);
const {workflowVersion} = projection;
const commandPending = commandMutation.isPending;
@@ -136,11 +144,26 @@ export function InvestigationHypotheses({
];
}
+ // The status block and the hypotheses are one object on the page: the block
+ // says what the run is doing and the cards are what it is doing it to. The
+ // panel is what makes that legible — without it the block reads as a
+ // page-level banner that happens to sit above an unrelated row.
return (
-
+
+
+ {statusBlock ? : null}
+
+
+
);
}
diff --git a/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx b/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx
new file mode 100644
index 000000000000..0067250bf5b9
--- /dev/null
+++ b/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx
@@ -0,0 +1,191 @@
+import {t, tn} from 'sentry/locale';
+import type {SeerStatusBlockVariant} from 'sentry/views/investigations/statusBlock/seerStatusBlock';
+import type {InvestigationOrchestration} from 'sentry/views/investigations/types';
+
+type SeerStatusBlockContent = {
+ statusLabel: string;
+ title: string;
+ variant: SeerStatusBlockVariant;
+ description?: string;
+ meta?: string;
+};
+
+/**
+ * How many checks have produced something across every hypothesis.
+ *
+ * A step counts as done once it has a result *or* an error — a check that broke
+ * still ran, and the tally is "how much work stands behind this", not "how much
+ * of it succeeded".
+ */
+function countCompletedChecks(projection: InvestigationOrchestration): number {
+ return projection.hypotheses.reduce(
+ (total, hypothesis) =>
+ total +
+ hypothesis.verificationSteps.filter(step => step.result || step.error).length,
+ 0
+ );
+}
+
+/**
+ * The tally under a finished or finishing run: how many explanations were
+ * weighed, and how much checking stands behind them.
+ *
+ * Absent until there is something to count, which is why the early running
+ * states render without it rather than claiming "0 possible causes".
+ */
+function getMeta(projection: InvestigationOrchestration): string | undefined {
+ const causeCount = projection.hypotheses.length;
+ if (causeCount === 0) {
+ return undefined;
+ }
+ const checkCount = countCompletedChecks(projection);
+ return t(
+ '%s • %s',
+ tn('%s possible cause', '%s possible causes', causeCount),
+ tn('%s check completed', '%s checks completed', checkCount)
+ );
+}
+
+/**
+ * The first error worth showing. The run-level list is the more specific of the
+ * two — `report.error` is whatever stopped the write-up, which is only the
+ * reason the run failed if nothing earlier did.
+ */
+function getFailureMessage(projection: InvestigationOrchestration): string | undefined {
+ return projection.errors[0]?.message ?? projection.report.error?.message ?? undefined;
+}
+
+/**
+ * What the status block should say for a run that is still moving.
+ *
+ * The phase is the only thing that separates these: `status` is `processing`
+ * for all of them. They are all the same `running` variant — the agent is
+ * working and the viewer has nothing to do — so only the words change.
+ */
+function getRunningContent(
+ projection: InvestigationOrchestration
+): SeerStatusBlockContent {
+ const causeCount = projection.hypotheses.length;
+
+ switch (projection.phase) {
+ case 'intake':
+ case 'broad_scan':
+ return {
+ variant: 'running',
+ title: t('Seer is gathering context'),
+ description: t(
+ 'Comparing the signals around the problem to work out where to look. No input needed.'
+ ),
+ statusLabel: t('Running…'),
+ };
+ case 'planning':
+ return {
+ variant: 'running',
+ title: t('Seer is looking for likely causes'),
+ description: t(
+ 'Possible causes will appear here as Seer connects the evidence. No input needed.'
+ ),
+ statusLabel: t('Running…'),
+ };
+ case 'investigating':
+ case 'judging':
+ return {
+ variant: 'running',
+ // Before the hypotheses land there is no count to quote, and "found 0
+ // possible causes" is worse than not saying it.
+ title: causeCount
+ ? tn(
+ 'Seer found %s possible cause and is checking for evidence',
+ 'Seer found %s possible causes and is checking for evidence',
+ causeCount
+ )
+ : t('Seer is checking for evidence'),
+ description: t(
+ 'Seer is checking for evidence to validate each possible cause. No input needed.'
+ ),
+ statusLabel: t('Running…'),
+ };
+ // The hypotheses are settled and the write-up is being assembled. Still the
+ // running variant, but the chip says so — this is the part that ends with
+ // the page changing under the viewer.
+ case 'reporting':
+ case 'metadata':
+ return {
+ variant: 'running',
+ title: t('Seer is bringing the findings together'),
+ meta: getMeta(projection),
+ description: t(
+ 'Organizing the explanation, supporting evidence, and next steps. Your investigation will open automatically.'
+ ),
+ statusLabel: t('Finalizing…'),
+ };
+ default:
+ return {
+ variant: 'running',
+ title: t('Seer is investigating'),
+ statusLabel: t('Running…'),
+ };
+ }
+}
+
+/**
+ * The status block's content for a run, or `null` when there is nothing to say.
+ *
+ * `status` decides the variant and `phase` decides the words, which is why this
+ * reads both: every in-flight phase shares one `processing` status, and every
+ * stopped run shares the `completed`/`failed`/`cancelled` phases with its
+ * status. Taking the variant from the status keeps the colour tied to whether
+ * the viewer has to do anything.
+ */
+export function getSeerStatusBlock(
+ projection: InvestigationOrchestration
+): SeerStatusBlockContent | null {
+ switch (projection.status) {
+ // Stopped, but recoverably, and the only state that asks for something
+ // back. The agent's own prompt is far more specific than anything that
+ // could be written here, so it wins when present.
+ case 'awaiting_input':
+ return {
+ variant: 'awaitingInput',
+ title: t('Seer needs more information to continue'),
+ description:
+ projection.pendingInput?.prompt ||
+ t('Seer is waiting on input before it can carry on.'),
+ statusLabel: t('Awaiting input'),
+ };
+ case 'failed':
+ return {
+ variant: 'failed',
+ title: t("Seer couldn't finish this investigation"),
+ description:
+ getFailureMessage(projection) ??
+ t('Checks that had already finished are saved.'),
+ statusLabel: t('Failed'),
+ };
+ case 'cancelled':
+ return {
+ variant: 'cancelled',
+ title: t('This investigation was stopped'),
+ description: t('Checks that had already finished are saved.'),
+ statusLabel: t('Cancelled'),
+ };
+ case 'completed':
+ return {
+ variant: 'complete',
+ title: t('Your investigation is ready'),
+ meta: getMeta(projection),
+ description: t(
+ 'Findings, supporting evidence, and recommended next steps are ready.'
+ ),
+ statusLabel: t('Complete'),
+ };
+ case 'pending':
+ case 'processing':
+ return getRunningContent(projection);
+ default:
+ // An unrecognized status is still a run in progress as far as the viewer
+ // is concerned — Seer can add one before this code knows the name, and a
+ // missing block reads as "nothing is happening", which is worse.
+ return getRunningContent(projection);
+ }
+}
diff --git a/static/app/views/investigations/statusBlock/seerStatusBlock.spec.tsx b/static/app/views/investigations/statusBlock/seerStatusBlock.spec.tsx
new file mode 100644
index 000000000000..544b29c48494
--- /dev/null
+++ b/static/app/views/investigations/statusBlock/seerStatusBlock.spec.tsx
@@ -0,0 +1,169 @@
+import {render, screen} from 'sentry-test/reactTestingLibrary';
+
+import {
+ InvestigationHypothesisFixture,
+ InvestigationOrchestrationFixture,
+ InvestigationVerificationStepFixture,
+} from 'sentry/views/investigations/fixtures';
+import {getSeerStatusBlock} from 'sentry/views/investigations/statusBlock/getSeerStatusBlock';
+import {SeerStatusBlock} from 'sentry/views/investigations/statusBlock/seerStatusBlock';
+
+describe('SeerStatusBlock', () => {
+ it('renders the sentence, the chip, and the elapsed time', () => {
+ render(
+
+ );
+
+ expect(screen.getByText('Seer is looking for likely causes')).toBeInTheDocument();
+ expect(screen.getByText('Possible causes will appear here.')).toBeInTheDocument();
+ expect(screen.getByText('Running…')).toBeInTheDocument();
+ expect(screen.getByText('101.5s')).toBeInTheDocument();
+ });
+
+ it('omits the elapsed time when there is nothing to count from', () => {
+ render(
+
+ );
+
+ expect(screen.queryByText(/\ds$/)).not.toBeInTheDocument();
+ });
+
+ it('renders an action only when one is supplied', () => {
+ const {rerender} = render(
+
+ );
+
+ expect(screen.queryByTestId('seer-status-block-action')).not.toBeInTheDocument();
+
+ rerender(
+ Connect Datadog}
+ />
+ );
+
+ expect(screen.getByTestId('seer-status-block-action')).toBeInTheDocument();
+ expect(screen.getByRole('button', {name: 'Connect Datadog'})).toBeInTheDocument();
+ });
+});
+
+describe('getSeerStatusBlock', () => {
+ it.each([
+ ['awaiting_input', 'awaitingInput', 'Awaiting input'],
+ ['failed', 'failed', 'Failed'],
+ ['cancelled', 'cancelled', 'Cancelled'],
+ ['completed', 'complete', 'Complete'],
+ ] as const)(
+ 'maps the %s run status to the %s variant',
+ (status, variant, statusLabel) => {
+ const block = getSeerStatusBlock(
+ InvestigationOrchestrationFixture({status, errors: []})
+ );
+
+ expect(block).toMatchObject({variant, statusLabel});
+ }
+ );
+
+ // Every in-flight phase shares one `processing` status, so the phase is the
+ // only thing that can tell them apart.
+ it.each([
+ ['intake', 'Seer is gathering context', 'Running…'],
+ ['broad_scan', 'Seer is gathering context', 'Running…'],
+ ['planning', 'Seer is looking for likely causes', 'Running…'],
+ ['reporting', 'Seer is bringing the findings together', 'Finalizing…'],
+ ['metadata', 'Seer is bringing the findings together', 'Finalizing…'],
+ ] as const)('reads the %s phase as "%s"', (phase, title, statusLabel) => {
+ const block = getSeerStatusBlock(
+ InvestigationOrchestrationFixture({status: 'processing', phase})
+ );
+
+ expect(block).toMatchObject({variant: 'running', title, statusLabel});
+ });
+
+ it('counts the hypotheses it is checking', () => {
+ const block = getSeerStatusBlock(
+ InvestigationOrchestrationFixture({status: 'processing', phase: 'investigating'})
+ );
+
+ expect(block?.title).toBe(
+ 'Seer found 3 possible causes and is checking for evidence'
+ );
+ });
+
+ it('does not quote a count before any hypothesis has landed', () => {
+ const block = getSeerStatusBlock(
+ InvestigationOrchestrationFixture({
+ status: 'processing',
+ phase: 'investigating',
+ hypotheses: [],
+ })
+ );
+
+ expect(block?.title).toBe('Seer is checking for evidence');
+ });
+
+ // A check that broke still ran: the tally is how much work stands behind the
+ // report, not how much of it succeeded.
+ it('tallies checks that errored alongside those that produced a result', () => {
+ const block = getSeerStatusBlock(
+ InvestigationOrchestrationFixture({
+ status: 'completed',
+ hypotheses: [
+ InvestigationHypothesisFixture({
+ verificationSteps: [
+ InvestigationVerificationStepFixture({id: 'a', result: 'Found it.'}),
+ InvestigationVerificationStepFixture({
+ id: 'b',
+ result: null,
+ error: {code: 'timeout', message: 'Timed out.', retryable: true},
+ }),
+ InvestigationVerificationStepFixture({id: 'c', result: null, error: null}),
+ ],
+ }),
+ ],
+ })
+ );
+
+ expect(block?.meta).toBe('1 possible cause • 2 checks completed');
+ });
+
+ it('prefers the run error over the report error', () => {
+ const block = getSeerStatusBlock(
+ InvestigationOrchestrationFixture({
+ status: 'failed',
+ errors: [
+ {code: 'timeout', message: 'The trace request timed out.', retryable: true},
+ ],
+ })
+ );
+
+ expect(block?.description).toBe('The trace request timed out.');
+ });
+
+ // Seer can introduce a status before this code knows the name. A missing
+ // block would read as "nothing is happening", which is worse than a generic
+ // one.
+ it('still renders a block for an unrecognized status', () => {
+ const block = getSeerStatusBlock(
+ InvestigationOrchestrationFixture({status: 'regrouping', phase: 'planning'})
+ );
+
+ expect(block).toMatchObject({variant: 'running'});
+ });
+});
diff --git a/static/app/views/investigations/statusBlock/seerStatusBlock.stories.tsx b/static/app/views/investigations/statusBlock/seerStatusBlock.stories.tsx
new file mode 100644
index 000000000000..ab0aa4317e22
--- /dev/null
+++ b/static/app/views/investigations/statusBlock/seerStatusBlock.stories.tsx
@@ -0,0 +1,201 @@
+import {Fragment} from 'react';
+
+import {Button} from '@sentry/scraps/button';
+import {Flex, Stack} from '@sentry/scraps/layout';
+import {Text} from '@sentry/scraps/text';
+
+import {IconAdd} from 'sentry/icons';
+import * as Storybook from 'sentry/stories';
+import {SeerStatusBlock} from 'sentry/views/investigations/statusBlock/seerStatusBlock';
+
+export default Storybook.story('Investigations — Seer status block', story => {
+ story('While the agent is working', () => (
+
+
+ The status block is the line above the hypotheses that says what Seer is doing.
+ There is exactly one on the page, in a fixed spot, for the whole life of a run —
+ so a reader who has learned where to look for "what is happening" never has to
+ relearn it when the run changes state.
+
+
+ Every phase the agent moves through on its own — gathering context, forming
+ hypotheses, checking evidence, composing the report — is the same{' '}
+ running variant. They differ in what they say, not how they look,
+ which is why the sentence and the chip label are props rather than another
+ variant. Note that "Finalizing…" is a running block too.
+
+ awaitingInput is the only state that has stopped recoverably
+ , and the only one that renders an action. Every other state is the
+ agent's to advance, so giving them a button would imply the viewer is holding
+ things up when they are not.
+
+
+ It is also one of only two states that colour their title. A run that is simply
+ working, or has finished cleanly, leaves the sentence in the ordinary heading
+ colour and lets the chip carry the state — otherwise every block on the page
+ shouts and none of them reads as urgent.
+
+
+
+
+
+ Datadog
+
+ Redis resource pressure and database latency
+
+ Aug 27, 09:00–11:00 UTC
+
+
+ }>
+ Connect Datadog
+
+
+ }
+ />
+
+
+ ));
+
+ story('When it has stopped', () => (
+
+
+ A failure is the other state that colours its title, because it is the only one
+ where nothing further will happen without someone reading the sentence.
+
+
+ cancelled is not in the design. It is here because the projection can
+ report it and the block still has to render something: falling through to{' '}
+ failed would paint a decision someone deliberately made bright red,
+ so it gets the neutral treatment instead.
+
+
+
+
+
+
+
+
+ ));
+
+ story('When it is done', () => (
+
+
+ The finished block is the one people scroll back to, so it carries a tally: how
+ many explanations were weighed, and how many checks stand behind them. The{' '}
+ meta line only appears once there is something to count, which is why
+ it is absent from the early running states above.
+
+ elapsed is optional, and the wired block currently leaves it out. The
+ projection carries no run-level start time — startedAt exists only on
+ individual block executions — so there is nothing honest to count from yet. The
+ prop is here because the design calls for it and the story can show it; the
+ component will start receiving a real value when the projection grows one.
+
+
+ When it is supplied it renders monospace and tabular, so a ticking counter does
+ not shuffle the chip sideways on every update.
+
+ The chip and the clock hold the top-right corner and never wrap under the
+ sentence: they are the part a viewer glances at. The title wraps around them
+ instead. Drag the demo's edge to watch it.
+
+
+
+
+
+ ));
+});
diff --git a/static/app/views/investigations/statusBlock/seerStatusBlock.tsx b/static/app/views/investigations/statusBlock/seerStatusBlock.tsx
new file mode 100644
index 000000000000..a4872ec4cab9
--- /dev/null
+++ b/static/app/views/investigations/statusBlock/seerStatusBlock.tsx
@@ -0,0 +1,204 @@
+import type {ReactNode} from 'react';
+
+import {Tag} from '@sentry/scraps/badge';
+import {Container, Flex, Stack} from '@sentry/scraps/layout';
+import {Text} from '@sentry/scraps/text';
+
+import {LoadingIndicator} from 'sentry/components/loadingIndicator';
+import {
+ IconCircleCheckmark,
+ IconCircleDashed,
+ IconFatal,
+ IconWarning,
+} from 'sentry/icons';
+
+/**
+ * Where an agentic run has got to, as one line the viewer can read without
+ * opening anything.
+ *
+ * `running` covers every phase the agent moves through on its own — gathering
+ * context, forming hypotheses, checking evidence, composing the report. They
+ * differ in what they *say*, not in how they look, so they share one variant
+ * and the caller supplies the sentence.
+ *
+ * The other three have all stopped. `awaitingInput` has stopped recoverably and
+ * is the only one that asks for something back, which is why it is the only one
+ * that renders an action.
+ */
+export type SeerStatusBlockVariant =
+ | 'running'
+ | 'awaitingInput'
+ | 'failed'
+ | 'complete'
+ // Not one of the designed states. A cancelled run still has to render, and
+ // falling through to `failed` would report a decision someone made as an
+ // error, so it gets the neutral treatment instead of a red one.
+ | 'cancelled';
+
+/**
+ * Only a state that has stopped and needs attention colours its title. A run
+ * that is simply working, or has finished cleanly, leaves the sentence in the
+ * ordinary heading colour and lets the chip carry the state — otherwise every
+ * status block on the page shouts.
+ */
+const TITLE_VARIANT = {
+ running: undefined,
+ awaitingInput: 'warning',
+ failed: 'danger',
+ complete: undefined,
+ cancelled: 'muted',
+} as const;
+
+const TAG_VARIANT = {
+ // `info` is the accent-purple pill: `content.accent` on
+ // `background.transparent.accent.muted`, which is what the design names.
+ running: 'info',
+ awaitingInput: 'warning',
+ failed: 'danger',
+ complete: 'success',
+ cancelled: 'muted',
+} as const;
+
+function StatusIcon({variant}: {variant: SeerStatusBlockVariant}) {
+ switch (variant) {
+ case 'running':
+ // A ring rather than a pulsing dot: the run is doing something, not
+ // sitting in a state.
+ return ;
+ case 'awaitingInput':
+ return ;
+ case 'failed':
+ return ;
+ case 'complete':
+ return ;
+ // A stopped run that is nobody's problem. A dashed ring reads as "this one
+ // is not going anywhere" without claiming anything went wrong.
+ case 'cancelled':
+ return ;
+ default:
+ return null;
+ }
+}
+
+type SeerStatusBlockProps = {
+ /** The short pill on the right, e.g. "Running…", "Awaiting input". */
+ statusLabel: string;
+ /** The sentence the block leads with, in the agent's voice. */
+ title: string;
+ variant: SeerStatusBlockVariant;
+ /**
+ * What the viewer can do about it. Only `awaitingInput` should supply one —
+ * every other state is the agent's to advance, and an action would imply
+ * otherwise.
+ */
+ action?: ReactNode;
+ className?: string;
+ /** The paragraph under the title. */
+ description?: string;
+ /**
+ * How long the run has been going, already formatted (e.g. "101.5s"). Left
+ * out when there is nothing to measure from: the projection carries no
+ * run-level start time, so the wired block omits this rather than invent one.
+ */
+ elapsed?: string;
+ /**
+ * A tally between the title and the description, e.g.
+ * "4 possible causes · 9 checks completed". Only worth showing once there is
+ * something to count.
+ */
+ meta?: string;
+};
+
+/**
+ * The status line above an agentic investigation's hypotheses.
+ *
+ * One component covers the whole run lifecycle because the shape never changes
+ * — icon, sentence, chip, elapsed time — only the words and the colour do. That
+ * is deliberate: the block sits in a fixed spot at the top of the panel, and a
+ * reader who has learned where to look for "what is Seer doing" should not have
+ * to relearn it when the run changes state.
+ *
+ * It is presentational and knows nothing about the projection, so it can be
+ * driven from a story, a fixture, or the live run.
+ */
+export function SeerStatusBlock({
+ action,
+ className,
+ description,
+ elapsed,
+ meta,
+ statusLabel,
+ title,
+ variant,
+}: SeerStatusBlockProps) {
+ return (
+
+
+ {/*
+ * A fixed column so the title, the description and the action all line
+ * up on the same left edge regardless of which icon is showing. `16px`
+ * is the title's line height, which centres the icon against the first
+ * line rather than the block.
+ */}
+
+
+
+
+
+
+
+ {title}
+
+ {/*
+ * The chip and the clock never wrap under the sentence: they are
+ * the part a viewer glances at, so they hold the top-right corner
+ * and the title wraps around them instead.
+ */}
+
+ {statusLabel}
+ {elapsed ? (
+ // Monospace and tabular so a ticking counter does not shuffle
+ // the chip sideways on every update.
+
+ {elapsed}
+
+ ) : null}
+
+
+
+ {meta ? (
+
+ {meta}
+
+ ) : null}
+
+ {description ? (
+
+ {description}
+
+ ) : null}
+
+ {action ? (
+
+ {action}
+
+ ) : null}
+
+
+
+ );
+}
From a21fca3763dbf3c5c070258ed3785802e991c3e8 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Mon, 14 Sep 2026 15:54:39 -0400
Subject: [PATCH 11/21] update status colors
---
.../hypotheses/hypothesisStatus.tsx | 77 ++++++++++++++-----
1 file changed, 59 insertions(+), 18 deletions(-)
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
index 009170e316cd..f0c019e19000 100644
--- a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -1,3 +1,5 @@
+import styled from '@emotion/styled';
+
import {Flex} from '@sentry/scraps/layout';
import {StatusIndicator} from '@sentry/scraps/statusIndicator';
import {Text} from '@sentry/scraps/text';
@@ -208,25 +210,64 @@ export function HypothesisStatus({hypothesis}: HypothesisStatusProps) {
const confidence = getHypothesisConfidencePercent(hypothesis);
return (
- // "Evidence checked" is both a status and the heading over the steps, so
- // this needs to be addressable on its own.
-
- {inFlight ? (
- // Live work gets a ring rather than a dot: the agent is doing
- // something, not resting in a state. Every other status is a place the
- // hypothesis has come to a stop, however briefly.
-
-
+ // The dot and its label are one statement, so they are one color.
+ //
+ // Left to themselves they disagree: `StatusIndicator` fills from the
+ // `background.*.vibrant` ramp while `Text` paints from `content.*`, which
+ // is darker in every variant — several steps for `muted`. Side by side the
+ // dot read as a lighter mark unrelated to the label it belongs to.
+ //
+ // `Text` already owns that variant-to-token mapping, `muted` ->
+ // `content.secondary` included, so its render-prop form hands the styling
+ // to the row itself rather than to a span inside it. The dot then picks the
+ // color up as `currentColor`, and there is no second copy of the table here
+ // to fall out of step with the design system.
+
+ {({className}) => (
+ // "Evidence checked" is both a status and the heading over the steps,
+ // so this needs to be addressable on its own.
+
+ {inFlight ? (
+ // Live work gets a ring rather than a dot: the agent is doing
+ // something, not resting in a state. Every other status is a place
+ // the hypothesis has come to a stop, however briefly.
+
+
+
+ ) : (
+
+
+
+ )}
+ {confidence === null
+ ? label
+ : // Translators: e.g. "Supported · 86% Confidence"
+ t('%s · %s%% Confidence', label, confidence)}
- ) : (
-
)}
-
- {confidence === null
- ? label
- : // Translators: e.g. "Supported · 86% Confidence"
- t('%s · %s%% Confidence', label, confidence)}
-
-
+
);
}
+
+/**
+ * Pins the dot to the line's color.
+ *
+ * `StatusIndicator` exposes no color of its own — the variant is the whole API
+ * — so this repaints the dot it draws in `::after`. The pulse behind it
+ * (`::before`) is deliberately left on its translucent token: it is a halo, and
+ * giving it the text color would make it a second, solid dot. `variant` is
+ * still passed through, so if this override ever stops matching, the dot falls
+ * back to its own ramp rather than disappearing.
+ */
+const StatusDot = styled('span')`
+ display: inline-flex;
+
+ & > span::after {
+ background-color: currentColor;
+ }
+`;
From 816328443b83b8906806da5d977005dbdc92a1c1 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Mon, 14 Sep 2026 10:32:25 -0400
Subject: [PATCH 12/21] feat(investigations): Create agentic runs from both
entry points
Neither entry point produced an investigation with hypotheses. The server
decides from the request body -- a `source` with no `templateKey` builds an
agentic run, anything else builds a plain notebook -- and both buttons sent
a shape that landed elsewhere.
"Investigate" on a metric issue now sends the metric snapshot alone,
dropping the template key. The candidates endpoint already matches agentic
and template lineage keys alike, so a breach that was investigated before
still resolves to "View" rather than offering a duplicate.
"New investigation" sends a manual source. That run opens `awaiting_input`
and stays there until someone supplies a prompt, which nothing in the UI
does yet, so the investigation behaves as before -- an empty notebook --
with an idle run attached.
That idle run is why polling changed. The predicate asked whether a run had
reached a terminal status, and `awaiting_input` has not; left alone, every
newly created investigation would have polled the detail and orchestration
endpoints every two seconds forever. It now asks whether the agent is
advancing, which `awaiting_input` is not: it is blocked on a person, and
supplying input writes the new projection into the cache and starts it
again.
Note that this does not gate the behaviour behind anything beyond the
existing organizations:investigations flag -- every investigation created
in a flagged org becomes agentic.
Claude-Session: https://claude.ai/code/session_012CtaiBZtJdz8uMRtUgvbmv
---
static/app/views/investigations/api.ts | 21 +++++++----
.../app/views/investigations/detail/index.tsx | 4 +--
.../investigationHypotheses.spec.tsx | 21 ++++++++++-
.../hypotheses/investigationHypotheses.tsx | 36 ++++++++++++++-----
.../app/views/investigations/index.spec.tsx | 5 ++-
.../metricDetectorTriggeredSection.spec.tsx | 4 +--
6 files changed, 70 insertions(+), 21 deletions(-)
diff --git a/static/app/views/investigations/api.ts b/static/app/views/investigations/api.ts
index 016d617510de..c05ccfb93d77 100644
--- a/static/app/views/investigations/api.ts
+++ b/static/app/views/investigations/api.ts
@@ -276,6 +276,14 @@ function useInvestigationMutation(
});
}
+/**
+ * Start an empty investigation.
+ *
+ * A `source` with no `templateKey` is what makes the server build an agentic
+ * run rather than a bare notebook, so this is the field that decides whether
+ * the investigation ever has hypotheses. A manual source carries no prompt yet,
+ * so the run opens `awaiting_input` and waits for one.
+ */
export function useCreateInvestigationMutation(
organizationSlug: string,
options?: MutationOptions
@@ -288,7 +296,7 @@ export function useCreateInvestigationMutation(
path: {organizationIdOrSlug: organizationSlug},
}),
method: 'POST',
- data: {title: 'Untitled investigation'},
+ data: {title: 'Untitled investigation', source: {type: 'manual'}},
}),
options
);
@@ -306,11 +314,12 @@ export function useLaunchInvestigationMutation(
path: {organizationIdOrSlug: organizationSlug},
}),
method: 'POST',
- data: {
- templateKey: 'breached_metric',
- templateVersion: 1,
- source,
- },
+ // No `templateKey`: the metric snapshot is enough for the server to
+ // build an agentic run, which is what gives this investigation
+ // hypotheses instead of a fixed sequence of notebook cells. The
+ // candidates endpoint matches agentic and template lineage keys alike,
+ // so an already-investigated breach still resolves to "View".
+ data: {source},
}),
options,
{invalidateCandidates: true}
diff --git a/static/app/views/investigations/detail/index.tsx b/static/app/views/investigations/detail/index.tsx
index 5bb08275242c..ae61a7927501 100644
--- a/static/app/views/investigations/detail/index.tsx
+++ b/static/app/views/investigations/detail/index.tsx
@@ -46,7 +46,7 @@ import {
} from 'sentry/views/investigations/detail/cell';
import {
InvestigationHypotheses,
- isInvestigationRunSettled,
+ shouldPollInvestigationRun,
} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
import {updateInvestigationCache} from 'sentry/views/investigations/investigationCache';
import {InvestigationSummaryCard} from 'sentry/views/investigations/investigationSummaryCard';
@@ -101,7 +101,7 @@ export function InvestigationBootstrapPage({investigationId}: {investigationId:
// row, so a stale copy would leave the row hidden or showing a run that
// has since finished.
const orchestrationActive =
- data?.orchestration && !isInvestigationRunSettled(data.orchestration.status);
+ data?.orchestration && shouldPollInvestigationRun(data.orchestration.status);
return orchestrationActive ||
shouldPollInvestigationBlocks(data?.blocks ?? []) ||
isTitleGenerationActive(data?.titleGeneration?.status)
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
index b660ee86b5b5..aa1bf9010eb1 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
@@ -5,7 +5,10 @@ import {makeTestQueryClient} from 'sentry-test/queryClient';
import {render, screen, userEvent, waitFor} from 'sentry-test/reactTestingLibrary';
import {InvestigationOrchestrationFixture} from 'sentry/views/investigations/fixtures';
-import {InvestigationHypotheses} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
+import {
+ InvestigationHypotheses,
+ shouldPollInvestigationRun,
+} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
import type {InvestigationOrchestration} from 'sentry/views/investigations/types';
const organization = OrganizationFixture({features: ['investigations']});
@@ -22,6 +25,22 @@ function renderHypotheses() {
});
}
+describe('shouldPollInvestigationRun', () => {
+ it.each([
+ ['pending', true],
+ ['processing', true],
+ [undefined, true],
+ // Blocked on a person, not on the agent. Every investigation created
+ // without a prompt starts here, so polling would never stop.
+ ['awaiting_input', false],
+ ['completed', false],
+ ['failed', false],
+ ['cancelled', false],
+ ] as const)('%s polls: %s', (status, expected) => {
+ expect(shouldPollInvestigationRun(status)).toBe(expected);
+ });
+});
+
describe('InvestigationHypotheses', () => {
it('renders the hypotheses carried on the projection', async () => {
MockApiClient.addMockResponse({
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
index e1f58eaee57c..1084a4a1cbaa 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
@@ -22,18 +22,34 @@ import type {
const POLL_INTERVAL_MS = 2000;
/**
- * Whether a workflow has stopped moving on its own.
+ * Statuses where the agent is not going to move on its own. The first three
+ * have stopped for good; `awaiting_input` has stopped recoverably, blocked on a
+ * person.
+ */
+const STOPPED_STATUSES = new Set([
+ 'completed',
+ 'failed',
+ 'cancelled',
+ 'awaiting_input',
+]);
+
+/**
+ * Whether a run is still advancing, and so worth polling.
+ *
+ * `awaiting_input` counts as stopped even though it can resume: an
+ * investigation created without a prompt starts there and stays there until
+ * someone supplies one, so polling it would be a permanent two-second request
+ * loop on a run nobody is driving. Supplying input from this client writes the
+ * new projection straight into the cache, which starts it again.
*
- * `awaiting_input` is deliberately not terminal: the run resumes as soon as
- * input arrives, which may happen from another surface, so polling has to
- * continue. Exported because the detail view decides from the summary served
- * alongside the investigation, and this component from the full projection —
- * the same three statuses either way.
+ * Exported because the detail view decides from the summary served alongside
+ * the investigation and this component from the full projection — the same
+ * statuses either way.
*/
-export function isInvestigationRunSettled(
+export function shouldPollInvestigationRun(
status: InvestigationOrchestrationStatus | undefined
): boolean {
- return status === 'completed' || status === 'failed' || status === 'cancelled';
+ return status === undefined || !STOPPED_STATUSES.has(status);
}
type InvestigationHypothesesProps = {
@@ -68,7 +84,9 @@ export function InvestigationHypotheses({
...investigationOrchestrationQueryOptions(organization.slug, investigationId),
enabled,
refetchInterval: query =>
- isInvestigationRunSettled(query.state.data?.json.status) ? false : POLL_INTERVAL_MS,
+ shouldPollInvestigationRun(query.state.data?.json.status)
+ ? POLL_INTERVAL_MS
+ : false,
});
const commandMutation = useInvestigationOrchestrationCommandMutation(
diff --git a/static/app/views/investigations/index.spec.tsx b/static/app/views/investigations/index.spec.tsx
index 3e01d7f0c84a..1acce5c09538 100644
--- a/static/app/views/investigations/index.spec.tsx
+++ b/static/app/views/investigations/index.spec.tsx
@@ -231,7 +231,10 @@ describe('Explore Investigations', () => {
await waitFor(() =>
expect(createRequest).toHaveBeenCalledWith(
listUrl,
- expect.objectContaining({data: {title: 'Untitled investigation'}})
+ expect.objectContaining({
+ // A source with no templateKey is what makes this agentic.
+ data: {title: 'Untitled investigation', source: {type: 'manual'}},
+ })
)
);
expect(await screen.findByText('Untitled investigation')).toBeInTheDocument();
diff --git a/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx b/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx
index b1f4f94edcd5..8c645f85bf31 100644
--- a/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx
+++ b/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx
@@ -293,9 +293,9 @@ describe('MetricDetectorTriggeredSection', () => {
expect(launchMock).toHaveBeenCalledWith(
expect.anything(),
expect.objectContaining({
+ // No templateKey: the server builds an agentic run from the metric
+ // snapshot rather than a fixed notebook.
data: {
- templateKey: 'breached_metric',
- templateVersion: 1,
source: {
type: 'metric_open_period',
ref: {groupId: defaultGroup.id, openPeriodId: '101'},
From 87ba476e625527c5af5d240295f0719034145992 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Mon, 14 Sep 2026 15:48:25 -0400
Subject: [PATCH 13/21] Revert "feat(investigations): Create agentic runs from
both entry points"
This reverts commit 02a6dba82d664381e1349bf7d380c2e487d3f389.
---
static/app/views/investigations/api.ts | 21 ++++-------
.../app/views/investigations/detail/index.tsx | 4 +--
.../investigationHypotheses.spec.tsx | 21 +----------
.../hypotheses/investigationHypotheses.tsx | 36 +++++--------------
.../app/views/investigations/index.spec.tsx | 5 +--
.../metricDetectorTriggeredSection.spec.tsx | 4 +--
6 files changed, 21 insertions(+), 70 deletions(-)
diff --git a/static/app/views/investigations/api.ts b/static/app/views/investigations/api.ts
index c05ccfb93d77..016d617510de 100644
--- a/static/app/views/investigations/api.ts
+++ b/static/app/views/investigations/api.ts
@@ -276,14 +276,6 @@ function useInvestigationMutation(
});
}
-/**
- * Start an empty investigation.
- *
- * A `source` with no `templateKey` is what makes the server build an agentic
- * run rather than a bare notebook, so this is the field that decides whether
- * the investigation ever has hypotheses. A manual source carries no prompt yet,
- * so the run opens `awaiting_input` and waits for one.
- */
export function useCreateInvestigationMutation(
organizationSlug: string,
options?: MutationOptions
@@ -296,7 +288,7 @@ export function useCreateInvestigationMutation(
path: {organizationIdOrSlug: organizationSlug},
}),
method: 'POST',
- data: {title: 'Untitled investigation', source: {type: 'manual'}},
+ data: {title: 'Untitled investigation'},
}),
options
);
@@ -314,12 +306,11 @@ export function useLaunchInvestigationMutation(
path: {organizationIdOrSlug: organizationSlug},
}),
method: 'POST',
- // No `templateKey`: the metric snapshot is enough for the server to
- // build an agentic run, which is what gives this investigation
- // hypotheses instead of a fixed sequence of notebook cells. The
- // candidates endpoint matches agentic and template lineage keys alike,
- // so an already-investigated breach still resolves to "View".
- data: {source},
+ data: {
+ templateKey: 'breached_metric',
+ templateVersion: 1,
+ source,
+ },
}),
options,
{invalidateCandidates: true}
diff --git a/static/app/views/investigations/detail/index.tsx b/static/app/views/investigations/detail/index.tsx
index ae61a7927501..5bb08275242c 100644
--- a/static/app/views/investigations/detail/index.tsx
+++ b/static/app/views/investigations/detail/index.tsx
@@ -46,7 +46,7 @@ import {
} from 'sentry/views/investigations/detail/cell';
import {
InvestigationHypotheses,
- shouldPollInvestigationRun,
+ isInvestigationRunSettled,
} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
import {updateInvestigationCache} from 'sentry/views/investigations/investigationCache';
import {InvestigationSummaryCard} from 'sentry/views/investigations/investigationSummaryCard';
@@ -101,7 +101,7 @@ export function InvestigationBootstrapPage({investigationId}: {investigationId:
// row, so a stale copy would leave the row hidden or showing a run that
// has since finished.
const orchestrationActive =
- data?.orchestration && shouldPollInvestigationRun(data.orchestration.status);
+ data?.orchestration && !isInvestigationRunSettled(data.orchestration.status);
return orchestrationActive ||
shouldPollInvestigationBlocks(data?.blocks ?? []) ||
isTitleGenerationActive(data?.titleGeneration?.status)
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
index aa1bf9010eb1..b660ee86b5b5 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
@@ -5,10 +5,7 @@ import {makeTestQueryClient} from 'sentry-test/queryClient';
import {render, screen, userEvent, waitFor} from 'sentry-test/reactTestingLibrary';
import {InvestigationOrchestrationFixture} from 'sentry/views/investigations/fixtures';
-import {
- InvestigationHypotheses,
- shouldPollInvestigationRun,
-} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
+import {InvestigationHypotheses} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
import type {InvestigationOrchestration} from 'sentry/views/investigations/types';
const organization = OrganizationFixture({features: ['investigations']});
@@ -25,22 +22,6 @@ function renderHypotheses() {
});
}
-describe('shouldPollInvestigationRun', () => {
- it.each([
- ['pending', true],
- ['processing', true],
- [undefined, true],
- // Blocked on a person, not on the agent. Every investigation created
- // without a prompt starts here, so polling would never stop.
- ['awaiting_input', false],
- ['completed', false],
- ['failed', false],
- ['cancelled', false],
- ] as const)('%s polls: %s', (status, expected) => {
- expect(shouldPollInvestigationRun(status)).toBe(expected);
- });
-});
-
describe('InvestigationHypotheses', () => {
it('renders the hypotheses carried on the projection', async () => {
MockApiClient.addMockResponse({
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
index 1084a4a1cbaa..e1f58eaee57c 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
@@ -22,34 +22,18 @@ import type {
const POLL_INTERVAL_MS = 2000;
/**
- * Statuses where the agent is not going to move on its own. The first three
- * have stopped for good; `awaiting_input` has stopped recoverably, blocked on a
- * person.
- */
-const STOPPED_STATUSES = new Set([
- 'completed',
- 'failed',
- 'cancelled',
- 'awaiting_input',
-]);
-
-/**
- * Whether a run is still advancing, and so worth polling.
- *
- * `awaiting_input` counts as stopped even though it can resume: an
- * investigation created without a prompt starts there and stays there until
- * someone supplies one, so polling it would be a permanent two-second request
- * loop on a run nobody is driving. Supplying input from this client writes the
- * new projection straight into the cache, which starts it again.
+ * Whether a workflow has stopped moving on its own.
*
- * Exported because the detail view decides from the summary served alongside
- * the investigation and this component from the full projection — the same
- * statuses either way.
+ * `awaiting_input` is deliberately not terminal: the run resumes as soon as
+ * input arrives, which may happen from another surface, so polling has to
+ * continue. Exported because the detail view decides from the summary served
+ * alongside the investigation, and this component from the full projection —
+ * the same three statuses either way.
*/
-export function shouldPollInvestigationRun(
+export function isInvestigationRunSettled(
status: InvestigationOrchestrationStatus | undefined
): boolean {
- return status === undefined || !STOPPED_STATUSES.has(status);
+ return status === 'completed' || status === 'failed' || status === 'cancelled';
}
type InvestigationHypothesesProps = {
@@ -84,9 +68,7 @@ export function InvestigationHypotheses({
...investigationOrchestrationQueryOptions(organization.slug, investigationId),
enabled,
refetchInterval: query =>
- shouldPollInvestigationRun(query.state.data?.json.status)
- ? POLL_INTERVAL_MS
- : false,
+ isInvestigationRunSettled(query.state.data?.json.status) ? false : POLL_INTERVAL_MS,
});
const commandMutation = useInvestigationOrchestrationCommandMutation(
diff --git a/static/app/views/investigations/index.spec.tsx b/static/app/views/investigations/index.spec.tsx
index 1acce5c09538..3e01d7f0c84a 100644
--- a/static/app/views/investigations/index.spec.tsx
+++ b/static/app/views/investigations/index.spec.tsx
@@ -231,10 +231,7 @@ describe('Explore Investigations', () => {
await waitFor(() =>
expect(createRequest).toHaveBeenCalledWith(
listUrl,
- expect.objectContaining({
- // A source with no templateKey is what makes this agentic.
- data: {title: 'Untitled investigation', source: {type: 'manual'}},
- })
+ expect.objectContaining({data: {title: 'Untitled investigation'}})
)
);
expect(await screen.findByText('Untitled investigation')).toBeInTheDocument();
diff --git a/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx b/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx
index 8c645f85bf31..b1f4f94edcd5 100644
--- a/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx
+++ b/static/app/views/issueDetails/sidebar/metricDetectorTriggeredSection.spec.tsx
@@ -293,9 +293,9 @@ describe('MetricDetectorTriggeredSection', () => {
expect(launchMock).toHaveBeenCalledWith(
expect.anything(),
expect.objectContaining({
- // No templateKey: the server builds an agentic run from the metric
- // snapshot rather than a fixed notebook.
data: {
+ templateKey: 'breached_metric',
+ templateVersion: 1,
source: {
type: 'metric_open_period',
ref: {groupId: defaultGroup.id, openPeriodId: '101'},
From 487c54c0be6411c77dc49277eb557905637ba2d5 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Mon, 14 Sep 2026 16:23:56 -0400
Subject: [PATCH 14/21] fix(investigations): Wrap long symbols in hypothesis
card text
The agent writes its findings in terms of what it read, so they are mostly symbols: module/file.py::function_name, dotted paths, issue short IDs. None of them carry a break opportunity, so a token wider than the column pushed its own text out through the card edge instead of wrapping inside it.
Set the scraps wordBreak prop on every agent-written string in the card, not just the evidence detail that surfaced it.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../hypotheses/hypothesisCard.tsx | 28 ++++++++++++++-----
1 file changed, 21 insertions(+), 7 deletions(-)
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
index aef444dccd4a..457fb561b78b 100644
--- a/static/app/views/investigations/hypotheses/hypothesisCard.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.tsx
@@ -91,18 +91,18 @@ export function HypothesisCard({
-
+
{hypothesis.statement}
{hypothesis.rationale ? (
-
+
{hypothesis.rationale}
) : null}
{hypothesis.error ? (
-
+
{hypothesis.error.message}
) : null}
@@ -137,8 +137,22 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
// A full flex-basis, because the chevron's button grows too — without this
// the two split the row and the text wraps in half the width it has.
- {step.title}
-
+
+ {step.title}
+
+ {/*
+ * The agent writes these in terms of what it read, so a finding is mostly
+ * symbols: `module/file.py::function_name`, dotted paths, issue short IDs.
+ * None of them carry a break opportunity, and one long enough to outrun
+ * the column would otherwise push its own text out through the card edge
+ * rather than wrap inside it.
+ */}
+
{detail}
@@ -174,7 +188,7 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
{t('Objective')}
-
+
{step.objective}
@@ -182,7 +196,7 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
{t('Method')}
-
+
{step.method}
From 45de4822ee616c314794703fa475741adbc72919 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Mon, 14 Sep 2026 16:24:13 -0400
Subject: [PATCH 15/21] ref(investigations): Simplify the detail header and
content layout
Drop the block count from the subheader, the status badge and the Seer mark from the header actions, and the debug-only add text/query cell composer.
The main content area now fills the page and left-aligns with the rest of the product rather than sitting in a centred 884px column. Three separate caps were producing that column: the header grid, the canvas wrapper, and the two stacks around the hypothesis row and the notebook.
Removing the composer left useAddInvestigationBlockMutation with no callers, so it goes too. The specs that reached the behaviour through it now exercise it directly: the never-run cell test uses the fixture's own unrun query block, and the fixture-API mutation test drives a rename instead.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../investigationFixtureApi.spec.tsx | 32 +--
static/app/views/investigations/api.ts | 62 ------
.../investigations/detail/index.spec.tsx | 121 ++----------
.../app/views/investigations/detail/index.tsx | 187 +-----------------
4 files changed, 39 insertions(+), 363 deletions(-)
diff --git a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
index 630200d99581..e8a5900f3e68 100644
--- a/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
+++ b/static/app/views/investigations/__stories__/investigationFixtureApi.spec.tsx
@@ -1,6 +1,6 @@
import {OrganizationFixture} from 'sentry-fixture/organization';
-import {render, screen, userEvent} from 'sentry-test/reactTestingLibrary';
+import {render, screen, userEvent, waitFor} from 'sentry-test/reactTestingLibrary';
import {QUERY_API_CLIENT} from 'sentry/utils/queryClient';
import {InvestigationsPage} from 'sentry/views/investigations';
@@ -83,22 +83,22 @@ describe('InvestigationFixtureApi', () => {
investigation.title
);
- await userEvent.click(
- screen.getByRole('button', {name: 'Add query cell (debug only)'})
- );
- await userEvent.type(
- screen.getByRole('textbox', {name: 'Cell title'}),
- 'Slow checkouts'
- );
- await userEvent.type(
- screen.getByRole('textbox', {name: 'Cell instructions'}),
- 'Compare checkout p95 before and after the deploy.'
- );
- await userEvent.click(screen.getByRole('button', {name: 'Add cell'}));
+ // Renaming is the detail mutation the page drives itself. Blurring the
+ // field cancels the debounce and writes immediately, so this needs no timer.
+ const titleField = screen.getByRole('textbox', {name: 'Investigation title'});
+ await userEvent.clear(titleField);
+ await userEvent.type(titleField, 'Invoice PDF timeouts everywhere');
+ await userEvent.tab();
- expect(
- await screen.findByRole('button', {name: 'Toggle Slow checkouts'})
- ).toBeInTheDocument();
+ // Read back through the fixture API rather than the field: the input would
+ // show the new title from the optimistic cache update either way, so only a
+ // fresh fetch proves the fixture backend actually stored it.
+ await waitFor(async () => {
+ const stored = await QUERY_API_CLIENT.requestPromise(
+ `/organizations/storybook-investigation-detail-test/investigations/${investigation.id}/`
+ );
+ expect(stored.title).toBe('Invoice PDF timeouts everywhere');
+ });
});
it('keeps fixture IDs and block positions unique across mutations', async () => {
diff --git a/static/app/views/investigations/api.ts b/static/app/views/investigations/api.ts
index 016d617510de..98525d8e46f5 100644
--- a/static/app/views/investigations/api.ts
+++ b/static/app/views/investigations/api.ts
@@ -11,7 +11,6 @@ import type {
InvestigationCandidate,
InvestigationBlock,
InvestigationBlockExecutionStart,
- InvestigationBlockKind,
InvestigationDetail,
InvestigationExecutionDetail,
InvestigationListItem,
@@ -211,13 +210,6 @@ type FavoriteVariables = {
shouldFavorite: boolean;
};
-type AddBlockVariables = {
- investigation: InvestigationDetail;
- kind: InvestigationBlockKind;
- prompt: string;
- title: string;
-};
-
type RunBlockVariables = {
block: InvestigationBlock;
investigationVersion: number;
@@ -378,60 +370,6 @@ export function useRenameInvestigationMutation(
});
}
-export function useAddInvestigationBlockMutation(
- organizationSlug: string,
- investigationId: string,
- options?: MutationOptions
-) {
- const queryClient = useQueryClient();
- const detailOptions = getInvestigationDetailQueryOptions(
- organizationSlug,
- investigationId
- );
-
- return useMutation({
- ...options,
- mutationFn: ({investigation, kind, prompt, title}) =>
- fetchMutation({
- url: getApiUrl(
- '/organizations/$organizationIdOrSlug/investigations/$investigationId/blocks/',
- {
- path: {
- organizationIdOrSlug: organizationSlug,
- investigationId,
- },
- }
- ),
- method: 'POST',
- data: {
- investigationVersion: investigation.version,
- kind,
- title,
- generationPrompt: prompt,
- },
- }),
- onSuccess: async (block, variables, onMutateResult, context) => {
- queryClient.setQueryData(detailOptions.queryKey, current =>
- current
- ? {
- ...current,
- json: {
- ...current.json,
- blockCount: current.json.blockCount + 1,
- blocks: [...(current.json.blocks ?? []), block],
- version: current.json.version + 1,
- },
- }
- : current
- );
- await queryClient.invalidateQueries({
- queryKey: investigationListQueryOptions({organizationSlug}).queryKey,
- });
- await options?.onSuccess?.(block, variables, onMutateResult, context);
- },
- });
-}
-
export function useDeleteInvestigationBlockMutation(
organizationSlug: string,
investigationId: string,
diff --git a/static/app/views/investigations/detail/index.spec.tsx b/static/app/views/investigations/detail/index.spec.tsx
index 77029714cc81..c8694507fc50 100644
--- a/static/app/views/investigations/detail/index.spec.tsx
+++ b/static/app/views/investigations/detail/index.spec.tsx
@@ -127,8 +127,6 @@ describe('Investigation detail', () => {
expect(screen.getByTestId('loading-indicator')).toBeInTheDocument();
expect(await screen.findByText('Investigate database latency')).toBeInTheDocument();
- expect(screen.getByText('Active')).toBeInTheDocument();
- expect(screen.queryByText('Completed')).not.toBeInTheDocument();
expect(screen.getByLabelText('Cell actions for Summary')).toBeInTheDocument();
expect(screen.getByLabelText('Cell actions for Latency query')).toBeInTheDocument();
expect(
@@ -988,106 +986,15 @@ describe('Investigation detail', () => {
).not.toBeInTheDocument();
});
- it('adds text and query cells and starts a never-run cell from its stored prompt', async () => {
- const fixture = InvestigationDetailFixture();
- const textTemplate = fixture.blocks[0];
- const queryTemplate = fixture.blocks[1];
- if (!textTemplate || !queryTemplate) {
- throw new Error('Expected text and query block fixtures.');
- }
- MockApiClient.addMockResponse({
- url: detailUrl,
- body: InvestigationDetailFixture({blocks: [], blockCount: 0}),
- });
- const blocksUrl = `${detailUrl}blocks/`;
- const textBlock = {
- ...textTemplate,
- id: 'text-block',
- title: 'Working theory',
- generationPrompt: 'Summarize the current evidence',
- };
- const textRequest = MockApiClient.addMockResponse({
- url: blocksUrl,
- method: 'POST',
- body: textBlock,
- });
-
- renderView();
- await userEvent.click(
- await screen.findByRole('button', {name: 'Add text cell (debug only)'})
- );
- await userEvent.type(screen.getByLabelText('Cell title'), 'Working theory');
- await userEvent.type(
- screen.getByLabelText('Cell instructions'),
- 'Summarize the current evidence'
- );
- await userEvent.click(screen.getByRole('button', {name: 'Add cell'}));
-
- await waitFor(() =>
- expect(textRequest).toHaveBeenCalledWith(
- blocksUrl,
- expect.objectContaining({
- data: {
- investigationVersion: 1,
- kind: 'text',
- title: 'Working theory',
- generationPrompt: 'Summarize the current evidence',
- },
- })
- )
- );
- expect(
- await screen.findByRole('button', {
- name: 'Cell actions for Working theory',
- })
- ).toBeInTheDocument();
- expect(screen.queryByDisplayValue('Working theory')).not.toBeInTheDocument();
-
- const queryBlock = {
- ...queryTemplate,
- id: 'query-block',
- title: 'Error volume',
- generationPrompt: 'Show errors over the last 24 hours',
- };
- const queryRequest = MockApiClient.addMockResponse({
- url: blocksUrl,
- method: 'POST',
- body: queryBlock,
- });
- await userEvent.click(
- screen.getByRole('button', {name: 'Add query cell (debug only)'})
- );
- await userEvent.type(screen.getByLabelText('Cell title'), 'Error volume');
- await userEvent.type(
- screen.getByLabelText('Cell instructions'),
- 'Show errors over the last 24 hours'
- );
- await userEvent.click(screen.getByRole('button', {name: 'Add cell'}));
-
- await waitFor(() =>
- expect(queryRequest).toHaveBeenCalledWith(
- blocksUrl,
- expect.objectContaining({
- data: {
- investigationVersion: 2,
- kind: 'query',
- title: 'Error volume',
- generationPrompt: 'Show errors over the last 24 hours',
- },
- })
- )
- );
- expect(
- (await screen.findAllByTestId('query-cell-title')).some(
- element => element.textContent === 'Error volume'
- )
- ).toBe(true);
-
- const updateUrl = `${blocksUrl}query-block/`;
+ it('starts a never-run cell from its stored prompt', async () => {
+ // `Latency query` arrives from the fixture never run, carrying the prompt
+ // it was created with — which is the state Refine is meant to pick up.
+ MockApiClient.addMockResponse({url: detailUrl, body: InvestigationDetailFixture()});
+ const updateUrl = `${detailUrl}blocks/block-2/`;
const updateRequest = MockApiClient.addMockResponse({
url: updateUrl,
method: 'PUT',
- body: queryBlock,
+ body: InvestigationDetailFixture().blocks[1],
});
const runUrl = `${updateUrl}executions/`;
const runRequest = MockApiClient.addMockResponse({
@@ -1107,10 +1014,12 @@ describe('Investigation detail', () => {
error: null,
},
});
- await chooseCellAction('Error volume', 'Refine');
- expect(screen.getByLabelText('Instructions for Seer')).toHaveValue(
- 'Show errors over the last 24 hours'
- );
+
+ renderView();
+ expect(await screen.findByText('Investigate database latency')).toBeInTheDocument();
+
+ await chooseCellAction('Latency query', 'Refine');
+ expect(screen.getByLabelText('Instructions for Seer')).toHaveValue('Find slow spans');
await userEvent.click(screen.getByRole('button', {name: 'Submit'}));
await waitFor(() =>
@@ -1118,9 +1027,9 @@ describe('Investigation detail', () => {
updateUrl,
expect.objectContaining({
data: {
- investigationVersion: 3,
+ investigationVersion: 1,
version: 1,
- generationPrompt: 'Show errors over the last 24 hours',
+ generationPrompt: 'Find slow spans',
},
})
)
@@ -1130,7 +1039,7 @@ describe('Investigation detail', () => {
expect(runRequest).toHaveBeenCalledWith(
runUrl,
expect.objectContaining({
- data: {investigationVersion: 3, version: 1},
+ data: {investigationVersion: 1, version: 1},
})
)
);
diff --git a/static/app/views/investigations/detail/index.tsx b/static/app/views/investigations/detail/index.tsx
index 5bb08275242c..ef404cb3388b 100644
--- a/static/app/views/investigations/detail/index.tsx
+++ b/static/app/views/investigations/detail/index.tsx
@@ -4,13 +4,10 @@ import {useDebouncer} from '@tanstack/react-pacer';
import {useQuery, useQueryClient} from '@tanstack/react-query';
import {Alert} from '@sentry/scraps/alert';
-import {Badge} from '@sentry/scraps/badge';
-import {Button} from '@sentry/scraps/button';
import {Input} from '@sentry/scraps/input';
import {Container, Flex, Grid, Stack} from '@sentry/scraps/layout';
import {Link} from '@sentry/scraps/link';
-import {Heading, Text} from '@sentry/scraps/text';
-import {TextArea} from '@sentry/scraps/textarea';
+import {Text} from '@sentry/scraps/text';
import {addErrorMessage, addSuccessMessage} from 'sentry/actionCreators/indicator';
import Feature from 'sentry/components/acl/feature';
@@ -22,7 +19,7 @@ import {FeedbackButton} from 'sentry/components/feedbackButton/feedbackButton';
import * as Layout from 'sentry/components/layouts/thirds';
import {LoadingIndicator} from 'sentry/components/loadingIndicator';
import {SentryDocumentTitle} from 'sentry/components/sentryDocumentTitle';
-import {IconAdd, IconSeer, IconStack} from 'sentry/icons';
+import {IconStack} from 'sentry/icons';
import {IconEllipsis} from 'sentry/icons/iconEllipsis';
import {t} from 'sentry/locale';
import {normalizeUrl} from 'sentry/utils/url/normalizeUrl';
@@ -34,7 +31,6 @@ import {
getInvestigationDetailQueryOptions,
investigationListQueryOptions,
investigationTitleGenerationQueryOptions,
- useAddInvestigationBlockMutation,
useDeleteInvestigationMutation,
useDuplicateInvestigationMutation,
useRenameInvestigationMutation,
@@ -50,10 +46,7 @@ import {
} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
import {updateInvestigationCache} from 'sentry/views/investigations/investigationCache';
import {InvestigationSummaryCard} from 'sentry/views/investigations/investigationSummaryCard';
-import type {
- InvestigationBlockKind,
- InvestigationDetail,
-} from 'sentry/views/investigations/types';
+import type {InvestigationDetail} from 'sentry/views/investigations/types';
import {RouteError} from 'sentry/views/routeError';
const DEFAULT_INVESTIGATION_TITLE = 'Untitled investigation';
@@ -225,12 +218,6 @@ function InvestigationPageContent({investigation}: {investigation: Investigation
},
onError: () => addErrorMessage(t('Unable to delete investigation.')),
});
- const addBlockMutation = useAddInvestigationBlockMutation(
- organization.slug,
- investigation.id,
- {onError: () => addErrorMessage(t('Unable to add cell.'))}
- );
-
function handleTitleChange(nextTitle: string) {
setDraftTitle(nextTitle);
updateInvestigationCache(
@@ -283,18 +270,6 @@ function InvestigationPageContent({investigation}: {investigation: Investigation
shouldDisplayInvestigationBlock(block, blocks)
);
- async function handleAddBlock({
- kind,
- prompt,
- title,
- }: {
- kind: InvestigationBlockKind;
- prompt: string;
- title: string;
- }) {
- await addBlockMutation.mutateAsync({investigation, kind, prompt, title});
- }
-
return (
@@ -359,14 +334,7 @@ function InvestigationPageContent({investigation}: {investigation: Investigation
-
+ {formatSourceType(investigation.sourceType)}
- {t('%s blocks', investigation.blockCount)}
-
{t('Last update: %s', formatNotebookDate(investigation.dateUpdated))}
@@ -404,16 +370,12 @@ function InvestigationPageContent({investigation}: {investigation: Investigation
>
{t('Give feedback')}
-
- {formatStatus(investigation.status)}
-
-
-
+
+
) : null}
-
+
{visibleSummaryBlock ? (
))}
- {investigation.status === 'active' ? (
-
- ) : null}
-
+
@@ -465,98 +421,6 @@ function InvestigationPageContent({investigation}: {investigation: Investigation
);
}
-function AddCellComposer({
- isAdding,
- onAdd,
-}: {
- isAdding: boolean;
- onAdd: (cell: {
- kind: InvestigationBlockKind;
- prompt: string;
- title: string;
- }) => Promise;
-}) {
- const [kind, setKind] = useState(null);
- const [title, setTitle] = useState('');
- const [prompt, setPrompt] = useState('');
-
- function reset() {
- setKind(null);
- setTitle('');
- setPrompt('');
- }
-
- async function handleAdd() {
- if (!kind || !prompt.trim()) {
- return;
- }
- try {
- await onAdd({kind, title: title.trim(), prompt: prompt.trim()});
- reset();
- } catch {
- // The mutation owns user-facing error handling and leaves the draft intact.
- }
- }
-
- if (!kind) {
- return (
-
- } onClick={() => setKind('text')}>
- {t('Add text cell (debug only)')}
-
- } onClick={() => setKind('query')}>
- {t('Add query cell (debug only)')}
-
-
- );
- }
-
- return (
-
-
-
- {kind === 'text'
- ? t('Add text cell (debug only)')
- : t('Add query cell (debug only)')}
-
- setTitle(event.target.value)}
- />
-
-
- );
-}
-
function isTitleGenerationActive(status: string | null | undefined) {
return status === 'pending' || status === 'running';
}
@@ -577,32 +441,10 @@ function formatSourceType(sourceType: string) {
return sourceType.replaceAll('_', ' ');
}
-function formatStatus(status: string) {
- if (status === 'active') {
- return t('Active');
- }
- return status.replaceAll('_', ' ').replace(/^./, character => character.toUpperCase());
-}
-
function formatNotebookDate(date: string) {
return new Date(date).toISOString().slice(0, 10).replaceAll('-', '.');
}
-function getStatusVariant(status: string): 'success' | 'warning' | 'muted' {
- if (status === 'completed' || status === 'active') {
- return 'success';
- }
- if (status === 'pending') {
- return 'warning';
- }
- return 'muted';
-}
-
-const InvestigationCanvas = styled(Stack)`
- width: min(100%, calc(884px + ${p => p.theme.space['2xl']}));
- margin: 0 auto;
-`;
-
const InvestigationHeader = styled(Container)`
position: relative;
@@ -682,19 +524,6 @@ const MetaDivider = styled('span')`
border-left: 1px solid ${p => p.theme.tokens.border.primary};
`;
-const AddCellActions = styled(Flex)`
- padding: ${p => p.theme.space.xl} 0;
-`;
-
-const CellComposer = styled('section')`
- width: min(100%, 862px);
- margin: ${p => p.theme.space.lg} auto 0;
- padding: ${p => p.theme.space.xl};
- background: ${p => p.theme.tokens.background.secondary};
- border: 1px solid ${p => p.theme.tokens.border.primary};
- border-radius: ${p => p.theme.radius.md};
-`;
-
export default function InvestigationDetailView() {
const organization = useOrganization();
const {investigationId} = useParams<{investigationId: string}>();
From 03cf76a68a43279ae4778b053aeb9dc3fbbad3fd Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Tue, 15 Sep 2026 10:52:27 -0400
Subject: [PATCH 16/21] ref(investigations): Keep the detail view in a measured
column
Reverts the fluid-width half of 45de4822ee6. Running the notebook to the full page width stretched the prose past a comfortable measure, and body text is most of what this view renders.
The header grid, the canvas, the hypothesis row and the notebook column go back to the centred 884px column. Everything else from that commit stands: the block count, the status badge, the Seer mark and the debug-only add-cell composer stay gone.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../app/views/investigations/detail/index.tsx | 22 ++++++++++++++-----
1 file changed, 17 insertions(+), 5 deletions(-)
diff --git a/static/app/views/investigations/detail/index.tsx b/static/app/views/investigations/detail/index.tsx
index ef404cb3388b..dd2a9170de8d 100644
--- a/static/app/views/investigations/detail/index.tsx
+++ b/static/app/views/investigations/detail/index.tsx
@@ -334,7 +334,14 @@ function InvestigationPageContent({investigation}: {investigation: Investigation
-
+
-
+
+
) : null}
-
+
{visibleSummaryBlock ? (
-
+
@@ -445,6 +452,11 @@ function formatNotebookDate(date: string) {
return new Date(date).toISOString().slice(0, 10).replaceAll('-', '.');
}
+const InvestigationCanvas = styled(Stack)`
+ width: min(100%, calc(884px + ${p => p.theme.space['2xl']}));
+ margin: 0 auto;
+`;
+
const InvestigationHeader = styled(Container)`
position: relative;
From 6365e30fc74f46c9d35b8f3abe1920e38f9deff3 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Tue, 15 Sep 2026 11:07:30 -0400
Subject: [PATCH 17/21] fix(investigations): Survive a run that settles before
a command lands
Two defects the review bot found in the hypothesis row.
A settled run stops polling, but accept, reject and retry stay on the menu -- and acting on a finished run is the main reason to open it. Sentry only queues a command: the response carries the projection it already had with nothing but workflowVersion moved on, and Seer rewrites the real one later. The card therefore kept its old disposition until someone reloaded. An accepted command now reopens polling for a bounded window, long enough for Seer to apply it and short enough not to poll a stopped run forever.
Separately, verificationSteps is declared required=False with no default on the contract, so DRF omits the key rather than sending an empty list. The frontend type claimed it was always there, and two call sites read .length and .filter straight off it -- a hypothesis the agent has only just formed would crash the row it appears in. The type now says what the wire says, which is what surfaced the third unguarded caller.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../__stories__/investigationFixtureApi.tsx | 2 +-
.../hypotheses/hypothesisStatus.tsx | 2 +-
.../investigationHypotheses.spec.tsx | 99 ++++++++++++++++++-
.../hypotheses/investigationHypotheses.tsx | 34 ++++++-
.../statusBlock/getSeerStatusBlock.tsx | 3 +-
static/app/views/investigations/types.ts | 7 +-
6 files changed, 138 insertions(+), 9 deletions(-)
diff --git a/static/app/views/investigations/__stories__/investigationFixtureApi.tsx b/static/app/views/investigations/__stories__/investigationFixtureApi.tsx
index 0930d53ebc2e..aa726e45d91b 100644
--- a/static/app/views/investigations/__stories__/investigationFixtureApi.tsx
+++ b/static/app/views/investigations/__stories__/investigationFixtureApi.tsx
@@ -645,7 +645,7 @@ function applyFixtureCommandToHypothesis(
decisionSource: 'none',
confidence: null,
agentVerdict: null,
- verificationSteps: hypothesis.verificationSteps.map(step => ({
+ verificationSteps: (hypothesis.verificationSteps ?? []).map(step => ({
...step,
status: 'queued',
result: null,
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
index f0c019e19000..f8195e33b10a 100644
--- a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -100,7 +100,7 @@ function getHypothesisStatusDisplay(
return {label: humanize(status), variant: 'muted', inFlight: false};
}
- const steps = hypothesis.verificationSteps;
+ const steps = hypothesis.verificationSteps ?? [];
if (steps.length === 0) {
// Proposed, with nothing planned to test it yet.
return {label: t('Formed'), variant: 'muted', inFlight: false};
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
index b660ee86b5b5..ba80c33259b4 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.spec.tsx
@@ -5,7 +5,10 @@ import {makeTestQueryClient} from 'sentry-test/queryClient';
import {render, screen, userEvent, waitFor} from 'sentry-test/reactTestingLibrary';
import {InvestigationOrchestrationFixture} from 'sentry/views/investigations/fixtures';
-import {InvestigationHypotheses} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
+import {
+ InvestigationHypotheses,
+ isInvestigationRunSettled,
+} from 'sentry/views/investigations/hypotheses/investigationHypotheses';
import type {InvestigationOrchestration} from 'sentry/views/investigations/types';
const organization = OrganizationFixture({features: ['investigations']});
@@ -175,9 +178,101 @@ describe('InvestigationHypotheses', () => {
);
await userEvent.click(await screen.findByRole('menuitemradio', {name: 'Accept'}));
- // No refetch is needed: the command response carries the new projection.
+ // A response that does carry the decision lands without a refetch. The
+ // server only does that once Seer has applied the command; see the settled
+ // run below for what happens in between.
expect(
await screen.findByText('Accepted by you · 86% Confidence')
).toBeInTheDocument();
});
+
+ it('keeps re-reading a settled run until Seer applies an accepted command', async () => {
+ // A finished run polls no more, which is the point of settling it. But
+ // Sentry only queues a command: the response echoes the projection it
+ // already had with nothing but `workflowVersion` moved on, and Seer
+ // rewrites the real one later. Without the command reopening the polling,
+ // the card would sit on its old disposition until someone reloaded.
+ const orchestrationRequest = MockApiClient.addMockResponse({
+ url: orchestrationUrl,
+ body: InvestigationOrchestrationFixture({
+ status: 'completed',
+ workflowVersion: 7,
+ }),
+ });
+ const commandRequest = MockApiClient.addMockResponse({
+ url: commandsUrl,
+ method: 'POST',
+ body: {
+ accepted: true,
+ duplicate: false,
+ requestId: 'request-1',
+ workflowVersion: 8,
+ commandStatus: 'accepted',
+ commandError: null,
+ runId: '9001',
+ // Unchanged apart from the version, exactly as the endpoint returns it.
+ projection: InvestigationOrchestrationFixture({
+ status: 'completed',
+ workflowVersion: 8,
+ }),
+ },
+ });
+
+ renderHypotheses();
+ await screen.findAllByTestId('investigation-hypothesis');
+ const callsWhileSettled = orchestrationRequest.mock.calls.length;
+
+ await userEvent.click(
+ await screen.findByRole('button', {
+ name: 'Actions for Database or cache degradation delayed the response',
+ })
+ );
+ await userEvent.click(await screen.findByRole('menuitemradio', {name: 'Accept'}));
+ await waitFor(() => expect(commandRequest).toHaveBeenCalled());
+
+ // Nothing else would ask again: the run is completed, so this only grows
+ // because the command put the query back on its interval.
+ await waitFor(
+ () =>
+ expect(orchestrationRequest.mock.calls.length).toBeGreaterThan(callsWhileSettled),
+ {timeout: 6000}
+ );
+ }, 15_000);
+
+ it('renders hypotheses whose verification steps have not been planned yet', async () => {
+ // `verificationSteps` is `required=False` with no default on the contract,
+ // so a hypothesis the agent has only just formed arrives without the key at
+ // all — not as an empty list.
+ const hypotheses = InvestigationOrchestrationFixture().hypotheses.map(hypothesis => {
+ const unplanned = {...hypothesis, effectiveStatus: 'pending' as const};
+ delete unplanned.verificationSteps;
+ return unplanned;
+ });
+ MockApiClient.addMockResponse({
+ url: orchestrationUrl,
+ body: InvestigationOrchestrationFixture({hypotheses}),
+ });
+
+ renderHypotheses();
+
+ // Both the cards and the status block's tally read the steps, so rendering
+ // at all is the assertion: either one throws on a missing list.
+ expect(await screen.findAllByTestId('investigation-hypothesis')).toHaveLength(3);
+ expect(screen.getAllByText('Formed')).toHaveLength(3);
+ });
+});
+
+describe('isInvestigationRunSettled', () => {
+ it.each([
+ ['completed', true],
+ ['failed', true],
+ ['cancelled', true],
+ // Not terminal: the run resumes as soon as input arrives, possibly from
+ // another surface, so the projection has to keep being read.
+ ['awaiting_input', false],
+ ['processing', false],
+ ['pending', false],
+ ] as const)('reads %s as %s', (status, expected) => {
+ expect(isInvestigationRunSettled(status)).toBe(expected);
+ });
});
diff --git a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
index e1f58eaee57c..a05d60c32d01 100644
--- a/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
+++ b/static/app/views/investigations/hypotheses/investigationHypotheses.tsx
@@ -1,3 +1,4 @@
+import {useState} from 'react';
import {uuid4} from '@sentry/core';
import {useQuery} from '@tanstack/react-query';
@@ -21,6 +22,21 @@ import type {
/** How often to re-read the projection while a workflow is still moving. */
const POLL_INTERVAL_MS = 2000;
+/**
+ * How long to keep re-reading a settled run after a command was accepted.
+ *
+ * Sentry only queues a command: the response carries the *existing* projection
+ * with nothing but `workflowVersion` bumped, and Seer rewrites the projection
+ * when it actually applies the decision. On a run that has already finished
+ * polling is off, so without this the card would keep the old disposition until
+ * someone reloaded the page — and accepting or rejecting a hypothesis on a
+ * finished run is the main reason to touch that menu at all.
+ *
+ * Bounded rather than open-ended: if Seer never applies the command, this stops
+ * asking instead of polling a stopped run forever.
+ */
+const COMMAND_SETTLE_MS = 30_000;
+
/**
* Whether a workflow has stopped moving on its own.
*
@@ -64,16 +80,28 @@ export function InvestigationHypotheses({
investigationId,
}: InvestigationHypothesesProps) {
const organization = useOrganization();
+ // When the last accepted command was sent, or null if none has been. A
+ // command makes a settled run interesting again, because Seer is about to
+ // rewrite the projection behind it.
+ const [commandSentAt, setCommandSentAt] = useState(null);
+
const {data: projection} = useQuery({
...investigationOrchestrationQueryOptions(organization.slug, investigationId),
enabled,
- refetchInterval: query =>
- isInvestigationRunSettled(query.state.data?.json.status) ? false : POLL_INTERVAL_MS,
+ refetchInterval: query => {
+ if (!isInvestigationRunSettled(query.state.data?.json.status)) {
+ return POLL_INTERVAL_MS;
+ }
+ const waitingOnCommand =
+ commandSentAt !== null && Date.now() - commandSentAt < COMMAND_SETTLE_MS;
+ return waitingOnCommand ? POLL_INTERVAL_MS : false;
+ },
});
const commandMutation = useInvestigationOrchestrationCommandMutation(
organization.slug,
- investigationId
+ investigationId,
+ {onSuccess: () => setCommandSentAt(Date.now())}
);
// The status block is the run talking, so it appears as soon as there is a
diff --git a/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx b/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx
index 0067250bf5b9..c5263b1e6f66 100644
--- a/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx
+++ b/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx
@@ -21,7 +21,8 @@ function countCompletedChecks(projection: InvestigationOrchestration): number {
return projection.hypotheses.reduce(
(total, hypothesis) =>
total +
- hypothesis.verificationSteps.filter(step => step.result || step.error).length,
+ (hypothesis.verificationSteps ?? []).filter(step => step.result || step.error)
+ .length,
0
);
}
diff --git a/static/app/views/investigations/types.ts b/static/app/views/investigations/types.ts
index 647cc9c92e75..3370b1e955d7 100644
--- a/static/app/views/investigations/types.ts
+++ b/static/app/views/investigations/types.ts
@@ -307,13 +307,18 @@ export type InvestigationHypothesis = {
rationale: string;
statement: string;
status: InvestigationOrchestrationWorkStatus;
- verificationSteps: InvestigationVerificationStep[];
agentVerdict?: InvestigationAgentVerdict | null;
attempt?: number;
automaticRetryCount?: number;
heartbeatAt?: string | null;
investigatorRunId?: number | null;
toolActivity?: InvestigationToolActivity[];
+ /**
+ * Absent, not empty, until the agent has planned any checks: the contract
+ * declares this `required=False` with no default, so DRF omits the key
+ * entirely rather than sending `[]`.
+ */
+ verificationSteps?: InvestigationVerificationStep[];
};
type InvestigationOrchestrationReport = {
From 4f89dc3bf23a75563891e45aa8a724f23fd8f03f Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Wed, 16 Sep 2026 12:11:09 -0400
Subject: [PATCH 18/21] ref(string): Extract the snake_case humanizer to
utils/string
Per review. `humanize` was local to the hypothesis status module, but there is
nothing investigation-specific about it: it is the fallback any open-set wire
value needs when the frontend meets a name it does not have a translated label
for.
It sits alongside `capitalize` rather than reusing it, because the rest of the
value is deliberately left alone -- an acronym the API sent in caps reads better
kept that way, and `capitalize` lowercases everything after the first letter.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
static/app/utils/string/humanize.spec.tsx | 21 +++++++++++++++++++
static/app/utils/string/humanize.tsx | 14 +++++++++++++
.../hypotheses/hypothesisStatus.tsx | 12 ++++-------
3 files changed, 39 insertions(+), 8 deletions(-)
create mode 100644 static/app/utils/string/humanize.spec.tsx
create mode 100644 static/app/utils/string/humanize.tsx
diff --git a/static/app/utils/string/humanize.spec.tsx b/static/app/utils/string/humanize.spec.tsx
new file mode 100644
index 000000000000..5cc42e2cd7a5
--- /dev/null
+++ b/static/app/utils/string/humanize.spec.tsx
@@ -0,0 +1,21 @@
+import {humanize} from 'sentry/utils/string/humanize';
+
+describe('humanize', () => {
+ it('replaces underscores with spaces', () => {
+ expect(humanize('reauth_required')).toBe('Reauth required');
+ });
+
+ it('capitalizes the first letter', () => {
+ expect(humanize('stalled')).toBe('Stalled');
+ });
+
+ // Unlike `capitalize`, the rest of the value is left alone: an acronym the
+ // API sent in caps is more readable kept that way.
+ it('leaves the rest of the casing alone', () => {
+ expect(humanize('needs_HTTP_retry')).toBe('Needs HTTP retry');
+ });
+
+ it('handles an empty string', () => {
+ expect(humanize('')).toBe('');
+ });
+});
diff --git a/static/app/utils/string/humanize.tsx b/static/app/utils/string/humanize.tsx
new file mode 100644
index 000000000000..9eaced0aa564
--- /dev/null
+++ b/static/app/utils/string/humanize.tsx
@@ -0,0 +1,14 @@
+/**
+ * Turn a snake_case wire value into something readable: underscores become
+ * spaces, and the first letter is capitalized.
+ *
+ * Meant for values from an open set — an API can introduce one before the
+ * frontend knows its name — where showing the raw value imperfectly beats
+ * dropping it. A value the frontend does recognize should get a translated
+ * label instead; this is the fallback for the ones it does not.
+ *
+ * @example humanize('reauth_required') // 'Reauth required'
+ */
+export function humanize(value: string): string {
+ return value.replaceAll('_', ' ').replace(/^./, character => character.toUpperCase());
+}
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
index f8195e33b10a..dfcd50484df4 100644
--- a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -6,6 +6,7 @@ import {Text} from '@sentry/scraps/text';
import {LoadingIndicator} from 'sentry/components/loadingIndicator';
import {t} from 'sentry/locale';
+import {humanize} from 'sentry/utils/string/humanize';
import type {
InvestigationHypothesis,
InvestigationHypothesisStatus,
@@ -31,14 +32,9 @@ function isHypothesisSettled(status: InvestigationHypothesisStatus): boolean {
return SETTLED_STATUSES.has(status);
}
-/**
- * Turn an unrecognized wire value into something readable rather than dropping
- * it. Statuses are an open set — Seer can introduce one before Sentry knows the
- * name — so every lookup here needs a fallback.
- */
-function humanize(status: string): string {
- return status.replaceAll('_', ' ').replace(/^./, character => character.toUpperCase());
-}
+// Statuses are an open set — Seer can introduce one before Sentry knows the
+// name — so every lookup below falls back to `humanize` rather than dropping
+// the value.
function hasRun(step: InvestigationVerificationStep): boolean {
return Boolean(step.result) || Boolean(step.error);
From 29dd50fd65f59ea7a4b1a2b30d1afbef18873e09 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Wed, 16 Sep 2026 12:11:17 -0400
Subject: [PATCH 19/21] fix(investigations): Open an evidence step from
anywhere on its row
Per review: the toggle on an evidence row was a ~24px chevron on a row several
hundred pixels wide.
`Disclosure.Title` renders `leadingItems` outside its button, so putting the
step's summary there left the chevron alone inside it. The summary is now the
title's children, which is what the component expects: one full-width stretched
button spanning the row, named by the text it contains rather than by a separate
aria-label. The chevron moves to the left of the summary, where every other
Disclosure in the app puts it.
The reason the summary was in the leading slot still holds -- `Button` is sized
as a single-line control, so a two-line block of title-plus-result needs the
fixed height, `nowrap` and centred contents overridden to sit inside one. That
is what the styled title does, and `&&` keeps those rules from depending on
emotion's insertion order against the button's own class.
While here, drop the card's `className` prop: nothing passes one.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../hypotheses/hypothesisCard.spec.tsx | 54 ++++++++++++++++
.../hypotheses/hypothesisCard.tsx | 64 ++++++++++++++-----
2 files changed, 101 insertions(+), 17 deletions(-)
diff --git a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
index d3a4b539d279..a08caabe2c2c 100644
--- a/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisCard.spec.tsx
@@ -150,6 +150,60 @@ describe('HypothesisCard', () => {
expect(steps[1]).toHaveTextContent('Second check');
});
+ // The summary is the toggle's children rather than a label sitting beside a
+ // chevron-only button, so the whole row opens the step.
+ it('makes the whole row of a step that has run the toggle', async () => {
+ render(
+
+ );
+
+ const toggle = screen.getByRole('button', {
+ name: /Compare FCP with server response time/,
+ });
+ expect(toggle).toHaveTextContent(
+ 'The delay begins before the document reaches the browser.'
+ );
+ expect(screen.getByText('Establish where the delay starts.')).not.toBeVisible();
+
+ // Clicking the result line — the far side of the row from the chevron —
+ // still toggles, because it is inside the button.
+ await userEvent.click(
+ screen.getByText('The delay begins before the document reaches the browser.')
+ );
+
+ expect(screen.getByText('Establish where the delay starts.')).toBeVisible();
+ });
+
+ it('leaves a step with nothing to unpack unopenable', () => {
+ render(
+
+ );
+
+ expect(
+ screen.queryByRole('button', {name: /Compare FCP with server response time/})
+ ).not.toBeInTheDocument();
+ });
+
it('describes a step that has not produced a result yet', () => {
render(
{step.title}
@@ -163,25 +160,23 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
as="li"
border={failed ? 'danger' : 'primary'}
radius="sm"
- // A Disclosure brings its own row padding; doubling it pushes the text
- // away from the edge the other rows sit against.
+ // The toggle owns the row padding for a step that can be opened, so the
+ // container only insets it far enough to keep the hover highlight off
+ // the border. A bare row has no toggle and pads itself.
padding={hasRun ? 'xs' : 'md lg'}
background="primary"
>
{hasRun ? (
{/*
- * The summary goes in `leadingItems`, not as the title's children:
- * children land inside a Button, which is one line tall and centres
- * what it holds. From the leading slot the summary lays out normally
- * and, taking the row's spare width, pushes the chevron to the edge.
+ * The summary is the toggle's children, not `leadingItems`: the
+ * leading slot renders outside the button, which would leave the
+ * chevron alone as the click target on a row several hundred pixels
+ * wide. As children it sits inside the full-width stretched button,
+ * so the whole row opens the step — and it names the toggle without
+ * a separate aria-label.
*/}
-
+ {summary}
@@ -210,6 +205,41 @@ function VerificationStepRow({step}: {step: InvestigationVerificationStep}) {
);
}
+/**
+ * Lets the step's toggle hold the two-line summary that makes the whole row
+ * clickable.
+ *
+ * `Button` is sized as a single-line control — fixed height, `nowrap`, contents
+ * centred — which is right for a label and wrong for a block of title-plus-
+ * result that wraps. `&&` rather than a plain rule because these compete with
+ * the button's own class at equal specificity, and emotion's insertion order
+ * between the two is not something to rely on.
+ */
+const StepDisclosureTitle = styled(Disclosure.Title)`
+ && {
+ height: auto;
+ min-height: 0;
+ padding-block: ${p => p.theme.space.xs};
+ white-space: normal;
+ text-align: left;
+ }
+
+ /* Button wraps its contents in a span carrying the same single-line sizing. */
+ && > span {
+ width: 100%;
+ height: auto;
+ white-space: normal;
+ align-items: flex-start;
+ justify-content: flex-start;
+ }
+
+ /* The chevron belongs beside the title, not centred against a block whose
+ * height depends on how far the result wraps. */
+ && > span > :first-child {
+ margin-top: 1px;
+ }
+`;
+
/**
* The card border carries the verdict, which is why it is CSS rather than the
* `border` prop: `getBorder` only ever emits `1px solid`, and a hypothesis that
From 0e0f63f4d57cde887a7c0be3c62def0ce4a965c2 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Wed, 16 Sep 2026 12:11:24 -0400
Subject: [PATCH 20/21] ref(investigations): Stop handing the tally separator
to translators
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Per review. The run tally was assembled with `t('%s • %s', ...)`, which puts a
format string whose only content is a bullet into the catalog. Each half is
already translated on its own; the bullet between them is punctuation, so the
two are joined directly.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../investigations/statusBlock/getSeerStatusBlock.tsx | 9 +++++----
1 file changed, 5 insertions(+), 4 deletions(-)
diff --git a/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx b/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx
index c5263b1e6f66..1464fb8979c1 100644
--- a/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx
+++ b/static/app/views/investigations/statusBlock/getSeerStatusBlock.tsx
@@ -40,11 +40,12 @@ function getMeta(projection: InvestigationOrchestration): string | undefined {
return undefined;
}
const checkCount = countCompletedChecks(projection);
- return t(
- '%s • %s',
+ // Each half is translated; the bullet between them is punctuation, not a
+ // string a translator has anything to do with.
+ return [
tn('%s possible cause', '%s possible causes', causeCount),
- tn('%s check completed', '%s checks completed', checkCount)
- );
+ tn('%s check completed', '%s checks completed', checkCount),
+ ].join(' • ');
}
/**
From 2fe5f785d463b7641967bef5ac17ea769bd79220 Mon Sep 17 00:00:00 2001
From: Billy Vong
Date: Wed, 16 Sep 2026 12:56:09 -0400
Subject: [PATCH 21/21] fix(investigations): Spread the props Text hands its
render function
CI failure, from a rule that landed on master after this branch: scraps'
`require-render-prop-spread` rejects destructuring the parameter a scraps render
function receives.
The status line took `{({className}) => ...}` and put it on the `Flex`. Today
that is the whole object -- `Text` calls its render function with exactly
`{className}` -- but naming the one prop is what the rule is about: the callee
decides what it hands down, and a destructure silently drops anything added to
it later. Spreading forwards whatever arrives.
Nothing about the rendered output changes, which is why the status specs are
untouched.
Claude-Session: https://claude.ai/code/session_014zh69vex76pjNTnarqVcXL
---
.../investigations/hypotheses/hypothesisStatus.tsx | 12 +++++-------
1 file changed, 5 insertions(+), 7 deletions(-)
diff --git a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
index dfcd50484df4..338ac471a3f1 100644
--- a/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
+++ b/static/app/views/investigations/hypotheses/hypothesisStatus.tsx
@@ -219,15 +219,13 @@ export function HypothesisStatus({hypothesis}: HypothesisStatusProps) {
// color up as `currentColor`, and there is no second copy of the table here
// to fall out of step with the design system.
- {({className}) => (
+ {textProps => (
+ // Spread rather than picking `className` off: `Text` decides what its
+ // render function hands down, and naming one prop drops the rest.
+ //
// "Evidence checked" is both a status and the heading over the steps,
// so this needs to be addressable on its own.
-
+
{inFlight ? (
// Live work gets a ring rather than a dot: the agent is doing
// something, not resting in a state. Every other status is a place