From 7d006a75f13486054d51f8f656330bb762aafdf8 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 13:33:06 -0700 Subject: [PATCH 01/26] test(replays): Rebuild replay fixtures from real SDK event shapes Replay recording fixtures used `tag: "ui.click"`, a shape the browser SDK never emits. Real user actions arrive as rrweb custom events with `tag: "breadcrumb"`, where the meaning lives in `payload.category`. Because the fixtures matched what the classifier expected rather than what Sentry sends, snapshots looked correct while real output degraded. Rebuild the segment fixtures from the recording spec and the `@sentry-internal/replay` frame types: breadcrumbs carry millisecond timestamps, performance spans carry seconds, and the set covers ui.click, a console error, a navigation breadcrumb, a navigation span, fetches at 200 and 500, options, a script resource, and two ui.slowClickDetected events that differ only by clickCount so rage and dead clicks are separable behaviorally. Reconcile the replay metadata counts with the recording. Upstream sets click_is_dead for both DEAD_CLICK and RAGE_CLICK, and count_dead_clicks sums that column, so one rage click plus one dead click is 2 dead and 1 rage. Add fixtures and handlers for endpoints the tools do not read yet: a multi-page recording served through a Link header cursor, replays-events-meta, and the experimental Seer summarize endpoint. Cover them with contract tests so the shapes cannot drift before the tools arrive. The get_replay_details snapshots are re-baselined against these fixtures and now record known-wrong output: every user action is labeled `breadcrumb`, four of six lines are session-boot noise, and the console error, failed checkout request, rage click, and dead click are dropped by the event cap. The snapshots are annotated as such so the taxonomy fix lands as a visible diff rather than a rewrite. Refs docs/specs/replay-review.md Co-Authored-By: Claude Opus 5 (1M context) --- .../tools/catalog/get-replay-details.test.ts | 28 +- .../tools/catalog/get-sentry-resource.test.ts | 8 +- .../src/tools/catalog/replay-fixtures.test.ts | 226 +++++++++++++++ .../src/evals/utils/fixtures.ts | 3 + .../src/fixtures/replay-details.json | 4 +- .../src/fixtures/replay-events-meta.json | 27 ++ .../replay-recording-segments-paged.json | 127 ++++++++ .../fixtures/replay-recording-segments.json | 272 +++++++++++++++++- .../fixtures/replay-summary-processing.json | 6 + .../src/fixtures/replay-summary.json | 25 ++ packages/mcp-server-mocks/src/index.ts | 113 +++++++- 11 files changed, 821 insertions(+), 18 deletions(-) create mode 100644 packages/mcp-core/src/tools/catalog/replay-fixtures.test.ts create mode 100644 packages/mcp-server-mocks/src/fixtures/replay-events-meta.json create mode 100644 packages/mcp-server-mocks/src/fixtures/replay-recording-segments-paged.json create mode 100644 packages/mcp-server-mocks/src/fixtures/replay-summary-processing.json create mode 100644 packages/mcp-server-mocks/src/fixtures/replay-summary.json diff --git a/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts index 1b7f59f8f..230fbffcc 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts @@ -9,6 +9,15 @@ import getReplayDetails, { resolveReplayParams } from "./get-replay-details.js"; import { getServerContext } from "../../test-setup.js"; describe("get_replay_details", () => { + // NOTE: The Activity snapshots below record known-wrong output. The fixtures + // now use the event shape the SDK actually emits (`tag: "breadcrumb"` with + // the meaning in `payload.category`), which the current classifier does not + // understand: it labels every user action `breadcrumb`, spends the six-event + // budget on session-boot noise, and drops the console error, the failed + // checkout request, the rage click, and the dead click entirely. + // + // These baselines exist to make the taxonomy fix visible as a diff. See + // docs/specs/replay-review.md. it("loads replay details from replayUrl", async () => { const result = await getReplayDetails.handler( { @@ -32,8 +41,8 @@ describe("get_replay_details", () => { - **Device**: MacBook Pro - **Release**: frontend@1.2.3 - **Errors**: 1 - - **Rage Clicks**: 0 - - **Dead Clicks**: 1 + - **Rage Clicks**: 1 + - **Dead Clicks**: 2 - **Warnings**: 2 - **Infos**: 3 - **Recording Segments**: 2 @@ -42,8 +51,11 @@ describe("get_replay_details", () => { ## Activity - T+0s · \`page.view\` · href=https://example.com/login - - T+10s · \`navigation.navigate\` · description=https://example.com/checkout · duration_ms=710 - - T+20s · \`ui.click\` · message="Clicked submit order" + - T+0s · \`options\` · payload="sessionSampleRate=0.1, errorSampleRate=1" + - T+1s · \`navigation.navigate\` · description=https://example.com/login · duration_ms=540 + - T+1s · \`resource.script\` · description=https://cdn.example.com/vendor.js + - T+12s · \`breadcrumb\` · message="body > div#root > form#login > button#sign-in" · category="ui.click" · type="default" · payload="timestamp=1744027212.4" + - T+13s · \`resource.fetch\` · description=https://example.com/api/login ## Related @@ -119,8 +131,8 @@ describe("get_replay_details", () => { - **Device**: MacBook Pro - **Release**: frontend@1.2.3 - **Errors**: 1 - - **Rage Clicks**: 0 - - **Dead Clicks**: 1 + - **Rage Clicks**: 1 + - **Dead Clicks**: 2 - **Warnings**: 2 - **Infos**: 3 - **Recording Segments**: 2 @@ -181,8 +193,8 @@ describe("get_replay_details", () => { - **Device**: MacBook Pro - **Release**: frontend@1.2.3 - **Errors**: 1 - - **Rage Clicks**: 0 - - **Dead Clicks**: 1 + - **Rage Clicks**: 1 + - **Dead Clicks**: 2 - **Warnings**: 2 - **Infos**: 3 - **Recording Segments**: 2 diff --git a/packages/mcp-core/src/tools/catalog/get-sentry-resource.test.ts b/packages/mcp-core/src/tools/catalog/get-sentry-resource.test.ts index 2225340ab..de961d343 100644 --- a/packages/mcp-core/src/tools/catalog/get-sentry-resource.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-sentry-resource.test.ts @@ -360,7 +360,10 @@ describe("get_sentry_resource", () => { expect(result).toContain( `# Replay ${replayDetailsFixture.id} in **sentry-mcp-evals**`, ); - expect(result).toContain("Clicked submit order"); + // Assert the recording was read, not how a signal is phrased — the + // activity rendering is being replaced (docs/specs/replay-review.md). + expect(result).toContain("## Activity"); + expect(result).toContain("button#sign-in"); }); it("dispatches monitor URL with a simple slug to get_monitor_details", async () => { @@ -716,7 +719,8 @@ describe("get_sentry_resource", () => { expect(result).toContain( `# Replay ${replayDetailsFixture.id} in **sentry-mcp-evals**`, ); - expect(result).toContain("Clicked submit order"); + expect(result).toContain("## Activity"); + expect(result).toContain("button#sign-in"); }); it("fetches snapshot by snapshot ID", async () => { diff --git a/packages/mcp-core/src/tools/catalog/replay-fixtures.test.ts b/packages/mcp-core/src/tools/catalog/replay-fixtures.test.ts new file mode 100644 index 000000000..a8373fc9b --- /dev/null +++ b/packages/mcp-core/src/tools/catalog/replay-fixtures.test.ts @@ -0,0 +1,226 @@ +/** + * Contract tests for the replay mock fixtures. + * + * The replay work depends on fixtures that reflect what Sentry and the browser + * SDK actually emit — the previous fixtures used an event shape the SDK never + * produces, so tests passed while real output degraded. These tests pin the + * fixture shapes and the mock endpoints that consume them, including the ones + * no tool reads yet (segment paging, `replays-events-meta`, and the summarize + * endpoint), so they cannot silently drift before the tools arrive. + */ +import { describe, expect, it } from "vitest"; +import { + PAGED_REPLAY_ID, + replayDetailsFixture, + replayRecordingSegmentsFixture, + replayRecordingSegmentsPagedFixture, +} from "@sentry/mcp-server-mocks"; + +const SEGMENTS_PATH = (replayId: string) => + `https://us.sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayId}/recording-segments/`; + +type ReplayEvent = { + type: number; + timestamp: number; + data?: { tag?: string; payload?: Record }; +}; + +function eventsOf(segments: unknown): ReplayEvent[] { + return (segments as ReplayEvent[][]).flat(); +} + +function categoriesOf(segments: unknown): string[] { + return eventsOf(segments) + .filter((event) => event.data?.tag === "breadcrumb") + .map((event) => event.data?.payload?.category as string); +} + +function opsOf(segments: unknown): string[] { + return eventsOf(segments) + .filter((event) => event.data?.tag === "performanceSpan") + .map((event) => event.data?.payload?.op as string); +} + +describe("replay recording segment fixtures", () => { + it("carries user actions as breadcrumb events, not as bare tags", () => { + // The SDK emits every user action as a custom event tagged `breadcrumb`, + // with the meaning in `payload.category`. A `tag` of `ui.click` is a shape + // the SDK never produces. + const tags = new Set( + eventsOf(replayRecordingSegmentsFixture) + .map((event) => event.data?.tag) + .filter(Boolean), + ); + + expect(tags).toEqual(new Set(["breadcrumb", "performanceSpan", "options"])); + expect(categoriesOf(replayRecordingSegmentsFixture)).toContain("ui.click"); + }); + + it("covers each event type the classifier must distinguish", () => { + const categories = categoriesOf(replayRecordingSegmentsFixture); + const ops = opsOf(replayRecordingSegmentsFixture); + + expect(categories).toEqual( + expect.arrayContaining([ + "ui.click", + "console", + "navigation", + "ui.slowClickDetected", + ]), + ); + expect(ops).toEqual( + expect.arrayContaining([ + "navigation.navigate", + "resource.fetch", + "resource.script", + ]), + ); + expect( + eventsOf(replayRecordingSegmentsFixture).some( + (event) => event.data?.tag === "options", + ), + ).toBe(true); + }); + + it("includes both a failed and a successful network request", () => { + // Upstream renders only failures but counts every request, so the fixture + // must contain both to tell those behaviours apart. + const statuses = eventsOf(replayRecordingSegmentsFixture) + .filter((event) => event.data?.payload?.op === "resource.fetch") + .map((event) => event.data?.payload?.data?.statusCode); + + expect(statuses).toEqual(expect.arrayContaining([200, 500])); + }); + + it("distinguishes a rage click from a plain dead click behaviorally", () => { + // Both are `ui.slowClickDetected`; only `clickCount` separates them. + const slowClicks = eventsOf(replayRecordingSegmentsFixture) + .filter( + (event) => event.data?.payload?.category === "ui.slowClickDetected", + ) + .map((event) => event.data?.payload?.data); + + expect(slowClicks).toHaveLength(2); + for (const click of slowClicks) { + // Dead-click conditions: timeout, interactive target, >= 7000ms. + expect(click.endReason).toBe("timeout"); + expect(["a", "button", "input"]).toContain(click.node.tagName); + expect(click.timeAfterClickMs).toBeGreaterThanOrEqual(7000); + } + + const clickCounts = slowClicks.map((click) => click.clickCount).sort(); + expect(clickCounts).toEqual([1, 5]); + }); + + it("uses second timestamps for spans and millisecond timestamps for breadcrumbs", () => { + // Timestamp unit is a function of event type, not magnitude — the fixture + // has to exercise both or the per-type unit fix is untestable. + for (const event of eventsOf(replayRecordingSegmentsFixture)) { + if (event.data?.tag === "performanceSpan") { + expect(event.timestamp).toBeLessThan(1e12); + } + if (event.data?.tag === "breadcrumb") { + expect(event.timestamp).toBeGreaterThan(1e12); + } + } + }); + + it("reports rage and dead click counts consistent with the recording", () => { + // Upstream counts a rage click as dead too (`count_dead_clicks` sums + // `click_is_dead`, which is set for DEAD_CLICK and RAGE_CLICK alike). + expect(replayDetailsFixture.count_rage_clicks).toBe(1); + expect(replayDetailsFixture.count_dead_clicks).toBe(2); + }); +}); + +describe("replay mock endpoints", () => { + it("serves the whole recording in one page when it fits", async () => { + const response = await fetch(SEGMENTS_PATH(replayDetailsFixture.id)); + + expect(await response.json()).toEqual(replayRecordingSegmentsFixture); + // Sentry always sends a Link header; `results="false"` — not an absent + // header — is how the last page is signalled. + expect(response.headers.get("Link")).toContain( + 'rel="next"; results="false"', + ); + }); + + it("pages a long recording through the Link header cursor", async () => { + const pages: unknown[][] = []; + let cursor: string | null = ""; + + while (cursor !== null) { + const url = new URL(SEGMENTS_PATH(PAGED_REPLAY_ID)); + if (cursor) { + url.searchParams.set("cursor", cursor); + } + + const response: Response = await fetch(url); + pages.push((await response.json()) as unknown[]); + + const nextLink = (response.headers.get("Link") ?? "") + .split(",") + .find( + (link) => + link.includes('rel="next"') && link.includes('results="true"'), + ); + cursor = nextLink?.match(/cursor="([^"]+)"/)?.[1] ?? null; + } + + expect(pages.length).toBeGreaterThan(1); + expect(pages.flat()).toEqual(replayRecordingSegmentsPagedFixture); + }); + + it("truncates to the first page when the cursor is ignored", async () => { + // This is the bug the paging fix has to close: without following the + // header, later segments are silently missing. + const response = await fetch(SEGMENTS_PATH(PAGED_REPLAY_ID)); + const firstPage = (await response.json()) as unknown[]; + + expect(firstPage.length).toBeLessThan( + replayRecordingSegmentsPagedFixture.length, + ); + }); + + it("resolves replay error ids to issue identity and a ms-precision timestamp", async () => { + const url = new URL( + "https://us.sentry.io/api/0/organizations/sentry-mcp-evals/replays-events-meta/", + ); + url.searchParams.set( + "query", + `id:[${replayDetailsFixture.error_ids.join(",")}]`, + ); + + const body = (await (await fetch(url)).json()) as { + data: Record[]; + }; + + expect(body.data).toHaveLength(1); + expect(body.data[0]).toMatchObject({ + id: replayDetailsFixture.error_ids[0], + issue: "CLOUDFLARE-MCP-41", + title: "Error: Tool list_organizations is already registered", + }); + // The endpoint folds millisecond precision into `timestamp` and deletes + // `timestamp_ms`, so the suggested window has to parse it from here. + expect(body.data[0]).not.toHaveProperty("timestamp_ms"); + expect(Date.parse(body.data[0].timestamp as string)).not.toBeNaN(); + }); + + it("returns a completed Seer summary with millisecond chapter windows", async () => { + const body = (await ( + await fetch( + `https://us.sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/summarize/`, + ) + ).json()) as { + status: string; + data: { time_ranges: { period_start: number; period_end: number }[] }; + }; + + expect(body.status).toBe("completed"); + for (const chapter of body.data.time_ranges) { + expect(chapter.period_start).toBeGreaterThan(1e12); + expect(chapter.period_end).toBeGreaterThan(chapter.period_start); + } + }); +}); diff --git a/packages/mcp-server-evals/src/evals/utils/fixtures.ts b/packages/mcp-server-evals/src/evals/utils/fixtures.ts index 82758ab35..8e0a43f73 100644 --- a/packages/mcp-server-evals/src/evals/utils/fixtures.ts +++ b/packages/mcp-server-evals/src/evals/utils/fixtures.ts @@ -21,5 +21,8 @@ export const FIXTURES = { traceId: "a4d1aae7216b47ff8117cf4e09ce9d0a", traceUrl: "https://sentry-mcp-evals.sentry.io/explore/traces/trace/a4d1aae7216b47ff8117cf4e09ce9d0a/", + replayId: "7e07485f-12f9-416b-8b14-26260799b51f", + replayUrl: + "https://sentry-mcp-evals.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/", dsn: "https://d20df0a1ab5031c7f3c7edca9c02814d@o4509106732793856.ingest.us.sentry.io/4509109104082945", }; diff --git a/packages/mcp-server-mocks/src/fixtures/replay-details.json b/packages/mcp-server-mocks/src/fixtures/replay-details.json index 9db146a4f..7ff2c4de7 100644 --- a/packages/mcp-server-mocks/src/fixtures/replay-details.json +++ b/packages/mcp-server-mocks/src/fixtures/replay-details.json @@ -10,8 +10,8 @@ "count_errors": 1, "count_warnings": 2, "count_infos": 3, - "count_dead_clicks": 1, - "count_rage_clicks": 0, + "count_dead_clicks": 2, + "count_rage_clicks": 1, "count_segments": 2, "count_urls": 2, "urls": ["/login", "/checkout"], diff --git a/packages/mcp-server-mocks/src/fixtures/replay-events-meta.json b/packages/mcp-server-mocks/src/fixtures/replay-events-meta.json new file mode 100644 index 000000000..3fd9a20db --- /dev/null +++ b/packages/mcp-server-mocks/src/fixtures/replay-events-meta.json @@ -0,0 +1,27 @@ +{ + "data": [ + { + "id": "7ca573c0f4814912aaa9bdc77d1a7d51", + "issue.id": 6507376925, + "issue": "CLOUDFLARE-MCP-41", + "title": "Error: Tool list_organizations is already registered", + "level": "error", + "error.type": ["TypeError"], + "project.name": "cloudflare-mcp", + "timestamp": "2025-04-07T12:03:01.300000+00:00" + } + ], + "meta": { + "fields": { + "id": "string", + "issue.id": "integer", + "issue": "string", + "title": "string", + "level": "string", + "error.type": "string", + "project.name": "string", + "timestamp": "date" + }, + "units": {} + } +} diff --git a/packages/mcp-server-mocks/src/fixtures/replay-recording-segments-paged.json b/packages/mcp-server-mocks/src/fixtures/replay-recording-segments-paged.json new file mode 100644 index 000000000..b145b4421 --- /dev/null +++ b/packages/mcp-server-mocks/src/fixtures/replay-recording-segments-paged.json @@ -0,0 +1,127 @@ +[ + [ + { + "type": 4, + "timestamp": 1744027200000, + "data": { + "href": "https://example.com/dashboard", + "width": 1440, + "height": 900 + } + }, + { + "type": 5, + "timestamp": 1744027201000, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "ui.click", + "message": "body > nav > button#page-1-action", + "timestamp": 1744027201, + "data": { + "nodeId": 11, + "node": { + "id": 11, + "tagName": "button", + "textContent": "Segment one action", + "attributes": { "id": "page-1-action" } + } + } + } + } + } + ], + [ + { + "type": 5, + "timestamp": 1744027260000, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "ui.click", + "message": "body > nav > button#page-2-action", + "timestamp": 1744027260, + "data": { + "nodeId": 12, + "node": { + "id": 12, + "tagName": "button", + "textContent": "Segment two action", + "attributes": { "id": "page-2-action" } + } + } + } + } + } + ], + [ + { + "type": 5, + "timestamp": 1744027320000, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "ui.click", + "message": "body > nav > button#page-3-action", + "timestamp": 1744027320, + "data": { + "nodeId": 13, + "node": { + "id": 13, + "tagName": "button", + "textContent": "Segment three action", + "attributes": { "id": "page-3-action" } + } + } + } + } + } + ], + [ + { + "type": 5, + "timestamp": 1744027380000, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "console", + "level": "error", + "message": "Error: checkout total mismatch", + "timestamp": 1744027380, + "data": { + "arguments": ["Error: checkout total mismatch"], + "logger": "console" + } + } + } + } + ], + [ + { + "type": 5, + "timestamp": 1744027440000, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "ui.click", + "message": "body > nav > button#page-5-action", + "timestamp": 1744027440, + "data": { + "nodeId": 15, + "node": { + "id": 15, + "tagName": "button", + "textContent": "Segment five action", + "attributes": { "id": "page-5-action" } + } + } + } + } + } + ] +] diff --git a/packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json b/packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json index f0066f10a..b43d514b3 100644 --- a/packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json +++ b/packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json @@ -11,14 +11,156 @@ }, { "type": 5, - "timestamp": 1744027210, + "timestamp": 1744027200100, + "data": { + "tag": "options", + "payload": { + "shouldRecordCanvas": false, + "sessionSampleRate": 0.1, + "errorSampleRate": 1, + "useCompressionOption": true, + "blockAllMedia": true, + "maskAllText": true, + "maskAllInputs": true, + "useCompression": true, + "networkDetailHasUrls": false, + "networkCaptureBodies": false, + "networkRequestHasHeaders": false, + "networkResponseHasHeaders": false + } + } + }, + { + "type": 5, + "timestamp": 1744027200.5, + "data": { + "tag": "performanceSpan", + "payload": { + "op": "navigation.navigate", + "description": "https://example.com/login", + "startTimestamp": 1744027200.5, + "endTimestamp": 1744027201.04, + "data": { + "size": 18420, + "decodedBodySize": 74210, + "encodedBodySize": 18420, + "duration": 540, + "domInteractive": 318, + "domContentLoadedEventStart": 402, + "domContentLoadedEventEnd": 407, + "loadEventStart": 531, + "loadEventEnd": 540, + "redirectCount": 0 + } + } + } + }, + { + "type": 5, + "timestamp": 1744027201.2, + "data": { + "tag": "performanceSpan", + "payload": { + "op": "resource.script", + "description": "https://cdn.example.com/vendor.js", + "startTimestamp": 1744027201.2, + "endTimestamp": 1744027201.58, + "data": { + "size": 145820, + "decodedBodySize": 402118, + "encodedBodySize": 145820, + "statusCode": 200 + } + } + } + }, + { + "type": 5, + "timestamp": 1744027212400, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "ui.click", + "message": "body > div#root > form#login > button#sign-in", + "timestamp": 1744027212.4, + "data": { + "nodeId": 42, + "node": { + "id": 42, + "tagName": "button", + "textContent": "Sign in", + "attributes": { + "id": "sign-in", + "type": "submit" + } + } + } + } + } + }, + { + "type": 5, + "timestamp": 1744027213.1, + "data": { + "tag": "performanceSpan", + "payload": { + "op": "resource.fetch", + "description": "https://example.com/api/login", + "startTimestamp": 1744027213.1, + "endTimestamp": 1744027213.34, + "data": { + "method": "POST", + "statusCode": 200, + "request": { + "size": 48, + "headers": {} + }, + "response": { + "size": 132, + "headers": {} + } + } + } + } + }, + { + "type": 5, + "timestamp": 1744027214000, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "navigation", + "timestamp": 1744027214, + "data": { + "from": "/login", + "to": "/checkout" + } + } + } + }, + { + "type": 5, + "timestamp": 1744027214.2, "data": { "tag": "performanceSpan", "payload": { "op": "navigation.navigate", "description": "https://example.com/checkout", + "startTimestamp": 1744027214.2, + "endTimestamp": 1744027214.91, "data": { - "duration": 710 + "size": 24680, + "decodedBodySize": 98304, + "encodedBodySize": 24680, + "duration": 710, + "domInteractive": 412, + "domContentLoadedEventStart": 480, + "domContentLoadedEventEnd": 486, + "loadEventStart": 700, + "loadEventEnd": 710, + "redirectCount": 0 } } } @@ -27,11 +169,131 @@ [ { "type": 5, - "timestamp": 1744027220, + "timestamp": 1744027380600, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "ui.click", + "message": "body > div#root > main > button#complete-order", + "timestamp": 1744027380.6, + "data": { + "nodeId": 96, + "node": { + "id": 96, + "tagName": "button", + "textContent": "Complete order", + "attributes": { + "id": "complete-order", + "type": "button" + } + } + } + } + } + }, + { + "type": 5, + "timestamp": 1744027381, "data": { - "tag": "ui.click", + "tag": "performanceSpan", "payload": { - "message": "Clicked submit order" + "op": "resource.fetch", + "description": "https://example.com/api/checkout", + "startTimestamp": 1744027381, + "endTimestamp": 1744027382.24, + "data": { + "method": "POST", + "statusCode": 500, + "request": { + "size": 214, + "headers": {} + }, + "response": { + "size": 87, + "headers": {} + } + } + } + } + }, + { + "type": 5, + "timestamp": 1744027381300, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "console", + "level": "error", + "message": "TypeError: Cannot read properties of undefined (reading 'id')", + "timestamp": 1744027381.3, + "data": { + "arguments": [ + "TypeError: Cannot read properties of undefined (reading 'id')" + ], + "logger": "console" + } + } + } + }, + { + "type": 5, + "timestamp": 1744027388400, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "ui.slowClickDetected", + "message": "body > div#root > main > button#complete-order", + "timestamp": 1744027388.4, + "data": { + "nodeId": 96, + "node": { + "id": 96, + "tagName": "button", + "textContent": "Complete order", + "attributes": { + "id": "complete-order", + "type": "button" + } + }, + "url": "https://example.com/checkout", + "route": "/checkout", + "timeAfterClickMs": 7000, + "endReason": "timeout", + "clickCount": 5 + } + } + } + }, + { + "type": 5, + "timestamp": 1744027421700, + "data": { + "tag": "breadcrumb", + "payload": { + "type": "default", + "category": "ui.slowClickDetected", + "message": "body > div#root > main > a#download-receipt", + "timestamp": 1744027421.7, + "data": { + "nodeId": 118, + "node": { + "id": 118, + "tagName": "a", + "textContent": "Download receipt", + "attributes": { + "id": "download-receipt", + "href": "/receipts/latest" + } + }, + "url": "https://example.com/checkout", + "route": "/checkout", + "timeAfterClickMs": 7000, + "endReason": "timeout", + "clickCount": 1 + } } } } diff --git a/packages/mcp-server-mocks/src/fixtures/replay-summary-processing.json b/packages/mcp-server-mocks/src/fixtures/replay-summary-processing.json new file mode 100644 index 000000000..36d2b870c --- /dev/null +++ b/packages/mcp-server-mocks/src/fixtures/replay-summary-processing.json @@ -0,0 +1,6 @@ +{ + "data": null, + "num_segments": 2, + "created_at": "2025-04-07T12:06:00.000Z", + "status": "processing" +} diff --git a/packages/mcp-server-mocks/src/fixtures/replay-summary.json b/packages/mcp-server-mocks/src/fixtures/replay-summary.json new file mode 100644 index 000000000..b63d756f9 --- /dev/null +++ b/packages/mcp-server-mocks/src/fixtures/replay-summary.json @@ -0,0 +1,25 @@ +{ + "data": { + "time_ranges": [ + { + "period_start": 1744027200500, + "period_end": 1744027214200, + "period_title": "Signed in and navigated to checkout" + }, + { + "period_start": 1744027380600, + "period_end": 1744027388400, + "period_title": "Complete order failed with a server error" + }, + { + "period_start": 1744027421700, + "period_end": 1744027428700, + "period_title": "Download receipt link did not respond" + } + ], + "summary": "The user signed in, moved to checkout, and tried to complete an order. The checkout request failed with a 500 and logged a TypeError, after which the user rage clicked the complete order button and then clicked an unresponsive download receipt link." + }, + "num_segments": 2, + "created_at": "2025-04-07T12:06:00.000Z", + "status": "completed" +} diff --git a/packages/mcp-server-mocks/src/index.ts b/packages/mcp-server-mocks/src/index.ts index 69bea77aa..b19171c23 100644 --- a/packages/mcp-server-mocks/src/index.ts +++ b/packages/mcp-server-mocks/src/index.ts @@ -108,9 +108,21 @@ import releaseDeploysFixture from "./fixtures/release-deploys.json" with { import replayDetailsFixture from "./fixtures/replay-details.json" with { type: "json", }; +import replayEventsMetaFixture from "./fixtures/replay-events-meta.json" with { + type: "json", +}; import replayRecordingSegmentsFixture from "./fixtures/replay-recording-segments.json" with { type: "json", }; +import replayRecordingSegmentsPagedFixture from "./fixtures/replay-recording-segments-paged.json" with { + type: "json", +}; +import replaySummaryFixture from "./fixtures/replay-summary.json" with { + type: "json", +}; +import replaySummaryProcessingFixture from "./fixtures/replay-summary-processing.json" with { + type: "json", +}; import tagsFixture from "./fixtures/tags.json" with { type: "json" }; import teamFixture from "./fixtures/team.json" with { type: "json" }; import traceFixture from "./fixtures/trace.json" with { type: "json" }; @@ -157,6 +169,17 @@ import uptimeMonitorFixture from "./fixtures/uptime-monitor.json" with { import userFixture from "./fixtures/user.json" with { type: "json" }; import { issueFixture2 } from "./payloads"; +/** + * Replay whose recording is served across multiple segment pages. + */ +export const PAGED_REPLAY_ID = "b81f2c3d-4e5a-6b7c-8d9e-0f1a2b3c4d5e"; + +/** + * Segments returned per page. Sentry's segment index caps `per_page` at 100; + * the mock uses a small page so paging is reachable with a few fixtures. + */ +const SEGMENTS_PER_PAGE = 2; + /** * Builds MSW handlers for both SaaS and self-hosted Sentry instances. * @@ -320,6 +343,53 @@ function buildUpdatedIssueResponse( }; } +/** + * A replay whose recording spans more segments than one page, so segment + * pagination is exercisable. Its counts match + * `replay-recording-segments-paged.json`. + */ +export const pagedReplayDetailsFixture = { + ...replayDetailsFixture, + id: PAGED_REPLAY_ID, + count_segments: replayRecordingSegmentsPagedFixture.length, + count_errors: 0, + count_dead_clicks: 0, + count_rage_clicks: 0, + count_urls: 1, + urls: ["/dashboard"], + trace_ids: [], + error_ids: [], +}; + +/** + * Mimic Sentry's segment index pagination. + * + * Sentry serves at most `SEGMENTS_PER_PAGE` segments per response and reports + * continuation through the `Link` header, which always carries both a `prev` + * and a `next` relation. `results="false"` — not an absent header — is how the + * final page is signalled. + */ +function respondWithSegmentPage( + request: Request, + segments: unknown[], +): HttpResponse { + const url = new URL(request.url); + const offset = Number(url.searchParams.get("cursor")?.split(":")[1] ?? "0"); + const page = segments.slice(offset, offset + SEGMENTS_PER_PAGE); + const nextOffset = offset + SEGMENTS_PER_PAGE; + const hasNext = nextOffset < segments.length; + const path = `${url.origin}${url.pathname}`; + + return HttpResponse.json(page, { + headers: { + Link: [ + `<${path}?cursor=0:0:1>; rel="previous"; results="${offset > 0}"; cursor="0:0:1"`, + `<${path}?cursor=0:${nextOffset}:0>; rel="next"; results="${hasNext}"; cursor="0:${nextOffset}:0"`, + ].join(", "), + }, + }); +} + /** * Complete set of Sentry API mock handlers. * @@ -895,7 +965,44 @@ export const restHandlers = buildHandlers([ { method: "get", path: `/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/recording-segments/`, - fetch: () => HttpResponse.json(replayRecordingSegmentsFixture), + fetch: ({ request }) => + respondWithSegmentPage(request, replayRecordingSegmentsFixture), + }, + { + method: "get", + path: `/api/0/organizations/sentry-mcp-evals/replays/${PAGED_REPLAY_ID}/`, + fetch: () => HttpResponse.json({ data: pagedReplayDetailsFixture }), + }, + { + method: "get", + path: `/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${PAGED_REPLAY_ID}/recording-segments/`, + // The paged replay has more segments than SEGMENTS_PER_PAGE, so this + // handler emits a `Link` header with a next cursor. Clients that ignore it + // see only the first page. + fetch: ({ request }) => + respondWithSegmentPage(request, replayRecordingSegmentsPagedFixture), + }, + { + method: "get", + path: "/api/0/organizations/sentry-mcp-evals/replays-events-meta/", + fetch: ({ request }) => { + const query = new URL(request.url).searchParams.get("query") ?? ""; + const requestedIds = new Set( + query.match(/id:\[([^\]]*)\]/)?.[1]?.split(",") ?? [], + ); + + return HttpResponse.json({ + ...replayEventsMetaFixture, + data: replayEventsMetaFixture.data.filter((event) => + requestedIds.has(event.id), + ), + }); + }, + }, + { + method: "get", + path: `/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/summarize/`, + fetch: () => HttpResponse.json(replaySummaryFixture), }, // Trace endpoints @@ -1986,7 +2093,11 @@ export { projectFixture, releaseFixture, replayDetailsFixture, + replayEventsMetaFixture, replayRecordingSegmentsFixture, + replayRecordingSegmentsPagedFixture, + replaySummaryFixture, + replaySummaryProcessingFixture, tagsFixture, teamFixture, traceEventFixture, From 836e67acadd1d7fcea15f77220baaae95cc1d16e Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 13:37:40 -0700 Subject: [PATCH 02/26] docs(replays): Add replay review spec and change proposal Document the map-then-zoom restructuring of replay retrieval: get_replay_details returns session shape instead of a fixed six-event prose sample, a new catalog-only get_replay_activity returns signals for a requested window and grain, and classification follows Sentry's own replay event taxonomy so MCP and Seer agree on what a session contains. Also records the adjacent correctness fixes the work depends on: segment pagination through the Link header cursor, per-event-type timestamp units, truncation reporting, replay sort reconciliation, capability gating on the replays search path, and distinguishing replay-count rate limiting from absence. Section 1 of tasks.md is checked off by the preceding commit, which rebuilt the fixtures from real SDK event shapes. Co-Authored-By: Claude Opus 5 (1M context) --- docs/specs/README.md | 1 + docs/specs/replay-review.md | 430 ++++++++++++++++++ .../improve-replay-review/.openspec.yaml | 2 + .../changes/improve-replay-review/design.md | 277 +++++++++++ .../changes/improve-replay-review/proposal.md | 80 ++++ .../specs/replay-review/spec.md | 205 +++++++++ .../changes/improve-replay-review/tasks.md | 84 ++++ 7 files changed, 1079 insertions(+) create mode 100644 docs/specs/replay-review.md create mode 100644 openspec/changes/improve-replay-review/.openspec.yaml create mode 100644 openspec/changes/improve-replay-review/design.md create mode 100644 openspec/changes/improve-replay-review/proposal.md create mode 100644 openspec/changes/improve-replay-review/specs/replay-review/spec.md create mode 100644 openspec/changes/improve-replay-review/tasks.md diff --git a/docs/specs/README.md b/docs/specs/README.md index b7d33cce0..3c83100c6 100644 --- a/docs/specs/README.md +++ b/docs/specs/README.md @@ -9,6 +9,7 @@ server. Each spec should live in a single Markdown file under `docs/specs/`. - [Embedded Agent OpenAI Routing](embedded-agent-openai-routing.md) - [Project Management Tools](project-management.md) - [Remembered OAuth Skill Defaults](remembered-oauth-skills.md) +- [Replay Review](replay-review.md) - [Search Events](search-events.md) - [Sentry-Bearer Cloudflare Auth](sentry-bearer-cloudflare-auth.md) - [Subpath Constraints](subpath-constraints.md) diff --git a/docs/specs/replay-review.md b/docs/specs/replay-review.md new file mode 100644 index 000000000..0d9c808fe --- /dev/null +++ b/docs/specs/replay-review.md @@ -0,0 +1,430 @@ +# Replay Review Specification + +## Overview + +Replay retrieval today returns a single fixed-grain summary capped at six +activity events. Agents cannot narrow to a time window, cannot request more or +less detail, and cannot tell truncation from absence. The cap is spent on +session-boot noise before the failure is reached. + +This spec restructures replay retrieval around three ideas: + +1. **Map before detail** — `get_replay_details` returns the shape of a session + (signal counts, page flow, error markers), not a prose sample of it. +2. **Windowed zoom** — a new catalog tool, `get_replay_activity`, returns + signals for a requested time window at a requested grain. +3. **Sentry's own taxonomy** — classify and phrase replay events the way + `sentry.replays.usecases.summarize` does, so MCP output matches what Sentry's + Seer summarizer sees. + +`replayId` plus a time window is the navigation handle. No server-side session +state is introduced. + +## Motivation + +### The current output is wrong, not merely terse + +Sentry's SDK emits user actions as rrweb custom events with +`data.tag === "breadcrumb"`, where the meaning lives in `payload.category` +(`ui.click`, `console`, `navigation`, `ui.slowClickDetected`). See "References" +below for the upstream spec and SDK types. + +`summarizeTaggedReplayEvent` in +`packages/mcp-core/src/tools/catalog/get-replay-details.ts` special-cases +`tag === "ui.click"` — a shape the SDK never produces. Feeding realistically +shaped segments through the current handler produces: + +```text +- T+0s · `options` · payload="sessionSampleRate=0.1, errorSampleRate=1" +- T+0s · `page.view` · href=https://example.com/checkout +- T+1s · `resource.script` · description=https://cdn.example.com/vendor.js +- T+5m 12s · `breadcrumb` · message="button#complete-order[...]" · category="ui.click" · type="default" · payload="timestamp=1744027511.8" +- T+5m 12s · `resource.fetch` · description=https://example.com/api/checkout +- T+5m 12s · `breadcrumb` · message="TypeError: Cannot read 'id' of undefined" · category="console" · type="default" · payload="timestamp=1744027512.09" +``` + +Four defects, none caught by existing tests: + +- Every user action is labeled `breadcrumb`; the real signal is demoted into a + details blob beside `type="default"` and a raw epoch timestamp. +- Half the six-event budget goes to `options` (which leaks sample rates), a + `` href, and a vendor script fetch. +- A `ui.slowClickDetected` rage click was dropped by the cap at 11 input events. + Real sessions contain thousands. +- `packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json` uses + `tag: "ui.click"`, so snapshots look correct while real output degrades. No + replay eval exists. + +### The API supports more than we ask of it + +Verified against Sentry source and public API docs: + +- `getReplayRecordingSegments` (`packages/mcp-core/src/api-client/client.ts`) + sends only `?download=true`. That parameter **does not exist on the segments + index endpoint**, which always downloads. Meanwhile `cursor` is omitted, so + replays longer than 100 segments are silently truncated. Note that `per_page` + is not the fix: its default *and* maximum are both 100, so only following the + `Link` header's cursor reaches later segments. +- The org replays index accepts a `field` allow-list, which `searchReplays` does + not send at all, so Sentry returns its default column set. Any allow-list must + be drawn from `VALID_FIELD_SET` in `sentry/replays/validators.py`, whose names + are coarser than the discovery list (`browser`, `user`, `device`, `sdk`, + `releases`, `trace_ids` — not `browser.name`). +- `POST`/`GET /projects/{org}/{project}/replays/{replay_id}/summarize/` exists + and returns a narrative plus chapters, but is unused. + +## Upstream Building Blocks + +`sentry.replays.usecases.ingest.event_parser` defines a 32-member `EventType` +enum and a `which()` classifier keyed on `payload.category` and span `op`. +`sentry.replays.usecases.summarize.as_log_message` converts each type into +agent-facing prose and returns `None` for noise. + +Behavior worth porting rather than reinventing: + +- Dropped entirely: `OPTIONS`, `MEMORY`, `MUTATIONS`, `CANVAS`, + `RESOURCE_SCRIPT`, `RESOURCE_IMAGE`, `UI_BLUR`, `UI_FOCUS`, `CLS`, + `SLOW_CLICK`, and `MULTI_CLICK`. +- **2xx network requests are skipped.** Only failures are narrated. Kind counts + still include them, so `network 58 (2 failed)` means 58 requests of which 2 + are rendered. +- **Web replays prefer `NAVIGATION_SPAN` over `NAVIGATION`.** Upstream drops the + navigation breadcrumb for web and keeps it only for mobile, where the span is + unavailable. +- Dead vs. rage click classification is behavioral, not a tag: + `ui.slowClickDetected` with `payload.data.endReason === "timeout"`, a + `payload.data.node.tagName` of `a`, `button`, or `input`, and + `timeAfterClickMs >= 7000` is a dead click; `clickCount >= 5` promotes it to a + rage click. Anything else is a plain slow click. Upstream also accepts the + lowercase spellings `timeafterclickms` and `clickcount`. +- **Timestamp units vary by event type.** `get_timestamp_unit()` returns `"s"` + for spans, web vitals, blur, and focus; `"ms"` for clicks, console, and + navigation. `getEventTimestampMillis` currently guesses from magnitude + (`value > 1e12`). That guess holds for present-day epochs by coincidence, not + by rule. + +## Design + +### Tool surface + +Add one catalog-only tool. Keep the top-level surface unchanged. + +```text +get_replay_activity +``` + +Existing tools change behavior, not identity: + +- `get_replay_details` returns a map plus a suggested next call. +- `get_sentry_resource` continues to route replay URLs to `get_replay_details`. +- `search_events` with `dataset="replays"` continues to list replays. + +Deliberate deviation from handle-based designs: MCP runs over stateless HTTP and +stdio. An `open`/`close` handle pair would require KV storage, TTLs, and +lifecycle handling to buy nothing that `replayId` plus a window does not already +provide. There is no `close` tool; usage telemetry belongs on existing spans +(grain requested, window width, result counts). + +### `get_replay_details` — the map + +Replace the prose activity sample with session shape: + +```text +# Replay 7e07485f… in **my-org** + +## Summary +… unchanged fields … + +## Map +- **Signals**: 1,182 across 0.0s–353.2s, 4 pages +- **Flow**: /login ▸ /cart ▸ /checkout ▸ /checkout/confirm +- **Kinds**: navigation 18 · click 36 (2 rage, 1 dead) · network 58 (2 failed) · console 4 (2 error) +- **Truncated**: no + +## Chapters +… present only when a Seer summary already exists … + +## Related +… unchanged … + +## Next +Error CLOUDFLARE-MCP-41 occurred at T+311.8s: +get_replay_activity(organizationSlug='my-org', replayId='7e07485f…', startMs=306000, endMs=316000, grain='detail') +``` + +The suggested window comes from +`GET /organizations/{org}/replays-events-meta/`, which resolves the replay's +`error_ids` in one batched call (`query=id:[a,b]`) and returns `id`, `issue`, +`issue.id`, `title`, and a millisecond-precision ISO `timestamp`. Note that the +endpoint deletes `timestamp_ms` from its own output and folds that precision +into `timestamp`. + +Because it returns issue identity alongside the timestamp, this call also +replaces the per-error `listIssues` lookups that populate the Related section — +one request instead of up to three. + +The endpoint is `ApiPublishStatus.PRIVATE` today, with the intent to make it +public. Until then, treat the response as untrusted: parse defensively, and omit +the suggested window rather than failing the call. + +### `get_replay_activity` — the zoom + +```typescript +inputSchema: { + organizationSlug: ParamOrganizationSlug.optional(), + replayId: ParamReplayId.optional(), + replayUrl: ParamReplayUrl.optional(), + regionUrl: ParamRegionUrl.nullable().optional(), + startMs: z.number().min(0).optional(), + endMs: z.number().min(0).optional(), + grain: z.enum(["digest", "standard", "detail"]).default("standard"), + kinds: z.array(z.enum([ + "navigation", "click", "dead-click", "rage-click", "slow-click", + "network", "console", "hydration-error", "feedback", "web-vital", + "tap", "scroll", "swipe", "app-lifecycle", "device", + ])).optional(), + limit: z.number().min(1).max(200).default(50), + cursor: z.string().optional(), +} +``` + +Semantics: + +- Omitting `startMs`/`endMs` selects the whole session. Offsets are measured + from the replay's `started_at`, not from the first recorded event. +- Omitting `kinds` includes every kind. Supplying it is an allow-list. +- `grain` controls rendering only: + - `digest` — one rollup line per kind: `network ×58 (2 failed)` + - `standard` — one line per signal, repeats merged with `×N` + - `detail` — one line per signal plus available payload (method, status, + duration, stack frames, selector attributes) +- Redaction is labeled, never inferred. `networkCaptureBodies` is opt-in, so + bodies are frequently absent; absent payload renders as + `body: `. Values equal to Relay's substitution marker + `[Filtered]` render as ``. Client-side SDK masking leaves no + marker and is therefore not detectable — such values render as delivered + rather than being claimed as redacted. +- Truncation is always stated, with the `cursor` needed to continue. Sentry + paginates segments rather than signals, so that `cursor` is synthetic: it + encodes the window, the `kinds` allow-list, and the offset, and each page + re-reads the recording. + +Gating matches `get_replay_details`: `skills: ["inspect"]`, +`requiredCapabilities: ["replays"]`, scopes `org:read`, `project:read`, +`event:read`. + +### Seer summary integration + +`get_replay_details` may call the summarize endpoint for chapters. It is +`ApiPublishStatus.EXPERIMENTAL` and triple-gated on the `session-replay` +feature, the `replay-ai-summaries` feature, and Seer access, returning 403 +otherwise. + +Requirements: + +- Treat it as strictly additive. A 403, a timeout, or a pending task must not + degrade the map. +- **Read it once.** `get_replay_details` issues a single `GET` and renders + chapters only if that response is already `completed`. It does not `POST` to + start a task, and it does not retry or wait for a running one. Starting would + spend a Seer LLM run per call for a section that would not be ready anyway; + retrying would put unbounded latency on the primary path. +- Because the read is one-shot, chapters appear only for replays already + summarized in the Sentry UI. +- The Seer response body is defined in `getsentry/seer` at + `src/seer/automation/summarize/replays.py`: + + ```text + { + data: {time_ranges: [{period_start, period_end, period_title}], summary} | null, + num_segments: int | null, + created_at: datetime | null, + status: "not_started" | "processing" | "completed" | "error", + } + ``` + + `period_start` and `period_end` are float UNIX timestamps in **milliseconds**, + so chapters carry windows usable for zoom, not just prose. Parse defensively + anyway — the endpoint is `EXPERIMENTAL` — and omit the section on any status + other than `completed`. +- Note the cold-start behavior: Seer's `/start` route returns an empty body and + enqueues a background task, so a replay nobody has summarized in the UI + returns `processing` on the first poll. Chapters render only for + already-summarized replays unless the tool starts a task and accepts the LLM + cost on every call. + +## Fixes Outside the Tool Surface + +These are correctness bugs in shipped code, independent of the new tool: + +- Classify replay events by `payload.category`/`op`; port the noise filter and + the dead/rage click rules. +- Replace the magnitude-based timestamp heuristic with per-type units. +- Follow the `Link` header's `cursor` on the segments index; drop the no-op + `download=true`. This needs the raw-response request path, since + `requestJSON` does not expose response headers. Read at most 150 segments or + 10MB of raw segment JSON, whichever comes first, and say which bound stopped + the read. 150 matches the clamp Sentry's own summarize endpoint applies; the + byte ceiling is the real guard, since segment sizes vary by orders of + magnitude, parsed rrweb objects expand well beyond their JSON size, and + `mcp-cloudflare` runs under a 128MB Workers limit. Both are provisional and + should be measured against real replays during QA. +- Rebuild `replay-recording-segments.json` from real SDK shapes, re-baseline + inline snapshots, and add a replay eval. +- Report every truncation. Today `MAX_ACTIVITY_EVENTS`, `MAX_RELATED_ERRORS`, + and `MAX_RELATED_TRACES` stop silently; only the issue-details replay list + prints an "and N more" line. +- Reconcile replay sorting against Sentry's actual sort configuration. The + authority is `sort_config` in + `sentry/replays/usecases/query/configs/aggregate_sort.py`; both the scalar and + aggregated query paths order by it, and `_get_sort_column` raises a + `ParseError` for anything absent. `REPLAY_SORT_FIELDS` in + `packages/mcp-core/src/tools/support/search-events/replays.ts` is missing only + `count_screens` and the aliases `browser`, `os`, and `os_name`. + `count_traces`, `count_segments`, and `viewed_by_me` are **not** sortable and + must not be added. `device.model` is already correct; the mismatch is that + discovery advertises `device.model_id`, which is filterable but not sortable. + The real defect is on the discovery side: `REPLAY_FIELDS` in + `packages/mcp-core/src/internal/agents/tools/dataset-fields.ts` presents + filterable fields as if they were sortable, so an agent picks a sort from + discovery output and gets a `UserInputError`. +- Gate the `dataset="replays"` path in `search_events` on the `replays` + capability, so replay search and replay details agree about availability. + A tool-level `requiredCapabilities` will not do: `search_events` serves six + datasets, and per `tools/catalog-runtime/availability.ts` the check only + applies when a `projectSlug` constraint is set. This needs a runtime rejection + in the handler plus removal of `replays` from the advertised dataset options. +- Distinguish rate limiting from absence in `listReplayIdsForIssue`. The + `replay-count` endpoint enforces 20 req/s per IP, per user, **and per + organization**. `get_issue_details` calls it on every lookup, and the current + `.catch(() => undefined)` makes throttling indistinguishable from "no + replays" — so parallel issue triage silently loses the Session Replay + section. + +## Examples + +Orientation, then a targeted read: + +```text +get_sentry_resource(url='https://my-org.sentry.io/explore/replays/7e07485f…/') +→ map: 1,182 signals, 4 pages, 2 failed network, 2 rage clicks + error at T+311.8s → suggested window + +get_replay_activity(replayId='7e07485f…', startMs=306000, endMs=316000, grain='detail') +→ T+311.8s click button#complete-order "Complete order" + T+312.0s network POST /api/checkout → 500 (1240ms) + body: + T+312.1s console error TypeError: Cannot read 'id' of undefined + at submitOrder (checkout.tsx:214) + T+312.3s click rage ×5 on button#complete-order (7000ms, timeout) +``` + +Cheap shape check on a long session: + +```text +get_replay_activity(replayId='7e07485f…', grain='digest', kinds=['network','console']) +→ network ×58 (2 failed) · console ×4 (2 error) +``` + +## Implementation + +1. Port the event taxonomy and log-message semantics into a shared internal + module; unit-test against realistically shaped fixtures. +2. Fix segment pagination and the timestamp units in the API client. +3. Rebuild replay fixtures; re-baseline snapshots. +4. Restructure `get_replay_details` into map form with a suggested next call + derived from error timestamps resolved through `replays-events-meta`. +5. Add `get_replay_activity` as a catalog-only tool. +6. Wire the Seer summary as an optional chapters section, read once. +7. Apply the sort-field, capability-gating, and rate-limit fixes. +8. Add a replay eval, then run + `pnpm run --filter @sentry/mcp-core generate-definitions`. + +## Testing + +- Unit tests per event type using SDK-shaped events, including + seconds-versus-milliseconds timestamps and the dead/rage click boundaries + (`timeAfterClickMs` at 6999 vs 7000, `clickCount` at 4 vs 5). +- Snapshot tests for each grain and for a windowed slice. +- Degradation tests: summarize 403, summarize timeout, segment fetch 404, + archived replay, replay with zero segments, and a replay exceeding one page of + segments. +- Redaction tests asserting `` and `` rendering. +- A replay eval covering map-then-zoom navigation. +- `pnpm run tsc && pnpm run lint && pnpm run test` must pass, plus + `pnpm run measure-tokens` to confirm tool-definition overhead. + +## Migration + +`get_replay_details` output changes shape; its inputs and name do not. Callers +passing replay URLs through `get_sentry_resource` are unaffected. Existing +inline snapshots must be re-baselined as part of the taxonomy fix, not +separately. Tool count rises by one, well inside the 20-tool target. + +## Out of Scope + +Web visual snapshots. Rendering a DOM snapshot requires running rrweb playback +in a browser, which this server has no place to host. The image plumbing already +exists (`get_snapshot_image`, `createImagePreview` in +`packages/mcp-core/src/internal/blob-utils.ts`), so this is a hosting gap, not a +protocol one. + +Mobile replay video is tractable — replays are captured as video and +`/projects/{org}/{project}/replays/{replay_id}/videos/{segment_id}/` exists — +but it is deferred rather than half-built alongside the web path. + +## Future Work + +- Friction analysis via `GET /organizations/{org}/replay-selectors/`, which + returns `count_dead_clicks`, `count_rage_clicks`, `dom_element`, and + `element.component_name` per selector. Component names come from Sentry + directly, with no separate corpus to author or upload. Likely belongs on the + `search_events` replay path rather than a new tool. +- Per-replay click detail via `/replays/{replay_id}/clicks/` (node IDs and click + timestamps only). +- `viewed-by` for review provenance. It returns an 18-key user serializer + including `experiments`, `has2fa`, and `isSuperuser`; trim to `id`, `name`, + and `email` before surfacing. +- Mobile replay video frames. + +## Constraints and Unverified Claims + +- The Seer summary response body is verified against `getsentry/seer` (see the + Seer section above), but the endpoint is `EXPERIMENTAL` on both sides; parse + defensively regardless. +- `replay_type` and `ota_updates` appear in the replay response but are **absent + from `VALID_FIELD_SET`**. Requesting them returns 400. Any `field` allow-list + must be explicit rather than derived from response keys. +- `data_source` on `replay-count` drifts between docs and runtime: docs declare + it required with `events`/`search_issues`/`spans`; the runtime validator + defaults to `discover` and also accepts `transactions`. The values currently + sent by `getReplayDataSource` (`discover`, `search_issues`) are valid against + the runtime validator. +- Replay retention is documented as 90 days paid and 30 days free. Only a flat + 90-day query window is visible in source; the per-plan split is a product-docs + claim. Hedge in user-facing copy or verify against a free-tier org. + +## References + +- Implementation: `packages/mcp-core/src/tools/catalog/get-replay-details.ts` +- Replay search: `packages/mcp-core/src/tools/support/search-events/replays.ts` +- API client: `packages/mcp-core/src/api-client/client.ts` +- Schemas: `packages/mcp-core/src/api-client/schema.ts` +- Issue-details integration: `packages/mcp-core/src/internal/formatting.ts` +- Field discovery: `packages/mcp-core/src/internal/agents/tools/dataset-fields.ts` +- Fixtures: `packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json` +- Upstream taxonomy: `getsentry/sentry` at + `src/sentry/replays/usecases/ingest/event_parser.py` and + `src/sentry/replays/usecases/summarize.py` +- Upstream summary endpoint: `getsentry/sentry` at + `src/sentry/replays/endpoints/project_replay_summary.py` +- Upstream summary response model: `getsentry/seer` at + `src/seer/automation/summarize/replays.py` +- Upstream error-event lookup: `getsentry/sentry` at + `src/sentry/replays/endpoints/organization_replay_events_meta.py` +- Upstream replay sort configuration: `getsentry/sentry` at + `src/sentry/replays/usecases/query/configs/aggregate_sort.py` +- [Replay recording event spec](https://develop.sentry.dev/sdk/data-model/event-payloads/replay-recording) +- [Sentry Replays API](https://docs.sentry.io/api/replays/) +- Tool authoring: [Adding Tools](../contributing/adding-tools.md), and + "Tool Output Policy" in [Tool Responses](../contributing/tool-responses.md) diff --git a/openspec/changes/improve-replay-review/.openspec.yaml b/openspec/changes/improve-replay-review/.openspec.yaml new file mode 100644 index 000000000..149631464 --- /dev/null +++ b/openspec/changes/improve-replay-review/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-08-17 diff --git a/openspec/changes/improve-replay-review/design.md b/openspec/changes/improve-replay-review/design.md new file mode 100644 index 000000000..9b12b8e39 --- /dev/null +++ b/openspec/changes/improve-replay-review/design.md @@ -0,0 +1,277 @@ +## Context + +`get_replay_details` is a catalog-only `inspect` tool gated on the `replays` +project capability. It fetches replay metadata, downloads recording segments, +and renders a fixed six-event activity list. `search_events` with +`dataset="replays"` lists replays, and `get_issue_details` surfaces related +replay IDs through `listReplayIdsForIssue`. + +Endpoint and payload validation against `getsentry/sentry@master` and the SDK +types vendored in `node_modules/@sentry-internal/replay`: + +- `src/sentry/replays/usecases/ingest/event_parser.py` defines a 32-member + `EventType` enum and a `which()` classifier keyed on `payload.category` for + `tag: "breadcrumb"` events and on `op` for `tag: "performanceSpan"` events. +- `src/sentry/replays/usecases/summarize.py` converts each type into + agent-facing prose through `as_log_message`, returning `None` for noise types + and skipping 2xx network requests. +- `get_timestamp_unit()` returns `"s"` for spans, web vitals, blur, and focus, + and `"ms"` for clicks, console, and navigation. The outer `timestamp` unit is + therefore a function of event type, not magnitude. +- `src/sentry/replays/endpoints/project_replay_summary.py` exposes + `POST`/`GET /projects/{org}/{project}/replays/{replay_id}/summarize/` with + `ApiPublishStatus.EXPERIMENTAL`, gated on the `session-replay` feature, the + `replay-ai-summaries` feature, and Seer access. It clamps `num_segments` to + 150 and proxies to Seer, returning Seer's JSON verbatim. +- The recording segments index takes only `cursor` and `per_page` and always + downloads bodies; it has no `download` parameter. The single-segment endpoint + does, and its check is presence-based. `per_page` cannot raise the page size: + its default and maximum are both 100, so `cursor` is the only way past the + first 100 segments. +- Seer's `/v1/automation/summarize/replay/breadcrumbs/start` returns an empty + body and enqueues a background task, so a first poll on a replay nobody has + summarized returns `processing`. +- `src/sentry/replays/endpoints/organization_replay_count.py` enforces + 20 req/s per IP, per user, and per organization. +- `GET /organizations/{org}/replays-events-meta/` resolves a batch of replay + error event IDs in one call via `query=id:[a,b]`, returning `id`, `issue`, + `issue.id`, `title`, `level`, `error.type`, `project.name`, and `timestamp`. + It is `ApiPublishStatus.PRIVATE`, returns 404 without the `session-replay` + feature and 403 without replay permission. Note the endpoint deletes + `timestamp_ms` from its output and rewrites `timestamp` as a + millisecond-precision ISO 8601 string, so the millisecond resolution the + suggested window needs arrives in `timestamp`, not in a separate field. + +## Goals / Non-Goals + +**Goals:** + +- Make replay output reflect what actually happened in a session, using + Sentry's own classification so MCP and Seer agree. +- Let agents choose a time window and a level of detail. +- Make truncation visible everywhere it occurs. +- Give agents a computed starting window derived from known error timestamps. +- Keep the direct top-level tool surface unchanged. + +**Non-Goals:** + +- Web visual snapshots. Rendering a DOM snapshot requires rrweb playback in a + browser, which this server has no place to host. +- Mobile replay video frames. +- Server-side session handles, open/close lifecycle, or usage-feedback tools. +- Changing Sentry's upstream API semantics or depending on unreleased + endpoints as a hard requirement. + +## Decisions + +### Port the upstream taxonomy instead of inventing one + +Classification and phrasing follow `event_parser.which()` and +`summarize.as_log_message()`. Dead versus rage clicks are behavioral, not +tagged: `ui.slowClickDetected` with `payload.data.endReason === "timeout"`, a +`payload.data.node.tagName` of `a`, `button`, or `input`, and +`timeAfterClickMs >= 7000` is a dead click, promoted to rage at +`clickCount >= 5`; anything else is a plain slow click. Upstream also accepts +the lowercase spellings `timeafterclickms` and `clickcount`, so the port reads +both. + +This inherits Sentry's noise filtering for free and keeps MCP output aligned +with what Sentry's own summarizer consumes. The alternative — a bespoke +classifier — was rejected because it would drift from upstream and reproduce +the current mismatch in a new form. + +The port lives in a shared internal module so `get_replay_details` and +`get_replay_activity` cannot diverge. + +### Split map from zoom rather than raising the cap + +`get_replay_details` returns session shape; `get_replay_activity` returns +signals for a window. Raising `MAX_ACTIVITY_EVENTS` was rejected: any fixed cap +either truncates long sessions or floods short ones, and neither tells the +agent where to look. + +### `replayId` plus a window is the handle + +No `client_id`, no open/close pair. MCP runs over stateless HTTP and stdio, so +handles would require KV storage, TTLs, and lifecycle handling to buy nothing +that `replayId` plus `startMs`/`endMs` does not already provide. Usage telemetry +goes on existing spans (grain requested, window width, result counts). + +Parameters are camelCase (`startMs`, `endMs`) to match every other tool in the +catalog. + +Statelessness has a cost worth naming: Sentry paginates recording segments, not +signals, so there is no server-side signal cursor to pass through. The `cursor` +returned by `get_replay_activity` is synthetic and must encode the window, the +`kinds` allow-list, and the offset so that continuing a page yields a stable +continuation of the same query. Each page therefore re-downloads and re-parses +the recording up to the segment budget below. + +### Grain controls rendering, `kinds` controls inclusion + +Keeping the two orthogonal avoids the ambiguity of a single `resolution` map +where one key both filters and sets fidelity. Omitting `kinds` includes +everything; supplying it is an allow-list. + +### Seer chapters are strictly additive + +The summarize endpoint is experimental and triple-gated. Its response body is +defined in `getsentry/seer` at `src/seer/automation/summarize/replays.py`: + +``` +SummarizeReplayBreadcrumbsStateResponse { + data: {time_ranges: [{period_start, period_end, period_title}], summary} | null + num_segments: int | null + created_at: datetime | null + status: not_started | processing | completed | error +} +``` + +`period_start` and `period_end` are float UNIX timestamps in milliseconds. +Chapters therefore carry windows usable for zoom, not merely prose. + +A 403, an error status, a timeout, a still-running task, or a parse failure +omits the section and never degrades the map. Parsing stays defensive because +the endpoint is experimental, but the field set is known rather than guessed. + +The section is read with a single request and never retried; see the decision +below. Blocking a tool call on repeated polling was rejected: it converts an +optional enhancement into a latency and failure risk on the primary path. + +### Error timestamps come from `replays-events-meta` + +The suggested window needs millisecond-resolution error timestamps. +`replay.error_ids` are event IDs, and resolving them through `listIssues` +returns issues, which carry no event timestamp. + +`GET /organizations/{org}/replays-events-meta/` resolves the whole batch in one +call and returns issue identity alongside the timestamp, so it also replaces the +per-error `listIssues` calls that populate the Related section. The alternative, +a `getEventForIssue` lookup per error, is public but costs two sequential calls +per error on the primary path. + +The accepted trade-off is that the endpoint is `ApiPublishStatus.PRIVATE` and +may change without notice; the intent is to make it public. Until then, treat +its response as untrusted: parse defensively and degrade to no suggested window +rather than failing the tool call. + +### The summary is read once, never started and never polled in a loop + +`get_replay_details` issues exactly one `GET` against the summarize endpoint and +renders chapters only if that single response is already `completed`. Any other +status is treated as "not available right now" and the section is omitted. + +No `POST`: Seer's start route enqueues a background task and returns an empty +body, so starting on every call would spend an LLM run per replay and still +return `processing` on the immediate read — paying the cost while rendering +nothing. + +No retry loop either. Waiting on a background task would put unbounded latency +on the primary path in exchange for a section that is optional by construction. +One read is bounded, cheap, and never worse than omitting the section. + +The consequence is that chapters appear only for replays already summarized in +the Sentry UI. That set is small today and grows with UI adoption. Revisit if +chapters prove valuable enough to justify starting tasks. + +### Segment budget is 150 segments with a byte ceiling + +150 matches the clamp Sentry's own summarize endpoint applies, so MCP reads no +more of a recording than Sentry's summarizer does. + +Segment count is a poor proxy for memory — segments vary in size by orders of +magnitude — so a byte ceiling is the real guard, with the segment count as a +cheap upper bound. This matters because `mcp-cloudflare` runs on Workers with a +128MB ceiling and each activity page re-reads the recording. + +The ceiling starts at **10MB of raw segment JSON**, measured on the downloaded +bytes before parsing. The headroom is deliberate: parsed rrweb objects expand +several times over their JSON representation, so 10MB of input can occupy +substantially more once decoded, and the budget must survive that expansion plus +the rendering pass. `MAX_PREVIEW_SOURCE_BYTES` in +`packages/mcp-core/src/internal/blob-utils.ts` is 20MB, but that bounds image +bytes that are never expanded into an object graph, so it is not a precedent +here. + +Both bounds are provisional. QA should measure actual parsed-heap cost against +real replays and adjust; the byte ceiling is the number most likely to move. + +### Redaction is labeled, never inferred + +`networkCaptureBodies` is opt-in, so bodies are frequently absent. Absent +payload renders as ``; silence would imply we could have retrieved +content that was never captured. + +Redaction is only claimed where it is detectable. Relay substitutes the literal +`[Filtered]` for scrubbed values, so that marker renders as ``. +Client-side SDK masking leaves no marker — a masked string is indistinguishable +from a real one — so such values render as delivered rather than being labeled. +Guessing would be worse than silence here: a wrong `` tells an agent +that content exists behind a mask when it may simply be the recorded value. + +### Distinguish rate limiting from absence + +`listReplayIdsForIssue` currently swallows every failure with +`.catch(() => undefined)`. Because `replay-count` is rate limited per +organization and `get_issue_details` calls it on every lookup, parallel issue +triage can silently lose the Session Replay section. Rate-limit responses are +reported as an unavailable-signal note rather than rendered as "no replays". + +## Risks / Trade-offs + +- **Snapshot churn.** The taxonomy fix changes nearly every replay snapshot. + Fixtures must be rebuilt before snapshots are re-baselined, otherwise the new + baselines encode the same fiction they do today. This forces task ordering. +- **Fixture realism is load-bearing.** Tests currently pass against an event + shape the SDK never emits. New fixtures must come from the documented + recording spec and the vendored SDK types. +- **Tool count.** Adds one catalog tool. The direct surface is unchanged and the + count stays well inside `PUBLIC_TOOL_HARD_LIMIT`. +- **Sort allow-list changes.** Widening `REPLAY_SORT_FIELDS` risks accepting a + sort Sentry rejects. Each added field is verified against Sentry's + `sort_config` rather than copied from `replayFields`. +- **Experimental Seer contract.** The response shape is verified against + `getsentry/seer`, but both endpoints are experimental. Mitigated by defensive + parsing and optional rendering; if the shape proves unstable, the section can + be dropped without touching the map. +- **Private `replays-events-meta` dependency.** The suggested window relies on a + `PRIVATE` endpoint that may change without notice. Mitigated by degrading to + no suggested window rather than failing the call, and by the intent to make + the endpoint public. +- **Chapters may rarely appear.** A single read means chapters show up only for + replays already summarized in the UI. Accepted deliberately: the alternatives + are spending an LLM run per call, or waiting on a background task, for a + section that is optional by construction. + +## Migration Plan + +`get_replay_details` keeps its name and inputs; only its output shape changes. +Callers passing replay URLs through `get_sentry_resource` are unaffected. +Existing inline snapshots are re-baselined as part of the taxonomy change, not +as a separate step. No API-client method is removed. + +## Resolved Questions + +- **Do Seer chapters carry usable timestamps?** Yes. `period_start` and + `period_end` are float UNIX milliseconds, so chapters can drive zoom windows. + They complement rather than replace the error-derived suggested window, + because the summary is triple-gated and frequently absent. +- **Should `get_replay_activity` accept an `errorId` shortcut?** No. It would + duplicate whatever timestamp resolution the suggested window uses, and the + suggested call in `get_replay_details` already covers the case. + +- **Where does an error's timestamp come from?** Resolved: the + `replays-events-meta` endpoint, accepting its private status. See the decision + above. +- **Does `get_replay_details` start a Seer summary, or only poll?** Resolved: + neither — one read, no start request and no retry loop. See the decision + above. +- **What is the segment download budget?** Resolved: 150 segments or 10MB of raw + segment JSON, whichever comes first. Both provisional. See the decision above. + +## Open Questions + +None blocking implementation. The three decisions above are settled. The +10MB byte ceiling and the 150-segment bound are starting values chosen from +upstream precedent and headroom reasoning rather than measurement, and QA +(task 9.6) should confirm or adjust them against real replays. diff --git a/openspec/changes/improve-replay-review/proposal.md b/openspec/changes/improve-replay-review/proposal.md new file mode 100644 index 000000000..acc8a9d74 --- /dev/null +++ b/openspec/changes/improve-replay-review/proposal.md @@ -0,0 +1,80 @@ +## Why + +Replay retrieval returns a single fixed-grain summary capped at six activity +events, and the classifier that builds it keys on an event shape the Sentry SDK +never emits. Real user actions arrive as rrweb custom events with +`data.tag === "breadcrumb"`, where meaning lives in `payload.category`, so every +click, console error, and rage click renders under the literal label +`breadcrumb` with its real content demoted into a details blob. + +Running the current handler against SDK-shaped segments produces six lines, of +which three are session-boot noise (`options`, which leaks sample rates; a +`` href; a vendor script fetch). A `ui.slowClickDetected` rage click was +dropped by the cap at only 11 input events; real sessions contain thousands. The +existing fixture uses `tag: "ui.click"`, so snapshots look correct while real +output degrades, and no replay eval exists. + +Agents cannot narrow to a time window, cannot trade detail for breadth, and +cannot distinguish truncation from absence. + +## What Changes + +- Port Sentry's own replay event taxonomy so classification matches what Seer + sees: classify by `payload.category` and span `op`, drop known noise types, + skip 2xx network requests, and apply the behavioral dead/rage click rules. +- Replace the magnitude-based timestamp heuristic with per-event-type units. +- Restructure `get_replay_details` from a prose sample into a map: signal + counts, page flow, kind breakdown with error markers, and a suggested + follow-up call windowed on the replay's own error timestamps, resolved in one + batched `replays-events-meta` call that also replaces the per-error issue + lookups behind the Related section. +- Add `get_replay_activity` as a catalog-only tool: a time window (`startMs`, + `endMs`), a grain (`digest`, `standard`, `detail`), an optional `kinds` + allow-list, and `limit`/`cursor` paging. +- Surface Sentry's experimental replay summary endpoint as an optional + `## Chapters` section, read once per call — never started, never retried — + strictly additive and degrading silently. +- Fix shipped API-client bugs: follow the `Link` header's `cursor` on the + recording segments index and drop the `download=true` parameter, which does + not exist on that endpoint. +- Report every truncation instead of stopping silently. +- Reconcile replay sorting against Sentry's `sort_config` — adding the sorts it + supports and no longer advertising filterable-but-unsortable fields as sort + options — gate the `dataset="replays"` search path on the `replays` + capability, and stop conflating `replay-count` rate limiting with "no + replays". +- Rebuild replay fixtures from real SDK shapes and add a replay eval. + +No server-side session state is introduced: `replayId` plus a window is the +navigation handle, so there is no open/close handle lifecycle. + +## Capabilities + +### New Capabilities + +- `replay-review`: Defines map-then-zoom replay retrieval, replay event + classification, and replay truncation reporting through MCP tools. + +### Modified Capabilities + +None. + +## Impact + +- `packages/mcp-core/src/tools/catalog/get-replay-details.ts` +- New catalog tool `packages/mcp-core/src/tools/catalog/get-replay-activity.ts` +- New shared module for replay event classification and rendering under + `packages/mcp-core/src/internal/` +- `packages/mcp-core/src/tools/catalog/index.ts` +- `packages/mcp-core/src/tools/catalog/search-events.ts` +- `packages/mcp-core/src/tools/support/search-events/replays.ts` +- `packages/mcp-core/src/tools/catalog/get-issue-details.ts` +- `packages/mcp-core/src/api-client/client.ts` +- `packages/mcp-core/src/api-client/schema.ts` +- `packages/mcp-core/src/internal/agents/tools/dataset-fields.ts` +- `packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json` +- `packages/mcp-server-mocks/src/index.ts` +- Replay tool tests, snapshots, and a new replay eval +- Generated definitions from + `pnpm run --filter @sentry/mcp-core generate-definitions` +- `docs/specs/replay-review.md` diff --git a/openspec/changes/improve-replay-review/specs/replay-review/spec.md b/openspec/changes/improve-replay-review/specs/replay-review/spec.md new file mode 100644 index 000000000..02f8a1162 --- /dev/null +++ b/openspec/changes/improve-replay-review/specs/replay-review/spec.md @@ -0,0 +1,205 @@ +## ADDED Requirements + +### Requirement: Replay events are classified by their payload discriminator +The system SHALL classify replay recording events using the event's +`payload.category` for breadcrumb events and `op` for performance span events, +matching Sentry's upstream replay event taxonomy. + +#### Scenario: Breadcrumb click event +- **WHEN** an event has `data.tag` of `breadcrumb` and `payload.category` of `ui.click` +- **THEN** it is classified as a click and rendered with its target, not under a literal `breadcrumb` label + +#### Scenario: Breadcrumb console error +- **WHEN** an event has `data.tag` of `breadcrumb` and `payload.category` of `console` +- **THEN** it is classified as a console signal and its level and message are rendered + +#### Scenario: Performance span network request +- **WHEN** an event has `data.tag` of `performanceSpan` and `op` of `resource.fetch` or `resource.xhr` +- **THEN** it is classified as a network signal and its method and status code are rendered + +#### Scenario: Session noise +- **WHEN** an event is an options, memory, mutations, canvas, script resource, or image resource event +- **THEN** it is excluded from rendered activity + +#### Scenario: Successful network request +- **WHEN** a network event has a 2xx status code +- **THEN** it is excluded from rendered signals but still included in kind counts, so a count of `network 58 (2 failed)` describes 58 requests of which 2 are rendered + +#### Scenario: Web navigation preference +- **WHEN** a web replay contains both a `navigation` breadcrumb and a `navigation.*` performance span for the same transition +- **THEN** the navigation span is used and the breadcrumb is excluded, matching upstream's web-replay preference + +#### Scenario: Unclassifiable event +- **WHEN** an event matches no known category or span `op` +- **THEN** it is excluded from rendered activity and its timestamp is interpreted as milliseconds + +### Requirement: Dead and rage clicks are classified behaviorally +The system SHALL distinguish slow, dead, and rage clicks using the slow-click +payload rather than treating every `ui.slowClickDetected` event alike. + +#### Scenario: Dead click +- **WHEN** a `ui.slowClickDetected` event has `endReason` of `timeout`, a target tag of `a`, `button`, or `input`, and `timeAfterClickMs` of at least 7000 +- **THEN** it is classified as a dead click + +#### Scenario: Rage click +- **WHEN** an event meets the dead click conditions and has `clickCount` of at least 5 +- **THEN** it is classified as a rage click + +#### Scenario: Slow click below threshold +- **WHEN** a `ui.slowClickDetected` event does not meet the dead click conditions +- **THEN** it is classified as a plain slow click and not counted as dead or rage + +### Requirement: Replay event timestamps use per-type units +The system SHALL resolve a replay event's timestamp unit from its event type +rather than inferring the unit from the value's magnitude. + +#### Scenario: Span-derived event +- **WHEN** the event is a navigation span, resource span, web vital, blur, or focus event +- **THEN** its outer timestamp is interpreted as seconds + +#### Scenario: Breadcrumb-derived event +- **WHEN** the event is a click, console, or navigation breadcrumb event +- **THEN** its outer timestamp is interpreted as milliseconds + +#### Scenario: Session time origin +- **WHEN** a signal's relative offset is computed +- **THEN** it is measured from the replay's `started_at`, not from the first recorded activity event, so offsets align with the replay metadata timeline + +### Requirement: Replay details returns a session map +The `get_replay_details` tool SHALL return the shape of a session rather than a +fixed-length sample of its events. + +#### Scenario: Map contents +- **WHEN** `get_replay_details` succeeds for a replay with recording segments +- **THEN** the response includes the total signal count, the covered time span, the page flow, and a per-kind breakdown with error and rage or dead click counts + +#### Scenario: Suggested follow-up window +- **WHEN** the replay has at least one associated error with a resolvable timestamp +- **THEN** the response includes a `get_replay_activity` call with a time window bracketing that error + +#### Scenario: Error timestamps unavailable +- **WHEN** the error-event lookup fails, is unauthorized, or returns no usable timestamp +- **THEN** the response omits the suggested window, suggests a whole-session digest instead, and still returns the map + +#### Scenario: Archived replay +- **WHEN** the replay is archived +- **THEN** the response states that the recording is unavailable and omits the map + +### Requirement: Replay activity supports windowed, graded retrieval +The system SHALL provide a `get_replay_activity` catalog tool that returns +replay signals for a requested time window at a requested grain. + +#### Scenario: Windowed request +- **WHEN** a caller supplies `startMs` and `endMs` +- **THEN** only signals within that window are returned, measured from the replay's `started_at` + +#### Scenario: Whole session request +- **WHEN** a caller omits `startMs` and `endMs` +- **THEN** signals from the entire session are considered + +#### Scenario: Digest grain +- **WHEN** a caller requests `grain` of `digest` +- **THEN** the response returns one rollup line per kind with counts, including error counts + +#### Scenario: Detail grain +- **WHEN** a caller requests `grain` of `detail` +- **THEN** the response includes available payload for each signal, such as request method, status code, duration, and stack frames + +#### Scenario: Kind filtering +- **WHEN** a caller supplies `kinds` +- **THEN** only those kinds are returned and all others are excluded + +#### Scenario: Available kinds +- **WHEN** a caller inspects the `kinds` allow-list +- **THEN** it offers `navigation`, `click`, `dead-click`, `rage-click`, `slow-click`, `network`, `console`, `hydration-error`, `feedback`, `web-vital`, and the mobile kinds `tap`, `scroll`, `swipe`, `app-lifecycle`, and `device`, each mapping to one or more upstream event types + +#### Scenario: Tool availability +- **WHEN** the MCP server registers tools +- **THEN** `get_replay_activity` is catalog-only, requires the `inspect` skill, and requires the `replays` project capability + +### Requirement: Unavailable replay payload is labeled +The system SHALL distinguish payload that was never captured from payload that +was removed by masking. + +#### Scenario: Body not captured +- **WHEN** a network signal is rendered at detail grain and the SDK did not capture bodies +- **THEN** the response marks the body as not captured + +#### Scenario: Scrubbed value +- **WHEN** a value equals Relay's PII substitution marker `[Filtered]` +- **THEN** the response marks the value as redacted + +#### Scenario: Client-masked value +- **WHEN** a value was masked client-side by the SDK, leaving no distinguishing marker +- **THEN** the value is rendered as delivered and is not claimed to be redacted, because client-side masking is not detectable from the payload + +### Requirement: Replay truncation is reported +The system SHALL state when replay output omits available data and how to +retrieve the remainder. + +#### Scenario: Signal limit reached +- **WHEN** more signals match a request than the response returns +- **THEN** the response states that results were truncated and includes a cursor for continuation + +#### Scenario: Segment paging +- **WHEN** a replay has more recording segments than one page +- **THEN** the system follows the `Link` response header's `cursor` to read subsequent pages rather than silently reading only the first 100 segments + +#### Scenario: Segment budget reached +- **WHEN** a replay has more segments than the configured download budget +- **THEN** the response states that the recording was read only up to that budget and that later signals are missing + +#### Scenario: Related resource limits +- **WHEN** related issues or traces exceed the display limit +- **THEN** the response states how many were omitted + +### Requirement: Replay summary chapters are optional +The system SHALL treat Sentry's experimental replay summary as an additive +enhancement that never degrades replay details. + +#### Scenario: Summary available +- **WHEN** the summary endpoint returns `status` of `completed` with a `data` body +- **THEN** `get_replay_details` includes a chapters section rendering each `time_ranges` entry's `period_title` with its `period_start` and `period_end` as relative offsets + +#### Scenario: Summary still running +- **WHEN** the summary endpoint returns `status` of `processing` or `not_started` +- **THEN** `get_replay_details` omits the chapters section and returns the map unchanged + +#### Scenario: Single read +- **WHEN** `get_replay_details` checks for a replay summary +- **THEN** it issues exactly one read, does not request that a summary be started, and does not retry or wait for a running task to finish + +#### Scenario: Summary unavailable +- **WHEN** the summary endpoint returns a permission error, an error status, times out, or returns a body that fails to parse +- **THEN** `get_replay_details` omits the chapters section and returns the map unchanged + +### Requirement: Replay search availability matches replay details +The system SHALL apply the `replays` project capability consistently across +replay retrieval paths. + +#### Scenario: Project without replays enabled +- **WHEN** a session is constrained to a project that does not have replays enabled +- **THEN** a `search_events` call with `dataset="replays"` is rejected at handler time, consistent with `get_replay_details` being unavailable + +#### Scenario: Dataset not advertised +- **WHEN** a session is constrained to a project that does not have replays enabled +- **THEN** the `search_events` tool description and dataset options omit `replays`, so the routing agent does not select a dataset that would be rejected + +### Requirement: Replay sort options match what Sentry can sort +The system SHALL accept exactly the replay sort values Sentry's replay sort +configuration supports, and SHALL NOT advertise unsortable fields as sortable. + +#### Scenario: Sortable field accepted +- **WHEN** an agent sorts by a replay field present in Sentry's replay sort configuration +- **THEN** the request is accepted rather than rejected as an invalid sort + +#### Scenario: Filterable but unsortable field +- **WHEN** a replay field is searchable but absent from Sentry's replay sort configuration +- **THEN** field discovery does not present it as a sort option, so an agent cannot pick a sort that Sentry would reject + +### Requirement: Replay lookup failures are distinguishable from absence +The system SHALL not report an unavailable replay signal as an absent one. + +#### Scenario: Replay count rate limited +- **WHEN** the replay count lookup for an issue is rate limited or otherwise fails +- **THEN** the issue response indicates that replay information was unavailable rather than implying the issue has no replays diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md new file mode 100644 index 000000000..3be04d247 --- /dev/null +++ b/openspec/changes/improve-replay-review/tasks.md @@ -0,0 +1,84 @@ +## 1. Fixtures First + +Fixtures must be realistic before any snapshot is re-baselined, otherwise new +baselines encode the same wrong event shape they do today. + +- [x] 1.1 Rebuild `packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json` from the documented recording spec and the `@sentry-internal/replay` frame types, using `tag: "breadcrumb"` with `payload.category` for user actions. Those types are present only transitively (`node_modules/.pnpm/@sentry-internal+replay@*/.../types/replayFrame.d.ts`), so read them for shape but do not import them — copy what is needed rather than adding a dependency on an internal package. +- [x] 1.2 Include at least one each of: `ui.click`, `console` error, `navigation` breadcrumb, `navigation.navigate` span, `resource.fetch` with a 5xx status, `resource.fetch` with a 2xx status, `ui.slowClickDetected` meeting rage thresholds, `options`, and `resource.script`. +- [x] 1.3 Add a multi-page segment fixture and MSW handlers that emit a `Link` header with a `cursor` so segment paging is exercisable. +- [x] 1.4 Add a `replayId` to `packages/mcp-server-evals/src/evals/utils/fixtures.ts`. +- [x] 1.5 Add summarize-endpoint fixtures and MSW handlers covering `completed`, `processing`, and 403. Default handler returns `completed`; `replaySummaryProcessingFixture` is exported and 403 needs no fixture, so both are applied as per-test `mswServer.use` overrides when task 4.6 lands. +- [x] 1.6 Add `replays-events-meta` fixtures and MSW handlers returning `id`, `issue`, `issue.id`, `title`, and a millisecond-precision ISO `timestamp` for the fixture replay's `error_ids`. + +## 2. Event Classification Module + +- [ ] 2.1 Re-verify `which()`, `as_log_message()`, and `get_timestamp_unit()` against `getsentry/sentry` before porting. +- [ ] 2.2 Add a shared internal module exposing replay event classification, per-type timestamp resolution, and signal rendering. +- [ ] 2.3 Port the event-type taxonomy and the noise exclusion list, including skipping 2xx network requests from rendering while keeping them in kind counts, and preferring `NAVIGATION_SPAN` over `NAVIGATION` for web replays. +- [ ] 2.4 Implement behavioral dead, rage, and slow click classification, reading `payload.data.node.tagName` and accepting the lowercase `timeafterclickms`/`clickcount` spellings. +- [ ] 2.5 Implement `digest`, `standard`, and `detail` rendering, labeling absent payload `` and values equal to `[Filtered]` as ``. +- [ ] 2.6 Unit-test each event type, both timestamp units, and the dead/rage boundaries at `timeAfterClickMs` 6999 vs 7000 and `clickCount` 4 vs 5. +- [ ] 2.7 Resolve signal offsets from the replay's `started_at` rather than the first recorded event. + +## 3. API Client + +- [ ] 3.1 Re-verify the recording segments index contract against `getsentry/sentry`. +- [ ] 3.2 Follow the `Link` header's `cursor` in `getReplayRecordingSegments` and remove the non-existent `download=true` parameter. Note `per_page` is a no-op (default and maximum are both 100), and that reading the header requires the raw-response request path rather than `requestJSON`. +- [ ] 3.3 Page through segments up to 150 segments or 10MB of raw segment JSON, whichever is hit first, and report which bound stopped the read. +- [ ] 3.4 Add a single-read replay summary method for the experimental summarize endpoint, parsing the known Seer response shape (`data.time_ranges[]`, `status`) defensively. No start method and no retry loop: one `GET`, then move on. +- [ ] 3.5 Add an explicit `field` allow-list to `searchReplays`, drawn from `VALID_FIELD_SET` in `sentry/replays/validators.py`. Use upstream's coarse names (`browser`, `user`, `device`, `sdk`, `releases`, `trace_ids`) and exclude `replay_type` and `ota_updates`, which return 400. +- [ ] 3.6 Add a `getReplayErrorEvents` method wrapping `GET /organizations/{org}/replays-events-meta/` with a batched `query=id:[...]`, returning `id`, `issue`, `issue.id`, `title`, and `timestamp`. Parse `timestamp` as a millisecond-precision ISO string; the endpoint deletes `timestamp_ms` from its own output. + +## 4. Replay Details as a Map + +- [ ] 4.1 Replace the activity sample with signal counts, time span, page flow, and per-kind breakdown including error and rage or dead click counts. +- [ ] 4.2 Derive a suggested `get_replay_activity` window from error timestamps resolved through `getReplayErrorEvents`, degrading to no suggested window if the endpoint fails or returns no usable timestamp. +- [ ] 4.3 Replace the per-error `listIssues` lookups in the Related section with the same batched `getReplayErrorEvents` call, which already returns issue identity. +- [ ] 4.4 Fall back to a whole-session digest suggestion when no error timestamp resolves. +- [ ] 4.5 Report truncation for related issues and traces. +- [ ] 4.6 Wire the summary chapters section as a single-read enhancement, rendering only on `status: completed` and degrading silently on 403, error status, `processing`, `not_started`, timeout, and parse failure. Assert in tests that exactly one request is issued and no start request is sent. +- [ ] 4.7 Preserve archived-replay and missing-segment behavior. + +## 5. Replay Activity Tool + +- [ ] 5.1 Add `get_replay_activity` as a catalog-only `inspect` tool with `requiredCapabilities: ["replays"]` and replay read scopes. +- [ ] 5.2 Implement `startMs`/`endMs` windowing, defaulting to the whole session. +- [ ] 5.3 Implement `grain` and the optional `kinds` allow-list. +- [ ] 5.4 Implement `limit`/`cursor` paging with explicit truncation reporting, encoding the window, `kinds`, and offset in the synthetic cursor so continuation is stable. +- [ ] 5.5 Accept `replayUrl` as well as `organizationSlug` plus `replayId`, reusing the existing parameter resolution and constraint checks. +- [ ] 5.6 Register in the catalog and confirm it is not added to the direct top-level surface. +- [ ] 5.7 Record telemetry on the existing span: grain requested, window width, and result counts. + +## 6. Adjacent Correctness Fixes + +- [ ] 6.1 Add the sorts Sentry supports but `REPLAY_SORT_FIELDS` omits: `count_screens` and the aliases `browser`, `os`, `os_name`. Do not add `count_traces`, `count_segments`, or `viewed_by_me`, which are absent from `sort_config` and would be rejected. +- [ ] 6.2 Stop advertising filterable-but-unsortable replay fields as sort options in `dataset-fields.ts`, so discovery output cannot produce an invalid sort. +- [ ] 6.3 Reject `dataset="replays"` in the `search_events` handler when the constrained project lacks the `replays` capability, and drop `replays` from the advertised dataset options in that case. +- [ ] 6.4 Distinguish rate-limit and error responses from absence in `listReplayIdsForIssue`, and report unavailability in the issue Session Replay section. + +## 7. Tests + +- [ ] 7.1 Re-baseline `get-replay-details.test.ts` snapshots against the rebuilt fixtures. +- [ ] 7.2 Add `get-replay-activity.test.ts` covering windowing, each grain, kind filtering, paging, and constraint injection. +- [ ] 7.3 Add degradation tests: summary 403, summary `processing`, summary timeout, summary unparseable, segment fetch 404, archived replay, zero segments, multi-page segments, and the segment budget being hit. +- [ ] 7.4 Add redaction tests asserting `` and `[Filtered]`-driven `` rendering. +- [ ] 7.5 Add search-events tests for the replay capability gate, the dataset options omitting `replays`, and the reconciled sort list. +- [ ] 7.6 Add a `get_issue_details` test asserting rate-limited replay lookup is reported as unavailable, not as no replays. +- [ ] 7.7 Add a replay eval covering map-then-zoom navigation. +- [ ] 7.8 Update registry, tool count, skill gating, and generated-definition tests. +- [ ] 7.9 Add suggested-window tests: a resolved error timestamp produces a bracketing window, and a failing or unauthorized `replays-events-meta` lookup degrades to the whole-session suggestion with the map intact. + +## 8. Documentation and Generated Definitions + +- [ ] 8.1 Update `docs/specs/replay-review.md` if the implemented contract diverges from the spec. +- [ ] 8.2 Update any docs describing replay tool output. +- [ ] 8.3 Run `pnpm run --filter @sentry/mcp-core generate-definitions`. + +## 9. Verification + +- [ ] 9.1 Run targeted replay tool tests and catalog availability tests. +- [ ] 9.2 Run `pnpm run tsc`. +- [ ] 9.3 Run `pnpm run lint`. +- [ ] 9.4 Run `pnpm run test`. +- [ ] 9.5 Run `pnpm run measure-tokens` and confirm the added tool definition stays within budget. +- [ ] 9.6 QA against a real organization with the `mcp-qa` skill, since mocks cannot prove the SDK-shape fix. Confirm on a long real replay that the 150-segment and 10MB bounds hold under the Workers memory ceiling, and adjust them if not. From 686471cff4d0c179f376c9e225388bf69a3ae8a7 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 13:49:29 -0700 Subject: [PATCH 03/26] feat(replays): Port Sentry's replay event taxonomy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replay events were classified by their rrweb `data.tag`, but the SDK emits every user action with `tag: "breadcrumb"` and puts the meaning in `payload.category`. Clicks, console errors, and rage clicks therefore all rendered under the literal label `breadcrumb`, with their real content demoted into a details blob. Add a shared internal module that ports Sentry's own taxonomy from `replays/usecases/ingest/event_parser.py` and `summarize.py`, so MCP output and Seer's summarizer agree about what a session contains: - Classify breadcrumbs by `payload.category` and spans by `op`, matching `which()`. Unrecognized events are `unknown` rather than guessed at. - Resolve timestamps by event type. Spans and web vitals are seconds; clicks, console, and navigation are milliseconds. The previous magnitude heuristic held only by coincidence of present-day epochs. - Drop the noise upstream refuses to narrate, and skip 2xx requests from the rendered set while still counting them, so `network 58 (2 failed)` means 58 requests of which 2 were narrated. - Prefer the navigation span on web and the navigation breadcrumb on mobile, where no span exists. - Separate dead, rage, and slow clicks behaviorally: all three arrive as `ui.slowClickDetected` and differ only by end reason, target element, stall duration, and click count. - Measure offsets from the replay's `started_at` rather than the first recorded event, so they align with replay metadata and error timestamps. Render at three grains, and label unavailable payload rather than passing over it in silence: bodies the SDK never captured are marked not captured, and only Relay's `[Filtered]` marker is reported as redacted. Client-side SDK masking leaves no marker, so masked values render as delivered instead of asserting a redaction we cannot detect. Widen the recording payload schema to carry the fields classification needs. Parsing stays permissive: recordings span many SDK versions and pass through PII scrubbing, which can replace a number with a marker string, so a malformed field degrades that field rather than dropping the event. Against the rebuilt fixtures this surfaces the failed checkout request, the console error, the rage click, and the dead click — all four of which the previous six-event cap dropped before reaching them. Refs docs/specs/replay-review.md Co-Authored-By: Claude Opus 5 (1M context) --- .../changes/improve-replay-review/tasks.md | 14 +- packages/mcp-core/src/api-client/schema.ts | 66 +- packages/mcp-core/src/api-client/types.ts | 7 + .../src/internal/replay-events.test.ts | 615 ++++++++++++ .../mcp-core/src/internal/replay-events.ts | 896 ++++++++++++++++++ 5 files changed, 1590 insertions(+), 8 deletions(-) create mode 100644 packages/mcp-core/src/internal/replay-events.test.ts create mode 100644 packages/mcp-core/src/internal/replay-events.ts diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index 3be04d247..55d62f240 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -12,13 +12,13 @@ baselines encode the same wrong event shape they do today. ## 2. Event Classification Module -- [ ] 2.1 Re-verify `which()`, `as_log_message()`, and `get_timestamp_unit()` against `getsentry/sentry` before porting. -- [ ] 2.2 Add a shared internal module exposing replay event classification, per-type timestamp resolution, and signal rendering. -- [ ] 2.3 Port the event-type taxonomy and the noise exclusion list, including skipping 2xx network requests from rendering while keeping them in kind counts, and preferring `NAVIGATION_SPAN` over `NAVIGATION` for web replays. -- [ ] 2.4 Implement behavioral dead, rage, and slow click classification, reading `payload.data.node.tagName` and accepting the lowercase `timeafterclickms`/`clickcount` spellings. -- [ ] 2.5 Implement `digest`, `standard`, and `detail` rendering, labeling absent payload `` and values equal to `[Filtered]` as ``. -- [ ] 2.6 Unit-test each event type, both timestamp units, and the dead/rage boundaries at `timeAfterClickMs` 6999 vs 7000 and `clickCount` 4 vs 5. -- [ ] 2.7 Resolve signal offsets from the replay's `started_at` rather than the first recorded event. +- [x] 2.1 Re-verify `which()`, `as_log_message()`, and `get_timestamp_unit()` against `getsentry/sentry` before porting. Two recent upstream fixes are folded in: missing `method`/`statusCode` on requests that never got a response (#120859), and PII-scrubbed values arriving where a number is expected (#121765). +- [x] 2.2 Add a shared internal module exposing replay event classification, per-type timestamp resolution, and signal rendering (`packages/mcp-core/src/internal/replay-events.ts`). +- [x] 2.3 Port the event-type taxonomy and the noise exclusion list, including skipping 2xx network requests from rendering while keeping them in kind counts, and preferring `NAVIGATION_SPAN` over `NAVIGATION` for web replays. +- [x] 2.4 Implement behavioral dead, rage, and slow click classification, reading `payload.data.node.tagName` and accepting the lowercase `timeafterclickms`/`clickcount` spellings. +- [x] 2.5 Implement `digest`, `standard`, and `detail` rendering, labeling absent payload `` and values equal to `[Filtered]` as ``. +- [x] 2.6 Unit-test each event type, both timestamp units, and the dead/rage boundaries at `timeAfterClickMs` 6999 vs 7000 and `clickCount` 4 vs 5. +- [x] 2.7 Resolve signal offsets from the replay's `started_at` rather than the first recorded event. ## 3. API Client diff --git a/packages/mcp-core/src/api-client/schema.ts b/packages/mcp-core/src/api-client/schema.ts index bd1f456e3..c2385148f 100644 --- a/packages/mcp-core/src/api-client/schema.ts +++ b/packages/mcp-core/src/api-client/schema.ts @@ -412,16 +412,80 @@ export const ReplayDetailsSchema = z }) .passthrough(); -const ReplayRecordingPayloadSchema = z +/** + * Request or response half of a replay network span. + * + * `size` is absent unless the SDK captured bodies (`networkCaptureBodies` is + * opt-in), so an absent value means "not captured", not "empty". + */ +const NetworkBodySchema = z .object({ + size: z.number().optional().catch(undefined), + headers: z.record(z.string(), z.unknown()).optional(), + body: z.unknown().optional(), + }) + .passthrough(); + +/** + * Body of a replay recording event. + * + * Every field is optional and permissively parsed: recordings are produced by + * many SDK versions and are subject to PII scrubbing, which can replace a + * numeric value with a marker string. Upstream reads these defensively for the + * same reason (see `sentry.replays.usecases.ingest.event_parser`), so a + * malformed field must degrade that one field rather than drop the event. + */ +export const ReplayRecordingPayloadSchema = z + .object({ + // Span events (`tag: "performanceSpan"`) carry the meaning in `op`; + // breadcrumb events carry it in `category`. op: z.string().optional().catch(undefined), description: z.string().optional().catch(undefined), message: z.string().optional().catch(undefined), category: z.string().optional().catch(undefined), type: z.string().optional().catch(undefined), + level: z.string().optional().catch(undefined), + startTimestamp: z.number().optional().catch(undefined), + endTimestamp: z.number().optional().catch(undefined), + timestamp: z.number().optional().catch(undefined), data: z .object({ duration: z.number().optional().catch(undefined), + // Network spans. `method` and `statusCode` are absent on requests that + // never got a response (CORS failures, for example). + method: z.string().optional().catch(undefined), + statusCode: z.number().optional().catch(undefined), + request: NetworkBodySchema.optional().catch(undefined), + response: NetworkBodySchema.optional().catch(undefined), + // SDK 7.44/7.45 reported body sizes at the top level instead. + requestBodySize: z.number().optional().catch(undefined), + responseBodySize: z.number().optional().catch(undefined), + // Click events. Upstream also accepts the lowercase spellings. + node: z + .object({ + id: z.number().optional().catch(undefined), + tagName: z.string().optional().catch(undefined), + textContent: z.string().optional().catch(undefined), + attributes: z.record(z.string(), z.unknown()).optional(), + }) + .passthrough() + .optional() + .catch(undefined), + endReason: z.string().optional().catch(undefined), + timeAfterClickMs: z.number().optional().catch(undefined), + timeafterclickms: z.number().optional().catch(undefined), + clickCount: z.number().optional().catch(undefined), + clickcount: z.number().optional().catch(undefined), + url: z.string().optional().catch(undefined), + // Navigation breadcrumbs (mobile keeps these; web prefers the span). + to: z.string().optional().catch(undefined), + from: z.string().optional().catch(undefined), + // Web vitals. + size: z.number().optional().catch(undefined), + rating: z.string().optional().catch(undefined), + // Mobile scroll/swipe events. + "view.id": z.string().optional().catch(undefined), + direction: z.string().optional().catch(undefined), }) .passthrough() .optional() diff --git a/packages/mcp-core/src/api-client/types.ts b/packages/mcp-core/src/api-client/types.ts index ffeda5280..cd388e4ed 100644 --- a/packages/mcp-core/src/api-client/types.ts +++ b/packages/mcp-core/src/api-client/types.ts @@ -109,6 +109,7 @@ import type { ReplayDetailsSchema, ReplayListResponseSchema, ReplayRecordingEventSchema, + ReplayRecordingPayloadSchema, ReplayRecordingSegmentsSchema, StacktraceLinkSchema, TagListSchema, @@ -179,6 +180,12 @@ export type AssignedTo = z.infer; export type ReplayDetails = z.infer; export type ReplayList = z.infer["data"]; export type ReplayRecordingEvent = z.infer; +export type ReplayRecordingPayload = z.infer< + typeof ReplayRecordingPayloadSchema +>; +export type ReplayRecordingPayloadData = NonNullable< + ReplayRecordingPayload["data"] +>; export type ReplayRecordingSegments = z.infer< typeof ReplayRecordingSegmentsSchema >; diff --git a/packages/mcp-core/src/internal/replay-events.test.ts b/packages/mcp-core/src/internal/replay-events.test.ts new file mode 100644 index 000000000..1d0f78198 --- /dev/null +++ b/packages/mcp-core/src/internal/replay-events.test.ts @@ -0,0 +1,615 @@ +import { describe, expect, it } from "vitest"; +import { replayRecordingSegmentsFixture } from "@sentry/mcp-server-mocks"; +import type { ReplayRecordingSegments } from "../api-client"; +import { + NOT_CAPTURED, + REDACTED, + classifyReplayEvent, + countReplayKinds, + extractReplaySignals, + formatReplayOffset, + isMobilePlatform, + renderReplaySignals, + resolveTimestampMs, +} from "./replay-events.js"; + +const SESSION_START = "2025-04-07T12:00:00.000Z"; +const SESSION_START_MS = Date.parse(SESSION_START); + +/** Build a breadcrumb custom event as the SDK emits it. */ +function breadcrumb( + category: string, + payload: Record = {}, + timestampMs = SESSION_START_MS, +) { + return { + type: 5, + timestamp: timestampMs, + data: { + tag: "breadcrumb", + payload: { type: "default", category, ...payload }, + }, + }; +} + +/** Build a performance span custom event. Span timestamps are in seconds. */ +function span( + op: string, + payload: Record = {}, + timestampSeconds = SESSION_START_MS / 1000, +) { + return { + type: 5, + timestamp: timestampSeconds, + data: { tag: "performanceSpan", payload: { op, ...payload } }, + }; +} + +/** Run events through the schema so tests exercise real parsed shapes. */ +function signalsFrom( + events: unknown[], + options: { platform?: string } = {}, +): ReturnType { + return extractReplaySignals([events] as ReplayRecordingSegments, { + startedAt: SESSION_START, + platform: options.platform ?? "javascript", + }); +} + +function slowClick(data: Record) { + return breadcrumb("ui.slowClickDetected", { + message: "button#submit", + data: { + node: { id: 1, tagName: "button", textContent: "Submit", attributes: {} }, + ...data, + }, + }); +} + +describe("classifyReplayEvent", () => { + it("classifies breadcrumbs by payload.category, not by tag", () => { + // The whole point of the port: `data.tag` is always "breadcrumb", so + // classifying on it collapses every user action into one label. + expect(classifyReplayEvent(breadcrumb("ui.click") as never)).toBe("click"); + expect(classifyReplayEvent(breadcrumb("console") as never)).toBe("console"); + expect(classifyReplayEvent(breadcrumb("navigation") as never)).toBe( + "navigation", + ); + expect( + classifyReplayEvent(breadcrumb("replay.hydrate-error") as never), + ).toBe("hydration-error"); + expect(classifyReplayEvent(breadcrumb("sentry.feedback") as never)).toBe( + "feedback", + ); + expect(classifyReplayEvent(breadcrumb("ui.multiClick") as never)).toBe( + "multi-click", + ); + }); + + it("classifies mobile breadcrumb categories", () => { + expect(classifyReplayEvent(breadcrumb("ui.tap") as never)).toBe("tap"); + expect(classifyReplayEvent(breadcrumb("ui.scroll") as never)).toBe( + "scroll", + ); + expect(classifyReplayEvent(breadcrumb("ui.swipe") as never)).toBe("swipe"); + expect(classifyReplayEvent(breadcrumb("app.background") as never)).toBe( + "background", + ); + expect(classifyReplayEvent(breadcrumb("device.battery") as never)).toBe( + "device-battery", + ); + }); + + it("classifies performance spans by op", () => { + expect(classifyReplayEvent(span("resource.fetch") as never)).toBe( + "resource-fetch", + ); + expect(classifyReplayEvent(span("resource.xhr") as never)).toBe( + "resource-xhr", + ); + expect(classifyReplayEvent(span("resource.script") as never)).toBe( + "resource-script", + ); + expect(classifyReplayEvent(span("memory") as never)).toBe("memory"); + }); + + it("treats every navigation* op as a navigation span", () => { + // Upstream matches on prefix, covering navigate, reload, back_forward, + // and the SPA-only navigation.push. + for (const op of [ + "navigation.navigate", + "navigation.reload", + "navigation.back_forward", + "navigation.push", + ]) { + expect(classifyReplayEvent(span(op) as never)).toBe("navigation-span"); + } + }); + + it("splits web vitals by description", () => { + expect( + classifyReplayEvent( + span("web-vital", { description: "largest-contentful-paint" }) as never, + ), + ).toBe("lcp"); + expect( + classifyReplayEvent( + span("web-vital", { description: "cumulative-layout-shift" }) as never, + ), + ).toBe("cls"); + expect( + classifyReplayEvent( + span("web-vital", { description: "first-input-delay" }) as never, + ), + ).toBe("unknown"); + }); + + it("classifies options events and canvas mutations", () => { + expect( + classifyReplayEvent({ + type: 5, + timestamp: 1, + data: { tag: "options", payload: {} }, + } as never), + ).toBe("options"); + expect( + classifyReplayEvent({ + type: 3, + timestamp: 1, + data: { source: 9 }, + } as never), + ).toBe("canvas"); + expect( + classifyReplayEvent({ + type: 3, + timestamp: 1, + data: { source: 2 }, + } as never), + ).toBe("unknown"); + }); + + it("returns unknown rather than guessing at unrecognized events", () => { + expect(classifyReplayEvent(breadcrumb("ui.somethingNew") as never)).toBe( + "unknown", + ); + expect(classifyReplayEvent(span("resource.websocket") as never)).toBe( + "unknown", + ); + // A full snapshot (type 4) carries no activity. + expect( + classifyReplayEvent({ + type: 4, + timestamp: 1, + data: { href: "/" }, + } as never), + ).toBe("unknown"); + }); +}); + +describe("dead and rage click classification", () => { + it("classifies a rage click at the clickCount threshold", () => { + expect( + classifyReplayEvent( + slowClick({ + endReason: "timeout", + timeAfterClickMs: 7000, + clickCount: 5, + }) as never, + ), + ).toBe("rage-click"); + }); + + it("stays a dead click one click below the rage threshold", () => { + expect( + classifyReplayEvent( + slowClick({ + endReason: "timeout", + timeAfterClickMs: 7000, + clickCount: 4, + }) as never, + ), + ).toBe("dead-click"); + }); + + it("is dead at exactly 7000ms and merely slow at 6999ms", () => { + // The boundary is inclusive upstream; off-by-one here would silently + // reclassify a whole category of clicks. + expect( + classifyReplayEvent( + slowClick({ endReason: "timeout", timeAfterClickMs: 7000 }) as never, + ), + ).toBe("dead-click"); + expect( + classifyReplayEvent( + slowClick({ endReason: "timeout", timeAfterClickMs: 6999 }) as never, + ), + ).toBe("slow-click"); + }); + + it("requires a timeout, not a late mutation", () => { + // endReason "mutation" means the page did respond, just slowly. + expect( + classifyReplayEvent( + slowClick({ + endReason: "mutation", + timeAfterClickMs: 9000, + clickCount: 9, + }) as never, + ), + ).toBe("slow-click"); + }); + + it("requires an interactive target", () => { + const onDiv = breadcrumb("ui.slowClickDetected", { + message: "div#panel", + data: { + node: { id: 2, tagName: "div", textContent: "", attributes: {} }, + endReason: "timeout", + timeAfterClickMs: 9000, + clickCount: 9, + }, + }); + expect(classifyReplayEvent(onDiv as never)).toBe("slow-click"); + + for (const tagName of ["a", "button", "input"]) { + const event = breadcrumb("ui.slowClickDetected", { + message: `${tagName}#target`, + data: { + node: { id: 3, tagName, textContent: "", attributes: {} }, + endReason: "timeout", + timeAfterClickMs: 7000, + }, + }); + expect(classifyReplayEvent(event as never)).toBe("dead-click"); + } + }); + + it("accepts the lowercase payload spellings", () => { + // Some SDKs lowercase payload keys; upstream reads both. + expect( + classifyReplayEvent( + slowClick({ + endReason: "timeout", + timeafterclickms: 7000, + clickcount: 5, + }) as never, + ), + ).toBe("rage-click"); + }); + + it("falls back to slow when the payload has no data at all", () => { + expect( + classifyReplayEvent(breadcrumb("ui.slowClickDetected") as never), + ).toBe("slow-click"); + }); +}); + +describe("resolveTimestampMs", () => { + it("reads span timestamps as seconds", () => { + const event = span("resource.fetch", {}, 1744027213.1); + expect(resolveTimestampMs(event as never, "resource-fetch")).toBe( + 1744027213100, + ); + }); + + it("reads breadcrumb timestamps as milliseconds", () => { + const event = breadcrumb("ui.click", {}, 1744027212400); + expect(resolveTimestampMs(event as never, "click")).toBe(1744027212400); + }); + + it("uses the event type rather than the value's magnitude", () => { + // A 1970s recording has a small ms timestamp. Magnitude-based guessing + // would read it as seconds and misplace it by decades. + const event = breadcrumb("ui.click", {}, 5000); + expect(resolveTimestampMs(event as never, "click")).toBe(5000); + expect(resolveTimestampMs(event as never, "resource-fetch")).toBe(5000000); + }); + + it("returns null when scrubbing removed the timestamp", () => { + const scrubbed = { type: 5, data: { tag: "breadcrumb" } }; + expect(resolveTimestampMs(scrubbed as never, "click")).toBeNull(); + }); +}); + +describe("noise exclusion", () => { + it("drops the event types upstream refuses to narrate", () => { + const noise = [ + { + type: 5, + timestamp: SESSION_START_MS, + data: { tag: "options", payload: {} }, + }, + span("memory"), + span("resource.script", { description: "https://cdn.example.com/a.js" }), + span("resource.img", { description: "https://cdn.example.com/a.png" }), + span("web-vital", { description: "cumulative-layout-shift" }), + breadcrumb("replay.mutations"), + breadcrumb("ui.blur"), + breadcrumb("ui.focus"), + breadcrumb("ui.multiClick", { data: { clickCount: 3 } }), + slowClick({ endReason: "mutation", timeAfterClickMs: 100 }), + { type: 3, timestamp: SESSION_START_MS, data: { source: 9 } }, + ]; + + expect(signalsFrom(noise)).toEqual([]); + }); + + it("counts successful requests but does not render them", () => { + const events = [ + span("resource.fetch", { + description: "https://example.com/api/ok", + data: { method: "GET", statusCode: 200 }, + }), + span("resource.fetch", { + description: "https://example.com/api/fail", + data: { method: "POST", statusCode: 500 }, + }), + ]; + + const signals = signalsFrom(events); + expect(signals).toHaveLength(1); + expect(signals[0].summary).toContain("500"); + + // The successful request is invisible in the signal list but real in the + // session, so the count must still see both. + const network = countReplayKinds([events] as ReplayRecordingSegments).find( + (entry) => entry.kind === "network", + ); + expect(network).toEqual({ kind: "network", total: 2, errors: 1 }); + }); + + it("reports a request that never got a response", () => { + // CORS failures arrive with no method and no statusCode. + const signals = signalsFrom([ + span("resource.fetch", { description: "https://other.example.com/x" }), + ]); + expect(signals[0].summary).toBe( + "Fetch other.example.com/x failed with no response", + ); + expect(signals[0].isError).toBe(true); + }); +}); + +describe("web versus mobile navigation", () => { + const events = [ + breadcrumb("navigation", { data: { from: "/login", to: "/checkout" } }), + span("navigation.navigate", { + description: "https://example.com/checkout", + }), + ]; + + it("prefers the navigation span on web", () => { + const signals = signalsFrom(events, { platform: "javascript" }); + expect(signals).toHaveLength(1); + expect(signals[0].type).toBe("navigation-span"); + expect(signals[0].summary).toBe("Navigated to example.com/checkout"); + }); + + it("uses the navigation breadcrumb on mobile, where no span exists", () => { + const signals = signalsFrom(events, { platform: "android" }); + expect(signals).toHaveLength(1); + expect(signals[0].type).toBe("navigation"); + expect(signals[0].summary).toBe("Navigated to /checkout"); + }); + + it("recognizes the mobile platform list", () => { + expect(isMobilePlatform("react-native")).toBe(true); + expect(isMobilePlatform("apple-ios")).toBe(true); + expect(isMobilePlatform("javascript")).toBe(false); + expect(isMobilePlatform(null)).toBe(false); + }); +}); + +describe("offsets", () => { + it("measures from the replay's started_at, not the first event", () => { + // The first recorded event here is 12s in; offsets must reflect that + // rather than resetting to zero. + const signals = signalsFrom([ + breadcrumb( + "ui.click", + { message: "button#a" }, + SESSION_START_MS + 12_000, + ), + breadcrumb( + "ui.click", + { message: "button#b" }, + SESSION_START_MS + 20_000, + ), + ]); + + expect(signals.map((signal) => signal.offsetMs)).toEqual([12_000, 20_000]); + }); + + it("yields a null offset when the session start is unknown", () => { + const signals = extractReplaySignals( + [ + [breadcrumb("ui.click", { message: "button#a" })], + ] as ReplayRecordingSegments, + { startedAt: null }, + ); + expect(signals[0].offsetMs).toBeNull(); + }); + + it("formats offsets with sub-second precision", () => { + // Failures cluster inside a single second; rounding loses the ordering. + expect(formatReplayOffset(0)).toBe("T+0.0s"); + expect(formatReplayOffset(311_800)).toBe("T+5m 11.8s"); + expect(formatReplayOffset(59_900)).toBe("T+59.9s"); + expect(formatReplayOffset(60_000)).toBe("T+1m 0.0s"); + expect(formatReplayOffset(null)).toBe("T+?"); + }); +}); + +describe("redaction labeling", () => { + it("marks an uncaptured body as not captured", () => { + // networkCaptureBodies is opt-in, so absence is the common case and + // silence would imply the content was retrievable. + const signals = signalsFrom([ + span("resource.fetch", { + description: "https://example.com/api/checkout", + data: { method: "POST", statusCode: 500 }, + }), + ]); + expect(signals[0].details).toContain(`request body: ${NOT_CAPTURED}`); + expect(signals[0].details).toContain(`response body: ${NOT_CAPTURED}`); + }); + + it("reports a known body size that was not captured", () => { + const signals = signalsFrom([ + span("resource.fetch", { + description: "https://example.com/api/checkout", + data: { + method: "POST", + statusCode: 500, + request: { size: 214, headers: {} }, + }, + }), + ]); + expect(signals[0].details).toContain( + `request body: 214 bytes ${NOT_CAPTURED}`, + ); + }); + + it("marks Relay-scrubbed values as redacted", () => { + const signals = signalsFrom([ + span("resource.fetch", { + description: "https://example.com/api/checkout", + data: { + method: "POST", + statusCode: 500, + request: { size: 8, body: "[Filtered]" }, + }, + }), + ]); + expect(signals[0].details).toContain(`request body: ${REDACTED}`); + }); + + it("renders client-masked values as delivered", () => { + // SDK masking leaves no marker, so claiming redaction would assert + // something we cannot know. + const signals = signalsFrom([ + breadcrumb("ui.click", { message: "input#card[value=****]" }), + ]); + expect(signals[0].summary).toBe("Clicked input#card[value=****]"); + expect(signals[0].summary).not.toContain(REDACTED); + }); +}); + +describe("rendering grains", () => { + const events = [ + breadcrumb( + "ui.click", + { message: "button#complete-order" }, + SESSION_START_MS + 180_600, + ), + span( + "resource.fetch", + { + description: "https://example.com/api/checkout", + data: { method: "POST", statusCode: 500, duration: 1240 }, + }, + (SESSION_START_MS + 181_000) / 1000, + ), + breadcrumb( + "console", + { level: "error", message: "TypeError: Cannot read 'id' of undefined" }, + SESSION_START_MS + 181_300, + ), + ]; + + it("renders one rollup line per kind at digest grain", () => { + expect(renderReplaySignals(signalsFrom(events), "digest")).toEqual([ + "click ×1", + "network ×1 (1 failed)", + "console ×1 (1 failed)", + ]); + }); + + it("renders one line per signal at standard grain", () => { + expect(renderReplaySignals(signalsFrom(events), "standard")).toEqual([ + "T+3m 0.6s click Clicked button#complete-order", + "T+3m 1.0s network Fetch POST example.com/api/checkout failed with 500", + "T+3m 1.3s console Console error: TypeError: Cannot read 'id' of undefined", + ]); + }); + + it("adds payload lines at detail grain", () => { + const lines = renderReplaySignals(signalsFrom(events), "detail"); + expect(lines).toContain(" duration: 1240ms"); + expect(lines).toContain(` request body: ${NOT_CAPTURED}`); + }); + + it("merges repeated signals at standard grain", () => { + const repeated = Array.from({ length: 3 }, (_, index) => + breadcrumb( + "ui.click", + { message: "button#retry" }, + SESSION_START_MS + index * 1000, + ), + ); + expect(renderReplaySignals(signalsFrom(repeated), "standard")).toEqual([ + "T+0.0s click Clicked button#retry ×3", + ]); + }); + + it("returns nothing for an empty signal list", () => { + expect(renderReplaySignals([], "standard")).toEqual([]); + expect(renderReplaySignals([], "digest")).toEqual([]); + }); +}); + +describe("against the recorded fixture", () => { + const signals = extractReplaySignals( + replayRecordingSegmentsFixture as ReplayRecordingSegments, + { startedAt: SESSION_START, platform: "javascript" }, + ); + + it("surfaces the failure the old classifier dropped", () => { + // The six-event cap previously spent its budget on session-boot noise and + // never reached any of these. + const summaries = signals.map((signal) => signal.summary); + expect(summaries).toEqual([ + "Navigated to example.com/login", + "Clicked body > div#root > form#login > button#sign-in", + "Navigated to example.com/checkout", + "Clicked body > div#root > main > button#complete-order", + "Fetch POST example.com/api/checkout failed with 500", + "Console error: TypeError: Cannot read properties of undefined (reading 'id')", + "Rage click on body > div#root > main > button#complete-order", + "Dead click — no response from body > div#root > main > a#download-receipt", + ]); + }); + + it("keeps the successful login request out of the rendered signals", () => { + expect( + signals.filter((signal) => signal.summary.includes("api/login")), + ).toEqual([]); + }); + + it("counts both requests including the successful one", () => { + expect( + countReplayKinds( + replayRecordingSegmentsFixture as ReplayRecordingSegments, + ), + ).toEqual( + expect.arrayContaining([ + { kind: "network", total: 2, errors: 1 }, + { kind: "console", total: 1, errors: 1 }, + ]), + ); + }); + + it("orders the checkout failure by its true offsets", () => { + const checkout = signals.slice(3, 6).map((signal) => ({ + offsetMs: signal.offsetMs, + kind: signal.kind, + })); + // Click, then request, then the resulting console error — all inside one + // second, which is exactly why sub-second offsets matter. + expect(checkout).toEqual([ + { offsetMs: 180_600, kind: "click" }, + { offsetMs: 181_000, kind: "network" }, + { offsetMs: 181_300, kind: "console" }, + ]); + }); +}); diff --git a/packages/mcp-core/src/internal/replay-events.ts b/packages/mcp-core/src/internal/replay-events.ts new file mode 100644 index 000000000..56b7095a2 --- /dev/null +++ b/packages/mcp-core/src/internal/replay-events.ts @@ -0,0 +1,896 @@ +/** + * Replay recording event classification and rendering. + * + * This is a port of Sentry's own replay event taxonomy so that MCP output and + * Sentry's Seer summarizer agree about what a session contains. The upstream + * sources are: + * + * - `sentry/replays/usecases/ingest/event_parser.py` — the `EventType` enum, + * the `which()` classifier, and `get_timestamp_unit()`. + * - `sentry/replays/usecases/summarize.py` — `as_log_message()`, which decides + * which events are worth narrating and which are noise. + * + * The rules that are easy to get wrong, and why they are what they are: + * + * - The SDK emits user actions as rrweb custom events (`type: 5`) tagged + * `breadcrumb`, where the meaning lives in `payload.category`. Classifying on + * `data.tag` alone labels everything `breadcrumb`. + * - Timestamp units are a function of event type, not magnitude. Spans and + * web vitals are seconds; clicks, console, and navigation are milliseconds. + * - Dead and rage clicks are not distinct categories. Both arrive as + * `ui.slowClickDetected` and are separated behaviorally. + * - Successful network requests are counted but not narrated. + * + * Parsing is deliberately forgiving. Recordings come from many SDK versions and + * pass through PII scrubbing, which can replace a numeric field with a marker + * string, so a malformed field degrades that field rather than dropping the + * event. + */ + +import type { + ReplayRecordingEvent, + ReplayRecordingPayload, + ReplayRecordingPayloadData, + ReplayRecordingSegments, +} from "../api-client"; +import { isPlainObject } from "./type-guards"; + +/** + * Replay event types, mirroring upstream's `EventType` enum. + * + * Names match upstream so the two can be diffed by eye. Upstream's deprecated + * `FCP` member is omitted. + */ +export type ReplayEventType = + | "canvas" + | "click" + | "console" + | "dead-click" + | "feedback" + | "hydration-error" + | "lcp" + | "memory" + | "mutations" + | "navigation" + | "options" + | "rage-click" + | "resource-fetch" + | "resource-image" + | "resource-script" + | "resource-xhr" + | "slow-click" + | "ui-blur" + | "ui-focus" + | "unknown" + | "cls" + | "navigation-span" + | "multi-click" + | "tap" + | "device-battery" + | "device-orientation" + | "device-connectivity" + | "scroll" + | "swipe" + | "background" + | "foreground"; + +/** + * Caller-facing grouping of event types, used by the `kinds` allow-list. + * + * Several event types collapse into one kind — `resource-fetch` and + * `resource-xhr` are both `network`, and the three device breadcrumbs are all + * `device` — because callers filter by what happened, not by which SDK API + * reported it. + */ +export type ReplaySignalKind = + | "navigation" + | "click" + | "dead-click" + | "rage-click" + | "slow-click" + | "network" + | "console" + | "hydration-error" + | "feedback" + | "web-vital" + | "tap" + | "scroll" + | "swipe" + | "app-lifecycle" + | "device"; + +export type ReplayGrain = "digest" | "standard" | "detail"; + +/** A classified, renderable event from a recording. */ +export interface ReplaySignal { + type: ReplayEventType; + kind: ReplaySignalKind; + /** Absolute event time in epoch milliseconds, or null when unresolvable. */ + timestampMs: number | null; + /** Offset from the replay's `started_at`, in milliseconds. */ + offsetMs: number | null; + /** One-line description of what happened. */ + summary: string; + /** Extra lines shown only at `detail` grain. */ + details: string[]; + /** True when this signal indicates a failure (error log, failed request). */ + isError: boolean; +} + +/** Rendered when the SDK never captured a value. */ +export const NOT_CAPTURED = ""; + +/** Rendered when Relay scrubbed a value. */ +export const REDACTED = ""; + +/** + * Relay substitutes this literal for scrubbed values. It is the only + * server-side redaction marker we can detect; client-side SDK masking leaves + * no marker at all. + */ +const RELAY_FILTERED_MARKER = "[Filtered]"; + +/** Upstream truncates console messages and resource URLs at this length. */ +const TRUNCATION_LENGTH = 200; + +/** A slow click is dead only if the action stalled for at least this long. */ +const DEAD_CLICK_THRESHOLD_MS = 7000; + +/** A dead click is promoted to a rage click at this many clicks. */ +const RAGE_CLICK_THRESHOLD = 5; + +/** Only clicks on these elements can be dead — a stalled div is not a defect. */ +const INTERACTIVE_TAG_NAMES = new Set(["a", "button", "input"]); + +/** + * Platforms whose replays are mobile, mirroring `MOBILE` in + * `sentry/utils/platform_categories.py`. + * + * This matters for one rule: web replays prefer the navigation *span* and drop + * the navigation breadcrumb, because the span is unavailable on mobile. + */ +const MOBILE_PLATFORMS = new Set([ + "android", + "apple-ios", + "cordova", + "capacitor", + "javascript-cordova", + "javascript-capacitor", + "ionic", + "react-native", + "flutter", + "dart-flutter", + "unity", + "dotnet-maui", + "dotnet-xamarin", + "unreal", + "java-android", + "cocoa-objc", + "cocoa-swift", +]); + +/** Event types whose outer `timestamp` is in seconds; everything else is ms. */ +const SECOND_TIMESTAMP_TYPES = new Set([ + "cls", + "lcp", + "memory", + "mutations", + "navigation-span", + "resource-fetch", + "resource-image", + "resource-script", + "resource-xhr", + "ui-blur", + "ui-focus", +]); + +/** Breadcrumb `payload.category` values, mapped to their event type. */ +const CATEGORY_TO_TYPE: Record = { + "ui.click": "click", + "ui.multiClick": "multi-click", + navigation: "navigation", + console: "console", + "ui.blur": "ui-blur", + "ui.focus": "ui-focus", + "replay.hydrate-error": "hydration-error", + "replay.mutations": "mutations", + "sentry.feedback": "feedback", + "ui.tap": "tap", + "device.battery": "device-battery", + "device.orientation": "device-orientation", + "device.connectivity": "device-connectivity", + "ui.scroll": "scroll", + "ui.swipe": "swipe", + "app.background": "background", + "app.foreground": "foreground", +}; + +/** Span `payload.op` values, mapped to their event type. */ +const OP_TO_TYPE: Record = { + "resource.fetch": "resource-fetch", + "resource.xhr": "resource-xhr", + "resource.script": "resource-script", + "resource.img": "resource-image", + memory: "memory", +}; + +const TYPE_TO_KIND: Partial> = { + click: "click", + "dead-click": "dead-click", + "rage-click": "rage-click", + "slow-click": "slow-click", + navigation: "navigation", + "navigation-span": "navigation", + "resource-fetch": "network", + "resource-xhr": "network", + console: "console", + "hydration-error": "hydration-error", + feedback: "feedback", + lcp: "web-vital", + cls: "web-vital", + tap: "tap", + scroll: "scroll", + swipe: "swipe", + background: "app-lifecycle", + foreground: "app-lifecycle", + "device-battery": "device", + "device-orientation": "device", + "device-connectivity": "device", +}; + +export function isMobilePlatform(platform?: string | null): boolean { + return platform ? MOBILE_PLATFORMS.has(platform) : false; +} + +/** + * Identify a replay recording event. + * + * Mirrors upstream's `which()`. Anything unrecognized is `unknown` rather than + * guessed at, so new SDK event types are ignored instead of mislabeled. + */ +export function classifyReplayEvent( + event: ReplayRecordingEvent, +): ReplayEventType { + // rrweb incremental snapshot; source 9 is a canvas mutation. + if (event.type === 3) { + const source = isPlainObject(event.data) ? event.data.source : undefined; + return source === 9 ? "canvas" : "unknown"; + } + + if (event.type !== 5) { + return "unknown"; + } + + const tag = event.data?.tag; + const payload = event.data?.payload; + + if (tag === "options") { + return "options"; + } + + if (tag === "breadcrumb") { + const category = payload?.category; + if (!category) { + return "unknown"; + } + if (category === "ui.slowClickDetected") { + return classifySlowClick(payload?.data); + } + return CATEGORY_TO_TYPE[category] ?? "unknown"; + } + + if (tag === "performanceSpan") { + const op = payload?.op; + if (!op) { + return "unknown"; + } + // Upstream matches any `navigation*` op, covering navigate, reload, + // back_forward, and push. + if (op.startsWith("navigation")) { + return "navigation-span"; + } + if (op === "web-vital") { + if (payload?.description === "largest-contentful-paint") return "lcp"; + if (payload?.description === "cumulative-layout-shift") return "cls"; + return "unknown"; + } + return OP_TO_TYPE[op] ?? "unknown"; + } + + return "unknown"; +} + +/** + * Separate a slow click into slow, dead, or rage. + * + * A click is dead when the action it triggered never completed: the SDK timed + * out (rather than observing a late mutation), the target was interactive, and + * the stall lasted at least 7 seconds. Repeating the click at least five times + * makes it a rage click. Anything weaker is a plain slow click, which upstream + * does not narrate. + */ +function classifySlowClick(data: unknown): ReplayEventType { + if (!isPlainObject(data)) { + return "slow-click"; + } + + const node = isPlainObject(data.node) ? data.node : null; + const tagName = + typeof node?.tagName === "string" ? node.tagName.toLowerCase() : null; + + // Upstream reads both spellings; some SDKs lowercase payload keys. + const timeAfterClickMs = + numberOrZero(data.timeAfterClickMs) || numberOrZero(data.timeafterclickms); + + const isDead = + data.endReason === "timeout" && + tagName !== null && + INTERACTIVE_TAG_NAMES.has(tagName) && + timeAfterClickMs >= DEAD_CLICK_THRESHOLD_MS; + + if (!isDead) { + return "slow-click"; + } + + const clickCount = + numberOrZero(data.clickCount) || numberOrZero(data.clickcount); + return clickCount >= RAGE_CLICK_THRESHOLD ? "rage-click" : "dead-click"; +} + +/** + * Resolve an event's absolute time in epoch milliseconds. + * + * The unit comes from the event type. Guessing from magnitude happens to work + * for present-day epochs but is not a rule, and would silently misplace events + * for any recording far from now. + */ +export function resolveTimestampMs( + event: ReplayRecordingEvent, + type: ReplayEventType, +): number | null { + const timestamp = event.timestamp; + if (typeof timestamp !== "number" || !Number.isFinite(timestamp)) { + return null; + } + return SECOND_TIMESTAMP_TYPES.has(type) ? timestamp * 1000 : timestamp; +} + +/** + * Convert a recording into classified signals. + * + * Noise types are dropped, successful network requests are dropped from the + * rendered set (callers count them separately), and navigation breadcrumbs are + * dropped for web replays in favor of the navigation span. + * + * Offsets are measured from `startedAt` — the replay's own start — rather than + * from the first recorded event, so they line up with the replay metadata + * timeline and with error timestamps resolved elsewhere. + */ +export function extractReplaySignals( + segments: ReplayRecordingSegments | null, + { + startedAt, + platform, + }: { startedAt?: string | null; platform?: string | null } = {}, +): ReplaySignal[] { + if (!segments) { + return []; + } + + const isMobile = isMobilePlatform(platform); + const startMs = startedAt ? Date.parse(startedAt) : Number.NaN; + const originMs = Number.isNaN(startMs) ? null : startMs; + const signals: ReplaySignal[] = []; + + for (const segment of segments) { + for (const event of segment) { + const type = classifyReplayEvent(event); + const kind = TYPE_TO_KIND[type]; + const summarized = kind ? summarizeEvent(event, type, isMobile) : null; + if (!summarized || !kind) { + continue; + } + + const timestampMs = resolveTimestampMs(event, type); + signals.push({ + type, + kind, + timestampMs, + offsetMs: + timestampMs !== null && originMs !== null + ? timestampMs - originMs + : null, + ...summarized, + }); + } + } + + return signals; +} + +/** + * Count every classified event by kind, including ones that are not rendered. + * + * Successful requests are invisible in the signal list but real in the session, + * so `network 58 (2 failed)` describes 58 requests of which 2 were narrated. + */ +export interface ReplayKindCount { + kind: ReplaySignalKind; + total: number; + errors: number; +} + +export function countReplayKinds( + segments: ReplayRecordingSegments | null, +): ReplayKindCount[] { + if (!segments) { + return []; + } + + const counts = new Map(); + + for (const segment of segments) { + for (const event of segment) { + const type = classifyReplayEvent(event); + const kind = TYPE_TO_KIND[type]; + if (!kind) { + continue; + } + + const entry = counts.get(kind) ?? { kind, total: 0, errors: 0 }; + entry.total += 1; + if (isErrorEvent(event, type)) { + entry.errors += 1; + } + counts.set(kind, entry); + } + } + + return [...counts.values()]; +} + +/** + * Format a millisecond offset as `T+1m 23.4s`. + * + * Sub-second precision is kept because replay failures cluster: a click, its + * request, and the resulting console error can land inside the same second, + * and rounding them together loses the ordering that explains the failure. + */ +export function formatReplayOffset(offsetMs: number | null): string { + if (offsetMs === null) { + return "T+?"; + } + + const totalSeconds = Math.max(0, offsetMs) / 1000; + if (totalSeconds < 60) { + return `T+${totalSeconds.toFixed(1)}s`; + } + + const minutes = Math.floor(totalSeconds / 60); + const seconds = totalSeconds - minutes * 60; + return `T+${minutes}m ${seconds.toFixed(1)}s`; +} + +/** + * Render signals at the requested grain. + * + * `digest` answers "what kind of session was this" in a handful of lines, + * `standard` lists what happened, and `detail` adds the payload needed to act + * on a specific failure. Grain controls rendering only — filtering is the + * caller's job, so a digest of a filtered window stays consistent with the + * standard rendering of the same window. + */ +export function renderReplaySignals( + signals: ReplaySignal[], + grain: ReplayGrain = "standard", +): string[] { + if (signals.length === 0) { + return []; + } + + if (grain === "digest") { + return renderDigest(signals); + } + + const lines: string[] = []; + for (const [index, signal] of signals.entries()) { + const repeats = grain === "standard" ? countRepeats(signals, index) : 0; + if (repeats < 0) { + continue; + } + + const suffix = repeats > 1 ? ` ×${repeats}` : ""; + lines.push( + `${formatReplayOffset(signal.offsetMs)} ${signal.kind} ${signal.summary}${suffix}`, + ); + + if (grain === "detail") { + for (const detail of signal.details) { + lines.push(` ${detail}`); + } + } + } + + return lines; +} + +/** + * One rollup line per kind, in first-appearance order. + */ +function renderDigest(signals: ReplaySignal[]): string[] { + const counts = new Map(); + + for (const signal of signals) { + const entry = counts.get(signal.kind) ?? { total: 0, errors: 0 }; + entry.total += 1; + if (signal.isError) { + entry.errors += 1; + } + counts.set(signal.kind, entry); + } + + return [...counts.entries()].map(([kind, { total, errors }]) => + errors > 0 ? `${kind} ×${total} (${errors} failed)` : `${kind} ×${total}`, + ); +} + +/** + * Collapse a run of identical consecutive signals. + * + * Returns the run length at its first element, and -1 for the rest so callers + * skip them. A user clicking the same dead button nine times is one fact, not + * nine lines. + */ +function countRepeats(signals: ReplaySignal[], index: number): number { + const signal = signals[index]; + const previous = signals[index - 1]; + if (previous && isSameSignal(previous, signal)) { + return -1; + } + + let count = 1; + while ( + index + count < signals.length && + isSameSignal(signals[index + count], signal) + ) { + count += 1; + } + return count; +} + +function isSameSignal(a: ReplaySignal, b: ReplaySignal): boolean { + return a.kind === b.kind && a.summary === b.summary; +} + +function isErrorEvent( + event: ReplayRecordingEvent, + type: ReplayEventType, +): boolean { + if (type === "console") { + return event.data?.payload?.level === "error"; + } + if (type === "resource-fetch" || type === "resource-xhr") { + const status = event.data?.payload?.data?.statusCode; + return typeof status === "number" && status >= 400; + } + return type === "hydration-error"; +} + +type SummarizedEvent = Pick; + +/** + * Describe an event, or return null when it should not be rendered. + * + * The exclusion list mirrors upstream's `as_log_message` returning `None`: + * options, memory, mutations, canvas, script and image resources, blur, focus, + * CLS, plain slow clicks, and multi-clicks are all noise. + */ +function summarizeEvent( + event: ReplayRecordingEvent, + type: ReplayEventType, + isMobile: boolean, +): SummarizedEvent | null { + const payload = event.data?.payload; + const data = payload?.data; + + switch (type) { + case "click": + return describeClick(payload?.message, data, "Clicked"); + case "dead-click": + return describeClick( + payload?.message, + data, + "Dead click — no response from", + ); + case "rage-click": + return describeClick(payload?.message, data, "Rage click on"); + + case "navigation-span": + // Web prefers the span; mobile has no span to prefer. + if (isMobile) return null; + return { + summary: `Navigated to ${formatUrl(payload?.description)}`, + details: [], + isError: false, + }; + + case "navigation": + // Mirror image of the rule above: web drops the breadcrumb. + if (!isMobile) return null; + return { + summary: data?.to ? `Navigated to ${formatUrl(data.to)}` : "Navigated", + details: [], + isError: false, + }; + + case "console": { + const level = payload?.level ?? "log"; + const message = truncate(redactable(payload?.message) ?? ""); + return { + summary: `Console ${level}: ${message}`, + details: [], + isError: level === "error", + }; + } + + case "resource-fetch": + case "resource-xhr": + return describeNetworkRequest(payload, type); + + case "hydration-error": + return { + summary: "Hydration error on the page", + details: data?.url ? [`url: ${formatUrl(data.url)}`] : [], + isError: true, + }; + + case "feedback": + return { + summary: "User submitted feedback", + details: [], + isError: false, + }; + + case "lcp": { + const size = data?.size; + const rating = data?.rating; + return { + summary: + size != null && rating != null + ? `Largest contentful paint: ${size}ms (${rating})` + : "Largest contentful paint", + details: [], + isError: false, + }; + } + + case "tap": { + const message = redactable(payload?.message); + // Upstream drops taps with no target. + if (!message) return null; + return { summary: `Tapped ${message}`, details: [], isError: false }; + } + + case "scroll": + case "swipe": { + const verb = type === "scroll" ? "Scrolled" : "Swiped"; + const target = [data?.["view.id"], data?.direction] + .filter(Boolean) + .join(" "); + return { + summary: target ? `${verb} ${target}` : verb, + details: [], + isError: false, + }; + } + + case "background": + return { + summary: "App moved to background", + details: [], + isError: false, + }; + case "foreground": + return { + summary: "App moved to foreground", + details: [], + isError: false, + }; + + case "device-battery": { + const level = data?.level; + const charging = data?.charging; + return { + summary: + level != null && charging != null + ? `Battery ${level}%, ${charging ? "charging" : "not charging"}` + : "Battery status changed", + details: [], + isError: false, + }; + } + case "device-orientation": + return { + summary: data?.position + ? `Orientation changed to ${data.position}` + : "Orientation changed", + details: [], + isError: false, + }; + case "device-connectivity": + return { + summary: data?.state + ? `Connectivity changed to ${data.state}` + : "Connectivity changed", + details: [], + isError: false, + }; + + // Noise. Upstream returns None for all of these. + default: + return null; + } +} + +function describeClick( + message: string | undefined, + data: ReplayRecordingPayloadData | undefined, + verb: string, +): SummarizedEvent { + const target = redactable(message) ?? describeNode(data) ?? "element"; + const details: string[] = []; + + const text = + typeof data?.node?.textContent === "string" + ? data.node.textContent.trim() + : ""; + if (text) { + details.push(`text: ${truncate(text)}`); + } + + const timeAfterClickMs = + numberOrZero(data?.timeAfterClickMs) || + numberOrZero(data?.timeafterclickms); + if (timeAfterClickMs > 0) { + details.push(`stalled: ${timeAfterClickMs}ms (${data?.endReason})`); + } + + const clickCount = + numberOrZero(data?.clickCount) || numberOrZero(data?.clickcount); + if (clickCount > 1) { + details.push(`clicks: ${clickCount}`); + } + + return { + summary: `${verb} ${target}`, + details, + // A dead or rage click is a failure of the page, not of the request. + isError: verb !== "Clicked", + }; +} + +/** + * Describe a network request. + * + * Successful requests return null: upstream narrates only failures, and callers + * that need the full count use `countReplayKinds`. A request with no status at + * all never got a response, which is itself worth reporting. + */ +function describeNetworkRequest( + payload: ReplayRecordingPayload | undefined, + type: ReplayEventType, +): SummarizedEvent | null { + const data = payload?.data; + const statusCode = data?.statusCode; + + if (typeof statusCode === "number" && statusCode >= 200 && statusCode < 300) { + return null; + } + + const label = type === "resource-fetch" ? "Fetch" : "XHR"; + const method = data?.method; + const url = formatUrl(payload?.description); + const request = method ? `${method} ${url}` : url; + const status = statusCode != null ? String(statusCode) : "no response"; + + const details: string[] = []; + if (typeof data?.duration === "number") { + details.push(`duration: ${data.duration}ms`); + } + details.push( + `request body: ${describeBody(data?.request, data?.requestBodySize)}`, + ); + details.push( + `response body: ${describeBody(data?.response, data?.responseBodySize)}`, + ); + + return { + summary: `${label} ${request} failed with ${status}`, + details, + isError: true, + }; +} + +/** + * Describe a captured request or response body. + * + * Absence is reported rather than passed over in silence: `networkCaptureBodies` + * is opt-in, so a missing body usually means the SDK was never asked to record + * one, and saying nothing would imply we could have retrieved it. + */ +function describeBody( + body: { size?: number; body?: unknown } | undefined, + legacySize: number | undefined, +): string { + const value = body?.body; + if (typeof value === "string") { + return value === RELAY_FILTERED_MARKER ? REDACTED : truncate(value); + } + + const size = body?.size ?? legacySize; + if (typeof size === "number") { + return `${size} bytes ${NOT_CAPTURED}`; + } + return NOT_CAPTURED; +} + +function describeNode(data: unknown): string | null { + if (!isPlainObject(data) || !isPlainObject(data.node)) { + return null; + } + const tagName = + typeof data.node.tagName === "string" ? data.node.tagName : null; + if (!tagName) { + return null; + } + const attributes = isPlainObject(data.node.attributes) + ? data.node.attributes + : {}; + const id = typeof attributes.id === "string" ? `#${attributes.id}` : ""; + return `${tagName.toLowerCase()}${id}`; +} + +/** + * Report a value as redacted only when Relay says so. + * + * Client-side SDK masking leaves no marker — a masked string is + * indistinguishable from a real one — so masked values are rendered as + * delivered. Guessing would be worse than silence: a wrong `` tells + * the reader content exists behind a mask when it may be the recorded value. + */ +function redactable(value: string | undefined): string | null { + if (typeof value !== "string" || !value.trim()) { + return null; + } + return value === RELAY_FILTERED_MARKER ? REDACTED : value.trim(); +} + +/** + * Shorten a URL to `host/path?query`, matching upstream's `_parse_url`. + */ +function formatUrl(value: string | undefined): string { + if (typeof value !== "string" || !value) { + return "unknown"; + } + if (value === RELAY_FILTERED_MARKER) { + return REDACTED; + } + + try { + const url = new URL(value); + const path = url.pathname.replace(/^\//, ""); + return `${url.host}/${path}${url.search}`; + } catch { + return truncate(value); + } +} + +function truncate(value: string): string { + return value.length > TRUNCATION_LENGTH + ? `${value.slice(0, TRUNCATION_LENGTH)} [truncated]` + : value; +} + +function numberOrZero(value: unknown): number { + return typeof value === "number" && Number.isFinite(value) ? value : 0; +} From 09218a56dbade8f11330d5ad5797ea1c497549c5 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 14:00:46 -0700 Subject: [PATCH 04/26] feat(replays): Page recording segments and add replay API methods MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The recording segments index was requested with `?download=true`, a parameter that does not exist on that endpoint, while the `Link` header's cursor was ignored. Sentry serves at most 100 segments per page — `per_page` cannot raise that, since its default and maximum are both 100 — so any replay longer than 100 segments was silently truncated to its first page. Follow the cursor, and bound the read twice over. Segments are capped at 150, matching the clamp Sentry's own summarize endpoint applies, and raw JSON at 10MB measured before parsing. The byte ceiling is the real guard: segment sizes vary by orders of magnitude, parsed rrweb objects expand well beyond their serialized size, and mcp-cloudflare runs under a 128MB Workers limit. Both bounds are provisional pending QA against real replays. The caller is told which bound stopped the read, and `get_replay_details` now says so rather than presenting a partial recording as a whole session. Add two endpoints the map will need: - `getReplayErrorEvents` resolves a replay's error ids through `replays-events-meta` in one batched call. Error ids are event ids, and resolving them through `listIssues` returns issues, which carry no event timestamp; this returns issue identity and a millisecond-precision timestamp together, so it also replaces the per-error issue lookups. The endpoint is PRIVATE, so its response is parsed defensively. - `getReplaySummary` reads Seer's replay summary state. Deliberately one GET: no start request, no retry loop. Seer's start route enqueues background work and returns an empty body, so starting would spend an LLM run per call and still report `processing` on the immediate read. Send an explicit `field` allow-list on the replays index, drawn from `VALID_FIELD_SET`. Omitting it returns Sentry's default column set, which is wider than anything rendered. `replay_type` and `ota_updates` appear in responses but are absent from that set and would 400, so they are excluded. Refs docs/specs/replay-review.md Co-Authored-By: Claude Opus 5 (1M context) --- .../changes/improve-replay-review/tasks.md | 12 +- .../mcp-core/src/api-client/client.test.ts | 17 +- packages/mcp-core/src/api-client/client.ts | 241 ++++++++++++++- .../src/api-client/replay-client.test.ts | 284 ++++++++++++++++++ packages/mcp-core/src/api-client/schema.ts | 70 +++++ packages/mcp-core/src/api-client/types.ts | 4 + .../src/tools/catalog/get-replay-details.ts | 71 +++-- 7 files changed, 662 insertions(+), 37 deletions(-) create mode 100644 packages/mcp-core/src/api-client/replay-client.test.ts diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index 55d62f240..5ec87b2b9 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -22,12 +22,12 @@ baselines encode the same wrong event shape they do today. ## 3. API Client -- [ ] 3.1 Re-verify the recording segments index contract against `getsentry/sentry`. -- [ ] 3.2 Follow the `Link` header's `cursor` in `getReplayRecordingSegments` and remove the non-existent `download=true` parameter. Note `per_page` is a no-op (default and maximum are both 100), and that reading the header requires the raw-response request path rather than `requestJSON`. -- [ ] 3.3 Page through segments up to 150 segments or 10MB of raw segment JSON, whichever is hit first, and report which bound stopped the read. -- [ ] 3.4 Add a single-read replay summary method for the experimental summarize endpoint, parsing the known Seer response shape (`data.time_ranges[]`, `status`) defensively. No start method and no retry loop: one `GET`, then move on. -- [ ] 3.5 Add an explicit `field` allow-list to `searchReplays`, drawn from `VALID_FIELD_SET` in `sentry/replays/validators.py`. Use upstream's coarse names (`browser`, `user`, `device`, `sdk`, `releases`, `trace_ids`) and exclude `replay_type` and `ota_updates`, which return 400. -- [ ] 3.6 Add a `getReplayErrorEvents` method wrapping `GET /organizations/{org}/replays-events-meta/` with a batched `query=id:[...]`, returning `id`, `issue`, `issue.id`, `title`, and `timestamp`. Parse `timestamp` as a millisecond-precision ISO string; the endpoint deletes `timestamp_ms` from its own output. +- [x] 3.1 Re-verify the recording segments index contract against `getsentry/sentry`. Confirmed: `GenericOffsetPaginator` emits `0::0` cursors, `PAGINATION_DEFAULT_PER_PAGE` and `max_per_page` are both 100, and the index has no `download` parameter. +- [x] 3.2 Follow the `Link` header's `cursor` in `getReplayRecordingSegments` and remove the non-existent `download=true` parameter. Note `per_page` is a no-op (default and maximum are both 100), and that reading the header requires the raw-response request path rather than `requestJSON`. +- [x] 3.3 Page through segments up to 150 segments or 10MB of raw segment JSON, whichever is hit first, and report which bound stopped the read. `get_replay_details` now prints the bound that stopped it; full truncation reporting lands with the map in section 4. +- [x] 3.4 Add a single-read replay summary method for the experimental summarize endpoint, parsing the known Seer response shape (`data.time_ranges[]`, `status`) defensively. No start method and no retry loop: one `GET`, then move on. +- [x] 3.5 Add an explicit `field` allow-list to `searchReplays`, drawn from `VALID_FIELD_SET` in `sentry/replays/validators.py`. Use upstream's coarse names (`browser`, `user`, `device`, `sdk`, `releases`, `trace_ids`) and exclude `replay_type` and `ota_updates`, which return 400. +- [x] 3.6 Add a `getReplayErrorEvents` method wrapping `GET /organizations/{org}/replays-events-meta/` with a batched `query=id:[...]`, returning `id`, `issue`, `issue.id`, `title`, and `timestamp`. Parse `timestamp` as a millisecond-precision ISO string; the endpoint deletes `timestamp_ms` from its own output. ## 4. Replay Details as a Map diff --git a/packages/mcp-core/src/api-client/client.test.ts b/packages/mcp-core/src/api-client/client.test.ts index dc8d3dfe1..41bece1eb 100644 --- a/packages/mcp-core/src/api-client/client.test.ts +++ b/packages/mcp-core/src/api-client/client.test.ts @@ -1451,10 +1451,21 @@ describe("API query builders", () => { expect(globalThis.fetch).toHaveBeenCalledWith( expect.stringContaining( - "/api/0/organizations/test-org/replays/?query=count_errors%3A%3E0&per_page=25&sort=-count_errors&environment=production&environment=staging&statsPeriod=24h", + "/api/0/organizations/test-org/replays/?query=count_errors%3A%3E0&per_page=25&sort=-count_errors&environment=production&environment=staging", ), expect.any(Object), ); + + const requestUrl = vi.mocked(globalThis.fetch).mock.calls[0][0] as string; + expect(requestUrl).toContain("statsPeriod=24h"); + // Without an explicit allow-list Sentry returns its own default column + // set, which is wider than anything rendered. + expect(requestUrl).toContain("field=id"); + expect(requestUrl).toContain("field=browser"); + // Present in responses but absent from VALID_FIELD_SET, so requesting + // them would 400. + expect(requestUrl).not.toContain("field=replay_type"); + expect(requestUrl).not.toContain("field=ota_updates"); }); }); @@ -2261,7 +2272,9 @@ describe("API query builders", () => { get: (key: string) => key === "content-type" ? "application/json" : null, }, - json: () => Promise.resolve([["segment-1"]]), + // Segments are read as text so the byte budget measures the bytes + // that crossed the wire rather than the parsed object graph. + text: () => Promise.resolve(JSON.stringify([[]])), }); }); diff --git a/packages/mcp-core/src/api-client/client.ts b/packages/mcp-core/src/api-client/client.ts index 2642b50d5..e6f5293ee 100644 --- a/packages/mcp-core/src/api-client/client.ts +++ b/packages/mcp-core/src/api-client/client.ts @@ -73,9 +73,11 @@ import { ReleaseDetailsSchema, ReleaseListSchema, ReplayDetailsSchema, + ReplayErrorEventsResponseSchema, ReplayIdsByResourceSchema, ReplayListResponseSchema, ReplayRecordingSegmentsSchema, + ReplaySummarySchema, RepositoryListSchema, SpansSearchResponseSchema, StacktraceLinkSchema, @@ -132,7 +134,9 @@ import type { ReleaseDetails, ReleaseList, ReplayDetails, + ReplayErrorEvent, ReplayList, + ReplaySummary, ReplayRecordingSegments, StacktraceLink, TagList, @@ -257,6 +261,74 @@ type RequestOptions = { allowStatuses?: number[]; }; +/** + * Segment ceiling for a single recording read. + * + * Matches the clamp Sentry's own summarize endpoint applies + * (`MAX_SEGMENTS_TO_SUMMARIZE`), so MCP never reads more of a recording than + * Sentry's summarizer does. + */ +export const MAX_REPLAY_SEGMENTS = 150; + +/** + * Byte ceiling for a single recording read, measured on raw JSON before + * parsing. + * + * Segment count is a poor proxy for memory — sizes vary by orders of magnitude + * — so this is the binding guard. The headroom below the Workers 128MB limit is + * deliberate: parsed rrweb objects expand several times over their serialized + * size, and the budget has to survive that expansion plus rendering. + * + * Provisional. QA against real replays should confirm or move it. + */ +export const MAX_REPLAY_SEGMENT_BYTES = 10 * 1024 * 1024; + +/** + * Fields requested from the replays index. + * + * Without an explicit allow-list Sentry returns its default column set, which + * is wider than anything rendered. Every name here must appear in + * `VALID_FIELD_SET` in `sentry/replays/validators.py` — invalid fields are + * rejected outright with a 400 — and those names are coarser than the ones + * discovery advertises (`browser`, not `browser.name`). + * + * `replay_type` and `ota_updates` appear in responses but are absent from + * `VALID_FIELD_SET`, so requesting them 400s. They are deliberately omitted. + */ +export const REPLAY_INDEX_FIELDS = [ + "id", + "project_id", + "started_at", + "finished_at", + "duration", + "environment", + "is_archived", + "count_errors", + "count_dead_clicks", + "count_rage_clicks", + "urls", + "browser", + "os", + "device", + "sdk", + "user", + "releases", + "trace_ids", +] as const; + +/** + * A recording read, plus whether it stopped early and why. + * + * `truncatedBy` is not an error: a partial read is useful, but callers must be + * able to say so rather than presenting it as the whole session. + */ +export type ReplayRecordingSegmentsResult = { + segments: ReplayRecordingSegments; + truncatedBy: "segments" | "bytes" | null; + segmentsRead: number; + bytesRead: number; +}; + export type TraceItemType = "spans" | "logs" | "tracemetrics"; type ClientKeyRateLimit = { @@ -3218,7 +3290,9 @@ export class SentryApiService { statsPeriod, start, end, - fields, + // Ask for exactly what is rendered. Omitting `field` entirely makes + // Sentry return its default column set instead. + fields = [...REPLAY_INDEX_FIELDS], }: { organizationSlug: string; query?: string; @@ -3983,7 +4057,164 @@ export class SentryApiService { return replayIdsByResource[normalizedIssueId] ?? []; } + /** + * Downloads a replay's recording segments. + * + * Sentry paginates this endpoint at 100 segments per page and reports + * continuation through the `Link` header. `per_page` cannot raise that — its + * default and maximum are both 100 — so following the header's cursor is the + * only way to read past segment 100. There is no `download` parameter on this + * endpoint; the index always downloads bodies. + * + * Reading is bounded twice over, because a long session can be arbitrarily + * large and `mcp-cloudflare` runs under a 128MB Workers ceiling: + * + * - {@link MAX_REPLAY_SEGMENTS} segments, matching the clamp Sentry's own + * summarize endpoint applies, so we read no more of a recording than + * Sentry's summarizer does. + * - {@link MAX_REPLAY_SEGMENT_BYTES} of raw JSON, measured before parsing. + * This is the real guard: segment sizes vary by orders of magnitude, and + * parsed rrweb objects expand well beyond their serialized size. + * + * The caller is told which bound stopped the read so it can report the gap + * rather than presenting a partial recording as complete. + */ async getReplayRecordingSegments( + { + organizationSlug, + projectSlugOrId, + replayId, + maxSegments = MAX_REPLAY_SEGMENTS, + maxBytes = MAX_REPLAY_SEGMENT_BYTES, + }: { + organizationSlug: string; + projectSlugOrId: string; + replayId: string; + maxSegments?: number; + maxBytes?: number; + }, + opts?: RequestOptions, + ): Promise { + const segments: ReplayRecordingSegments = []; + let cursor: string | null = null; + let bytesRead = 0; + let truncatedBy: ReplayRecordingSegmentsResult["truncatedBy"] = null; + + do { + const params = new URLSearchParams(); + if (cursor) { + params.set("cursor", cursor); + } + + const path = apiPath`/projects/${organizationSlug}/${projectSlugOrId}/replays/${replayId}/recording-segments/`; + const response = await this.request( + params.toString() ? `${path}?${params.toString()}` : path, + { method: "GET" }, + opts, + ); + + // Read as text so the byte budget measures what actually crossed the + // wire, before parsing expands it into an object graph. + const text = await response.text(); + bytesRead += text.length; + + let body: unknown; + try { + body = JSON.parse(text); + } catch (error) { + throw new Error( + `Failed to parse replay recording segments: ${error instanceof Error ? error.message : String(error)}`, + ); + } + + const page = ReplayRecordingSegmentsSchema.parse(body); + for (const segment of page) { + if (segments.length >= maxSegments) { + truncatedBy = "segments"; + break; + } + segments.push(segment); + } + + if (truncatedBy) { + break; + } + if (bytesRead >= maxBytes) { + // Keep what this page delivered — it is already parsed — but stop + // before requesting another. + truncatedBy = "bytes"; + break; + } + + cursor = getNextCursor(response.headers.get("link")); + } while (cursor); + + return { segments, truncatedBy, segmentsRead: segments.length, bytesRead }; + } + + /** + * Resolves a replay's error event IDs to issue identity and event timestamps. + * + * `replay.error_ids` are event IDs, and resolving them through `listIssues` + * returns issues, which carry no event timestamp. This endpoint resolves the + * whole batch in one call and returns both, so it replaces the per-error + * issue lookups as well as supplying the millisecond-resolution timestamps a + * suggested zoom window needs. + * + * The endpoint is `ApiPublishStatus.PRIVATE` and may change without notice, + * so callers should degrade rather than fail when it errors. + * + * @returns Error events for the requested IDs; empty when none are given + */ + async getReplayErrorEvents( + { + organizationSlug, + errorIds, + projectId, + statsPeriod = "90d", + }: { + organizationSlug: string; + errorIds: string[]; + projectId?: string; + statsPeriod?: string; + }, + opts?: RequestOptions, + ): Promise { + if (errorIds.length === 0) { + return []; + } + + const params = new URLSearchParams(); + // One batched lookup rather than a request per error. + params.set("query", `id:[${errorIds.join(",")}]`); + params.set("statsPeriod", statsPeriod); + params.append("project", projectId ?? "-1"); + + const body = await this.requestJSON( + apiPath`/organizations/${organizationSlug}/replays-events-meta/` + + `?${params.toString()}`, + undefined, + opts, + ); + + return ReplayErrorEventsResponseSchema.parse(body).data; + } + + /** + * Reads the state of a replay's AI summary. + * + * Deliberately a single read. It does not POST to start a summary task and + * does not poll: Seer's start route enqueues background work and returns an + * empty body, so starting would spend an LLM run per call and still report + * `processing` on the immediate read, while polling would put unbounded + * latency on a section that is optional by construction. + * + * The endpoint is `ApiPublishStatus.EXPERIMENTAL` and triple-gated on the + * `session-replay` feature, the `replay-ai-summaries` feature, and Seer + * access, returning 403 otherwise. Callers must treat every non-`completed` + * outcome as "not available right now" and omit the section. + */ + async getReplaySummary( { organizationSlug, projectSlugOrId, @@ -3994,14 +4225,14 @@ export class SentryApiService { replayId: string; }, opts?: RequestOptions, - ): Promise { + ): Promise { const body = await this.requestJSON( - apiPath`/projects/${organizationSlug}/${projectSlugOrId}/replays/${replayId}/recording-segments/` + - `?download=true`, + apiPath`/projects/${organizationSlug}/${projectSlugOrId}/replays/${replayId}/summarize/`, undefined, opts, ); - return ReplayRecordingSegmentsSchema.parse(body); + + return ReplaySummarySchema.parse(body); } async updateIssue( diff --git a/packages/mcp-core/src/api-client/replay-client.test.ts b/packages/mcp-core/src/api-client/replay-client.test.ts new file mode 100644 index 000000000..2a3a5cf13 --- /dev/null +++ b/packages/mcp-core/src/api-client/replay-client.test.ts @@ -0,0 +1,284 @@ +/** + * Tests for the replay-specific API client methods. + * + * These cover the contracts that are easy to get wrong and expensive to get + * wrong silently: segment pagination (whose absence truncated long recordings + * without saying so), the read budgets, and the two endpoints whose responses + * are experimental or private. + */ +import { afterEach, describe, expect, it, vi } from "vitest"; +import { http, HttpResponse } from "msw"; +import { + PAGED_REPLAY_ID, + mswServer, + replayDetailsFixture, + replayRecordingSegmentsFixture, + replayRecordingSegmentsPagedFixture, + replaySummaryProcessingFixture, +} from "@sentry/mcp-server-mocks"; +import { SentryApiService } from "./client.js"; + +const apiService = new SentryApiService({ + host: "sentry.io", + accessToken: "test-token", +}); + +const SEGMENTS_URL = (replayId: string) => + `https://sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayId}/recording-segments/`; + +const SUMMARIZE_URL = `https://sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/summarize/`; + +function readSegments(replayId: string, overrides = {}) { + return apiService.getReplayRecordingSegments({ + organizationSlug: "sentry-mcp-evals", + projectSlugOrId: String(replayDetailsFixture.project_id), + replayId, + ...overrides, + }); +} + +afterEach(() => { + // Per-test handler overrides are not reset automatically, and a leaked stub + // silently changes what a later test is measuring. + mswServer.resetHandlers(); + vi.restoreAllMocks(); +}); + +describe("getReplayRecordingSegments", () => { + it("returns a single-page recording whole", async () => { + const result = await readSegments(replayDetailsFixture.id); + + expect(result.segments).toEqual(replayRecordingSegmentsFixture); + expect(result.truncatedBy).toBeNull(); + expect(result.segmentsRead).toBe(replayRecordingSegmentsFixture.length); + }); + + it("follows the Link header cursor across pages", async () => { + // The bug this closes: without following the header, a recording longer + // than one page is silently cut off at the first page. + const result = await readSegments(PAGED_REPLAY_ID); + + expect(result.segments).toEqual(replayRecordingSegmentsPagedFixture); + expect(result.segmentsRead).toBe( + replayRecordingSegmentsPagedFixture.length, + ); + expect(result.truncatedBy).toBeNull(); + }); + + it("does not send the download parameter, which this endpoint has no concept of", async () => { + const requested: string[] = []; + mswServer.use( + http.get(SEGMENTS_URL(PAGED_REPLAY_ID), ({ request }) => { + requested.push(request.url); + return HttpResponse.json([]); + }), + ); + + await readSegments(PAGED_REPLAY_ID); + + expect(requested).toHaveLength(1); + expect(requested[0]).not.toContain("download"); + }); + + it("stops at the segment budget and says so", async () => { + const result = await readSegments(PAGED_REPLAY_ID, { maxSegments: 3 }); + + expect(result.segmentsRead).toBe(3); + expect(result.truncatedBy).toBe("segments"); + // The caller must be able to distinguish this from a short recording. + expect(result.segments).toHaveLength(3); + }); + + it("stops at the byte budget and says so", async () => { + const result = await readSegments(PAGED_REPLAY_ID, { maxBytes: 1 }); + + expect(result.truncatedBy).toBe("bytes"); + expect(result.bytesRead).toBeGreaterThan(0); + // The page that tripped the budget is already parsed, so it is kept + // rather than thrown away. + expect(result.segments.length).toBeGreaterThan(0); + expect(result.segments.length).toBeLessThan( + replayRecordingSegmentsPagedFixture.length, + ); + }); + + it("reports how many bytes were read", async () => { + const result = await readSegments(replayDetailsFixture.id); + expect(result.bytesRead).toBe( + JSON.stringify(replayRecordingSegmentsFixture).length, + ); + }); + + it("surfaces a malformed body as an error rather than an empty recording", async () => { + mswServer.use( + http.get(SEGMENTS_URL(replayDetailsFixture.id), () => + HttpResponse.text("not json", { + headers: { "Content-Type": "application/json" }, + }), + ), + ); + + await expect(readSegments(replayDetailsFixture.id)).rejects.toThrow( + /Failed to parse replay recording segments/, + ); + }); +}); + +describe("getReplayErrorEvents", () => { + it("resolves a batch of error ids in one request", async () => { + const requested: string[] = []; + mswServer.use( + http.get( + "https://sentry.io/api/0/organizations/sentry-mcp-evals/replays-events-meta/", + ({ request }) => { + requested.push(request.url); + return HttpResponse.json({ + data: [ + { + id: "aaa", + issue: "CLOUDFLARE-MCP-41", + "issue.id": 6507376925, + title: "Error: boom", + timestamp: "2025-04-07T12:03:01.300000+00:00", + }, + ], + }); + }, + ), + ); + + const events = await apiService.getReplayErrorEvents({ + organizationSlug: "sentry-mcp-evals", + errorIds: ["aaa", "bbb"], + }); + + expect(requested).toHaveLength(1); + expect(decodeURIComponent(requested[0])).toContain("query=id:[aaa,bbb]"); + // issue.id arrives as a number from Snuba; every other issue identifier in + // this client is a string. + expect(events[0]["issue.id"]).toBe("6507376925"); + expect(Date.parse(events[0].timestamp as string)).not.toBeNaN(); + }); + + it("skips the request entirely when there are no error ids", async () => { + const fetchSpy = vi.spyOn(globalThis, "fetch"); + + await expect( + apiService.getReplayErrorEvents({ + organizationSlug: "sentry-mcp-evals", + errorIds: [], + }), + ).resolves.toEqual([]); + + expect(fetchSpy).not.toHaveBeenCalled(); + }); + + it("tolerates the private endpoint omitting optional fields", async () => { + mswServer.use( + http.get( + "https://sentry.io/api/0/organizations/sentry-mcp-evals/replays-events-meta/", + () => HttpResponse.json({ data: [{ id: "aaa" }] }), + ), + ); + + const events = await apiService.getReplayErrorEvents({ + organizationSlug: "sentry-mcp-evals", + errorIds: ["aaa"], + }); + + // A missing issue.id normalizes to null rather than being absent, so + // callers have one shape to handle instead of two. + expect(events).toEqual([{ id: "aaa", "issue.id": null }]); + }); +}); + +describe("getReplaySummary", () => { + it("reads a completed summary with millisecond chapter windows", async () => { + const summary = await apiService.getReplaySummary({ + organizationSlug: "sentry-mcp-evals", + projectSlugOrId: String(replayDetailsFixture.project_id), + replayId: replayDetailsFixture.id, + }); + + expect(summary.status).toBe("completed"); + expect(summary.data?.time_ranges?.[0].period_start).toBeGreaterThan(1e12); + }); + + it("issues exactly one GET and never starts a task", async () => { + // Starting would spend a Seer LLM run per call and still report + // `processing` on the immediate read; polling would put unbounded latency + // on an optional section. + const calls: { method: string; url: string }[] = []; + mswServer.use( + http.all(SUMMARIZE_URL, ({ request }) => { + calls.push({ method: request.method, url: request.url }); + return HttpResponse.json(replaySummaryProcessingFixture); + }), + ); + + await apiService.getReplaySummary({ + organizationSlug: "sentry-mcp-evals", + projectSlugOrId: String(replayDetailsFixture.project_id), + replayId: replayDetailsFixture.id, + }); + + expect(calls).toHaveLength(1); + expect(calls[0].method).toBe("GET"); + }); + + it("parses a still-running summary without inventing data", async () => { + mswServer.use( + http.get(SUMMARIZE_URL, () => + HttpResponse.json(replaySummaryProcessingFixture), + ), + ); + + const summary = await apiService.getReplaySummary({ + organizationSlug: "sentry-mcp-evals", + projectSlugOrId: String(replayDetailsFixture.project_id), + replayId: replayDetailsFixture.id, + }); + + expect(summary.status).toBe("processing"); + expect(summary.data).toBeNull(); + }); + + it("degrades an unrecognized status to error rather than throwing", async () => { + // Both sides of this endpoint are experimental, so an unknown status is a + // reason to omit the section, not to fail the call. + mswServer.use( + http.get(SUMMARIZE_URL, () => + HttpResponse.json({ data: null, status: "something-new" }), + ), + ); + + const summary = await apiService.getReplaySummary({ + organizationSlug: "sentry-mcp-evals", + projectSlugOrId: String(replayDetailsFixture.project_id), + replayId: replayDetailsFixture.id, + }); + + expect(summary.status).toBe("error"); + }); + + it("propagates a 403 so the caller can omit the section", async () => { + mswServer.use( + http.get(SUMMARIZE_URL, () => + HttpResponse.json( + { + detail: "Replay summaries are not available for this organization.", + }, + { status: 403 }, + ), + ), + ); + + await expect( + apiService.getReplaySummary({ + organizationSlug: "sentry-mcp-evals", + projectSlugOrId: String(replayDetailsFixture.project_id), + replayId: replayDetailsFixture.id, + }), + ).rejects.toThrow(); + }); +}); diff --git a/packages/mcp-core/src/api-client/schema.ts b/packages/mcp-core/src/api-client/schema.ts index c2385148f..7a8e209a7 100644 --- a/packages/mcp-core/src/api-client/schema.ts +++ b/packages/mcp-core/src/api-client/schema.ts @@ -522,6 +522,76 @@ export const ReplayListResponseSchema = z.object({ data: z.array(ReplayDetailsSchema), }); +/** + * One error event associated with a replay, from + * `GET /organizations/{org}/replays-events-meta/`. + * + * The endpoint deletes `timestamp_ms` from its own output and folds that + * millisecond precision into `timestamp` as an ISO 8601 string, so the + * resolution a zoom window needs arrives there and nowhere else. + * + * `issue.id` arrives as a number from Snuba but is coerced to a string, since + * every other issue identifier in this client is a string. + */ +export const ReplayErrorEventSchema = z + .object({ + id: z.string(), + issue: z.string().nullish(), + "issue.id": z + .union([z.string(), z.number()]) + .nullish() + .transform((value) => (value == null ? null : String(value))), + title: z.string().nullish(), + timestamp: z.string().nullish(), + }) + .passthrough(); + +export const ReplayErrorEventsResponseSchema = z.object({ + data: z.array(ReplayErrorEventSchema), +}); + +/** + * Seer's replay summary state, proxied verbatim by Sentry's summarize endpoint. + * + * Both sides are experimental, so every field below `status` is optional and + * the shape is parsed defensively: the caller renders chapters only when the + * status is `completed` and the body actually parses, and omits the section + * otherwise. + * + * Defined upstream in `getsentry/seer` at + * `src/seer/automation/summarize/replays.py`. + */ +export const ReplaySummarySchema = z + .object({ + data: z + .object({ + time_ranges: z + .array( + z + .object({ + // Float UNIX timestamps in milliseconds, so chapters carry + // windows usable for zoom rather than only prose. + period_start: z.number(), + period_end: z.number(), + period_title: z.string(), + }) + .passthrough(), + ) + .optional() + .catch(undefined), + summary: z.string().optional().catch(undefined), + }) + .passthrough() + .nullish() + .catch(null), + num_segments: z.number().nullish().catch(null), + created_at: z.string().nullish().catch(null), + status: z + .enum(["not_started", "processing", "completed", "error"]) + .catch("error"), + }) + .passthrough(); + export const ReplayIdsByResourceSchema = z.record( z.string(), z.array(z.string()), diff --git a/packages/mcp-core/src/api-client/types.ts b/packages/mcp-core/src/api-client/types.ts index cd388e4ed..ec043e471 100644 --- a/packages/mcp-core/src/api-client/types.ts +++ b/packages/mcp-core/src/api-client/types.ts @@ -107,7 +107,9 @@ import type { ReleaseListSchema, ReleaseSchema, ReplayDetailsSchema, + ReplayErrorEventSchema, ReplayListResponseSchema, + ReplaySummarySchema, ReplayRecordingEventSchema, ReplayRecordingPayloadSchema, ReplayRecordingSegmentsSchema, @@ -179,6 +181,8 @@ export type AutofixRunState = z.infer; export type AssignedTo = z.infer; export type ReplayDetails = z.infer; export type ReplayList = z.infer["data"]; +export type ReplayErrorEvent = z.infer; +export type ReplaySummary = z.infer; export type ReplayRecordingEvent = z.infer; export type ReplayRecordingPayload = z.infer< typeof ReplayRecordingPayloadSchema diff --git a/packages/mcp-core/src/tools/catalog/get-replay-details.ts b/packages/mcp-core/src/tools/catalog/get-replay-details.ts index 241a89ca1..c70ad9e2a 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-details.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-details.ts @@ -4,9 +4,14 @@ import type { ReplayDetails, ReplayRecordingEvent, ReplayRecordingSegments, + ReplayRecordingSegmentsResult, SentryApiService, TraceMeta, } from "../../api-client"; +import { + MAX_REPLAY_SEGMENTS, + MAX_REPLAY_SEGMENT_BYTES, +} from "../../api-client"; import { defineTool } from "../../internal/tool-helpers/define"; import { apiServiceFromContext } from "../../internal/tool-helpers/api"; import { resolveRegionUrlForOrganization } from "../../internal/tool-helpers/resolve-region-url"; @@ -116,26 +121,27 @@ export default defineTool({ replay.project_id != null ? String(replay.project_id) : null; const hasSegments = (replay.count_segments ?? 0) > 0; - const [{ segments }, relatedIssues, relatedTraces] = await Promise.all([ - fetchReplaySegments({ - apiService, - organizationSlug: resolved.organizationSlug, - replayId: resolved.replayId, - projectId, - isArchived, - hasSegments, - }), - fetchReplayIssues({ - apiService, - organizationSlug: resolved.organizationSlug, - errorIds: replay.error_ids, - }), - fetchReplayTraces({ - apiService, - organizationSlug: resolved.organizationSlug, - traceIds: replay.trace_ids, - }), - ]); + const [{ segments, truncatedBy }, relatedIssues, relatedTraces] = + await Promise.all([ + fetchReplaySegments({ + apiService, + organizationSlug: resolved.organizationSlug, + replayId: resolved.replayId, + projectId, + isArchived, + hasSegments, + }), + fetchReplayIssues({ + apiService, + organizationSlug: resolved.organizationSlug, + errorIds: replay.error_ids, + }), + fetchReplayTraces({ + apiService, + organizationSlug: resolved.organizationSlug, + traceIds: replay.trace_ids, + }), + ]); return formatReplayOutput({ replay, @@ -144,6 +150,7 @@ export default defineTool({ params.replayUrl ?? apiService.getReplayUrl(resolved.organizationSlug, replay.id), segments, + truncatedBy, isArchived, relatedIssues, relatedTraces, @@ -223,6 +230,7 @@ function formatReplayOutput({ organizationSlug, replayUrl, segments, + truncatedBy, isArchived, relatedIssues, relatedTraces, @@ -231,6 +239,7 @@ function formatReplayOutput({ organizationSlug: string; replayUrl: string; segments: ReplayRecordingSegments | null; + truncatedBy: ReplayRecordingSegmentsResult["truncatedBy"]; isArchived: boolean; relatedIssues: RelatedReplayIssue[]; relatedTraces: RelatedReplayTrace[]; @@ -313,6 +322,19 @@ function formatReplayOutput({ lines.push("No activity events recorded."); } + // A partial read must say so rather than reading as a complete session. + if (truncatedBy === "segments") { + lines.push(""); + lines.push( + `Recording was read up to the first ${MAX_REPLAY_SEGMENTS} segments; later activity is not included.`, + ); + } else if (truncatedBy === "bytes") { + lines.push(""); + lines.push( + `Recording was read up to ${MAX_REPLAY_SEGMENT_BYTES / (1024 * 1024)}MB; later activity is not included.`, + ); + } + // Related const hasRelated = relatedIssues.length > 0 || relatedTraces.length > 0; if (hasRelated) { @@ -370,20 +392,21 @@ async function fetchReplaySegments({ hasSegments: boolean; }): Promise<{ segments: ReplayRecordingSegments | null; + truncatedBy: ReplayRecordingSegmentsResult["truncatedBy"]; }> { if (isArchived || !projectId || !hasSegments) { - return { segments: null }; + return { segments: null, truncatedBy: null }; } try { - const segments = await apiService.getReplayRecordingSegments({ + const result = await apiService.getReplayRecordingSegments({ organizationSlug, projectSlugOrId: projectId, replayId, }); - return { segments }; + return { segments: result.segments, truncatedBy: result.truncatedBy }; } catch { - return { segments: null }; + return { segments: null, truncatedBy: null }; } } From 66d80bbfaa660b7ebd5aa4ab3debd93dac7ae8d9 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 14:16:24 -0700 Subject: [PATCH 05/26] feat(replays): Return a session map from get_replay_details MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replay details returned a six-event prose sample, which either truncates a long session or floods a short one, and in neither case tells the reader where to look. The cap was routinely spent on session-boot noise before reaching the failure the user cared about. Return the shape of the session instead: signal counts and time span, the page flow, and a per-kind breakdown carrying error, rage, and dead click counts. Counts come from every classified event including ones that are never rendered, so `network 2 (1 failed)` describes two requests of which one is shown. Truncation is stated rather than implied. Add a suggested next call windowed on the replay's own first resolvable error, so the reader gets a concrete zoom target instead of guessing. Error timestamps come from one batched `replays-events-meta` lookup, which also replaces the per-error `listIssues` calls behind the Related section — up to three sequential requests become one. Related entries are still driven by `replay.error_ids`, so an id that private endpoint cannot resolve is listed by id rather than disappearing. When no timestamp resolves, the suggestion falls back to a whole-session digest and the map is unaffected. Render Sentry's AI summary as an optional Chapters section, read once. It is strictly additive: a 403, a server failure, a still-running or not-started task, an unparseable body, or a non-completed status carrying stale chapter data all omit the section and leave the map intact. Status decides, not the presence of data, so a superseded run cannot present itself as current. Report how many related issues and traces were omitted; both previously stopped at their display limit in silence. Delete the old classifier, which keyed on a `data.tag` shape the SDK never emits. Its snapshots recorded that defect deliberately; they now record correct output, and two tests that pinned the old rendering are rewritten to assert what replaced it. Refs docs/specs/replay-review.md Co-Authored-By: Claude Opus 5 (1M context) --- .../changes/improve-replay-review/tasks.md | 16 +- .../tools/catalog/get-replay-details.test.ts | 255 ++++++- .../src/tools/catalog/get-replay-details.ts | 669 ++++++++++++------ .../tools/catalog/get-sentry-resource.test.ts | 12 +- 4 files changed, 701 insertions(+), 251 deletions(-) diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index 5ec87b2b9..8f84bb460 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -31,13 +31,13 @@ baselines encode the same wrong event shape they do today. ## 4. Replay Details as a Map -- [ ] 4.1 Replace the activity sample with signal counts, time span, page flow, and per-kind breakdown including error and rage or dead click counts. -- [ ] 4.2 Derive a suggested `get_replay_activity` window from error timestamps resolved through `getReplayErrorEvents`, degrading to no suggested window if the endpoint fails or returns no usable timestamp. -- [ ] 4.3 Replace the per-error `listIssues` lookups in the Related section with the same batched `getReplayErrorEvents` call, which already returns issue identity. -- [ ] 4.4 Fall back to a whole-session digest suggestion when no error timestamp resolves. -- [ ] 4.5 Report truncation for related issues and traces. -- [ ] 4.6 Wire the summary chapters section as a single-read enhancement, rendering only on `status: completed` and degrading silently on 403, error status, `processing`, `not_started`, timeout, and parse failure. Assert in tests that exactly one request is issued and no start request is sent. -- [ ] 4.7 Preserve archived-replay and missing-segment behavior. +- [x] 4.1 Replace the activity sample with signal counts, time span, page flow, and per-kind breakdown including error and rage or dead click counts. +- [x] 4.2 Derive a suggested `get_replay_activity` window from error timestamps resolved through `getReplayErrorEvents`, degrading to no suggested window if the endpoint fails or returns no usable timestamp. +- [x] 4.3 Replace the per-error `listIssues` lookups in the Related section with the same batched `getReplayErrorEvents` call, which already returns issue identity. Related entries are still driven by `replay.error_ids`, so an id the private endpoint cannot resolve is listed by id rather than dropped. +- [x] 4.4 Fall back to a whole-session digest suggestion when no error timestamp resolves. +- [x] 4.5 Report truncation for related issues and traces. +- [x] 4.6 Wire the summary chapters section as a single-read enhancement, rendering only on `status: completed` and degrading silently on 403, error status, `processing`, `not_started`, timeout, and parse failure. Assert in tests that exactly one request is issued and no start request is sent. Also covered: a non-completed status that carries stale chapter data, verified by mutation to fail without the status guard. +- [x] 4.7 Preserve archived-replay and missing-segment behavior. ## 5. Replay Activity Tool @@ -58,7 +58,7 @@ baselines encode the same wrong event shape they do today. ## 7. Tests -- [ ] 7.1 Re-baseline `get-replay-details.test.ts` snapshots against the rebuilt fixtures. +- [x] 7.1 Re-baseline `get-replay-details.test.ts` snapshots against the rebuilt fixtures. - [ ] 7.2 Add `get-replay-activity.test.ts` covering windowing, each grain, kind filtering, paging, and constraint injection. - [ ] 7.3 Add degradation tests: summary 403, summary `processing`, summary timeout, summary unparseable, segment fetch 404, archived replay, zero segments, multi-page segments, and the segment budget being hit. - [ ] 7.4 Add redaction tests asserting `` and `[Filtered]`-driven `` rendering. diff --git a/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts index 230fbffcc..8b568db89 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts @@ -9,15 +9,6 @@ import getReplayDetails, { resolveReplayParams } from "./get-replay-details.js"; import { getServerContext } from "../../test-setup.js"; describe("get_replay_details", () => { - // NOTE: The Activity snapshots below record known-wrong output. The fixtures - // now use the event shape the SDK actually emits (`tag: "breadcrumb"` with - // the meaning in `payload.category`), which the current classifier does not - // understand: it labels every user action `breadcrumb`, spends the six-event - // budget on session-boot noise, and drops the console error, the failed - // checkout request, the rage click, and the dead click entirely. - // - // These baselines exist to make the taxonomy fix visible as a diff. See - // docs/specs/replay-review.md. it("loads replay details from replayUrl", async () => { const result = await getReplayDetails.handler( { @@ -48,21 +39,30 @@ describe("get_replay_details", () => { - **Recording Segments**: 2 - **Archived**: No - ## Activity + ## Map - - T+0s · \`page.view\` · href=https://example.com/login - - T+0s · \`options\` · payload="sessionSampleRate=0.1, errorSampleRate=1" - - T+1s · \`navigation.navigate\` · description=https://example.com/login · duration_ms=540 - - T+1s · \`resource.script\` · description=https://cdn.example.com/vendor.js - - T+12s · \`breadcrumb\` · message="body > div#root > form#login > button#sign-in" · category="ui.click" · type="default" · payload="timestamp=1744027212.4" - - T+13s · \`resource.fetch\` · description=https://example.com/api/login + - **Signals**: 8 signals across T+0.5s–T+3m 41.7s + - **Flow**: /login ▸ /checkout + - **Kinds**: navigation 3 · click 4 (1 rage, 1 dead) · network 2 (1 failed) · console 1 (1 error) + - **Truncated**: no + + ## Chapters + + - T+0.5s–T+14.2s Signed in and navigated to checkout + - T+3m 0.6s–T+3m 8.4s Complete order failed with a server error + - T+3m 41.7s–T+3m 48.7s Download receipt link did not respond ## Related - **CLOUDFLARE-MCP-41**: Error: Tool list_organizations is already registered - Trace \`a4d1aae7216b47ff8117cf4e09ce9d0a\` (112 spans) - Use \`get_sentry_resource\` to inspect any issue or trace listed above." + Use \`get_sentry_resource\` to inspect any issue or trace listed above. + + ## Next + + Error CLOUDFLARE-MCP-41 occurred at T+3m 1.3s. Use the Sentry tool \`get_replay_activity\` to read the signals in a time window: + get_replay_activity(organizationSlug='sentry-mcp-evals', replayId='7e07485f-12f9-416b-8b14-26260799b51f', startMs=176300, endMs=186300, grain='detail')" `); }); @@ -138,7 +138,7 @@ describe("get_replay_details", () => { - **Recording Segments**: 2 - **Archived**: Yes - ## Activity + ## Map Recording is archived and not available for playback. @@ -200,9 +200,15 @@ describe("get_replay_details", () => { - **Recording Segments**: 2 - **Archived**: No - ## Activity + ## Map - No activity events recorded. + Recording is unavailable. + + ## Chapters + + - T+0.5s–T+14.2s Signed in and navigated to checkout + - T+3m 0.6s–T+3m 8.4s Complete order failed with a server error + - T+3m 41.7s–T+3m 48.7s Download receipt link did not respond ## Related @@ -342,7 +348,11 @@ describe("get_replay_details", () => { ); }); - it("does not repeat explicit payload fields in generic replay events", async () => { + it("ignores events whose tag is not one the SDK emits", async () => { + // `tag: "console"` is not a shape the SDK produces — meaning lives in + // `payload.category` under `tag: "breadcrumb"`. Such an event is + // unclassifiable, and inventing a rendering for it is what produced the + // old `breadcrumb`-labeled output. mswServer.use( http.get( `https://us.sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/recording-segments/`, @@ -356,11 +366,8 @@ describe("get_replay_details", () => { tag: "console", payload: { message: "Payment request failed", - description: "POST /api/orders returned 500", category: "network", type: "error", - endpoint: "/api/orders", - status: 500, }, }, }, @@ -379,9 +386,47 @@ describe("get_replay_details", () => { getServerContext(), ); - expect(result).toContain( - '- T+0s · `console` · message="Payment request failed" · description="POST /api/orders returned 500" · category="network" · type="error" · payload="endpoint=/api/orders, status=500"', + expect(result).toContain("No activity recorded."); + expect(result).not.toContain("Payment request failed"); + }); + + it("classifies a real SDK-shaped console error", async () => { + mswServer.use( + http.get( + `https://us.sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/recording-segments/`, + () => + HttpResponse.json([ + [ + { + type: 5, + timestamp: 1744027205000, + data: { + tag: "breadcrumb", + payload: { + type: "default", + category: "console", + level: "error", + message: "Payment request failed", + data: { logger: "console" }, + }, + }, + }, + ], + ]), + { once: true }, + ), + ); + + const result = await getReplayDetails.handler( + { + organizationSlug: "sentry-mcp-evals", + replayId: replayDetailsFixture.id, + regionUrl: "https://us.sentry.io", + }, + getServerContext(), ); + + expect(result).toContain("**Kinds**: console 1 (1 error)"); }); it("ignores array payloads instead of rendering numeric keys", async () => { @@ -418,6 +463,164 @@ describe("get_replay_details", () => { expect(result).not.toContain('payload="0='); }); + describe("summary chapters", () => { + const SUMMARIZE_URL = `https://us.sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/summarize/`; + + async function loadReplay() { + return getReplayDetails.handler( + { + organizationSlug: "sentry-mcp-evals", + replayId: replayDetailsFixture.id, + regionUrl: "https://us.sentry.io", + }, + getServerContext(), + ); + } + + it("issues exactly one read and never starts a summary task", async () => { + // Starting would spend a Seer LLM run per call for a section that would + // not be ready anyway; polling would put unbounded latency on the map. + const calls: string[] = []; + mswServer.use( + http.all(SUMMARIZE_URL, ({ request }) => { + calls.push(request.method); + return HttpResponse.json({ data: null, status: "processing" }); + }), + ); + + await loadReplay(); + + expect(calls).toEqual(["GET"]); + }); + + // A stale summary can carry chapter data alongside a non-completed + // status. Rendering it would present partial or superseded analysis as + // current, so the status — not the presence of data — decides. + const staleChapters = { + time_ranges: [ + { + period_start: 1744027200500, + period_end: 1744027214200, + period_title: "Stale chapter from a superseded run", + }, + ], + summary: "Superseded.", + }; + + for (const [label, respond] of [ + [ + "a permission error", + () => HttpResponse.json({ detail: "no" }, { status: 403 }), + ], + [ + "a still-running task", + () => HttpResponse.json({ data: null, status: "processing" }), + ], + [ + "a not-started task", + () => HttpResponse.json({ data: null, status: "not_started" }), + ], + [ + "an error status", + () => HttpResponse.json({ data: null, status: "error" }), + ], + [ + "a still-running task that carries stale chapter data", + () => HttpResponse.json({ data: staleChapters, status: "processing" }), + ], + [ + "an errored task that carries stale chapter data", + () => HttpResponse.json({ data: staleChapters, status: "error" }), + ], + ["an unparseable body", () => HttpResponse.json({ nonsense: true })], + [ + "a server failure", + () => HttpResponse.json({ detail: "boom" }, { status: 500 }), + ], + ] as const) { + it(`omits chapters on ${label} without degrading the map`, async () => { + mswServer.use(http.get(SUMMARIZE_URL, respond)); + + const result = await loadReplay(); + + expect(result).not.toContain("## Chapters"); + // The map is the primary content; chapters are additive by + // construction and must never take it down with them. + expect(result).toContain("## Map"); + expect(result).toContain( + "**Kinds**: navigation 3 · click 4 (1 rage, 1 dead) · network 2 (1 failed) · console 1 (1 error)", + ); + }); + } + }); + + describe("suggested next call", () => { + const EVENTS_META_URL = + "https://us.sentry.io/api/0/organizations/sentry-mcp-evals/replays-events-meta/"; + + async function loadReplay() { + return getReplayDetails.handler( + { + organizationSlug: "sentry-mcp-evals", + replayId: replayDetailsFixture.id, + regionUrl: "https://us.sentry.io", + }, + getServerContext(), + ); + } + + it("brackets a resolved error timestamp", async () => { + const result = await loadReplay(); + + // The fixture error lands at T+3m 1.3s (181,300ms), so the window is + // that offset padded either side. + expect(result).toContain("Error CLOUDFLARE-MCP-41 occurred at T+3m 1.3s"); + expect(result).toContain("startMs=176300, endMs=186300, grain='detail'"); + }); + + for (const [label, respond] of [ + [ + "the private endpoint is unauthorized", + () => HttpResponse.json({ detail: "no" }, { status: 403 }), + ], + [ + "the lookup fails", + () => HttpResponse.json({ detail: "boom" }, { status: 500 }), + ], + ["no events resolve", () => HttpResponse.json({ data: [] })], + [ + "the timestamp is unusable", + () => + HttpResponse.json({ + data: [{ id: "7ca573c0f4814912aaa9bdc77d1a7d51", timestamp: "?" }], + }), + ], + ] as const) { + it(`falls back to a whole-session digest when ${label}`, async () => { + mswServer.use(http.get(EVENTS_META_URL, respond)); + + const result = await loadReplay(); + + expect(result).toContain("grain='digest'"); + expect(result).not.toContain("startMs="); + // Degrading the suggestion must not degrade the map. + expect(result).toContain("**Signals**: 8 signals"); + }); + } + + it("still lists an error id the lookup could not resolve", async () => { + // `error_ids` is the replay's own record that an error occurred, so + // dropping it would understate the session. + mswServer.use( + http.get(EVENTS_META_URL, () => HttpResponse.json({ data: [] })), + ); + + const result = await loadReplay(); + + expect(result).toContain("- Event `7ca573c0f4814912aaa9bdc77d1a7d51`"); + }); + }); + describe("tool definition", () => { it("requires the replay read scopes used by the backend endpoints", () => { expect(getReplayDetails.requiredScopes).toEqual([ diff --git a/packages/mcp-core/src/tools/catalog/get-replay-details.ts b/packages/mcp-core/src/tools/catalog/get-replay-details.ts index c70ad9e2a..99d02c990 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-details.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-details.ts @@ -1,8 +1,7 @@ import { setTag } from "@sentry/core"; import type { - Issue, ReplayDetails, - ReplayRecordingEvent, + ReplayErrorEvent, ReplayRecordingSegments, ReplayRecordingSegmentsResult, SentryApiService, @@ -12,6 +11,20 @@ import { MAX_REPLAY_SEGMENTS, MAX_REPLAY_SEGMENT_BYTES, } from "../../api-client"; +import type { + ReplayKindCount, + ReplaySignal, + ReplaySignalKind, +} from "../../internal/replay-events"; +import { + countReplayKinds, + extractReplaySignals, + formatReplayOffset, +} from "../../internal/replay-events"; +import { + formatToolCall, + formatToolCallInstruction, +} from "../../internal/tool-helpers/tool-call-formatting"; import { defineTool } from "../../internal/tool-helpers/define"; import { apiServiceFromContext } from "../../internal/tool-helpers/api"; import { resolveRegionUrlForOrganization } from "../../internal/tool-helpers/resolve-region-url"; @@ -31,19 +44,10 @@ interface ResolvedReplayParams { replayId: string; } -interface ReplayActivityEvent { - timestampMs: number | null; - label: string; - details: string[]; -} - -type ReplayRecordingPayload = NonNullable< - NonNullable["payload"] ->; - interface RelatedReplayIssue { eventId: string; - issue: Issue | null; + shortId: string | null; + title: string | null; } interface RelatedReplayTrace { @@ -51,10 +55,27 @@ interface RelatedReplayTrace { traceMeta: TraceMeta | null; } -const MAX_ACTIVITY_EVENTS = 6; +/** A time-bounded section of the session, from Sentry's AI summary. */ +interface ReplayChapter { + startMs: number; + endMs: number; + title: string; +} + const MAX_RELATED_ERRORS = 3; const MAX_RELATED_TRACES = 2; +/** Pages shown in the flow line before it is summarized with a count. */ +const MAX_FLOW_PAGES = 6; + +/** + * Half-width of the suggested zoom window around an error. + * + * Wide enough to include the interaction that caused the failure and the + * fallout after it, narrow enough that the window is still a zoom. + */ +const ERROR_WINDOW_PADDING_MS = 5000; + export default defineTool({ name: "get_replay_details", skills: ["inspect"], @@ -121,7 +142,7 @@ export default defineTool({ replay.project_id != null ? String(replay.project_id) : null; const hasSegments = (replay.count_segments ?? 0) > 0; - const [{ segments, truncatedBy }, relatedIssues, relatedTraces] = + const [{ segments, truncatedBy }, errorEvents, relatedTraces, chapters] = await Promise.all([ fetchReplaySegments({ apiService, @@ -131,18 +152,33 @@ export default defineTool({ isArchived, hasSegments, }), - fetchReplayIssues({ + fetchReplayErrorEvents({ apiService, organizationSlug: resolved.organizationSlug, errorIds: replay.error_ids, + projectId, }), fetchReplayTraces({ apiService, organizationSlug: resolved.organizationSlug, traceIds: replay.trace_ids, }), + fetchReplayChapters({ + apiService, + organizationSlug: resolved.organizationSlug, + projectId, + replayId: resolved.replayId, + startedAt: replay.started_at, + isArchived, + }), ]); + const signals = extractReplaySignals(segments, { + startedAt: replay.started_at, + platform: replay.platform, + }); + const kindCounts = countReplayKinds(segments); + return formatReplayOutput({ replay, organizationSlug: resolved.organizationSlug, @@ -150,10 +186,25 @@ export default defineTool({ params.replayUrl ?? apiService.getReplayUrl(resolved.organizationSlug, replay.id), segments, + signals, + kindCounts, + chapters, + nextStepLines: buildNextStepLines({ + organizationSlug: resolved.organizationSlug, + replayId: replay.id, + errorEvents, + startedAt: replay.started_at, + context, + }), truncatedBy, isArchived, - relatedIssues, + relatedIssues: toRelatedIssues(replay.error_ids, errorEvents).slice( + 0, + MAX_RELATED_ERRORS, + ), + omittedIssues: Math.max(0, replay.error_ids.length - MAX_RELATED_ERRORS), relatedTraces, + omittedTraces: Math.max(0, replay.trace_ids.length - MAX_RELATED_TRACES), }); }, }); @@ -230,19 +281,31 @@ function formatReplayOutput({ organizationSlug, replayUrl, segments, + signals, + kindCounts, + chapters, + nextStepLines, truncatedBy, isArchived, relatedIssues, + omittedIssues, relatedTraces, + omittedTraces, }: { replay: ReplayDetails; organizationSlug: string; replayUrl: string; segments: ReplayRecordingSegments | null; + signals: ReplaySignal[]; + kindCounts: ReplayKindCount[]; + chapters: ReplayChapter[]; + nextStepLines: string[]; truncatedBy: ReplayRecordingSegmentsResult["truncatedBy"]; isArchived: boolean; relatedIssues: RelatedReplayIssue[]; + omittedIssues: number; relatedTraces: RelatedReplayTrace[]; + omittedTraces: number; }): string { const lines: string[] = []; const user = @@ -256,7 +319,6 @@ function formatReplayOutput({ replay.device?.model ?? replay.device?.family ?? null; - const activityEvents = extractReplayActivityEvents(segments); // Summary lines.push(`# Replay ${replay.id} in **${organizationSlug}**`); @@ -300,39 +362,41 @@ function formatReplayOutput({ lines.push(`- **Viewed**: ${replay.has_viewed ? "Yes" : "No"}`); } - // Activity + // Map — the shape of the session rather than a sample of it. A fixed-length + // prose sample either truncates a long session or floods a short one, and + // neither tells the reader where to look. lines.push(""); - lines.push("## Activity"); + lines.push("## Map"); lines.push(""); if (isArchived) { lines.push("Recording is archived and not available for playback."); - } else if (activityEvents.length > 0) { - const startTime = activityEvents[0]?.timestampMs ?? null; - for (const event of activityEvents) { - const prefix = - event.timestampMs !== null && startTime !== null - ? `${formatRelativeTime(event.timestampMs - startTime)} · ` - : ""; - const details = - event.details.length > 0 ? ` · ${event.details.join(" · ")}` : ""; - lines.push(`- ${prefix}\`${event.label}\`${details}`); - } + } else if (segments === null) { + lines.push("Recording is unavailable."); + } else if (signals.length === 0) { + lines.push("No activity recorded."); } else { - lines.push("No activity events recorded."); + lines.push(`- **Signals**: ${formatSignalSpan(signals)}`); + + const flow = buildPageFlow(signals, replay.urls); + if (flow) { + lines.push(`- **Flow**: ${flow}`); + } + + lines.push(`- **Kinds**: ${formatKindBreakdown(kindCounts, signals)}`); + lines.push(`- **Truncated**: ${formatTruncation(truncatedBy)}`); } - // A partial read must say so rather than reading as a complete session. - if (truncatedBy === "segments") { + // Chapters — present only when Sentry already has a summary for this replay. + if (chapters.length > 0) { lines.push(""); - lines.push( - `Recording was read up to the first ${MAX_REPLAY_SEGMENTS} segments; later activity is not included.`, - ); - } else if (truncatedBy === "bytes") { + lines.push("## Chapters"); lines.push(""); - lines.push( - `Recording was read up to ${MAX_REPLAY_SEGMENT_BYTES / (1024 * 1024)}MB; later activity is not included.`, - ); + for (const chapter of chapters) { + lines.push( + `- ${formatReplayOffset(chapter.startMs)}–${formatReplayOffset(chapter.endMs)} ${chapter.title}`, + ); + } } // Related @@ -343,12 +407,19 @@ function formatReplayOutput({ lines.push(""); for (const ri of relatedIssues) { - if (ri.issue) { - lines.push(`- **${ri.issue.shortId}**: ${ri.issue.title}`); + if (ri.shortId) { + lines.push(`- **${ri.shortId}**: ${ri.title ?? "Unknown error"}`); } else { lines.push(`- Event \`${ri.eventId}\``); } } + // Stopping at a display limit without saying so reads as "this is all of + // them". + if (omittedIssues > 0) { + lines.push( + `- …and ${omittedIssues} more error${omittedIssues === 1 ? "" : "s"}`, + ); + } for (const rt of relatedTraces) { const spanInfo = rt.traceMeta @@ -356,6 +427,11 @@ function formatReplayOutput({ : ""; lines.push(`- Trace \`${rt.traceId}\`${spanInfo}`); } + if (omittedTraces > 0) { + lines.push( + `- …and ${omittedTraces} more trace${omittedTraces === 1 ? "" : "s"}`, + ); + } lines.push(""); lines.push( @@ -363,9 +439,158 @@ function formatReplayOutput({ ); } + // Next — a concrete call, windowed on this replay's own failure when one is + // resolvable, so the reader does not have to guess where to zoom. + if (!isArchived && segments !== null && signals.length > 0) { + lines.push(""); + lines.push("## Next"); + lines.push(""); + lines.push(...nextStepLines); + } + return lines.join("\n"); } +/** + * Describe how much happened and over what span. + */ +function formatSignalSpan(signals: ReplaySignal[]): string { + const offsets = signals + .map((signal) => signal.offsetMs) + .filter((offset): offset is number => offset !== null); + + const count = `${signals.length.toLocaleString("en-US")} signal${signals.length === 1 ? "" : "s"}`; + if (offsets.length === 0) { + return count; + } + + const first = formatReplayOffset(Math.min(...offsets)); + const last = formatReplayOffset(Math.max(...offsets)); + return `${count} across ${first}–${last}`; +} + +/** + * Render the page flow as an ordered path. + * + * Built from navigation signals when the recording has them, since those carry + * ordering. `replay.urls` is a fallback: it lists the pages visited but is + * metadata rather than a timeline. + */ +function buildPageFlow(signals: ReplaySignal[], urls: string[]): string | null { + const visited: string[] = []; + + for (const signal of signals) { + if (signal.kind !== "navigation") { + continue; + } + const page = toPagePath(signal.summary.replace(/^Navigated to /, "")); + if (page !== visited.at(-1)) { + visited.push(page); + } + } + + const flow = visited.length > 0 ? visited : urls; + if (flow.length === 0) { + return null; + } + + const shown = flow.slice(0, MAX_FLOW_PAGES); + const suffix = + flow.length > MAX_FLOW_PAGES ? ` …+${flow.length - MAX_FLOW_PAGES}` : ""; + return `${shown.join(" ▸ ")}${suffix}`; +} + +/** + * Reduce a navigation target to its path. + * + * The flow line is about where the user went, and repeating the host on every + * hop crowds out that shape. `replay.urls` is already path-only, so this also + * keeps both sources of the flow consistent. + */ +function toPagePath(page: string): string { + const withoutHost = page.replace(/^[^/]+/, ""); + return withoutHost.startsWith("/") ? withoutHost : `/${page}`; +} + +/** + * Summarize each kind with its failure count. + * + * Counts come from every classified event, including ones that are never + * rendered, so `network 58 (2 failed)` means 58 requests of which 2 are shown. + * Click kinds are folded into one entry with their rage and dead counts, which + * is how the Sentry UI presents them. + */ +function formatKindBreakdown( + kindCounts: ReplayKindCount[], + signals: ReplaySignal[], +): string { + const byKind = new Map(kindCounts.map((entry) => [entry.kind, entry])); + const parts: string[] = []; + + const navigation = byKind.get("navigation"); + if (navigation) { + parts.push(`navigation ${navigation.total}`); + } + + const clicks = ["click", "dead-click", "rage-click", "slow-click"] as const; + const clickTotal = clicks.reduce( + (total, kind) => total + (byKind.get(kind)?.total ?? 0), + 0, + ); + if (clickTotal > 0) { + const rage = byKind.get("rage-click")?.total ?? 0; + const dead = byKind.get("dead-click")?.total ?? 0; + const notes = [ + rage > 0 ? `${rage} rage` : null, + dead > 0 ? `${dead} dead` : null, + ].filter(Boolean); + parts.push( + `click ${clickTotal}${notes.length > 0 ? ` (${notes.join(", ")})` : ""}`, + ); + } + + const network = byKind.get("network"); + if (network) { + parts.push( + `network ${network.total}${network.errors > 0 ? ` (${network.errors} failed)` : ""}`, + ); + } + + const consoleCount = byKind.get("console"); + if (consoleCount) { + parts.push( + `console ${consoleCount.total}${consoleCount.errors > 0 ? ` (${consoleCount.errors} error)` : ""}`, + ); + } + + // Anything not called out above, so no kind disappears from the breakdown. + const named = new Set([ + "navigation", + ...clicks, + "network", + "console", + ]); + for (const entry of kindCounts) { + if (!named.has(entry.kind)) { + parts.push(`${entry.kind} ${entry.total}`); + } + } + + return parts.length > 0 ? parts.join(" · ") : `${signals.length} signals`; +} + +function formatTruncation( + truncatedBy: ReplayRecordingSegmentsResult["truncatedBy"], +): string { + if (truncatedBy === "segments") { + return `yes — read the first ${MAX_REPLAY_SEGMENTS} segments; later activity is not included`; + } + if (truncatedBy === "bytes") { + return `yes — read the first ${MAX_REPLAY_SEGMENT_BYTES / (1024 * 1024)}MB of recording; later activity is not included`; + } + return "no"; +} + function formatDurationSeconds(durationSeconds: number): string { if (durationSeconds < 60) { return `${durationSeconds}s`; @@ -410,31 +635,205 @@ async function fetchReplaySegments({ } } -async function fetchReplayIssues({ +/** + * Resolve the replay's error events. + * + * One batched call replaces the per-error `listIssues` lookups this section + * used to make — up to three sequential requests — and returns event + * timestamps as well as issue identity, which `listIssues` cannot provide. + * + * The endpoint is PRIVATE, so failure degrades to an empty list: the map is + * worth returning without a Related section, but not worth failing over one. + */ +async function fetchReplayErrorEvents({ apiService, organizationSlug, errorIds, + projectId, }: { apiService: SentryApiService; organizationSlug: string; errorIds: string[]; -}): Promise { - const ids = errorIds.slice(0, MAX_RELATED_ERRORS); + projectId: string | null; +}): Promise { + if (errorIds.length === 0) { + return []; + } - return Promise.all( - ids.map(async (eventId) => { - try { - const [issue] = await apiService.listIssues({ - organizationSlug, - query: eventId, - limit: 1, - }); - return { eventId, issue: issue ?? null }; - } catch { - return { eventId, issue: null }; - } + try { + return await apiService.getReplayErrorEvents({ + organizationSlug, + errorIds, + projectId: projectId ?? undefined, + }); + } catch { + return []; + } +} + +/** + * Read Sentry's AI summary and convert it into chapters. + * + * Strictly additive: any failure, any non-`completed` status, or an + * unparseable body yields no chapters and leaves the map untouched. Chapters + * therefore appear only for replays already summarized in the Sentry UI, which + * is the accepted cost of never starting a Seer run from here. + */ +async function fetchReplayChapters({ + apiService, + organizationSlug, + projectId, + replayId, + startedAt, + isArchived, +}: { + apiService: SentryApiService; + organizationSlug: string; + projectId: string | null; + replayId: string; + startedAt?: string | null; + isArchived: boolean; +}): Promise { + if (isArchived || !projectId) { + return []; + } + + const originMs = startedAt ? Date.parse(startedAt) : Number.NaN; + if (Number.isNaN(originMs)) { + // Without a session origin the chapter windows cannot be placed on the + // replay's timeline, and an absolute epoch is not useful to the reader. + return []; + } + + try { + const summary = await apiService.getReplaySummary({ + organizationSlug, + projectSlugOrId: projectId, + replayId, + }); + + if (summary.status !== "completed") { + return []; + } + + return (summary.data?.time_ranges ?? []).map((range) => ({ + startMs: range.period_start - originMs, + endMs: range.period_end - originMs, + title: range.period_title, + })); + } catch { + return []; + } +} + +/** + * Pair each of the replay's error ids with whatever the lookup resolved. + * + * `error_ids` is the source of truth for which errors this replay has; the + * meta lookup only enriches them with issue identity. An id that fails to + * resolve is still listed by id — dropping it would understate the session, + * and the endpoint is private enough to fail on its own. + */ +function toRelatedIssues( + errorIds: string[], + errorEvents: ReplayErrorEvent[], +): RelatedReplayIssue[] { + const byId = new Map(errorEvents.map((event) => [event.id, event])); + + return errorIds.map((eventId) => { + const event = byId.get(eventId); + return { + eventId, + shortId: event?.issue ?? null, + title: event?.title ?? null, + }; + }); +} + +/** + * Build the suggested follow-up call. + * + * Prefers a window bracketing the replay's first resolvable error, since that + * is almost always what the reader is looking for. Falls back to a + * whole-session digest when no error timestamp resolves — the endpoint is + * private and may be unavailable — so there is always a concrete next call. + */ +function buildNextStepLines({ + organizationSlug, + replayId, + errorEvents, + startedAt, + context, +}: { + organizationSlug: string; + replayId: string; + errorEvents: ReplayErrorEvent[]; + startedAt?: string | null; + context: ServerContext; +}): string[] { + const instruction = formatToolCallInstruction({ + toolName: "get_replay_activity", + experimentalMode: context.experimentalMode ?? false, + availableToolNames: context.availableToolNames, + directToolNames: context.directToolNames, + fallbackInstruction: + "Replay activity lookup is not available in this session", + purpose: "to read the signals in a time window", + }); + + const anchor = findErrorAnchor(errorEvents, startedAt); + if (!anchor) { + return [ + `${instruction}:`, + formatToolCall({ + toolName: "get_replay_activity", + arguments: { organizationSlug, replayId, grain: "digest" }, + }), + ]; + } + + const startMs = Math.max(0, anchor.offsetMs - ERROR_WINDOW_PADDING_MS); + const endMs = anchor.offsetMs + ERROR_WINDOW_PADDING_MS; + + return [ + `${anchor.label} occurred at ${formatReplayOffset(anchor.offsetMs)}. ${instruction}:`, + formatToolCall({ + toolName: "get_replay_activity", + arguments: { + organizationSlug, + replayId, + startMs, + endMs, + grain: "detail", + }, }), - ); + ]; +} + +function findErrorAnchor( + errorEvents: ReplayErrorEvent[], + startedAt?: string | null, +): { offsetMs: number; label: string } | null { + const originMs = startedAt ? Date.parse(startedAt) : Number.NaN; + if (Number.isNaN(originMs)) { + return null; + } + + for (const event of errorEvents) { + if (!event.timestamp) { + continue; + } + const timestampMs = Date.parse(event.timestamp); + if (Number.isNaN(timestampMs)) { + continue; + } + return { + offsetMs: timestampMs - originMs, + label: event.issue ? `Error ${event.issue}` : `Error \`${event.id}\``, + }; + } + + return null; } async function fetchReplayTraces({ @@ -472,155 +871,3 @@ function formatNameVersion( } return name ?? version ?? "Unknown"; } - -function extractReplayActivityEvents( - segments: ReplayRecordingSegments | null, -): ReplayActivityEvent[] { - if (!segments) { - return []; - } - - const events: ReplayActivityEvent[] = []; - - for (const segment of segments) { - for (const event of segment) { - const replayEvent = summarizeReplayEvent(event); - if (replayEvent) { - events.push(replayEvent); - } - if (events.length >= MAX_ACTIVITY_EVENTS) { - return events; - } - } - } - - return events; -} - -function summarizeReplayEvent( - event: ReplayRecordingEvent, -): ReplayActivityEvent | null { - const timestampMs = getEventTimestampMillis(event.timestamp); - const data = event.data; - const tag = data?.tag ?? ""; - const payload = data?.payload ?? null; - - if (tag) { - const replayEvent = summarizeTaggedReplayEvent(tag, payload); - if (replayEvent) { - return { timestampMs, ...replayEvent }; - } - } - - if (event.type !== undefined && data) { - const href = data.href ?? null; - if (href) { - return { - timestampMs, - label: "page.view", - details: [`href=${href}`], - }; - } - } - - return null; -} - -function summarizeTaggedReplayEvent( - tag: string, - payload: ReplayRecordingPayload | null, -): Omit | null { - if (tag === "performanceSpan") { - const op = firstString(payload?.op); - const description = firstString(payload?.description); - const durationMs = payload?.data?.duration ?? null; - - if (description || op) { - return { - label: op ?? "performanceSpan", - details: [ - description ? `description=${description}` : null, - durationMs !== null ? `duration_ms=${durationMs}` : null, - ].filter((value): value is string => value !== null), - }; - } - } - - if (tag === "ui.click") { - const message = firstString(payload?.message, payload?.description); - if (message) { - return { - label: tag, - details: [`message=${quoteDetail(message)}`], - }; - } - } - - const knownKeys = ["message", "description", "category", "type"] as const; - const details: string[] = []; - for (const key of knownKeys) { - const value = firstString(payload?.[key]); - if (value) { - details.push(`${key}=${quoteDetail(value)}`); - } - } - const extra = summarizeObject(payload, new Set(knownKeys)); - if (extra) { - details.push(extra); - } - - return details.length > 0 ? { label: tag, details } : null; -} - -function getEventTimestampMillis(value: unknown): number | null { - if (typeof value !== "number") { - return null; - } - return value > 1e12 ? value : value * 1000; -} - -function formatRelativeTime(offsetMs: number): string { - const offsetSeconds = Math.max(0, Math.round(offsetMs / 1000)); - if (offsetSeconds < 60) { - return `T+${offsetSeconds}s`; - } - - const minutes = Math.floor(offsetSeconds / 60); - const seconds = offsetSeconds % 60; - return seconds > 0 ? `T+${minutes}m ${seconds}s` : `T+${minutes}m`; -} - -function summarizeObject( - value: Record | null, - excludedKeys: ReadonlySet = new Set(), -): string | null { - if (!value) { - return null; - } - - const entries = Object.entries(value) - .filter( - ([key, nested]) => - !excludedKeys.has(key) && - (typeof nested === "string" || typeof nested === "number"), - ) - .slice(0, 3) - .map(([key, nested]) => `${key}=${nested}`); - - return entries.length > 0 - ? `payload=${quoteDetail(entries.join(", "))}` - : null; -} - -function firstString(...values: unknown[]): string | null { - for (const value of values) { - if (typeof value === "string" && value.trim()) { - return value.trim(); - } - } - return null; -} - -function quoteDetail(value: string): string { - return JSON.stringify(value.trim()); -} diff --git a/packages/mcp-core/src/tools/catalog/get-sentry-resource.test.ts b/packages/mcp-core/src/tools/catalog/get-sentry-resource.test.ts index de961d343..36a40b183 100644 --- a/packages/mcp-core/src/tools/catalog/get-sentry-resource.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-sentry-resource.test.ts @@ -360,10 +360,10 @@ describe("get_sentry_resource", () => { expect(result).toContain( `# Replay ${replayDetailsFixture.id} in **sentry-mcp-evals**`, ); - // Assert the recording was read, not how a signal is phrased — the - // activity rendering is being replaced (docs/specs/replay-review.md). - expect(result).toContain("## Activity"); - expect(result).toContain("button#sign-in"); + // Assert the recording was read and mapped, not how a signal is + // phrased — that belongs to the replay tool's own tests. + expect(result).toContain("## Map"); + expect(result).toContain("**Signals**:"); }); it("dispatches monitor URL with a simple slug to get_monitor_details", async () => { @@ -719,8 +719,8 @@ describe("get_sentry_resource", () => { expect(result).toContain( `# Replay ${replayDetailsFixture.id} in **sentry-mcp-evals**`, ); - expect(result).toContain("## Activity"); - expect(result).toContain("button#sign-in"); + expect(result).toContain("## Map"); + expect(result).toContain("**Signals**:"); }); it("fetches snapshot by snapshot ID", async () => { From 3333e5729ae34770a2df61444bb628bc4c866b4a Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 14:29:02 -0700 Subject: [PATCH 06/26] feat(replays): Add get_replay_activity for windowed replay reads The replay map tells an agent where to look but gives it no way to look there. Add a catalog-only tool that returns the signals in a time window at a requested level of detail, completing the map-then-zoom pair: the map's suggested call now resolves. Windowing is by `startMs`/`endMs` measured from the replay's `started_at`, defaulting to the whole session. Bounds are inclusive. A signal whose timestamp could not be resolved is excluded from a windowed read rather than guessed into one, and stays available in a whole-session read. `grain` controls rendering and `kinds` controls inclusion, kept orthogonal so a digest of a filtered window stays consistent with the same window read at standard grain. Paging is by `limit`/`cursor`. Sentry paginates recording segments rather than signals, so there is no server-side signal cursor to pass through: each page re-reads the recording and skips a signal offset. The cursor is therefore synthetic and encodes the window and kind filter alongside that offset, so a continuation resumes the same query rather than a differently-filtered one. It also overrides any window or filter passed beside it, since honouring both would silently change what the caller is paging through. Truncation always reports the cursor needed to continue, and a partial recording read is reported separately from a paged result. Extract replay parameter resolution and the project-constraint check into a shared helper so the two replay tools cannot disagree about what a valid replay reference is, or about which replays a constrained session may read. The direct tool surface is unchanged: this is catalog-only, and `measure-tokens` still reports 5,598 tokens across 9 direct tools. Refs docs/specs/replay-review.md Co-Authored-By: Claude Opus 5 (1M context) --- .../changes/improve-replay-review/tasks.md | 16 +- .../mcp-core/src/internal/replay-events.ts | 35 +- .../src/internal/tool-helpers/replay.ts | 90 +++++ packages/mcp-core/src/server.test.ts | 6 +- packages/mcp-core/src/skillDefinitions.json | 7 +- packages/mcp-core/src/toolDefinitions.json | 86 +++++ .../tools/catalog/get-replay-activity.test.ts | 271 ++++++++++++++ .../src/tools/catalog/get-replay-activity.ts | 346 ++++++++++++++++++ .../tools/catalog/get-replay-details.test.ts | 3 +- .../src/tools/catalog/get-replay-details.ts | 74 +--- packages/mcp-core/src/tools/catalog/index.ts | 2 + 11 files changed, 839 insertions(+), 97 deletions(-) create mode 100644 packages/mcp-core/src/internal/tool-helpers/replay.ts create mode 100644 packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts create mode 100644 packages/mcp-core/src/tools/catalog/get-replay-activity.ts diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index 8f84bb460..c4a70066a 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -41,13 +41,13 @@ baselines encode the same wrong event shape they do today. ## 5. Replay Activity Tool -- [ ] 5.1 Add `get_replay_activity` as a catalog-only `inspect` tool with `requiredCapabilities: ["replays"]` and replay read scopes. -- [ ] 5.2 Implement `startMs`/`endMs` windowing, defaulting to the whole session. -- [ ] 5.3 Implement `grain` and the optional `kinds` allow-list. -- [ ] 5.4 Implement `limit`/`cursor` paging with explicit truncation reporting, encoding the window, `kinds`, and offset in the synthetic cursor so continuation is stable. -- [ ] 5.5 Accept `replayUrl` as well as `organizationSlug` plus `replayId`, reusing the existing parameter resolution and constraint checks. -- [ ] 5.6 Register in the catalog and confirm it is not added to the direct top-level surface. -- [ ] 5.7 Record telemetry on the existing span: grain requested, window width, and result counts. +- [x] 5.1 Add `get_replay_activity` as a catalog-only `inspect` tool with `requiredCapabilities: ["replays"]` and replay read scopes. +- [x] 5.2 Implement `startMs`/`endMs` windowing, defaulting to the whole session. Bounds are inclusive; a signal whose timestamp did not resolve is excluded from a windowed read rather than guessed into one. +- [x] 5.3 Implement `grain` and the optional `kinds` allow-list. +- [x] 5.4 Implement `limit`/`cursor` paging with explicit truncation reporting, encoding the window, `kinds`, and offset in the synthetic cursor so continuation is stable. A cursor fully describes its query and overrides any window or filter passed alongside it, verified by mutation. +- [x] 5.5 Accept `replayUrl` as well as `organizationSlug` plus `replayId`, reusing the existing parameter resolution and constraint checks. Both were extracted to `internal/tool-helpers/replay.ts` so the two replay tools cannot diverge. +- [x] 5.6 Register in the catalog and confirm it is not added to the direct top-level surface. `pnpm run measure-tokens` is unchanged at 5,598 tokens across 9 direct tools. +- [x] 5.7 Record telemetry on the existing span: grain requested, window width, and result counts. ## 6. Adjacent Correctness Fixes @@ -59,7 +59,7 @@ baselines encode the same wrong event shape they do today. ## 7. Tests - [x] 7.1 Re-baseline `get-replay-details.test.ts` snapshots against the rebuilt fixtures. -- [ ] 7.2 Add `get-replay-activity.test.ts` covering windowing, each grain, kind filtering, paging, and constraint injection. +- [x] 7.2 Add `get-replay-activity.test.ts` covering windowing, each grain, kind filtering, paging, and constraint injection. - [ ] 7.3 Add degradation tests: summary 403, summary `processing`, summary timeout, summary unparseable, segment fetch 404, archived replay, zero segments, multi-page segments, and the segment budget being hit. - [ ] 7.4 Add redaction tests asserting `` and `[Filtered]`-driven `` rendering. - [ ] 7.5 Add search-events tests for the replay capability gate, the dataset options omitting `replays`, and the reconciled sort list. diff --git a/packages/mcp-core/src/internal/replay-events.ts b/packages/mcp-core/src/internal/replay-events.ts index 56b7095a2..e4586ec06 100644 --- a/packages/mcp-core/src/internal/replay-events.ts +++ b/packages/mcp-core/src/internal/replay-events.ts @@ -82,22 +82,25 @@ export type ReplayEventType = * `device` — because callers filter by what happened, not by which SDK API * reported it. */ -export type ReplaySignalKind = - | "navigation" - | "click" - | "dead-click" - | "rage-click" - | "slow-click" - | "network" - | "console" - | "hydration-error" - | "feedback" - | "web-vital" - | "tap" - | "scroll" - | "swipe" - | "app-lifecycle" - | "device"; +export const REPLAY_SIGNAL_KINDS = [ + "navigation", + "click", + "dead-click", + "rage-click", + "slow-click", + "network", + "console", + "hydration-error", + "feedback", + "web-vital", + "tap", + "scroll", + "swipe", + "app-lifecycle", + "device", +] as const; + +export type ReplaySignalKind = (typeof REPLAY_SIGNAL_KINDS)[number]; export type ReplayGrain = "digest" | "standard" | "detail"; diff --git a/packages/mcp-core/src/internal/tool-helpers/replay.ts b/packages/mcp-core/src/internal/tool-helpers/replay.ts new file mode 100644 index 000000000..2da38cb32 --- /dev/null +++ b/packages/mcp-core/src/internal/tool-helpers/replay.ts @@ -0,0 +1,90 @@ +/** + * Shared replay parameter resolution and constraint checks. + * + * Both replay tools accept either a replay URL or an organization plus replay + * ID, and both must honour a session's project constraint. Keeping that in one + * place means the two cannot disagree about what a valid replay reference is, + * or about which replays a constrained session may read. + */ +import type { ReplayDetails, SentryApiService } from "../../api-client"; +import { UserInputError } from "../../errors"; +import { parseSentryUrl } from "../url-helpers"; +import { resolveScopedOrganizationSlug } from "../url-scope"; + +export interface ResolvedReplayParams { + organizationSlug: string; + replayId: string; +} + +export function resolveReplayParams(params: { + replayUrl?: string | null; + organizationSlug?: string | null; + replayId?: string | null; +}): ResolvedReplayParams { + if (params.replayUrl) { + const parsed = parseSentryUrl(params.replayUrl); + if (parsed.type !== "replay" || !parsed.replayId) { + throw new UserInputError( + "Invalid replay URL. URL must point to a Sentry replay resource.", + ); + } + return { + organizationSlug: resolveScopedOrganizationSlug({ + resourceLabel: "Replay", + scopedOrganizationSlug: params.organizationSlug, + urlOrganizationSlug: parsed.organizationSlug, + }), + replayId: parsed.replayId, + }; + } + + if (!params.organizationSlug || !params.replayId) { + throw new UserInputError( + "Provide either `replayUrl` or both `organizationSlug` and `replayId`.", + ); + } + + return { + organizationSlug: params.organizationSlug, + replayId: params.replayId, + }; +} + +/** + * Reject a replay that falls outside the session's project constraint. + * + * A replay with no project cannot be shown to satisfy the constraint, so it is + * rejected rather than assumed to be in scope. + */ +export async function assertReplayWithinProjectConstraint({ + apiService, + organizationSlug, + replay, + projectSlug, +}: { + apiService: SentryApiService; + organizationSlug: string; + replay: ReplayDetails; + projectSlug?: string | null; +}): Promise { + if (!projectSlug) { + return; + } + + if (replay.project_id == null) { + throw new UserInputError( + `Replay is outside the active project constraint. Expected project "${projectSlug}".`, + ); + } + + const project = await apiService.getProject({ + organizationSlug, + projectSlugOrId: projectSlug, + }); + + if (String(project.id) !== String(replay.project_id)) { + throw new UserInputError( + `Replay is outside the active project constraint. Expected project "${projectSlug}".`, + ); + } +} diff --git a/packages/mcp-core/src/server.test.ts b/packages/mcp-core/src/server.test.ts index 59a5ac5e0..ea5127032 100644 --- a/packages/mcp-core/src/server.test.ts +++ b/packages/mcp-core/src/server.test.ts @@ -1024,7 +1024,11 @@ describe("buildServer", () => { ); expect(getTextContent(result)).not.toContain("# Tool Search Results"); expect(getTextContent(result)).not.toContain("```json"); - expect(firstResult?.name).toBe("get_replay_details"); + // This test is about the shape of a search result and the hiding of + // constraint-injected parameters, not about which replay tool ranks + // first, so it asserts the match is a replay tool rather than pinning + // the ranking. + expect(firstResult?.name).toMatch(/^get_replay_/); expect(Object.keys(firstResult ?? {}).sort()).toEqual([ "annotations", "description", diff --git a/packages/mcp-core/src/skillDefinitions.json b/packages/mcp-core/src/skillDefinitions.json index 3d833c054..51c1e7c2d 100644 --- a/packages/mcp-core/src/skillDefinitions.json +++ b/packages/mcp-core/src/skillDefinitions.json @@ -5,7 +5,7 @@ "description": "Read-only access to core Sentry data: issues, events, traces, replays, releases, cron monitors, uptime monitors, profiles, documentation, and project metadata", "defaultEnabled": true, "order": 1, - "toolCount": 37, + "toolCount": 38, "tools": [ { "name": "find_alert_rules", @@ -127,6 +127,11 @@ "description": "Get details for a Sentry release.\n\nUse this tool when you need to:\n- Inspect an exact release version\n- Scope a release lookup to a specific project\n- See deploys and environments for a release\n- See recent commits attached to a release\n- Gather release health metadata when a project is known\n\n\nget_release_details(organizationSlug='my-organization', releaseVersion='1.2.3')\nget_release_details(organizationSlug='my-organization', releaseVersion='1.2.3', projectSlugOrId='backend')\nget_release_details(organizationSlug='my-organization', releaseVersion='1.2.3', includeHealth=true, projectSlugOrId='backend')\n", "requiredScopes": ["project:read"] }, + { + "name": "get_replay_activity", + "description": "Read what happened during a window of a Sentry replay session.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what happened around a specific moment in a replay\n- Need the requests, clicks, or console output behind a replay failure\n- Want more or less detail than the replay map provides\n\nCall `get_replay_details` first for the session map; it suggests a window.\nOffsets are milliseconds from the start of the replay.\n\n\n### Zoom into a failure\n```\nget_replay_activity(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', startMs=306000, endMs=316000, grain='detail')\n```\n\n### Cheap shape check on a long session\n```\nget_replay_activity(replayUrl='https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/', grain='digest', kinds=['network','console'])\n```\n", + "requiredScopes": ["org:read", "project:read", "event:read"] + }, { "name": "get_replay_details", "description": "Get high-level information about a specific Sentry replay by URL or replay ID.\n\nUSE THIS TOOL WHEN USERS:\n- Share a replay URL\n- Ask what happened in a specific replay\n- Want a concise replay summary plus the next issue or trace lookups to run\n\n\n### With replay URL\n```\nget_replay_details(replayUrl='https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/')\n```\n\n### With organization and replay ID\n```\nget_replay_details(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f')\n```\n", diff --git a/packages/mcp-core/src/toolDefinitions.json b/packages/mcp-core/src/toolDefinitions.json index 1bb49a1f4..bbcf60c85 100644 --- a/packages/mcp-core/src/toolDefinitions.json +++ b/packages/mcp-core/src/toolDefinitions.json @@ -3481,6 +3481,92 @@ "skills": ["inspect"], "surface": "catalog" }, + { + "name": "get_replay_activity", + "description": "Read what happened during a window of a Sentry replay session.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what happened around a specific moment in a replay\n- Need the requests, clicks, or console output behind a replay failure\n- Want more or less detail than the replay map provides\n\nCall `get_replay_details` first for the session map; it suggests a window.\nOffsets are milliseconds from the start of the replay.\n\n\n### Zoom into a failure\n```\nget_replay_activity(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', startMs=306000, endMs=316000, grain='detail')\n```\n\n### Cheap shape check on a long session\n```\nget_replay_activity(replayUrl='https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/', grain='digest', kinds=['network','console'])\n```\n", + "inputSchema": { + "type": "object", + "properties": { + "replayUrl": { + "type": "string", + "format": "uri", + "description": "The URL of the replay. e.g. https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/" + }, + "organizationSlug": { + "type": "string", + "description": "The organization's slug. You can find a existing list of organizations you have access to using the `find_organizations()` tool." + }, + "replayId": { + "type": "string", + "description": "The replay ID. e.g. `7e07485f-12f9-416b-8b14-26260799b51f`" + }, + "regionUrl": { + "anyOf": [ + { + "type": "string", + "description": "The region URL for the organization you're querying, if known. For Sentry's Cloud Service (sentry.io), this is typically the region-specific URL like 'https://us.sentry.io'. For self-hosted Sentry installations, this parameter is usually not needed and should be omitted. You can find the correct regionUrl from the organization details using the `find_organizations()` tool." + }, + { + "type": "null" + } + ] + }, + "startMs": { + "description": "Window start, in milliseconds from the start of the replay. Omit for the whole session.", + "type": "number", + "minimum": 0 + }, + "endMs": { + "description": "Window end, in milliseconds from the start of the replay. Omit for the whole session.", + "type": "number", + "minimum": 0 + }, + "grain": { + "default": "standard", + "description": "How much to render per signal: `digest` is one rollup line per kind, `standard` one line per signal, `detail` adds payload such as status codes and durations.", + "type": "string", + "enum": ["digest", "standard", "detail"] + }, + "kinds": { + "description": "Only return these kinds of signal. Omit to include everything.", + "type": "array", + "items": { + "type": "string", + "enum": [ + "navigation", + "click", + "dead-click", + "rage-click", + "slow-click", + "network", + "console", + "hydration-error", + "feedback", + "web-vital", + "tap", + "scroll", + "swipe", + "app-lifecycle", + "device" + ] + } + }, + "limit": { + "default": 50, + "type": "number", + "minimum": 1, + "maximum": 200 + }, + "cursor": { + "description": "Continue a truncated result, using the cursor it returned.", + "type": "string" + } + } + }, + "requiredScopes": ["org:read", "project:read", "event:read"], + "skills": ["inspect"], + "surface": "catalog" + }, { "name": "get_replay_details", "description": "Get high-level information about a specific Sentry replay by URL or replay ID.\n\nUSE THIS TOOL WHEN USERS:\n- Share a replay URL\n- Ask what happened in a specific replay\n- Want a concise replay summary plus the next issue or trace lookups to run\n\n\n### With replay URL\n```\nget_replay_details(replayUrl='https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/')\n```\n\n### With organization and replay ID\n```\nget_replay_details(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f')\n```\n", diff --git a/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts new file mode 100644 index 000000000..792b309d7 --- /dev/null +++ b/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts @@ -0,0 +1,271 @@ +import { afterEach, describe, expect, it } from "vitest"; +import { http, HttpResponse } from "msw"; +import { mswServer, replayDetailsFixture } from "@sentry/mcp-server-mocks"; +import getReplayActivity from "./get-replay-activity.js"; +import { getServerContext } from "../../test-setup.js"; + +const REPLAY_URL = `https://us.sentry.io/api/0/organizations/sentry-mcp-evals/replays/${replayDetailsFixture.id}/`; + +function callTool( + params: Record = {}, + context = getServerContext(), +) { + return getReplayActivity.handler( + { + organizationSlug: "sentry-mcp-evals", + replayId: replayDetailsFixture.id, + regionUrl: "https://us.sentry.io", + grain: "standard", + limit: 50, + ...params, + } as never, + context, + ); +} + +/** Pull the cursor out of a truncated result so paging can be followed. */ +function cursorFrom(output: string): string { + const match = output.match(/cursor='([^']+)'/); + if (!match) { + throw new Error(`expected a cursor in output:\n${output}`); + } + return match[1]; +} + +afterEach(() => { + mswServer.resetHandlers(); +}); + +describe("get_replay_activity", () => { + it("returns the whole session when no window is given", async () => { + const result = await callTool(); + + expect(result).toMatchInlineSnapshot(` + "# Replay 7e07485f-12f9-416b-8b14-26260799b51f activity + + Window: whole session + + T+0.5s navigation Navigated to example.com/login + T+12.4s click Clicked body > div#root > form#login > button#sign-in + T+14.2s navigation Navigated to example.com/checkout + T+3m 0.6s click Clicked body > div#root > main > button#complete-order + T+3m 1.0s network Fetch POST example.com/api/checkout failed with 500 + T+3m 1.3s console Console error: TypeError: Cannot read properties of undefined (reading 'id') + T+3m 8.4s rage-click Rage click on body > div#root > main > button#complete-order + T+3m 41.7s dead-click Dead click — no response from body > div#root > main > a#download-receipt" + `); + }); + + it("returns only signals inside the requested window", async () => { + // The window suggested by get_replay_details for the fixture's error. + const result = await callTool({ startMs: 176300, endMs: 186300 }); + + expect(result).toContain("Window: T+2m 56.3s–T+3m 6.3s"); + expect(result).toContain( + "Clicked body > div#root > main > button#complete-order", + ); + expect(result).toContain( + "Fetch POST example.com/api/checkout failed with 500", + ); + expect(result).toContain("Console error: TypeError"); + // Outside the window on either side. + expect(result).not.toContain("button#sign-in"); + expect(result).not.toContain("a#download-receipt"); + }); + + it("treats window bounds as inclusive", async () => { + const result = await callTool({ startMs: 181300, endMs: 181300 }); + + expect(result).toContain("Console error: TypeError"); + expect(result).not.toContain("network"); + }); + + it("rejects a window that ends before it starts", async () => { + await expect(callTool({ startMs: 5000, endMs: 1000 })).rejects.toThrow( + "`endMs` must be greater than or equal to `startMs`.", + ); + }); + + describe("grain", () => { + it("rolls up one line per kind at digest grain", async () => { + const result = await callTool({ grain: "digest" }); + + expect(result).toContain("navigation ×2"); + expect(result).toContain("network ×1 (1 failed)"); + expect(result).toContain("console ×1 (1 failed)"); + }); + + it("adds payload lines at detail grain", async () => { + const result = await callTool({ + grain: "detail", + kinds: ["network"], + }); + + // The SDK reported body sizes but did not capture the bodies + // themselves, which is the common case since networkCaptureBodies is + // opt-in. Saying so beats silence, which would imply we could have + // retrieved them. + expect(result).toContain("request body: 214 bytes "); + expect(result).toContain("response body: 87 bytes "); + }); + + it("omits payload lines at standard grain", async () => { + const result = await callTool({ kinds: ["network"] }); + + expect(result).not.toContain("request body:"); + }); + }); + + describe("kind filtering", () => { + it("returns only the requested kinds", async () => { + const result = await callTool({ kinds: ["console", "network"] }); + + expect(result).toContain("Console error: TypeError"); + expect(result).toContain("Fetch POST example.com/api/checkout"); + expect(result).not.toContain("Clicked "); + expect(result).not.toContain("Navigated to"); + }); + + it("distinguishes rage and dead clicks from ordinary clicks", async () => { + const result = await callTool({ kinds: ["rage-click"] }); + + expect(result).toContain("Rage click on"); + expect(result).not.toContain("Dead click"); + expect(result).not.toContain("Clicked body"); + }); + + it("reports an empty result rather than pretending nothing happened", async () => { + const result = await callTool({ kinds: ["feedback"] }); + + expect(result).toContain("No signals matched."); + }); + }); + + describe("paging", () => { + it("states truncation and returns a usable cursor", async () => { + const result = await callTool({ limit: 3 }); + + expect(result).toContain("Showing 1–3 of 8 matching signals"); + expect(result).toContain("cursor='"); + }); + + it("continues from the cursor without repeating or skipping signals", async () => { + const first = await callTool({ limit: 3 }); + const second = await callTool({ cursor: cursorFrom(first), limit: 3 }); + const third = await callTool({ cursor: cursorFrom(second), limit: 3 }); + + expect(second).toContain("Showing 4–6 of 8"); + expect(third).toContain("T+3m 41.7s"); + // The last page is complete, so it offers no further cursor. + expect(third).not.toContain("cursor='"); + }); + + it("carries the window and kind filter through the cursor", async () => { + // Each page re-reads the recording, so a cursor that did not encode the + // query would silently continue a different one. + const first = await callTool({ + kinds: ["click", "rage-click", "dead-click"], + limit: 1, + }); + const second = await callTool({ cursor: cursorFrom(first), limit: 1 }); + + expect(second).toContain("kinds: click, rage-click, dead-click"); + expect(second).toContain("Showing 2–2 of 4"); + expect(second).not.toContain("Navigated to"); + }); + + it("ignores window and kind arguments passed alongside a cursor", async () => { + // Honouring both would silently change what the caller is paging through. + const first = await callTool({ kinds: ["click"], limit: 1 }); + const second = await callTool({ + cursor: cursorFrom(first), + kinds: ["network"], + limit: 1, + }); + + expect(second).toContain("kinds: click"); + expect(second).not.toContain("Fetch POST"); + }); + + it("rejects a malformed cursor", async () => { + await expect(callTool({ cursor: "not-a-cursor" })).rejects.toThrow( + "Invalid `cursor`", + ); + }); + }); + + describe("degradation", () => { + it("reports an archived replay instead of an empty window", async () => { + mswServer.use( + http.get(REPLAY_URL, () => + HttpResponse.json({ + data: { ...replayDetailsFixture, is_archived: true }, + }), + ), + ); + + const result = await callTool(); + + expect(result).toContain("Recording is archived"); + }); + + it("reports a replay with no recording segments", async () => { + mswServer.use( + http.get(REPLAY_URL, () => + HttpResponse.json({ + data: { ...replayDetailsFixture, count_segments: 0 }, + }), + ), + ); + + const result = await callTool(); + + expect(result).toContain("No recording segments are available"); + }); + + it("rejects a replay outside the active project constraint", async () => { + // The constrained project resolves to a different id than the replay's, + // so the replay is out of scope for this session. + mswServer.use( + http.get( + "https://us.sentry.io/api/0/projects/sentry-mcp-evals/frontend/", + () => + HttpResponse.json({ + id: "9999999999999999", + slug: "frontend", + name: "frontend", + }), + ), + ); + + await expect( + callTool( + {}, + getServerContext({ constraints: { projectSlug: "frontend" } }), + ), + ).rejects.toThrow("outside the active project constraint"); + }); + }); + + describe("tool definition", () => { + it("is gated like get_replay_details", async () => { + expect(getReplayActivity.requiredScopes).toEqual([ + "org:read", + "project:read", + "event:read", + ]); + expect(getReplayActivity.requiredCapabilities).toEqual(["replays"]); + expect(getReplayActivity.skills).toEqual(["inspect"]); + }); + + it("stays off the direct top-level surface", async () => { + const { isDefaultTopLevelToolName } = await import("../surfaces.js"); + expect(isDefaultTopLevelToolName("get_replay_activity")).toBe(false); + }); + + it("is reachable through the catalog", async () => { + const { default: catalog } = await import("./index.js"); + expect(catalog.get_replay_activity).toBe(getReplayActivity); + }); + }); +}); diff --git a/packages/mcp-core/src/tools/catalog/get-replay-activity.ts b/packages/mcp-core/src/tools/catalog/get-replay-activity.ts new file mode 100644 index 000000000..de02c8e71 --- /dev/null +++ b/packages/mcp-core/src/tools/catalog/get-replay-activity.ts @@ -0,0 +1,346 @@ +import { getActiveSpan, setTag } from "@sentry/core"; +import type { ReplayRecordingSegmentsResult } from "../../api-client"; +import { + MAX_REPLAY_SEGMENTS, + MAX_REPLAY_SEGMENT_BYTES, +} from "../../api-client"; +import type { ReplayGrain, ReplaySignal } from "../../internal/replay-events"; +import { + REPLAY_SIGNAL_KINDS, + extractReplaySignals, + formatReplayOffset, + renderReplaySignals, +} from "../../internal/replay-events"; +import { defineTool } from "../../internal/tool-helpers/define"; +import { apiServiceFromContext } from "../../internal/tool-helpers/api"; +import { + assertReplayWithinProjectConstraint, + resolveReplayParams, +} from "../../internal/tool-helpers/replay"; +import { resolveRegionUrlForOrganization } from "../../internal/tool-helpers/resolve-region-url"; +import { UserInputError } from "../../errors"; +import type { ServerContext } from "../../types"; +import { z } from "zod"; +import { + ParamOrganizationSlug, + ParamReplayId, + ParamRegionUrl, + ParamReplayUrl, +} from "../../schema"; + +/** + * Encodes the query a cursor continues. + * + * Sentry paginates recording segments, not signals, so there is no server-side + * signal cursor to pass through. This cursor is synthetic: each page re-reads + * the recording and skips `offset` matching signals. It carries the window and + * the kind filter so that continuing a page yields a stable continuation of the + * same query rather than a differently-filtered one. + */ +interface ActivityCursor { + startMs?: number; + endMs?: number; + kinds?: string[]; + offset: number; +} + +const DEFAULT_LIMIT = 50; + +export default defineTool({ + name: "get_replay_activity", + skills: ["inspect"], + requiredScopes: ["org:read", "project:read", "event:read"], + requiredCapabilities: ["replays"], + description: [ + "Read what happened during a window of a Sentry replay session.", + "", + "USE THIS TOOL WHEN USERS:", + "- Ask what happened around a specific moment in a replay", + "- Need the requests, clicks, or console output behind a replay failure", + "- Want more or less detail than the replay map provides", + "", + "Call `get_replay_details` first for the session map; it suggests a window.", + "Offsets are milliseconds from the start of the replay.", + "", + "", + "### Zoom into a failure", + "```", + "get_replay_activity(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', startMs=306000, endMs=316000, grain='detail')", + "```", + "", + "### Cheap shape check on a long session", + "```", + "get_replay_activity(replayUrl='https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/', grain='digest', kinds=['network','console'])", + "```", + "", + ].join("\n"), + inputSchema: { + replayUrl: ParamReplayUrl.optional(), + organizationSlug: ParamOrganizationSlug.optional(), + replayId: ParamReplayId.optional(), + regionUrl: ParamRegionUrl.nullable().optional(), + startMs: z + .number() + .min(0) + .optional() + .describe( + "Window start, in milliseconds from the start of the replay. Omit for the whole session.", + ), + endMs: z + .number() + .min(0) + .optional() + .describe( + "Window end, in milliseconds from the start of the replay. Omit for the whole session.", + ), + grain: z + .enum(["digest", "standard", "detail"]) + .default("standard") + .describe( + "How much to render per signal: `digest` is one rollup line per kind, `standard` one line per signal, `detail` adds payload such as status codes and durations.", + ), + kinds: z + .array(z.enum(REPLAY_SIGNAL_KINDS)) + .optional() + .describe( + "Only return these kinds of signal. Omit to include everything.", + ), + limit: z.number().min(1).max(200).default(DEFAULT_LIMIT), + cursor: z + .string() + .optional() + .describe("Continue a truncated result, using the cursor it returned."), + }, + annotations: { + readOnlyHint: true, + destructiveHint: false, + openWorldHint: true, + }, + async handler(params, context: ServerContext) { + const resolved = resolveReplayParams(params); + const regionUrl = await resolveRegionUrlForOrganization({ + context, + organizationSlug: resolved.organizationSlug, + regionUrl: params.regionUrl, + }); + const apiService = apiServiceFromContext(context, { + regionUrl: regionUrl ?? undefined, + }); + + setTag("organization.slug", resolved.organizationSlug); + setTag("replay.id", resolved.replayId); + + // A cursor fully describes the query it continues, so it overrides any + // window or filter passed alongside it. Honouring both would silently + // change what the caller is paging through. + const cursor = params.cursor ? decodeCursor(params.cursor) : null; + const startMs = cursor ? cursor.startMs : params.startMs; + const endMs = cursor ? cursor.endMs : params.endMs; + const kinds = cursor ? cursor.kinds : params.kinds; + const offset = cursor?.offset ?? 0; + + if (startMs !== undefined && endMs !== undefined && endMs < startMs) { + throw new UserInputError( + "`endMs` must be greater than or equal to `startMs`.", + ); + } + + const replay = await apiService.getReplayDetails({ + organizationSlug: resolved.organizationSlug, + replayId: resolved.replayId, + }); + await assertReplayWithinProjectConstraint({ + apiService, + organizationSlug: resolved.organizationSlug, + replay, + projectSlug: context.constraints.projectSlug, + }); + + if (replay.is_archived === true) { + return `# Replay ${replay.id} activity\n\nRecording is archived and not available for playback.`; + } + + const projectId = + replay.project_id != null ? String(replay.project_id) : null; + if (!projectId || (replay.count_segments ?? 0) === 0) { + return `# Replay ${replay.id} activity\n\nNo recording segments are available for this replay.`; + } + + const recording = await apiService.getReplayRecordingSegments({ + organizationSlug: resolved.organizationSlug, + projectSlugOrId: projectId, + replayId: resolved.replayId, + }); + + const allSignals = extractReplaySignals(recording.segments, { + startedAt: replay.started_at, + platform: replay.platform, + }); + const matching = allSignals.filter((signal) => + matchesQuery(signal, { startMs, endMs, kinds }), + ); + const page = matching.slice(offset, offset + params.limit); + + const span = getActiveSpan(); + span?.setAttribute("replay.grain", params.grain); + span?.setAttribute("replay.kinds", (kinds ?? []).join(",") || "all"); + span?.setAttribute( + "replay.window_ms", + startMs !== undefined || endMs !== undefined + ? (endMs ?? Number.POSITIVE_INFINITY) - (startMs ?? 0) + : -1, + ); + span?.setAttribute("replay.signals_matched", matching.length); + span?.setAttribute("gen_ai.tool.call.result.count", page.length); + + return formatActivityOutput({ + replayId: replay.id, + signals: page, + matchedCount: matching.length, + offset, + limit: params.limit, + grain: params.grain, + startMs, + endMs, + kinds, + truncatedBy: recording.truncatedBy, + }); + }, +}); + +function matchesQuery( + signal: ReplaySignal, + { + startMs, + endMs, + kinds, + }: { startMs?: number; endMs?: number; kinds?: string[] }, +): boolean { + if (kinds && kinds.length > 0 && !kinds.includes(signal.kind)) { + return false; + } + + if (startMs === undefined && endMs === undefined) { + return true; + } + + // A signal whose timestamp could not be resolved cannot be placed in a + // window, so it is excluded from windowed queries rather than guessed into + // one. It remains available in a whole-session read. + if (signal.offsetMs === null) { + return false; + } + + if (startMs !== undefined && signal.offsetMs < startMs) { + return false; + } + if (endMs !== undefined && signal.offsetMs > endMs) { + return false; + } + return true; +} + +function formatActivityOutput({ + replayId, + signals, + matchedCount, + offset, + limit, + grain, + startMs, + endMs, + kinds, + truncatedBy, +}: { + replayId: string; + signals: ReplaySignal[]; + matchedCount: number; + offset: number; + limit: number; + grain: ReplayGrain; + startMs?: number; + endMs?: number; + kinds?: string[]; + truncatedBy: ReplayRecordingSegmentsResult["truncatedBy"]; +}): string { + const lines: string[] = []; + const window = + startMs !== undefined || endMs !== undefined + ? `${formatReplayOffset(startMs ?? 0)}–${endMs !== undefined ? formatReplayOffset(endMs) : "end"}` + : "whole session"; + + lines.push(`# Replay ${replayId} activity`); + lines.push(""); + lines.push( + `Window: ${window}${kinds?.length ? ` · kinds: ${kinds.join(", ")}` : ""}`, + ); + lines.push(""); + + if (signals.length === 0) { + lines.push( + matchedCount === 0 + ? "No signals matched." + : "No further signals in this window.", + ); + } else { + lines.push(...renderReplaySignals(signals, grain)); + } + + // Truncation is always stated, and always with the means to continue. + const nextOffset = offset + signals.length; + if (nextOffset < matchedCount) { + lines.push(""); + lines.push( + `Showing ${offset + 1}–${nextOffset} of ${matchedCount} matching signals. Continue with:`, + ); + lines.push( + `cursor='${encodeCursor({ startMs, endMs, kinds, offset: nextOffset })}'`, + ); + } + + // A partial recording read is a different kind of gap from a paged result, + // and hiding it would make a truncated session look complete. + if (truncatedBy === "segments") { + lines.push(""); + lines.push( + `Note: the recording was read up to the first ${MAX_REPLAY_SEGMENTS} segments; later activity is not included.`, + ); + } else if (truncatedBy === "bytes") { + lines.push(""); + lines.push( + `Note: the recording was read up to ${MAX_REPLAY_SEGMENT_BYTES / (1024 * 1024)}MB; later activity is not included.`, + ); + } + + return lines.join("\n"); +} + +function encodeCursor(cursor: ActivityCursor): string { + return Buffer.from(JSON.stringify(cursor), "utf8").toString("base64url"); +} + +function decodeCursor(value: string): ActivityCursor { + let parsed: unknown; + try { + parsed = JSON.parse(Buffer.from(value, "base64url").toString("utf8")); + } catch { + throw new UserInputError( + "Invalid `cursor`. Use the cursor returned by a previous call.", + ); + } + + const result = CursorSchema.safeParse(parsed); + if (!result.success) { + throw new UserInputError( + "Invalid `cursor`. Use the cursor returned by a previous call.", + ); + } + return result.data; +} + +const CursorSchema = z.object({ + startMs: z.number().min(0).optional(), + endMs: z.number().min(0).optional(), + kinds: z.array(z.string()).optional(), + offset: z.number().min(0), +}); diff --git a/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts index 8b568db89..fb525e69f 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts @@ -5,7 +5,8 @@ import { organizationFixture, replayDetailsFixture, } from "@sentry/mcp-server-mocks"; -import getReplayDetails, { resolveReplayParams } from "./get-replay-details.js"; +import getReplayDetails from "./get-replay-details.js"; +import { resolveReplayParams } from "../../internal/tool-helpers/replay.js"; import { getServerContext } from "../../test-setup.js"; describe("get_replay_details", () => { diff --git a/packages/mcp-core/src/tools/catalog/get-replay-details.ts b/packages/mcp-core/src/tools/catalog/get-replay-details.ts index 99d02c990..07b7b6ee1 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-details.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-details.ts @@ -25,12 +25,13 @@ import { formatToolCall, formatToolCallInstruction, } from "../../internal/tool-helpers/tool-call-formatting"; +import { + assertReplayWithinProjectConstraint, + resolveReplayParams, +} from "../../internal/tool-helpers/replay"; import { defineTool } from "../../internal/tool-helpers/define"; import { apiServiceFromContext } from "../../internal/tool-helpers/api"; import { resolveRegionUrlForOrganization } from "../../internal/tool-helpers/resolve-region-url"; -import { parseSentryUrl } from "../../internal/url-helpers"; -import { resolveScopedOrganizationSlug } from "../../internal/url-scope"; -import { UserInputError } from "../../errors"; import type { ServerContext } from "../../types"; import { ParamOrganizationSlug, @@ -209,73 +210,6 @@ export default defineTool({ }, }); -export function resolveReplayParams(params: { - replayUrl?: string | null; - organizationSlug?: string | null; - replayId?: string | null; -}): ResolvedReplayParams { - if (params.replayUrl) { - const parsed = parseSentryUrl(params.replayUrl); - if (parsed.type !== "replay" || !parsed.replayId) { - throw new UserInputError( - "Invalid replay URL. URL must point to a Sentry replay resource.", - ); - } - return { - organizationSlug: resolveScopedOrganizationSlug({ - resourceLabel: "Replay", - scopedOrganizationSlug: params.organizationSlug, - urlOrganizationSlug: parsed.organizationSlug, - }), - replayId: parsed.replayId, - }; - } - - if (!params.organizationSlug || !params.replayId) { - throw new UserInputError( - "Provide either `replayUrl` or both `organizationSlug` and `replayId`.", - ); - } - - return { - organizationSlug: params.organizationSlug, - replayId: params.replayId, - }; -} - -async function assertReplayWithinProjectConstraint({ - apiService, - organizationSlug, - replay, - projectSlug, -}: { - apiService: SentryApiService; - organizationSlug: string; - replay: ReplayDetails; - projectSlug?: string | null; -}): Promise { - if (!projectSlug) { - return; - } - - if (replay.project_id == null) { - throw new UserInputError( - `Replay is outside the active project constraint. Expected project "${projectSlug}".`, - ); - } - - const project = await apiService.getProject({ - organizationSlug, - projectSlugOrId: projectSlug, - }); - - if (String(project.id) !== String(replay.project_id)) { - throw new UserInputError( - `Replay is outside the active project constraint. Expected project "${projectSlug}".`, - ); - } -} - function formatReplayOutput({ replay, organizationSlug, diff --git a/packages/mcp-core/src/tools/catalog/index.ts b/packages/mcp-core/src/tools/catalog/index.ts index e519aefea..478bfda0f 100644 --- a/packages/mcp-core/src/tools/catalog/index.ts +++ b/packages/mcp-core/src/tools/catalog/index.ts @@ -23,6 +23,7 @@ import getIssueTagValues from "./get-issue-tag-values"; import getIssueUserReports from "./get-issue-user-reports"; import getTraceDetails from "./get-trace-details"; import getSpanDetails from "./get-span-details"; +import getReplayActivity from "./get-replay-activity"; import getReplayDetails from "./get-replay-details"; import getEventAttachment from "./get-event-attachment"; import updateIssue from "./update-issue"; @@ -88,6 +89,7 @@ const catalogTools = { get_trace_details: getTraceDetails, get_span_details: getSpanDetails, get_replay_details: getReplayDetails, + get_replay_activity: getReplayActivity, get_event_attachment: getEventAttachment, update_issue: updateIssue, search_events: searchEvents, From cf74dfd727993098c925b62ca705945acb36d9d2 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 14:47:41 -0700 Subject: [PATCH 07/26] fix(replays): Reconcile replay sorts, capability gating, and lookup failures MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four correctness fixes on the replay paths outside the two replay tools. Replay sorting drifted from Sentry's `sort_config`, which is the authority for both the scalar and aggregated query paths and raises a ParseError for anything absent from it. Add the four sorts it supports that we omitted: `count_screens` and the aliases `browser`, `os`, and `os_name`. A test now asserts exact parity with upstream's 29 keys, in both directions — a sort we advertise but Sentry rejects fails at query time, and one Sentry supports but we omit is simply unreachable. Field discovery had no notion of sortability at all, so it presented every searchable replay field as if a sort on it would work. Most replay fields are filterable but not sortable, which meant an agent could take a sort straight from discovery output and get a UserInputError. Replay fields now carry a `sortable` flag derived from the same allow-list, and the routing prompt tells the agent to respect it. `search_events` served the replays dataset regardless of whether the constrained project had Session Replay, so replay search and replay details disagreed about availability. Reject it at handler time, checking the resolved dataset so an agent-chosen route is caught as well as an explicit request, and drop `replays` from the advertised dataset options so the router cannot pick what would be rejected. The schema narrowing goes through a new general `refineInputSchema` hook rather than putting replay logic in shared infrastructure: `requiredCapabilities` gates a tool as a whole, and search_events serves six datasets of which only one needs replays. `listReplayIdsForIssue` swallowed every failure with `.catch(() => undefined)`. Because `replay-count` is rate limited per organization and `get_issue_details` calls it on every lookup, parallel issue triage silently lost the Session Replay section — a throttled request was indistinguishable from an issue with no replays. Report the lookup as unavailable instead, both when nothing else is known and when an attached replay was found but related ones could not be. Refs docs/specs/replay-review.md Co-Authored-By: Claude Opus 5 (1M context) --- .../changes/improve-replay-review/tasks.md | 12 +-- .../agents/tools/dataset-fields.test.ts | 8 ++ .../internal/agents/tools/dataset-fields.ts | 24 +++++ packages/mcp-core/src/internal/formatting.ts | 25 +++++- .../src/tools/catalog-runtime/availability.ts | 6 +- .../tools/catalog/get-issue-details.test.ts | 60 +++++++++++++ .../src/tools/catalog/get-issue-details.ts | 21 ++++- .../src/tools/catalog/search-events.test.ts | 77 ++++++++++++++++ .../src/tools/catalog/search-events.ts | 49 +++++++++++ .../src/tools/support/search-events/config.ts | 1 + .../search-events/replay-sorts.test.ts | 87 +++++++++++++++++++ .../tools/support/search-events/replays.ts | 15 ++++ packages/mcp-core/src/tools/types.ts | 16 ++++ 13 files changed, 387 insertions(+), 14 deletions(-) create mode 100644 packages/mcp-core/src/tools/support/search-events/replay-sorts.test.ts diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index c4a70066a..26a4e470d 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -51,10 +51,10 @@ baselines encode the same wrong event shape they do today. ## 6. Adjacent Correctness Fixes -- [ ] 6.1 Add the sorts Sentry supports but `REPLAY_SORT_FIELDS` omits: `count_screens` and the aliases `browser`, `os`, `os_name`. Do not add `count_traces`, `count_segments`, or `viewed_by_me`, which are absent from `sort_config` and would be rejected. -- [ ] 6.2 Stop advertising filterable-but-unsortable replay fields as sort options in `dataset-fields.ts`, so discovery output cannot produce an invalid sort. -- [ ] 6.3 Reject `dataset="replays"` in the `search_events` handler when the constrained project lacks the `replays` capability, and drop `replays` from the advertised dataset options in that case. -- [ ] 6.4 Distinguish rate-limit and error responses from absence in `listReplayIdsForIssue`, and report unavailability in the issue Session Replay section. +- [x] 6.1 Add the sorts Sentry supports but `REPLAY_SORT_FIELDS` omits: `count_screens` and the aliases `browser`, `os`, `os_name`. Do not add `count_traces`, `count_segments`, or `viewed_by_me`, which are absent from `sort_config` and would be rejected. The list is now verified against upstream's 29 keys with a test asserting exact parity. +- [x] 6.2 Stop advertising filterable-but-unsortable replay fields as sort options in `dataset-fields.ts`, so discovery output cannot produce an invalid sort. Discovery had no sortable signal at all, so replay fields now carry a `sortable` flag derived from the same allow-list, and the agent prompt tells the router to respect it. +- [x] 6.3 Reject `dataset="replays"` in the `search_events` handler when the constrained project lacks the `replays` capability, and drop `replays` from the advertised dataset options in that case. The check runs on the resolved dataset so an agent-chosen route is rejected too; schema narrowing uses a new general `refineInputSchema` hook rather than replay-specific logic in shared infrastructure. +- [x] 6.4 Distinguish rate-limit and error responses from absence in `listReplayIdsForIssue`, and report unavailability in the issue Session Replay section. Covered for both the no-replays-found and attached-replay-only cases, verified by mutation. ## 7. Tests @@ -62,8 +62,8 @@ baselines encode the same wrong event shape they do today. - [x] 7.2 Add `get-replay-activity.test.ts` covering windowing, each grain, kind filtering, paging, and constraint injection. - [ ] 7.3 Add degradation tests: summary 403, summary `processing`, summary timeout, summary unparseable, segment fetch 404, archived replay, zero segments, multi-page segments, and the segment budget being hit. - [ ] 7.4 Add redaction tests asserting `` and `[Filtered]`-driven `` rendering. -- [ ] 7.5 Add search-events tests for the replay capability gate, the dataset options omitting `replays`, and the reconciled sort list. -- [ ] 7.6 Add a `get_issue_details` test asserting rate-limited replay lookup is reported as unavailable, not as no replays. +- [x] 7.5 Add search-events tests for the replay capability gate, the dataset options omitting `replays`, and the reconciled sort list. +- [x] 7.6 Add a `get_issue_details` test asserting rate-limited replay lookup is reported as unavailable, not as no replays. - [ ] 7.7 Add a replay eval covering map-then-zoom navigation. - [ ] 7.8 Update registry, tool count, skill gating, and generated-definition tests. - [ ] 7.9 Add suggested-window tests: a resolved error timestamp produces a bracketing window, and a failing or unauthorized `replays-events-meta` lookup degrades to the whole-session suggestion with the map intact. diff --git a/packages/mcp-core/src/internal/agents/tools/dataset-fields.test.ts b/packages/mcp-core/src/internal/agents/tools/dataset-fields.test.ts index b60ad3cac..9fca0dc35 100644 --- a/packages/mcp-core/src/internal/agents/tools/dataset-fields.test.ts +++ b/packages/mcp-core/src/internal/agents/tools/dataset-fields.test.ts @@ -203,6 +203,14 @@ describe("dataset-fields agent tool", () => { key: "viewed_by_me", totalValues: 2, }); + + // Replay fields are mostly searchable but not sortable, and Sentry + // rejects a sort outside its sort_config. A flat list invites an agent + // to pick one that cannot work, so discovery marks which are sortable. + expect(fields.get("count_screens")?.sortable).toBe(true); + expect(fields.get("viewed_by_me")?.sortable).toBe(false); + expect(fields.get("click.textContent")?.sortable).toBe(false); + expect(fields.get("user.segment")?.sortable).toBe(false); expect(fields.get("user.segment")).toMatchObject({ key: "user.segment", name: "User Segment", diff --git a/packages/mcp-core/src/internal/agents/tools/dataset-fields.ts b/packages/mcp-core/src/internal/agents/tools/dataset-fields.ts index 21528fe1e..b81713a73 100644 --- a/packages/mcp-core/src/internal/agents/tools/dataset-fields.ts +++ b/packages/mcp-core/src/internal/agents/tools/dataset-fields.ts @@ -1,5 +1,6 @@ import { z } from "zod"; import type { SentryApiService } from "../../../api-client"; +import { REPLAY_SORT_FIELDS } from "../../../tools/support/search-events/replays"; import { agentTool } from "./utils"; export type DatasetType = "events" | "errors" | "replays" | "search_issues"; @@ -10,6 +11,18 @@ export interface DatasetField { name: string; totalValues: number; examples?: string[]; + /** + * Whether this field can be sorted on, when that differs from being + * searchable. + * + * Replay search is the case that motivates this: most replay fields are + * filterable but only a subset appears in Sentry's replay sort + * configuration, and a sort outside it is rejected outright. Presenting a + * flat field list invites an agent to pick a sort that cannot work. + * + * Left undefined for datasets where the distinction does not apply. + */ + sortable?: boolean; } export interface DatasetFieldsResult { @@ -26,6 +39,14 @@ interface CommonPattern { const REPLAY_EXCLUDED_TAGS = new Set(["browser", "device", "os", "user"]); +/** + * Replay fields Sentry can sort on. + * + * Drawn from the single allow-list validated against Sentry's own + * `sort_config`, so discovery cannot advertise a sort the query layer rejects. + */ +const REPLAY_SORTABLE_KEYS = new Set(REPLAY_SORT_FIELDS); + const REPLAY_FIELDS = [ "activity", "browser.name", @@ -384,6 +405,9 @@ function createDatasetField( name: tag?.name ?? humanizeFieldName(key), totalValues: tag?.totalValues ?? 0, examples: getFieldExamples(key, dataset), + ...(dataset === "replays" + ? { sortable: REPLAY_SORTABLE_KEYS.has(key) } + : {}), }; } diff --git a/packages/mcp-core/src/internal/formatting.ts b/packages/mcp-core/src/internal/formatting.ts index cce671a87..39f9490a5 100644 --- a/packages/mcp-core/src/internal/formatting.ts +++ b/packages/mcp-core/src/internal/formatting.ts @@ -195,6 +195,7 @@ export function formatEventOutput( apiService: SentryApiService; organizationSlug: string; relatedReplayIds?: string[]; + replayLookupFailed?: boolean; experimentalMode?: boolean; availableToolNames?: ReadonlySet; directToolNames?: ReadonlySet; @@ -218,6 +219,7 @@ export function formatEventOutput( organizationSlug: options.replaySummary.organizationSlug, event, relatedReplayIds: options.replaySummary.relatedReplayIds, + replayLookupFailed: options.replaySummary.replayLookupFailed, experimentalMode: options.replaySummary.experimentalMode ?? false, availableToolNames: options.replaySummary.availableToolNames, directToolNames: options.replaySummary.directToolNames, @@ -1956,6 +1958,7 @@ export function formatIssueOutput({ performanceTrace, externalIssues, relatedReplayIds, + replayLookupFailed, aiConversations, codeLocation, experimentalMode, @@ -1970,6 +1973,7 @@ export function formatIssueOutput({ performanceTrace?: Trace; externalIssues?: ExternalIssueList; relatedReplayIds?: string[]; + replayLookupFailed?: boolean; aiConversations?: AIConversationReference[]; codeLocation?: CodeLocation; experimentalMode?: boolean; @@ -2138,6 +2142,7 @@ export function formatIssueOutput({ apiService, organizationSlug, relatedReplayIds, + replayLookupFailed, experimentalMode: experimentalMode ?? false, availableToolNames, directToolNames, @@ -2323,6 +2328,7 @@ function formatIssueReplayOutput({ organizationSlug, event, relatedReplayIds, + replayLookupFailed, experimentalMode, availableToolNames, directToolNames, @@ -2331,6 +2337,7 @@ function formatIssueReplayOutput({ organizationSlug: string; event: Event; relatedReplayIds?: string[]; + replayLookupFailed?: boolean; experimentalMode: boolean; availableToolNames?: ReadonlySet; directToolNames?: ReadonlySet; @@ -2342,7 +2349,17 @@ function formatIssueReplayOutput({ ); if (!attachedReplayId && normalizedRelatedReplayIds.length === 0) { - return ""; + // The replay-count lookup is rate limited per organization, so a failure + // here is common under parallel triage. Reporting nothing would assert + // this issue has no replays, which we do not actually know. + return replayLookupFailed + ? [ + "## Session Replay", + "", + "Related replays could not be looked up (the replay count endpoint is rate limited or unavailable). This issue may still have replays.", + "", + ].join("\n") + : ""; } const lines: string[] = ["## Session Replay", ""]; @@ -2357,6 +2374,12 @@ function formatIssueReplayOutput({ lines.push( `**Related Replay Count**: ${normalizedRelatedReplayIds.length}`, ); + } else if (replayLookupFailed) { + // An attached replay was found on the event, but the lookup for others + // failed. Saying nothing would imply this is the only one. + lines.push( + "**Related Replays**: could not be looked up (the replay count endpoint is rate limited or unavailable)", + ); } if (additionalRelatedReplayIds.length > 0) { diff --git a/packages/mcp-core/src/tools/catalog-runtime/availability.ts b/packages/mcp-core/src/tools/catalog-runtime/availability.ts index 930f27da1..25c12aa84 100644 --- a/packages/mcp-core/src/tools/catalog-runtime/availability.ts +++ b/packages/mcp-core/src/tools/catalog-runtime/availability.ts @@ -90,7 +90,6 @@ function isAllowedBySkills({ tool: ToolConfig; context: ServerContext; }): boolean { - const grantedSkills: Set | undefined = context.grantedSkills ? new Set(context.grantedSkills) : undefined; @@ -126,11 +125,13 @@ export function getFilteredInputSchema( getConstraintKeysToFilter(context.constraints, tool.inputSchema), ); - return Object.fromEntries( + const schema = Object.fromEntries( Object.entries(tool.inputSchema).filter( ([key]) => !constraintKeysToFilter.has(key), ), ) as Record; + + return tool.refineInputSchema?.(schema, context) ?? schema; } export function injectConstraintParams( @@ -202,7 +203,6 @@ export function getToolsForMcpRegistration({ useDefaultSurfacePolicy, }); - return availableTools.filter(({ isTopLevel }) => isTopLevel); } diff --git a/packages/mcp-core/src/tools/catalog/get-issue-details.test.ts b/packages/mcp-core/src/tools/catalog/get-issue-details.test.ts index 4668138a8..ca2a04afa 100644 --- a/packages/mcp-core/src/tools/catalog/get-issue-details.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-issue-details.test.ts @@ -849,6 +849,66 @@ describe("get_issue_details", () => { expect(result).not.toContain("**replayId**:"); }); + it("reports a rate-limited replay lookup as unavailable, not as no replays", async () => { + // `replay-count` is rate limited per organization and called on every + // issue lookup, so under parallel triage this is a routine failure. + // Rendering nothing would assert the issue has no replays. + mswServer.use( + http.get( + "https://sentry.io/api/0/organizations/sentry-mcp-evals/replay-count/", + () => + HttpResponse.json({ detail: "Too many requests" }, { status: 429 }), + ), + ); + + const result = await getIssueDetails.handler( + { + organizationSlug: "sentry-mcp-evals", + issueId: "CLOUDFLARE-MCP-41", + eventId: undefined, + issueUrl: undefined, + regionUrl: null, + }, + baseContext, + ); + + if (typeof result !== "string") { + throw new Error("Expected string result"); + } + + expect(result).toContain("## Session Replay"); + expect(result).toContain("could not be looked up"); + expect(result).toContain("may still have replays"); + }); + + it("stays silent when the lookup succeeds and finds nothing", async () => { + // The absence of replays is a real answer and should not be dressed up as + // a failure. + mswServer.use( + http.get( + "https://sentry.io/api/0/organizations/sentry-mcp-evals/replay-count/", + () => HttpResponse.json({}), + ), + ); + + const result = await getIssueDetails.handler( + { + organizationSlug: "sentry-mcp-evals", + issueId: "CLOUDFLARE-MCP-41", + eventId: undefined, + issueUrl: undefined, + regionUrl: null, + }, + baseContext, + ); + + if (typeof result !== "string") { + throw new Error("Expected string result"); + } + + expect(result).not.toContain("## Session Replay"); + }); + it("serializes with issueUrl", async () => { const result = await getIssueDetails.handler( { diff --git a/packages/mcp-core/src/tools/catalog/get-issue-details.ts b/packages/mcp-core/src/tools/catalog/get-issue-details.ts index d2f069de4..62151d861 100644 --- a/packages/mcp-core/src/tools/catalog/get-issue-details.ts +++ b/packages/mcp-core/src/tools/catalog/get-issue-details.ts @@ -135,7 +135,7 @@ export default defineTool({ // For this call, we might want to provide context if it fails const [ { event, performanceTrace, aiConversations, codeLocation }, - { autofixState, externalIssues, relatedReplayIds }, + { autofixState, externalIssues, relatedReplayIds, replayLookupFailed }, ] = await Promise.all([ apiService .getEventForIssue({ @@ -180,6 +180,7 @@ export default defineTool({ performanceTrace, externalIssues, relatedReplayIds, + replayLookupFailed, aiConversations, codeLocation, experimentalMode: context.experimentalMode, @@ -239,7 +240,7 @@ export default defineTool({ const [ { event, performanceTrace, aiConversations, codeLocation }, - { autofixState, externalIssues, relatedReplayIds }, + { autofixState, externalIssues, relatedReplayIds, replayLookupFailed }, ] = await Promise.all([ apiService .getLatestEventForIssue({ @@ -272,6 +273,7 @@ export default defineTool({ performanceTrace, externalIssues, relatedReplayIds, + replayLookupFailed, aiConversations, codeLocation, experimentalMode: context.experimentalMode, @@ -341,6 +343,7 @@ async function fetchIssueEnrichmentData({ autofixState: AutofixRunState | undefined; externalIssues: ExternalIssueList | undefined; relatedReplayIds: string[] | undefined; + replayLookupFailed: boolean; }> { const issueId = String(issue.id); const [autofixState, externalIssues, relatedReplayIds] = await Promise.all([ @@ -350,16 +353,26 @@ async function fetchIssueEnrichmentData({ apiService .getIssueExternalLinks({ organizationSlug, issueId: issue.shortId }) .catch(() => undefined), + // `replay-count` is rate limited per organization as well as per user and + // IP, and this runs on every issue lookup. Swallowing the failure would + // make throttling indistinguishable from "this issue has no replays", so + // parallel issue triage would silently lose the Session Replay section. apiService .listReplayIdsForIssue({ organizationSlug, issueId, dataSource: getReplayDataSource(issue), }) - .catch(() => undefined), + .then((ids) => ({ ok: true as const, ids })) + .catch(() => ({ ok: false as const })), ]); - return { autofixState, externalIssues, relatedReplayIds }; + return { + autofixState, + externalIssues, + relatedReplayIds: relatedReplayIds.ok ? relatedReplayIds.ids : undefined, + replayLookupFailed: !relatedReplayIds.ok, + }; } async function maybeFetchPerformanceTrace({ diff --git a/packages/mcp-core/src/tools/catalog/search-events.test.ts b/packages/mcp-core/src/tools/catalog/search-events.test.ts index dbf487f81..4dd748a19 100644 --- a/packages/mcp-core/src/tools/catalog/search-events.test.ts +++ b/packages/mcp-core/src/tools/catalog/search-events.test.ts @@ -3156,3 +3156,80 @@ describe("search_events", () => { expect(mockGenerateText).not.toHaveBeenCalled(); }); }); + +describe("replays capability gate", () => { + const constrainedContext = (replays: boolean | undefined) => ({ + constraints: { + organizationSlug: "test-org", + regionUrl: null, + projectSlug: "cloudflare-mcp", + projectCapabilities: replays === undefined ? undefined : { replays }, + }, + accessToken: "test-token", + userId: "1", + }); + + const replayParams = { + organizationSlug: "test-org", + regionUrl: null, + projectSlug: "cloudflare-mcp", + dataset: "replays" as const, + query: "count_errors:>0", + fields: null, + sort: null, + period: "24h", + limit: 10, + includeExplanation: false, + }; + + it("rejects an explicit replays dataset when the project lacks the capability", async () => { + await expect( + searchEvents.handler(replayParams, constrainedContext(false) as never), + ).rejects.toThrow(/Session Replay is not enabled for project/); + }); + + it("allows replays when capabilities are unknown", async () => { + // An unconstrained session may span projects with different capabilities, + // so Sentry decides per request rather than us guessing. + mswServer.use( + http.get("https://sentry.io/api/0/organizations/test-org/replays/", () => + HttpResponse.json({ data: [] }), + ), + ); + + await expect( + searchEvents.handler( + replayParams, + constrainedContext(undefined) as never, + ), + ).resolves.toContain("replay"); + }); + + describe("advertised dataset options", () => { + const datasetOptions = (replays: boolean) => { + const schema = searchEvents.refineInputSchema?.( + searchEvents.inputSchema as never, + constrainedContext(replays) as never, + ); + if (!schema?.dataset) { + throw new Error("expected a dataset parameter in the refined schema"); + } + // `.unwrap()` drops the `.optional()` wrapper; `.options` is the + // public accessor for an enum's values. + return ( + schema.dataset as never as { unwrap: () => { options: string[] } } + ).unwrap().options; + }; + + it("omits replays when the constrained project lacks the capability", () => { + // A routing agent that cannot see the option cannot choose one the + // handler would reject. + expect(datasetOptions(false)).not.toContain("replays"); + expect(datasetOptions(false)).toContain("errors"); + }); + + it("keeps replays when the capability is present", () => { + expect(datasetOptions(true)).toContain("replays"); + }); + }); +}); diff --git a/packages/mcp-core/src/tools/catalog/search-events.ts b/packages/mcp-core/src/tools/catalog/search-events.ts index 8d3d41c1e..4fc51458d 100644 --- a/packages/mcp-core/src/tools/catalog/search-events.ts +++ b/packages/mcp-core/src/tools/catalog/search-events.ts @@ -54,6 +54,31 @@ const DEFAULT_EVENTS_SORT = "-timestamp"; type SearchEventsAgentResult = z.output; +/** + * Whether replay search is available to this session. + * + * Only meaningful when the session is constrained to a project, since that is + * the only time project capabilities are known. An unconstrained session may + * span projects with different capabilities, so replay search stays available + * and Sentry decides per request. + */ +function hasReplaysCapability(context: ServerContext): boolean { + const { projectSlug, projectCapabilities } = context.constraints; + if (!projectSlug || !projectCapabilities) { + return true; + } + return projectCapabilities.replays === true; +} + +function assertReplaysAvailable(context: ServerContext): void { + if (hasReplaysCapability(context)) { + return; + } + throw new UserInputError( + `Session Replay is not enabled for project "${context.constraints.projectSlug}", so replay search is unavailable. Choose a different dataset.`, + ); +} + function defaultSortForDataset(dataset: PublicEventsDataset | "replays") { return dataset === "replays" ? DEFAULT_REPLAY_SORT : DEFAULT_EVENTS_SORT; } @@ -402,6 +427,24 @@ export default defineTool({ destructiveHint: false, openWorldHint: true, }, + // Drop `replays` from the advertised dataset options when the constrained + // project has no Session Replay, so the routing agent cannot pick a dataset + // the handler would reject. + refineInputSchema(schema, context) { + if (hasReplaysCapability(context)) { + return schema; + } + + return { + ...schema, + dataset: z + .enum(PUBLIC_EVENTS_DATASETS) + .optional() + .describe( + "Initial dataset hint: errors, logs, spans, metrics, or profiles. The agent may correct this when configured.", + ), + }; + }, async handler(params, context: ServerContext) { const apiService = apiServiceFromContext(context, { regionUrl: params.regionUrl ?? undefined, @@ -556,6 +599,12 @@ export default defineTool({ } if (dataset === "replays") { + // Checked on the resolved dataset, not the requested one, so an agent + // that routes to replays is rejected the same way an explicit request + // is. A tool-level `requiredCapabilities` cannot do this: search_events + // serves six datasets, and only one of them needs replays. + assertReplaysAvailable(context); + const replaySort = sortParam || DEFAULT_REPLAY_SORT; if (!isValidReplaySort(replaySort)) { throw new UserInputError( diff --git a/packages/mcp-core/src/tools/support/search-events/config.ts b/packages/mcp-core/src/tools/support/search-events/config.ts index 2a9d6823d..367333289 100644 --- a/packages/mcp-core/src/tools/support/search-events/config.ts +++ b/packages/mcp-core/src/tools/support/search-events/config.ts @@ -98,6 +98,7 @@ QUERY MODES: - For replays, put environment in the separate "environment" field, NOT inside query - For replay environment filters, use a string for one environment or an array of strings for multiple environments - For replays, use sorts like -started_at, -count_errors, -count_rage_clicks, -count_dead_clicks, -duration + - For replays, only sort by a field the field-discovery tool marks sortable; most replay fields are searchable but not sortable, and Sentry rejects the rest - Replays do NOT support count()/avg()/sum() aggregations through this path CRITICAL LIMITATION - TIME SERIES NOT SUPPORTED: diff --git a/packages/mcp-core/src/tools/support/search-events/replay-sorts.test.ts b/packages/mcp-core/src/tools/support/search-events/replay-sorts.test.ts new file mode 100644 index 000000000..dd24400f5 --- /dev/null +++ b/packages/mcp-core/src/tools/support/search-events/replay-sorts.test.ts @@ -0,0 +1,87 @@ +/** + * Guards the replay sort allow-list against Sentry's own sort configuration. + * + * The failure mode this prevents is asymmetric and easy to miss: a sort we + * advertise but Sentry rejects surfaces as a `UserInputError` at query time, + * and a sort Sentry supports but we omit is simply unreachable. Both are + * invisible until someone tries the exact field. + */ +import { describe, expect, it } from "vitest"; +import { REPLAY_SORT_FIELDS, isValidReplaySort } from "./replays.js"; + +/** + * `sort_config` from + * `sentry/replays/usecases/query/configs/aggregate_sort.py`, including the + * four aliases assigned below the literal. + */ +const UPSTREAM_SORT_CONFIG = [ + "activity", + "browser.name", + "browser.version", + "count_dead_clicks", + "count_errors", + "count_warnings", + "count_infos", + "count_rage_clicks", + "count_urls", + "device.brand", + "device.family", + "device.model", + "device.name", + "dist", + "duration", + "finished_at", + "os.name", + "os.version", + "platform", + "project_id", + "started_at", + "sdk.name", + "user.email", + "user.id", + "user.username", + // Aliases. + "browser", + "os", + "os_name", + "count_screens", +]; + +describe("replay sort fields", () => { + it("matches Sentry's sort configuration exactly", () => { + expect([...REPLAY_SORT_FIELDS].sort()).toEqual( + [...UPSTREAM_SORT_CONFIG].sort(), + ); + }); + + it("accepts the sorts that were previously missing", () => { + // `_get_sort_column` raises a ParseError for anything absent from + // sort_config, so these were reachable in Sentry but rejected by us. + for (const field of ["count_screens", "browser", "os", "os_name"]) { + expect(isValidReplaySort(field)).toBe(true); + expect(isValidReplaySort(`-${field}`)).toBe(true); + } + }); + + it("rejects fields that are searchable but not sortable", () => { + // Present in replay search and in field discovery, but absent from + // sort_config — Sentry would reject a sort on any of them. + for (const field of [ + "count_traces", + "count_segments", + "viewed_by_me", + "device.model_id", + "replay_type", + "urls", + ]) { + expect(isValidReplaySort(field)).toBe(false); + expect(isValidReplaySort(`-${field}`)).toBe(false); + } + }); + + it("accepts both ascending and descending forms", () => { + expect(isValidReplaySort("started_at")).toBe(true); + expect(isValidReplaySort("-started_at")).toBe(true); + expect(isValidReplaySort("--started_at")).toBe(false); + }); +}); diff --git a/packages/mcp-core/src/tools/support/search-events/replays.ts b/packages/mcp-core/src/tools/support/search-events/replays.ts index 1f8ada598..9afd1fed0 100644 --- a/packages/mcp-core/src/tools/support/search-events/replays.ts +++ b/packages/mcp-core/src/tools/support/search-events/replays.ts @@ -10,10 +10,23 @@ import { export const DEFAULT_REPLAY_SORT = "-started_at"; export const DEFAULT_REPLAY_STATS_PERIOD = "14d"; +/** + * Replay sorts Sentry accepts. + * + * Mirrors `sort_config` in + * `sentry/replays/usecases/query/configs/aggregate_sort.py`, which is the + * authority for both the scalar and aggregated query paths; anything absent + * from it raises a `ParseError`. Fields that are searchable but not in that + * config — `count_traces`, `count_segments`, `viewed_by_me`, + * `device.model_id` — are deliberately excluded. + */ export const REPLAY_SORT_FIELDS = [ "activity", + // Upstream aliases `browser`, `os`, and `os_name` onto the `.name` variants. + "browser", "browser.name", "browser.version", + "count_screens", "count_dead_clicks", "count_errors", "count_infos", @@ -27,8 +40,10 @@ export const REPLAY_SORT_FIELDS = [ "dist", "duration", "finished_at", + "os", "os.name", "os.version", + "os_name", "platform", "project_id", "sdk.name", diff --git a/packages/mcp-core/src/tools/types.ts b/packages/mcp-core/src/tools/types.ts index 5f956de38..261a2b74b 100644 --- a/packages/mcp-core/src/tools/types.ts +++ b/packages/mcp-core/src/tools/types.ts @@ -90,6 +90,22 @@ export interface ToolConfig< experimental?: boolean; // Mark tool as experimental (only shown in experimental mode) hideInExperimentalMode?: boolean; // Hide tool when experimental mode is active (for tools replaced by unified tools) requiredCapabilities?: (keyof ProjectCapabilities)[]; // Project capabilities required for this tool + /** + * Narrows the advertised input schema for a specific session. + * + * `requiredCapabilities` gates a tool as a whole. This is for tools that + * remain available but whose options do not all apply — `search_events` + * serves six datasets, and only the replay dataset needs the `replays` + * capability. Narrowing the advertised options keeps a routing agent from + * choosing one that would be rejected at handler time. + * + * Called after constraint-injected parameters are removed. Returning the + * schema unchanged is always valid. + */ + refineInputSchema?: ( + schema: Record, + context: ServerContext, + ) => Record; outputSchema?: z.ZodType; annotations: { // readOnlyHint, destructiveHint, and openWorldHint are required so every From 4ecc8da9524763d8e1c62e986e42006d49d3292d Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 15:07:42 -0700 Subject: [PATCH 08/26] test(replays): Close replay degradation, redaction, and gating gaps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Section 7 of the replay review change. The gaps that remained were the ones mocks make easy to skip: paths that render nothing, where a passing test and a silently broken one look identical. Cover the two truncation paths end to end. A recording spanning three segment pages is only complete if both `Link` cursors are followed, and the byte budget is tripped with a real oversized page rather than by injecting a bound, so the "Truncated" line is driven by the same code that runs in production. Silence here would present a partial recording as a whole one, turning "no errors in this replay" into a false negative. Extend summary degradation to a connection failure and a non-JSON body. Seer takes tens of seconds on a long replay, and a timeout raises a ConfigurationError that would abort the whole tool if it escaped the chapters lookup. Assert redaction survives to tool output. `` and `` mean different things — enable networkCaptureBodies, or relax a scrubbing rule — so they appear on one signal where conflating them would fail. Cover capability gating, which no tool tested. Verified by mutation: stubbing the check out fails the replays-absent case. Add a replay eval for map-then-zoom. It executes against the mocks, so the zoom can only pass by reading the window off the map. Also reset MSW handlers between replay detail tests; overrides are not reset automatically here, and a leaked stub changes what a later test measures. Co-Authored-By: Claude Opus 5 --- .../changes/improve-replay-review/tasks.md | 10 +- .../catalog-runtime/availability.test.ts | 83 +++++++++++++ .../tools/catalog/get-replay-activity.test.ts | 58 +++++++++ .../tools/catalog/get-replay-details.test.ts | 113 +++++++++++++++++- .../src/evals/get-replay.eval.ts | 83 +++++++++++++ 5 files changed, 341 insertions(+), 6 deletions(-) create mode 100644 packages/mcp-server-evals/src/evals/get-replay.eval.ts diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index 26a4e470d..8f2f2187d 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -60,13 +60,13 @@ baselines encode the same wrong event shape they do today. - [x] 7.1 Re-baseline `get-replay-details.test.ts` snapshots against the rebuilt fixtures. - [x] 7.2 Add `get-replay-activity.test.ts` covering windowing, each grain, kind filtering, paging, and constraint injection. -- [ ] 7.3 Add degradation tests: summary 403, summary `processing`, summary timeout, summary unparseable, segment fetch 404, archived replay, zero segments, multi-page segments, and the segment budget being hit. -- [ ] 7.4 Add redaction tests asserting `` and `[Filtered]`-driven `` rendering. +- [x] 7.3 Add degradation tests: summary 403, summary `processing`, summary timeout, summary unparseable, segment fetch 404, archived replay, zero segments, multi-page segments, and the segment budget being hit. Summary degradation covers eight responses including a connection failure and a non-JSON body; the byte budget is tripped end-to-end rather than by injecting a bound. +- [x] 7.4 Add redaction tests asserting `` and `[Filtered]`-driven `` rendering. Unit-level in `replay-events.test.ts`, plus end-to-end through `get_replay_activity` at detail grain, where the two labels appear on the same signal so they cannot be conflated. - [x] 7.5 Add search-events tests for the replay capability gate, the dataset options omitting `replays`, and the reconciled sort list. - [x] 7.6 Add a `get_issue_details` test asserting rate-limited replay lookup is reported as unavailable, not as no replays. -- [ ] 7.7 Add a replay eval covering map-then-zoom navigation. -- [ ] 7.8 Update registry, tool count, skill gating, and generated-definition tests. -- [ ] 7.9 Add suggested-window tests: a resolved error timestamp produces a bracketing window, and a failing or unauthorized `replays-events-meta` lookup degrades to the whole-session suggestion with the map intact. +- [x] 7.7 Add a replay eval covering map-then-zoom navigation. `get-replay.eval.ts` runs the tools for real against the mocks, so the second case can only pass by reading the window off the map. Not executed here — evals need `OPENROUTER_API_KEY`. +- [x] 7.8 Update registry, tool count, skill gating, and generated-definition tests. Capability gating was untested for any tool; `availability.test.ts` now covers replays-present, replays-absent, unconstrained, direct-surface, and skill-gated cases, verified by mutation. Generated definitions are current and the token budget is unchanged at 5,598. +- [x] 7.9 Add suggested-window tests: a resolved error timestamp produces a bracketing window, and a failing or unauthorized `replays-events-meta` lookup degrades to the whole-session suggestion with the map intact. Landed with task 4.2. ## 8. Documentation and Generated Definitions diff --git a/packages/mcp-core/src/tools/catalog-runtime/availability.test.ts b/packages/mcp-core/src/tools/catalog-runtime/availability.test.ts index b627354c8..4651e4c79 100644 --- a/packages/mcp-core/src/tools/catalog-runtime/availability.test.ts +++ b/packages/mcp-core/src/tools/catalog-runtime/availability.test.ts @@ -74,6 +74,89 @@ describe("catalog availability", () => { ); }); + describe("replay capability gating", () => { + const REPLAY_TOOLS = ["get_replay_details", "get_replay_activity"]; + + function getInspectContext( + constraints: Partial = {}, + ): ServerContext { + return getServerContext({ + grantedSkills: new Set(["inspect"]), + constraints, + }); + } + + it("offers both replay tools when the constrained project has replays", () => { + const names = getSearchableToolNames( + getInspectContext({ + organizationSlug: "my-org", + projectSlug: "my-project", + projectCapabilities: { replays: true }, + }), + ); + + expect(names).toEqual(expect.arrayContaining(REPLAY_TOOLS)); + }); + + it("hides both replay tools when the constrained project has no replays", () => { + // Advertising a tool that can only fail costs a tool slot and invites a + // call that returns nothing useful. + const names = getSearchableToolNames( + getInspectContext({ + organizationSlug: "my-org", + projectSlug: "my-project", + projectCapabilities: { replays: false }, + }), + ); + + for (const toolName of REPLAY_TOOLS) { + expect(names).not.toContain(toolName); + } + }); + + it("keeps replay tools available when no project constrains the session", () => { + // Capabilities are a property of a project. Without one there is + // nothing to check, and hiding the tools would break unconstrained + // sessions. + const names = getSearchableToolNames(getInspectContext()); + + expect(names).toEqual(expect.arrayContaining(REPLAY_TOOLS)); + }); + + it("keeps both replay tools off the direct surface", () => { + // They are catalog-only by design: the direct surface is budgeted, and + // replay review starts from a search or a pasted URL either way. + const directToolNames = getToolsForMcpRegistration({ + tools: catalogTools, + context: getInspectContext({ + organizationSlug: "my-org", + projectSlug: "my-project", + projectCapabilities: { replays: true }, + }), + experimentalMode: false, + useDefaultSurfacePolicy: true, + }).map(({ tool }) => tool.name); + + for (const toolName of REPLAY_TOOLS) { + expect(directToolNames).not.toContain(toolName); + } + }); + + it("hides replay tools from sessions without the inspect skill", () => { + const names = getSearchableToolNames( + getProjectManagementContext({ + organizationSlug: "my-org", + projectSlug: "my-project", + projectCapabilities: { replays: true }, + }), + ); + + for (const toolName of REPLAY_TOOLS) { + expect(names).not.toContain(toolName); + } + }); + }); + it("hides create_project from project-scoped project-management sessions", () => { const context = getProjectManagementContext({ organizationSlug: "my-org", diff --git a/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts index 792b309d7..575dc80b5 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts @@ -5,6 +5,7 @@ import getReplayActivity from "./get-replay-activity.js"; import { getServerContext } from "../../test-setup.js"; const REPLAY_URL = `https://us.sentry.io/api/0/organizations/sentry-mcp-evals/replays/${replayDetailsFixture.id}/`; +const SEGMENTS_URL = `https://us.sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/recording-segments/`; function callTool( params: Record = {}, @@ -116,6 +117,63 @@ describe("get_replay_activity", () => { }); }); + describe("redaction", () => { + /** Serve a single fetch span with the given `data` payload. */ + function serveNetworkSpan(data: Record) { + mswServer.use( + http.get(SEGMENTS_URL, () => + HttpResponse.json([ + [ + { + type: 5, + timestamp: 1744027201, + data: { + tag: "performanceSpan", + payload: { + op: "resource.fetch", + description: "https://example.com/api/pay", + startTimestamp: 1744027201, + endTimestamp: 1744027201.4, + data, + }, + }, + }, + ], + ]), + ), + ); + } + + it("distinguishes a value Relay scrubbed from one never captured", async () => { + // Both are absent from the output, but for different reasons, and the + // difference decides what the reader does next: enable + // networkCaptureBodies, or relax a server-side scrubbing rule. + serveNetworkSpan({ + method: "POST", + statusCode: 500, + request: { size: 64, body: "[Filtered]" }, + response: { size: 32 }, + }); + + const result = await callTool({ grain: "detail", kinds: ["network"] }); + + expect(result).toContain("request body: "); + expect(result).toContain("response body: 32 bytes "); + }); + + it("never prints a body Relay scrubbed", async () => { + serveNetworkSpan({ + method: "POST", + statusCode: 500, + request: { size: 64, body: "[Filtered]" }, + }); + + const result = await callTool({ grain: "detail", kinds: ["network"] }); + + expect(result).not.toContain("[Filtered]"); + }); + }); + describe("kind filtering", () => { it("returns only the requested kinds", async () => { const result = await callTool({ kinds: ["console", "network"] }); diff --git a/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts index fb525e69f..af16d0a7a 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts @@ -1,8 +1,9 @@ -import { describe, expect, it } from "vitest"; +import { afterEach, describe, expect, it } from "vitest"; import { http, HttpResponse } from "msw"; import { mswServer, organizationFixture, + PAGED_REPLAY_ID, replayDetailsFixture, } from "@sentry/mcp-server-mocks"; import getReplayDetails from "./get-replay-details.js"; @@ -10,6 +11,12 @@ import { resolveReplayParams } from "../../internal/tool-helpers/replay.js"; import { getServerContext } from "../../test-setup.js"; describe("get_replay_details", () => { + // Overrides are not reset automatically, and a leaked stub silently changes + // what a later test is measuring. + afterEach(() => { + mswServer.resetHandlers(); + }); + it("loads replay details from replayUrl", async () => { const result = await getReplayDetails.handler( { @@ -349,6 +356,102 @@ describe("get_replay_details", () => { ); }); + it("reports a replay whose recording has no segments", async () => { + // Distinct from a failed fetch: the request succeeded and the recording + // is genuinely empty, which the reader should not confuse with an error. + mswServer.use( + http.get( + `https://us.sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/recording-segments/`, + () => HttpResponse.json([]), + ), + ); + + const result = await getReplayDetails.handler( + { + organizationSlug: "sentry-mcp-evals", + replayId: replayDetailsFixture.id, + regionUrl: "https://us.sentry.io", + }, + getServerContext(), + ); + + expect(result).toContain("No activity recorded."); + expect(result).not.toContain("Recording is unavailable."); + // With nothing to zoom into, suggesting a window would be noise. + expect(result).not.toContain("## Next"); + }); + + it("reads a recording that spans multiple segment pages", async () => { + // A recording longer than one page was previously cut off at the page + // boundary without any indication, so the map understated the session. + const result = await getReplayDetails.handler( + { + organizationSlug: "sentry-mcp-evals", + replayId: PAGED_REPLAY_ID, + regionUrl: "https://us.sentry.io", + }, + getServerContext(), + ); + + // The fixture's five segments span three pages, so the last signal is + // only reachable by following two `Link` cursors. Reading the first page + // alone would report a session ending at T+1m 0.0s. + expect(result).toContain( + "- **Signals**: 5 signals across T+1.0s–T+4m 0.0s", + ); + expect(result).toContain("- **Kinds**: click 4 · console 1 (1 error)"); + // Following the header is not the same as hitting a bound. + expect(result).toContain("- **Truncated**: no"); + }); + + it("says the map is partial when the byte budget stops the read", async () => { + // Silence here is the dangerous failure: a truncated recording renders as + // a complete one, and "no errors in this replay" becomes a false negative. + // The budget is enforced on bytes off the wire, so one oversized page + // trips it — padding an ignored field keeps the signals intact. + const oversizedPage = [ + [ + { + type: 5, + timestamp: 1744027200500, + data: { + tag: "breadcrumb", + payload: { + type: "default", + category: "ui.click", + timestamp: 1744027200500, + message: "button#pay", + // Pushes the page past MAX_REPLAY_SEGMENT_BYTES on its own. + padding: "x".repeat(11 * 1024 * 1024), + }, + }, + }, + ], + ]; + + mswServer.use( + http.get( + `https://us.sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/recording-segments/`, + () => HttpResponse.json(oversizedPage), + ), + ); + + const result = await getReplayDetails.handler( + { + organizationSlug: "sentry-mcp-evals", + replayId: replayDetailsFixture.id, + regionUrl: "https://us.sentry.io", + }, + getServerContext(), + ); + + expect(result).toContain("- **Truncated**: yes"); + expect(result).toContain("later activity is not included"); + // What was read is still reported; truncation degrades the map's + // completeness, not its usefulness. + expect(result).toContain("- **Signals**: 1 signal"); + }); + it("ignores events whose tag is not one the SDK emits", async () => { // `tag: "console"` is not a shape the SDK produces — meaning lives in // `payload.category` under `tag: "breadcrumb"`. Such an event is @@ -538,6 +641,14 @@ describe("get_replay_details", () => { "a server failure", () => HttpResponse.json({ detail: "boom" }, { status: 500 }), ], + // Seer can take tens of seconds on a long replay. A connection timeout + // is raised as a ConfigurationError, which would abort the whole tool + // if it escaped the chapters lookup. + ["the connection times out", () => HttpResponse.error()], + [ + "the body is not JSON at all", + () => HttpResponse.text("gateway"), + ], ] as const) { it(`omits chapters on ${label} without degrading the map`, async () => { mswServer.use(http.get(SUMMARIZE_URL, respond)); diff --git a/packages/mcp-server-evals/src/evals/get-replay.eval.ts b/packages/mcp-server-evals/src/evals/get-replay.eval.ts new file mode 100644 index 000000000..e5c611bd7 --- /dev/null +++ b/packages/mcp-server-evals/src/evals/get-replay.eval.ts @@ -0,0 +1,83 @@ +import { describeEval, ToolCallScorer } from "vitest-evals"; +import { FIXTURES, McpToolCallTaskRunner } from "./utils"; + +/** + * Replay review is a two-step read: `get_replay_details` returns the session + * map, and `get_replay_activity` zooms into a window of it. The map is what + * makes the second call answerable — it reports where the failure is and + * prints the exact call to reach it — so this exercises the handoff rather + * than either tool alone. + * + * The tools execute against the mocks here, so the model sees the real map + * output and has to act on the window it suggests. + */ +describeEval("get-replay", { + data: async () => { + return [ + { + // Map first: a replay URL with no stated moment of interest has no + // window to zoom into yet. + input: `What happened in this Sentry replay? ${FIXTURES.replayUrl}`, + expectedTools: [ + { + name: "search_sentry_tools", + arguments: { + query: "replay", + }, + }, + { + name: "execute_sentry_tool", + arguments: { + name: "get_replay_details", + arguments: { + replayUrl: FIXTURES.replayUrl, + }, + }, + }, + ], + }, + { + // Map, then zoom. Reading the map is not optional: the offsets the + // second call needs are milliseconds from the start of the replay, + // which nothing in the prompt supplies. + input: `Something went wrong near the end of replay ${FIXTURES.replayId} in ${FIXTURES.organizationSlug}. Find the error and show me the requests and console output around it.`, + expectedTools: [ + { + name: "search_sentry_tools", + arguments: { + query: "replay", + }, + }, + { + name: "execute_sentry_tool", + arguments: { + name: "get_replay_details", + arguments: { + organizationSlug: FIXTURES.organizationSlug, + replayId: FIXTURES.replayId, + }, + }, + }, + { + name: "execute_sentry_tool", + arguments: { + name: "get_replay_activity", + arguments: { + organizationSlug: FIXTURES.organizationSlug, + replayId: FIXTURES.replayId, + // The window the map suggests for the fixture's error. + startMs: 176300, + endMs: 186300, + grain: "detail", + }, + }, + }, + ], + }, + ]; + }, + task: McpToolCallTaskRunner(), + scorers: [ToolCallScorer({ ordered: true, params: "fuzzy" })], + threshold: 0.6, + timeout: 90000, +}); From 6ffbaf325610c53897e6d6763e249d5a13affcd4 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 15:39:11 -0700 Subject: [PATCH 09/26] test(replays): Assert the replay eval's zoom, not its route to the map MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Running the eval showed both cases scoring 0.5 while the model's answers were correct — it found the 500, the console error, and the uncaptured body sizes. The spec was wrong, not the behavior. Two problems. The eval required routing to the map through search_sentry_tools, but get_sentry_resource is a top-level tool that delegates to get_replay_details, so a pasted replay URL reaches the map without catalog discovery. Both routes are correct, and asserting one scores a routing preference rather than the handoff under test. The first case also asserted a map read with nothing to hand off to, which get_sentry_resource already covers. Assert only the zoom, and let allowExtras absorb whichever route the model takes to the map. The zoom is the real subject: its offsets are milliseconds from the start of the replay and appear nowhere in the prompt, so passing means the map's suggested window was read and followed. Now scores 1.00. Mutating the expected startMs drops it to 0.5, confirming the offsets are load-bearing rather than incidentally matched. Co-Authored-By: Claude Opus 5 --- .../changes/improve-replay-review/tasks.md | 2 +- .../src/evals/get-replay.eval.ts | 60 +++++-------------- 2 files changed, 16 insertions(+), 46 deletions(-) diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index 8f2f2187d..00bd31ec2 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -64,7 +64,7 @@ baselines encode the same wrong event shape they do today. - [x] 7.4 Add redaction tests asserting `` and `[Filtered]`-driven `` rendering. Unit-level in `replay-events.test.ts`, plus end-to-end through `get_replay_activity` at detail grain, where the two labels appear on the same signal so they cannot be conflated. - [x] 7.5 Add search-events tests for the replay capability gate, the dataset options omitting `replays`, and the reconciled sort list. - [x] 7.6 Add a `get_issue_details` test asserting rate-limited replay lookup is reported as unavailable, not as no replays. -- [x] 7.7 Add a replay eval covering map-then-zoom navigation. `get-replay.eval.ts` runs the tools for real against the mocks, so the second case can only pass by reading the window off the map. Not executed here — evals need `OPENROUTER_API_KEY`. +- [x] 7.7 Add a replay eval covering map-then-zoom navigation. `get-replay.eval.ts` runs the tools for real against the mocks, so it can only pass by reading the suggested window off the map. Scores 1.00; mutating the expected offset drops it to 0.5, confirming the offsets are load-bearing. Only the zoom is asserted — the map is reachable through either `get_sentry_resource` or the catalog, and both are correct. - [x] 7.8 Update registry, tool count, skill gating, and generated-definition tests. Capability gating was untested for any tool; `availability.test.ts` now covers replays-present, replays-absent, unconstrained, direct-surface, and skill-gated cases, verified by mutation. Generated definitions are current and the token budget is unchanged at 5,598. - [x] 7.9 Add suggested-window tests: a resolved error timestamp produces a bracketing window, and a failing or unauthorized `replays-events-meta` lookup degrades to the whole-session suggestion with the map intact. Landed with task 4.2. diff --git a/packages/mcp-server-evals/src/evals/get-replay.eval.ts b/packages/mcp-server-evals/src/evals/get-replay.eval.ts index e5c611bd7..661b7ed3c 100644 --- a/packages/mcp-server-evals/src/evals/get-replay.eval.ts +++ b/packages/mcp-server-evals/src/evals/get-replay.eval.ts @@ -2,11 +2,16 @@ import { describeEval, ToolCallScorer } from "vitest-evals"; import { FIXTURES, McpToolCallTaskRunner } from "./utils"; /** - * Replay review is a two-step read: `get_replay_details` returns the session - * map, and `get_replay_activity` zooms into a window of it. The map is what - * makes the second call answerable — it reports where the failure is and - * prints the exact call to reach it — so this exercises the handoff rather - * than either tool alone. + * Replay review is a two-step read: the session map first, then a zoom into a + * window of it. The map is what makes the second call answerable — it reports + * where the failure is and prints the exact call to reach it — so this + * exercises the handoff rather than either tool alone. + * + * Only the zoom is asserted. The map is reachable two ways — `get_replay_details` + * through the catalog, or the top-level `get_sentry_resource`, which delegates + * to it — and both are correct, so pinning one would score a routing preference + * rather than the behavior under test. `allowExtras` lets whichever route the + * model picks pass through. * * The tools execute against the mocks here, so the model sees the real map * output and has to act on the window it suggests. @@ -14,50 +19,13 @@ import { FIXTURES, McpToolCallTaskRunner } from "./utils"; describeEval("get-replay", { data: async () => { return [ - { - // Map first: a replay URL with no stated moment of interest has no - // window to zoom into yet. - input: `What happened in this Sentry replay? ${FIXTURES.replayUrl}`, - expectedTools: [ - { - name: "search_sentry_tools", - arguments: { - query: "replay", - }, - }, - { - name: "execute_sentry_tool", - arguments: { - name: "get_replay_details", - arguments: { - replayUrl: FIXTURES.replayUrl, - }, - }, - }, - ], - }, { // Map, then zoom. Reading the map is not optional: the offsets the - // second call needs are milliseconds from the start of the replay, - // which nothing in the prompt supplies. + // zoom needs are milliseconds from the start of the replay, which + // nothing in the prompt supplies. Getting them right means the map's + // suggested window was read and followed. input: `Something went wrong near the end of replay ${FIXTURES.replayId} in ${FIXTURES.organizationSlug}. Find the error and show me the requests and console output around it.`, expectedTools: [ - { - name: "search_sentry_tools", - arguments: { - query: "replay", - }, - }, - { - name: "execute_sentry_tool", - arguments: { - name: "get_replay_details", - arguments: { - organizationSlug: FIXTURES.organizationSlug, - replayId: FIXTURES.replayId, - }, - }, - }, { name: "execute_sentry_tool", arguments: { @@ -68,6 +36,8 @@ describeEval("get-replay", { // The window the map suggests for the fixture's error. startMs: 176300, endMs: 186300, + // Payload detail is what the prompt asks for; the map's + // suggestion names this grain. grain: "detail", }, }, From 6b10d06da08d8345ed1f8187904ed95618c58e89 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 15:47:00 -0700 Subject: [PATCH 10/26] docs(replays): Record where the replay implementation diverged from its spec Section 8 of the replay review change. The spec was written before the work and read as though it still described pending behavior, so mark it implemented and record where the shipped code took a different path. Keep the Motivation section in its original present tense. It is the record of what was wrong and why the work was justified; rewriting it to past tense would lose the evidence and leave assertions no one can check. Correct the Map example. Offsets render in a consistent `T+` form rather than mixing bare seconds, and the page count is gone from the Signals line, where a count beside a time span read as something temporal. Add a divergences section covering the seven places the implementation differs: discovery gained a sortability flag rather than losing fields, dataset narrowing uses a general `refineInputSchema` hook instead of replay-specific logic in shared infrastructure, both replay tools share extracted parameter resolution, a cursor overrides arguments passed beside it, unresolvable error ids are listed rather than dropped, `count_dead_clicks` includes rage clicks upstream, and the segment bounds remain unmeasured against real replays. Also correct the two upstream fix references. The descriptions of #120859 and #121765 were swapped in the task notes; verified against the commits in getsentry/sentry and confirmed both behaviors are implemented and tested here. No other doc describes replay tool output. Generated definitions were already current. Co-Authored-By: Claude Opus 5 --- docs/specs/replay-review.md | 101 +++++++++++++++++- .../changes/improve-replay-review/tasks.md | 8 +- 2 files changed, 102 insertions(+), 7 deletions(-) diff --git a/docs/specs/replay-review.md b/docs/specs/replay-review.md index 0d9c808fe..0188621fe 100644 --- a/docs/specs/replay-review.md +++ b/docs/specs/replay-review.md @@ -1,5 +1,11 @@ # Replay Review Specification +> **Status: implemented.** The Motivation section below describes the behavior +> that prompted this work and is retained as the record of what was wrong; it is +> written in the present tense of that time. See "Divergences from the As-Built +> Implementation" at the end for the points where the shipped code differs from +> what this spec proposed. + ## Overview Replay retrieval today returns a single fixed-grain summary capped at six @@ -136,7 +142,7 @@ Replace the prose activity sample with session shape: … unchanged fields … ## Map -- **Signals**: 1,182 across 0.0s–353.2s, 4 pages +- **Signals**: 1,182 signals across T+0.0s–T+5m 53.2s - **Flow**: /login ▸ /cart ▸ /checkout ▸ /checkout/confirm - **Kinds**: navigation 18 · click 36 (2 rage, 1 dead) · network 58 (2 failed) · console 4 (2 error) - **Truncated**: no @@ -148,10 +154,15 @@ Replace the prose activity sample with session shape: … unchanged … ## Next -Error CLOUDFLARE-MCP-41 occurred at T+311.8s: +Error CLOUDFLARE-MCP-41 occurred at T+5m 11.8s. Use the Sentry tool `get_replay_activity` to read the signals in a time window: get_replay_activity(organizationSlug='my-org', replayId='7e07485f…', startMs=306000, endMs=316000, grain='detail') ``` +Offsets render in the same `T+` form throughout, rather than mixing bare +seconds into the Map and Next lines. The page count was dropped from the +Signals line: the Flow line already lists the pages, and a count beside a time +span read as a count of something temporal. + The suggested window comes from `GET /organizations/{org}/replays-events-meta/`, which resolves the replay's `error_ids` in one batched call (`query=id:[a,b]`) and returns `id`, `issue`, @@ -287,7 +298,9 @@ These are correctness bugs in shipped code, independent of the new tool: The real defect is on the discovery side: `REPLAY_FIELDS` in `packages/mcp-core/src/internal/agents/tools/dataset-fields.ts` presents filterable fields as if they were sortable, so an agent picks a sort from - discovery output and gets a `UserInputError`. + discovery output and gets a `UserInputError`. Note that discovery carries no + sortability signal at all, so the fix is to add one rather than to correct an + existing claim — see the divergences section. - Gate the `dataset="replays"` path in `search_events` on the `replays` capability, so replay search and replay details agree about availability. A tool-level `requiredCapabilities` will not do: `search_events` serves six @@ -404,6 +417,88 @@ but it is deferred rather than half-built alongside the web path. 90-day query window is visible in source; the per-plan split is a product-docs claim. Hedge in user-facing copy or verify against a free-tier org. +## Divergences from the As-Built Implementation + +Where the shipped code differs from what this spec proposed, and why. + +### Discovery gained a sortability flag rather than losing fields + +The spec framed `REPLAY_FIELDS` as advertising filterable fields "as if they +were sortable". It carried no sortability signal at all — a flat list that +invited an agent to sort by anything on it. Removing the unsortable entries +would have been wrong, since they are legitimately searchable; the fix adds a +`sortable` flag drawn from the same allow-list as `REPLAY_SORT_FIELDS`, plus a +routing-prompt line telling the agent to respect it. + +The sort list also turned out to be short by exactly the four the spec named. +It is now verified against upstream's 29 keys by a test that fails in both +directions, so a sort we advertise and Sentry rejects is as visible as one +Sentry supports and we omit. + +### Dataset narrowing uses a general hook, not replay-specific logic + +The spec called for "removal of `replays` from the advertised dataset options" +without saying where that lives. Special-casing replays inside +`getFilteredInputSchema` would have put one tool's concern in shared +infrastructure, so `ToolConfig` gained a general `refineInputSchema(schema, +context)` hook and `search_events` supplies the replay-specific narrowing. +`requiredCapabilities` still gates whole tools; this covers the case where a +tool stays available but not all of its options apply. + +### Both replay tools share extracted parameter resolution + +`get_replay_activity` accepts the same `replayUrl`-or-`organizationSlug`-plus- +`replayId` shape as `get_replay_details`. Rather than reimplement it, the +resolution and the project-constraint check were extracted to +`packages/mcp-core/src/internal/tool-helpers/replay.ts`, so the two tools cannot +drift apart on which replays a constrained session may read. + +### The activity cursor overrides its sibling arguments + +The spec said the synthetic cursor encodes the window, `kinds`, and offset. It +did not say what happens when a caller passes a cursor *and* a conflicting +window. A cursor fully describes its own query and wins over any window or +filter passed beside it — otherwise a continuation could silently page through +different criteria than the first request. It is base64url-encoded JSON, opaque +by intent. + +### The Related section lists unresolvable error ids + +The spec had `replays-events-meta` replacing the per-error `listIssues` lookups +for issue identity. Related entries are still driven by `replay.error_ids`, so +an id the private endpoint cannot resolve is listed by id rather than dropped — +`error_ids` is the replay's own record that an error occurred, and omitting it +would understate the session. + +### `count_dead_clicks` includes rage clicks + +Not a divergence in our code, but a fixture correction worth recording: upstream +sets `click_is_dead` for both `DEAD_CLICK` and `RAGE_CLICK`, and +`count_dead_clicks` sums that column. A replay with one rage click and one dead +click therefore reports `count_dead_clicks: 2`, not 1. The Summary section +mirrors Sentry's own counts, so it inherits this; the Map's `click 4 (1 rage, 1 +dead)` breakdown is computed from the recording and counts them separately. + +### Two upstream fixes folded in + +Beyond the spec's list, the taxonomy port includes two upstream fixes: + +- #120859 — `as_log_message` indexed `payload["data"]["method"]` and + `["statusCode"]` directly, which raises for a request that never got a + response. Upstream now renders `no response` in place of a status; we do the + same, and report the request rather than dropping it. +- #121765 — PII scrubbing can replace a click breadcrumb's numeric `timestamp` + with a marker string, and `int()` raised `ValueError` on it. Our equivalent is + `resolveTimestampMs` returning `null` for a non-finite timestamp, so one + scrubbed click cannot take out the events around it. + +### Still provisional + +The 150-segment and 10MB bounds remain unmeasured against real replays. Mocks +cannot exercise them meaningfully — the fixtures are orders of magnitude smaller +than a real recording — so they stand as reasoned defaults until QA against a +real organization confirms or moves them. + ## References - Implementation: `packages/mcp-core/src/tools/catalog/get-replay-details.ts` diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index 00bd31ec2..b30743fea 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -12,7 +12,7 @@ baselines encode the same wrong event shape they do today. ## 2. Event Classification Module -- [x] 2.1 Re-verify `which()`, `as_log_message()`, and `get_timestamp_unit()` against `getsentry/sentry` before porting. Two recent upstream fixes are folded in: missing `method`/`statusCode` on requests that never got a response (#120859), and PII-scrubbed values arriving where a number is expected (#121765). +- [x] 2.1 Re-verify `which()`, `as_log_message()`, and `get_timestamp_unit()` against `getsentry/sentry` before porting. Two recent upstream fixes are folded in: missing `method`/`statusCode` on requests that never got a response (#120859), and PII-scrubbed values arriving where a number is expected (#121765). Verified against the commits in `getsentry/sentry`. - [x] 2.2 Add a shared internal module exposing replay event classification, per-type timestamp resolution, and signal rendering (`packages/mcp-core/src/internal/replay-events.ts`). - [x] 2.3 Port the event-type taxonomy and the noise exclusion list, including skipping 2xx network requests from rendering while keeping them in kind counts, and preferring `NAVIGATION_SPAN` over `NAVIGATION` for web replays. - [x] 2.4 Implement behavioral dead, rage, and slow click classification, reading `payload.data.node.tagName` and accepting the lowercase `timeafterclickms`/`clickcount` spellings. @@ -70,9 +70,9 @@ baselines encode the same wrong event shape they do today. ## 8. Documentation and Generated Definitions -- [ ] 8.1 Update `docs/specs/replay-review.md` if the implemented contract diverges from the spec. -- [ ] 8.2 Update any docs describing replay tool output. -- [ ] 8.3 Run `pnpm run --filter @sentry/mcp-core generate-definitions`. +- [x] 8.1 Update `docs/specs/replay-review.md` if the implemented contract diverges from the spec. Added a status banner (Motivation is retained as the record of what was wrong, in its original present tense), corrected the Map example to the shipped `T+` offset format, and added a "Divergences from the As-Built Implementation" section covering the sortability flag, the `refineInputSchema` hook, shared parameter resolution, cursor precedence, unresolvable error ids, `count_dead_clicks` including rage clicks, and the two upstream fixes. The #120859/#121765 descriptions were swapped in the task notes; verified against the commits and corrected. +- [x] 8.2 Update any docs describing replay tool output. Nothing else describes it: `docs/operations/embedded-agents.md` and `docs/contributing/adding-tools.md` mention replays only in passing and both remain accurate. `proposal.md` and the delta spec are the historical record of the change and are left as written. +- [x] 8.3 Run `pnpm run --filter @sentry/mcp-core generate-definitions`. Already up to date — no regeneration needed. ## 9. Verification From f4915c6212bc662b022e97102608a57b3358eacc Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 15:56:10 -0700 Subject: [PATCH 11/26] chore(replays): Record the automated verification gate for replay review Section 9's automated portion. Targeted replay and catalog tests (167 across 9 files), tsc, lint, full suite (1481 in mcp-core), and token measurement all pass. Token cost is unchanged at 5,598 across 9 direct tools, since get_replay_activity is catalog-only. The remaining task, QA against a real organization, needs a Sentry token and a long real replay; it cannot be satisfied from mocks. Co-Authored-By: Claude Opus 5 --- openspec/changes/improve-replay-review/tasks.md | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index b30743fea..00e25374f 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -76,9 +76,9 @@ baselines encode the same wrong event shape they do today. ## 9. Verification -- [ ] 9.1 Run targeted replay tool tests and catalog availability tests. -- [ ] 9.2 Run `pnpm run tsc`. -- [ ] 9.3 Run `pnpm run lint`. -- [ ] 9.4 Run `pnpm run test`. -- [ ] 9.5 Run `pnpm run measure-tokens` and confirm the added tool definition stays within budget. +- [x] 9.1 Run targeted replay tool tests and catalog availability tests. 167 passed across 9 files. +- [x] 9.2 Run `pnpm run tsc`. Clean. +- [x] 9.3 Run `pnpm run lint`. 2 warnings, both pre-existing and unrelated (`mcp-cloudflare/index.html` noDescendingSpecificity, `oauth/helpers.test.ts` noUnusedImports). +- [x] 9.4 Run `pnpm run test`. 8/8 tasks; mcp-core 1481 passed / 6 skipped. The `smoke-tests` package skips without a deployed `PREVIEW_URL`, which is expected locally and unrelated to replays. +- [x] 9.5 Run `pnpm run measure-tokens` and confirm the added tool definition stays within budget. Unchanged at 5,598 tokens across 9 direct tools — `get_replay_activity` is catalog-only, so it costs no direct-surface budget. - [ ] 9.6 QA against a real organization with the `mcp-qa` skill, since mocks cannot prove the SDK-shape fix. Confirm on a long real replay that the 150-segment and 10MB bounds hold under the Workers memory ceiling, and adjust them if not. From 89b68963da5ce80f749d6272f98e3ab3467d0fd5 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 16:20:11 -0700 Subject: [PATCH 12/26] docs(replays): Specify DOM tree reads as deferred replay work MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Out of Scope section ruled out "web visual snapshots" on the grounds that rendering needs a browser. That is right about images and wrong as a blanket statement: reading DOM structure needs no layout engine, and the structure is already in segments this change downloads and discards. Record what is actually reachable. rrweb FullSnapshot (type 2) carries the serialized DOM, and IncrementalSnapshot (type 3, source 0) carries the mutations to replay it forward to a timestamp. Upstream's which() ignores both, so matching it lost nothing for behavioral summarization, but it leaves structure on the floor. Rooting is close to free: click breadcrumbs already carry payload.data.node.id, and rrweb ids are stable within a recording, so the identifier a subtree read needs is one get_replay_activity already surfaces. Bounding boxes and a visible lens stay out, for the same reason images do — both are layout properties, not recorded facts. An interactive lens is decidable from tags and attributes alone. Two constraints are recorded because they shape any implementation: replaying mutations to a timestamp cannot reuse the existing segment reader without blowing the Workers memory ceiling, and a full tree exceeds any reasonable response size, so rooting plus a lens is the default rather than an option. Also note that SDK masking defaults to on and leaves no marker, so tree text must render as delivered rather than be claimed as redacted — consistent with the redaction rule this spec already sets. Verified against rrweb's type definitions and Sentry's pinned EventType enum rather than from memory; both are now cited. Co-Authored-By: Claude Opus 5 --- docs/specs/replay-review.md | 84 ++++++++++++++++++- .../changes/improve-replay-review/tasks.md | 2 +- 2 files changed, 84 insertions(+), 2 deletions(-) diff --git a/docs/specs/replay-review.md b/docs/specs/replay-review.md index 0188621fe..5e9b68c6b 100644 --- a/docs/specs/replay-review.md +++ b/docs/specs/replay-review.md @@ -376,18 +376,96 @@ separately. Tool count rises by one, well inside the 20-tool target. ## Out of Scope -Web visual snapshots. Rendering a DOM snapshot requires running rrweb playback +Web visual snapshots. *Rendering* a DOM snapshot requires running rrweb playback in a browser, which this server has no place to host. The image plumbing already exists (`get_snapshot_image`, `createImagePreview` in `packages/mcp-core/src/internal/blob-utils.ts`), so this is a hosting gap, not a protocol one. +Note the distinction from the DOM *tree*, which is a different problem with a +different answer. Rendering needs a layout engine; reading the structure does +not, and the structure is already in the segments we download. See "DOM tree +reads" under Future Work. + Mobile replay video is tractable — replays are captured as video and `/projects/{org}/{project}/replays/{replay_id}/videos/{segment_id}/` exists — but it is deferred rather than half-built alongside the web path. ## Future Work +### DOM tree reads + +The largest capability gap against comparable tools. Peers expose a +`review-snapshot`-style call returning a screenshot, a component tree, and +bounding boxes at a timestamp, rooted at an optional component id. Of those four +pieces, the tree and the rooting are reachable here; the image and the boxes are +not, and for the same reason. + +**What the segments already contain.** A recording is rrweb events, and the ones +this spec's taxonomy ignores are exactly the ones carrying structure: + +| `type` | rrweb name | Currently | +|---|---|---| +| 2 | `FullSnapshot` | Ignored — carries the complete serialized DOM | +| 3 | `IncrementalSnapshot` | Ignored except `source: 9` (canvas) | +| 5 | `Custom` | Everything this spec classifies | + +Upstream's `which()` ignores types 2 and 3 too, so nothing was lost by matching +it — Seer summarizes behavior, not structure. But the data is already downloaded +and discarded. + +A `FullSnapshot` holds `serializedNodeWithId` recursively: `id`, `type` +(`NodeType`: Document 0, DocumentType 1, Element 2, Text 3, CDATA 4, Comment 5), +`tagName` and `attributes` on elements, `textContent` on text nodes, and +`childNodes`. Reconstructing state at an arbitrary timestamp means applying each +subsequent `IncrementalSnapshot` with `source: 0` (`Mutation`), whose payload is +four arrays — `adds` (`parentId`, `nextId`, serialized `node`), `removes` +(`parentId`, `id`), `attributes` (`id` plus a partial attribute map, `null` +meaning removal), and `texts` (`id`, `value`). + +**Why rooting is nearly free.** rrweb node ids are stable within a recording and +already reach us: every click breadcrumb carries `payload.data.node.id` +alongside the `tagName` and `attributes` the classifier renders today. So "show +me the DOM around the element that was rage-clicked" needs no new identifier +scheme — the id in the signal is the handle, and `get_replay_activity` already +surfaces those signals. + +**What is not reachable.** Bounding boxes require layout, and rrweb records no +geometry beyond the `Meta` event's viewport width and height. A `visible` lens +is the same problem: visibility is a computed style, not a recorded fact. Both +need a browser, which puts them with the screenshot in Out of Scope. An +`interactive` lens — form controls, buttons, links, ARIA roles — is decidable +from tag and attributes alone, and is the useful one regardless. + +**The two hard constraints.** + +- *Memory.* Replaying mutations to a timestamp means holding a DOM snapshot plus + every mutation up to that point. Real snapshots run to megabytes of JSON and + parsed rrweb objects expand well beyond their serialized size — the same + pressure that produced this spec's 10MB read budget under a 128MB Workers + ceiling. A viable implementation streams segments and applies mutations into a + node map, discarding raw text as it goes, rather than parsing the recording + whole. This is why it cannot simply reuse `getReplayRecordingSegments`. +- *Output size.* A full tree is far past any reasonable tool response. Rooting + at a node id plus an `interactive` lens is the default that keeps output + bounded; a `full` lens needs a depth limit and truncation reporting consistent + with the rest of this spec. + +**Masking is not redaction.** `maskAllText` and `maskAllInputs` default to on, +so a real tree arrives with text and input values already replaced by the SDK, +client-side, with no marker. Per this spec's redaction rule, such values render +as delivered and must not be labeled `` — the tool cannot distinguish +"masked at capture" from "genuinely this text". Only Relay's `[Filtered]` marker +supports that claim. + +**Open questions for QA to answer.** Whether real segments carry a +`FullSnapshot` in the first page or only later — which decides whether a tree +read can be cheap — and whether real clicks carry `node.id` consistently or only +sometimes, which decides whether subtree rooting is a reliable entry point or a +best-effort one. + +### Other + - Friction analysis via `GET /organizations/{org}/replay-selectors/`, which returns `count_dead_clicks`, `count_rage_clicks`, `dom_element`, and `element.component_name` per selector. Component names come from Sentry @@ -520,6 +598,10 @@ real organization confirms or moves them. - Upstream replay sort configuration: `getsentry/sentry` at `src/sentry/replays/usecases/query/configs/aggregate_sort.py` - [Replay recording event spec](https://develop.sentry.dev/sdk/data-model/event-payloads/replay-recording) +- rrweb event, mutation, and serialized-node types: + [`rrweb-io/rrweb` `packages/types/src/index.ts`](https://github.com/rrweb-io/rrweb/blob/master/packages/types/src/index.ts). + Sentry pins its own copy of the `EventType` enum in + `src/sentry/replays/testutils.py`, which is the version to match. - [Sentry Replays API](https://docs.sentry.io/api/replays/) - Tool authoring: [Adding Tools](../contributing/adding-tools.md), and "Tool Output Policy" in [Tool Responses](../contributing/tool-responses.md) diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index 00e25374f..53e09c8c6 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -81,4 +81,4 @@ baselines encode the same wrong event shape they do today. - [x] 9.3 Run `pnpm run lint`. 2 warnings, both pre-existing and unrelated (`mcp-cloudflare/index.html` noDescendingSpecificity, `oauth/helpers.test.ts` noUnusedImports). - [x] 9.4 Run `pnpm run test`. 8/8 tasks; mcp-core 1481 passed / 6 skipped. The `smoke-tests` package skips without a deployed `PREVIEW_URL`, which is expected locally and unrelated to replays. - [x] 9.5 Run `pnpm run measure-tokens` and confirm the added tool definition stays within budget. Unchanged at 5,598 tokens across 9 direct tools — `get_replay_activity` is catalog-only, so it costs no direct-surface budget. -- [ ] 9.6 QA against a real organization with the `mcp-qa` skill, since mocks cannot prove the SDK-shape fix. Confirm on a long real replay that the 150-segment and 10MB bounds hold under the Workers memory ceiling, and adjust them if not. +- [ ] 9.6 QA against a real organization with the `mcp-qa` skill, since mocks cannot prove the SDK-shape fix. Confirm on a long real replay that the 150-segment and 10MB bounds hold under the Workers memory ceiling, and adjust them if not. While there, capture two observations for the deferred DOM tree work (see "DOM tree reads" in `docs/specs/replay-review.md`): whether a `FullSnapshot` (`type: 2`) appears in the first page of segments or only later, and whether real click breadcrumbs carry `payload.data.node.id` consistently. Neither blocks this change. From a57ab723b41429ca45b6023b7233f68ef8c79e32 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 16:47:02 -0700 Subject: [PATCH 13/26] docs(replays): Expand the DOM tree design with reconstruction semantics MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The prior entry established that tree reads are reachable. This works out what a correct one requires, and surfaces two problems the sketch had glossed over. Input values arrive as rrweb source 5 (Input), not as attribute mutations. An implementation applying only source 0 renders every field at its initial value — a stale tree that looks authoritative, which is worse than no tree. Called out explicitly and given a test case, since it would otherwise ship silently. Reconstruction cost is not known to be bounded. rrweb re-snapshots on checkoutEveryNms, and Sentry's SDK treats a recording's first event as a checkout, so a read at time T may only need the nearest preceding snapshot — or, if a recording carries exactly one, may have to apply every mutation in the session. Those two cases have different feature shapes, and building for the wrong one yields a tool that works on short replays and fails on the long ones that matter. Promoted to the first question QA should answer. Also record that element nodes carry no textContent (a label is a child text node, not a property), that reconstruction fidelity must be reported rather than assumed, and that a visible lens is out for the same reason boxes are — both need the cascade resolved. Argue the tool surface three ways and recommend a separate catalog tool: a point-in-time structural read has a different return shape and cost model than a windowed signal list, and only a separate tool can refuse on budget grounds without complicating a contract agents call routinely. Verified against rrweb's type definitions and the Sentry SDK's recording emit handler rather than from memory. Co-Authored-By: Claude Opus 5 --- docs/specs/replay-review.md | 233 ++++++++++++++---- .../changes/improve-replay-review/tasks.md | 2 +- 2 files changed, 185 insertions(+), 50 deletions(-) diff --git a/docs/specs/replay-review.md b/docs/specs/replay-review.md index 5e9b68c6b..6dac6ee41 100644 --- a/docs/specs/replay-review.md +++ b/docs/specs/replay-review.md @@ -414,55 +414,190 @@ Upstream's `which()` ignores types 2 and 3 too, so nothing was lost by matching it — Seer summarizes behavior, not structure. But the data is already downloaded and discarded. -A `FullSnapshot` holds `serializedNodeWithId` recursively: `id`, `type` -(`NodeType`: Document 0, DocumentType 1, Element 2, Text 3, CDATA 4, Comment 5), -`tagName` and `attributes` on elements, `textContent` on text nodes, and -`childNodes`. Reconstructing state at an arbitrary timestamp means applying each -subsequent `IncrementalSnapshot` with `source: 0` (`Mutation`), whose payload is -four arrays — `adds` (`parentId`, `nextId`, serialized `node`), `removes` -(`parentId`, `id`), `attributes` (`id` plus a partial attribute map, `null` -meaning removal), and `texts` (`id`, `value`). - -**Why rooting is nearly free.** rrweb node ids are stable within a recording and -already reach us: every click breadcrumb carries `payload.data.node.id` -alongside the `tagName` and `attributes` the classifier renders today. So "show -me the DOM around the element that was rage-clicked" needs no new identifier -scheme — the id in the signal is the handle, and `get_replay_activity` already -surfaces those signals. - -**What is not reachable.** Bounding boxes require layout, and rrweb records no -geometry beyond the `Meta` event's viewport width and height. A `visible` lens -is the same problem: visibility is a computed style, not a recorded fact. Both -need a browser, which puts them with the screenshot in Out of Scope. An -`interactive` lens — form controls, buttons, links, ARIA roles — is decidable -from tag and attributes alone, and is the useful one regardless. - -**The two hard constraints.** - -- *Memory.* Replaying mutations to a timestamp means holding a DOM snapshot plus - every mutation up to that point. Real snapshots run to megabytes of JSON and - parsed rrweb objects expand well beyond their serialized size — the same - pressure that produced this spec's 10MB read budget under a 128MB Workers - ceiling. A viable implementation streams segments and applies mutations into a - node map, discarding raw text as it goes, rather than parsing the recording - whole. This is why it cannot simply reuse `getReplayRecordingSegments`. -- *Output size.* A full tree is far past any reasonable tool response. Rooting - at a node id plus an `interactive` lens is the default that keeps output - bounded; a `full` lens needs a depth limit and truncation reporting consistent - with the rest of this spec. - -**Masking is not redaction.** `maskAllText` and `maskAllInputs` default to on, -so a real tree arrives with text and input values already replaced by the SDK, -client-side, with no marker. Per this spec's redaction rule, such values render -as delivered and must not be labeled `` — the tool cannot distinguish -"masked at capture" from "genuinely this text". Only Relay's `[Filtered]` marker -supports that claim. - -**Open questions for QA to answer.** Whether real segments carry a -`FullSnapshot` in the first page or only later — which decides whether a tree -read can be cheap — and whether real clicks carry `node.id` consistently or only -sometimes, which decides whether subtree rooting is a reliable entry point or a -best-effort one. +#### The reconstruction model + +A `FullSnapshot` holds `serializedNodeWithId` recursively. Node shapes differ by +`NodeType` — Document 0, DocumentType 1, Element 2, Text 3, CDATA 4, Comment 5 — +and the distinction that matters for rendering is that **element nodes have no +`textContent`**. Their text lives in child text nodes, so a button's label is a +child, not a property. Elements carry `tagName`, `attributes`, and `childNodes`; +text, CDATA, and comment nodes carry `textContent`. + +State at time *T* is the last `FullSnapshot` at or before *T*, with every +intervening `IncrementalSnapshot` applied in order. Two sources mutate structure +or content: + +- `source: 0` (`Mutation`) — four arrays. `adds` (`parentId`, `nextId`, + serialized `node`), `removes` (`parentId`, `id`), `attributes` (`id` plus a + partial map whose `null` values mean removal and whose values may be a + `styleOMValue` rather than a string), and `texts` (`id`, `value`). +- `source: 5` (`Input`) — `id`, `text`, `isChecked`. **Input values do not + arrive as attribute mutations.** A tree that applies only `source: 0` shows + every field at its initial value, which is worse than showing nothing: it + looks authoritative and is stale. + +`nextId` is the ordering handle for `adds`; `previousId` exists only for +backward compatibility and should be ignored. An `add` whose `parentId` is +unknown — because it was pruned, or the snapshot was truncated — must be +dropped rather than reparented, and the drop counted toward the fidelity report +below. + +#### Reconstruction is not guaranteed to be cheap + +The naive read is "start at segment zero, apply everything". Whether that is +necessary depends on how often full snapshots appear, and the answer is not +fixed: rrweb re-snapshots on `checkoutEveryNms`/`checkoutEveryNth`, and Sentry's +SDK additionally treats the first event of a recording as a checkout +(`handleRecordingEmit`). So a recording may contain several `FullSnapshot` +events, and a read at *T* only needs the nearest one at or before it. + +This is the single largest open question, and it decides the shape of the +feature: + +- **If checkouts are frequent** — a tree read pages segments until it passes *T*, + keeping only the most recent snapshot seen, then applies the remaining + mutations. Cost is bounded by checkout spacing, not by session length. +- **If a recording has exactly one snapshot at segment zero** — a read at the end + of a long session must apply every mutation in between, and cost grows with + session length. That is the case where the memory ceiling binds and where a + read may have to refuse rather than truncate. + +QA should answer this before the interface is settled. Implementing for the +frequent-checkout case and discovering the single-snapshot case in production +means a tool that works on short replays and OOMs on the long ones that matter. + +#### Fidelity must be reported, not assumed + +A reconstruction can be complete, partial, or wrong, and the three are +indistinguishable from the output alone. Consistent with this spec's rule that +truncation is always stated, a tree read reports what it did: + +- Which `FullSnapshot` it started from, as an offset. +- How many mutation events it applied. +- How many operations it dropped, and why (unknown `parentId`, unknown `id`, + malformed payload). +- Whether it stopped early against a budget. + +A read that dropped a meaningful fraction of its mutations is a read whose tree +should not be trusted for the element in question, and the caller cannot know +that unless told. + +#### What is not reachable + +Bounding boxes require layout, and rrweb records no geometry beyond the `Meta` +event's viewport width and height. A `visible` lens is the same problem: +visibility is a computed style, not a recorded fact — `display: none` on an +ancestor is knowable only by resolving the cascade, which is a browser's job. +Both belong with the screenshot in Out of Scope. + +An `interactive` lens is decidable from tag and attributes alone — form +controls, buttons, links, `[role]`, `[onclick]`, `[tabindex]` — and is the +useful one regardless, since it is the set a user can act on. + +#### Rooting is nearly free + +rrweb node ids are stable within a recording and already reach us: every click +breadcrumb carries `payload.data.node.id` alongside the `tagName` and +`attributes` the classifier renders today. So "show me the DOM around the +element that was rage-clicked" needs no new identifier scheme — the id in the +signal is the handle, and `get_replay_activity` already surfaces those signals. + +This is what makes the feature coherent rather than a curiosity: the map finds +the failure, the activity read names the element, and the tree read explains +what was around it. Each step hands the next a concrete handle. + +#### Tool surface + +Three options, with the tradeoff that decides it. + +**A. A separate `get_replay_dom` tool.** A point-in-time structural read is a +different operation from a windowed signal list: different return shape, +different cost model, different failure modes. It is also the only option where +a read can refuse on budget grounds without complicating an existing tool's +contract. + +**B. A `grain: "dom"` on `get_replay_activity`.** Reuses the tool, but `grain` +currently means "how verbose", not "what kind of thing". A `dom` grain would +change the return type rather than its verbosity, and would need `startMs` +and `endMs` collapsed to a point, which the window semantics do not express. + +**C. An `include: ["tree"]` parameter on `get_replay_activity`.** Closest to the +peer design, but pushes a second, much more expensive operation behind a +parameter on a tool agents already call routinely. A caller asking for signals +should not risk a multi-megabyte reconstruction because a default changed. + +**Recommendation: A.** The catalog is the right home — it costs no direct-surface +budget, which is why `get_replay_activity` landed there too. Sketch: + +```typescript +get_replay_dom({ + organizationSlug, replayId, regionUrl, // or replayUrl + atMs: number, // required; no sensible default + rootNodeId?: number, // from a click signal + lens?: "interactive" | "full", // default "interactive" + maxDepth?: number, + maxNodes?: number, +}) +``` + +`atMs` is deliberately required. A tree with no timestamp would default to +either end of the session, and both defaults are wrong often enough that +guessing is worse than asking. + +Expected output, rooted at a click's node id: + +```text +# Replay 7e07485f… DOM at T+3m 1.3s + +Reconstructed from the snapshot at T+0.4s, applying 1,847 mutations. +Dropped 3 operations (unknown parentId). + +form#checkout-form + ├─ div.address-block + │ ├─ input#unit [value="***"] + │ └─ input#zip [value="***"] + └─ button#complete-order "Complete order" [disabled] + +Node ids: form#checkout-form=88, button#complete-order=96 +``` + +#### Masking is not redaction + +`maskAllText` and `maskAllInputs` default to on, so a real tree arrives with +text and input values already replaced by the SDK, client-side, with no marker. +Per this spec's redaction rule, such values render as delivered and must not be +labeled `` — the tool cannot distinguish "masked at capture" from +"genuinely this text". Only Relay's `[Filtered]` marker supports that claim. + +The practical consequence is that a tree is a **structural** artifact, not a +content one. It answers "was the button disabled", "did the error node exist +yet", "what was around the element the user rage-clicked". It does not answer +"what did the user type", and should not be built as though it might. + +#### Testing + +Beyond the usual unit coverage, three cases carry the risk: + +- **Reconstruction correctness.** Apply a known mutation sequence to a known + snapshot and assert the resulting tree — including that an `add` with an + unknown `parentId` is dropped and counted, not reparented. +- **Input values come from `source: 5`.** A fixture where a field's value + changes only via an input event must render the new value. This is the + failure that would otherwise ship silently, since a `source: 0`-only + implementation produces a plausible-looking stale tree. +- **Budget refusal.** A recording that cannot be reconstructed within budget + must say so rather than return a partial tree that reads as complete. + +#### Open questions for QA + +- Does a `FullSnapshot` appear only at segment zero, or periodically? This + decides whether reconstruction cost scales with session length or with + checkout spacing — see above. +- Do real click breadcrumbs carry `payload.data.node.id` consistently, or only + sometimes? This decides whether subtree rooting is a reliable entry point or + best-effort. +- How large is a real `FullSnapshot` in practice? The 10MB segment budget was + set for signal extraction; a tree read has a different profile. ### Other diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index 53e09c8c6..b868276a9 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -81,4 +81,4 @@ baselines encode the same wrong event shape they do today. - [x] 9.3 Run `pnpm run lint`. 2 warnings, both pre-existing and unrelated (`mcp-cloudflare/index.html` noDescendingSpecificity, `oauth/helpers.test.ts` noUnusedImports). - [x] 9.4 Run `pnpm run test`. 8/8 tasks; mcp-core 1481 passed / 6 skipped. The `smoke-tests` package skips without a deployed `PREVIEW_URL`, which is expected locally and unrelated to replays. - [x] 9.5 Run `pnpm run measure-tokens` and confirm the added tool definition stays within budget. Unchanged at 5,598 tokens across 9 direct tools — `get_replay_activity` is catalog-only, so it costs no direct-surface budget. -- [ ] 9.6 QA against a real organization with the `mcp-qa` skill, since mocks cannot prove the SDK-shape fix. Confirm on a long real replay that the 150-segment and 10MB bounds hold under the Workers memory ceiling, and adjust them if not. While there, capture two observations for the deferred DOM tree work (see "DOM tree reads" in `docs/specs/replay-review.md`): whether a `FullSnapshot` (`type: 2`) appears in the first page of segments or only later, and whether real click breadcrumbs carry `payload.data.node.id` consistently. Neither blocks this change. +- [ ] 9.6 QA against a real organization with the `mcp-qa` skill, since mocks cannot prove the SDK-shape fix. Confirm on a long real replay that the 150-segment and 10MB bounds hold under the Workers memory ceiling, and adjust them if not. While there, capture three observations for the deferred DOM tree work (see "DOM tree reads" in `docs/specs/replay-review.md`): whether `FullSnapshot` (`type: 2`) events appear periodically or only once at segment zero — this decides whether reconstruction cost scales with checkout spacing or with session length, and so decides the feature's shape; whether real click breadcrumbs carry `payload.data.node.id` consistently; and how large a real `FullSnapshot` is. None of these block this change. From d5d0a68240bb97cc5e5c16e542714401592ad918 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Mon, 17 Aug 2026 17:04:24 -0700 Subject: [PATCH 14/26] docs(replays): Settle the DOM tree surface on a separate catalog tool MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Turn the three-way comparison into a decision. get_replay_dom is a separate catalog-only tool: a point-in-time structural read differs from a windowed signal list in return shape, cost model, and failure modes, and it is the only option where a read can refuse on budget grounds without complicating a contract agents call routinely. Keep the rejected options as a short record so the choice is not relitigated from scratch, and note why a third replay tool is affordable: the ≤20 target governs the direct surface, which stands at 9, while catalog tools carry no per-session token overhead. Two consequences are written down because they are easy to drop under implementation pressure. Refusal is a supported outcome rather than a failure, since a partial tree is indistinguishable from a complete one at the point of use. And the tool returns structure only, so it composes with get_replay_activity instead of drifting into a competing account of the same moment. Gating, parameter resolution, and the constraint check follow the existing replay tools; internal/tool-helpers/replay.ts was extracted for this case when the second one landed. Co-Authored-By: Claude Opus 5 --- docs/specs/replay-review.md | 68 +++++++++++++++++++++++++------------ 1 file changed, 46 insertions(+), 22 deletions(-) diff --git a/docs/specs/replay-review.md b/docs/specs/replay-review.md index 6dac6ee41..2aa38f1ea 100644 --- a/docs/specs/replay-review.md +++ b/docs/specs/replay-review.md @@ -506,28 +506,30 @@ This is what makes the feature coherent rather than a curiosity: the map finds the failure, the activity read names the element, and the tree read explains what was around it. Each step hands the next a concrete handle. -#### Tool surface - -Three options, with the tradeoff that decides it. - -**A. A separate `get_replay_dom` tool.** A point-in-time structural read is a -different operation from a windowed signal list: different return shape, -different cost model, different failure modes. It is also the only option where -a read can refuse on budget grounds without complicating an existing tool's -contract. - -**B. A `grain: "dom"` on `get_replay_activity`.** Reuses the tool, but `grain` -currently means "how verbose", not "what kind of thing". A `dom` grain would -change the return type rather than its verbosity, and would need `startMs` -and `endMs` collapsed to a point, which the window semantics do not express. - -**C. An `include: ["tree"]` parameter on `get_replay_activity`.** Closest to the -peer design, but pushes a second, much more expensive operation behind a -parameter on a tool agents already call routinely. A caller asking for signals -should not risk a multi-megabyte reconstruction because a default changed. - -**Recommendation: A.** The catalog is the right home — it costs no direct-surface -budget, which is why `get_replay_activity` landed there too. Sketch: +#### Tool surface: a separate `get_replay_dom` + +**Decided: a separate catalog-only tool.** A point-in-time structural read is a +different operation from a windowed signal list — different return shape, +different cost model, different failure modes — and it is the only option where +a read can refuse on budget grounds without complicating a contract agents call +routinely. + +The alternatives, recorded so the decision is not relitigated from scratch: + +- *A `grain: "dom"` on `get_replay_activity`.* `grain` means "how verbose", not + "what kind of thing". A `dom` grain would change the return type rather than + its verbosity, and needs `startMs`/`endMs` collapsed to a point, which the + window semantics do not express. +- *An `include: ["tree"]` parameter on `get_replay_activity`.* Closest to + comparable tools, but puts a second, far more expensive operation behind a + parameter on a routinely-called tool. A caller asking for signals should not + risk a multi-megabyte reconstruction because a default moved. + +Catalog-only matters for cost: the tool-count target of ≤20 (hard limit 25) +governs the direct surface, which stands at 9. Catalog tools are discovered +through `search_sentry_tools` and add no per-session token overhead, which is +why `get_replay_activity` landed there and why a third replay tool is +affordable at all. ```typescript get_replay_dom({ @@ -540,10 +542,28 @@ get_replay_dom({ }) ``` +Gating matches the other two replay tools: `skills: ["inspect"]`, +`requiredCapabilities: ["replays"]`, scopes `org:read`, `project:read`, +`event:read`. Parameter resolution and the project-constraint check reuse +`internal/tool-helpers/replay.ts`, which was extracted for exactly this reason +when the second replay tool landed. + `atMs` is deliberately required. A tree with no timestamp would default to either end of the session, and both defaults are wrong often enough that guessing is worse than asking. +Two behaviors follow from the decision and should not be quietly dropped in +implementation: + +- **Refusal is a supported outcome.** When a reconstruction cannot complete + within budget, the tool says so and explains what would help — a nearer + `atMs`, or a `rootNodeId`. It does not return a partial tree, because a + partial tree is indistinguishable from a complete one at the point of use. +- **It composes with the other two rather than duplicating them.** The tool + returns structure only. Anything about *what happened* stays in + `get_replay_activity`, so the two cannot drift into competing accounts of the + same moment. + Expected output, rooted at a click's node id: ```text @@ -587,6 +607,10 @@ Beyond the usual unit coverage, three cases carry the risk: implementation produces a plausible-looking stale tree. - **Budget refusal.** A recording that cannot be reconstructed within budget must say so rather than return a partial tree that reads as complete. +- **Gating and surface.** Like the other replay tools: hidden when the + constrained project lacks the `replays` capability, absent from the direct + top-level surface, reachable through the catalog. The existing + `availability.test.ts` cases extend to cover it. #### Open questions for QA From 535a3dd21b97a62b61d5aa2b4f8567b275ddf74d Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 10:13:29 -0700 Subject: [PATCH 15/26] feat(replays): Add a streaming segment reader MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit DOM reconstruction folds a whole recording into a running node map and keeps only the result. The existing reader accumulates every segment before returning, so a tree read would hold the raw events and the derived state at once — double the peak for no benefit, under a 128MB Workers ceiling that the byte budget already exists to respect. Add streamReplayRecordingSegments, which invokes a callback per segment in wire order and retains nothing. Returning "stop" ends the read immediately, so a caller that has passed the moment it cares about does not page the rest of the session. Stopping by choice is reported as truncatedBy: null. Conflating it with a budget stop would make a deliberate early exit look like a lost tail. getReplayRecordingSegments becomes a thin wrapper that pushes into an array, so signal extraction is unchanged — its 111 existing tests pass untouched. A test asserts the two agree on segments, count, and bytes, since any drift between them would now be a bug in the wrapper. The stop path is mutation-verified: forcing the branch false fails the early-exit test rather than passing quietly. Co-Authored-By: Claude Opus 5 --- packages/mcp-core/src/api-client/client.ts | 64 ++++++++++++-- .../src/api-client/replay-client.test.ts | 88 +++++++++++++++++++ 2 files changed, 146 insertions(+), 6 deletions(-) diff --git a/packages/mcp-core/src/api-client/client.ts b/packages/mcp-core/src/api-client/client.ts index e6f5293ee..4ad3ad7ed 100644 --- a/packages/mcp-core/src/api-client/client.ts +++ b/packages/mcp-core/src/api-client/client.ts @@ -4080,6 +4080,46 @@ export class SentryApiService { * rather than presenting a partial recording as complete. */ async getReplayRecordingSegments( + params: { + organizationSlug: string; + projectSlugOrId: string; + replayId: string; + maxSegments?: number; + maxBytes?: number; + }, + opts?: RequestOptions, + ): Promise { + const segments: ReplayRecordingSegments = []; + const stats = await this.streamReplayRecordingSegments( + params, + (segment) => { + segments.push(segment); + }, + opts, + ); + + return { ...stats, segments }; + } + + /** + * Reads a recording segment by segment, without retaining it. + * + * {@link getReplayRecordingSegments} accumulates every segment before + * returning, which is fine for signal extraction — signals are a tiny + * projection of the events they come from — but not for anything that must + * fold a whole recording into a running state. DOM reconstruction applies + * thousands of mutations and keeps only the resulting node map, so holding + * the raw events as well would double the peak for no benefit. + * + * `onSegment` is called once per segment in wire order. Returning `"stop"` + * ends the read immediately, which lets a caller that has passed the moment + * it cares about avoid paging the rest of the session. + * + * The budgets and their reporting are identical to the buffering read: the + * caller is told which bound stopped it, so a partial fold is never + * presented as a whole one. + */ + async streamReplayRecordingSegments( { organizationSlug, projectSlugOrId, @@ -4093,11 +4133,16 @@ export class SentryApiService { maxSegments?: number; maxBytes?: number; }, + onSegment: ( + segment: ReplayRecordingSegments[number], + index: number, + ) => "stop" | void, opts?: RequestOptions, - ): Promise { - const segments: ReplayRecordingSegments = []; + ): Promise> { let cursor: string | null = null; let bytesRead = 0; + let segmentsRead = 0; + let stopped = false; let truncatedBy: ReplayRecordingSegmentsResult["truncatedBy"] = null; do { @@ -4129,14 +4174,21 @@ export class SentryApiService { const page = ReplayRecordingSegmentsSchema.parse(body); for (const segment of page) { - if (segments.length >= maxSegments) { + if (segmentsRead >= maxSegments) { truncatedBy = "segments"; break; } - segments.push(segment); + const outcome = onSegment(segment, segmentsRead); + segmentsRead += 1; + if (outcome === "stop") { + // The caller has what it needs. This is not truncation: the read + // ended by choice, with nothing missing from what was asked for. + stopped = true; + break; + } } - if (truncatedBy) { + if (truncatedBy || stopped) { break; } if (bytesRead >= maxBytes) { @@ -4149,7 +4201,7 @@ export class SentryApiService { cursor = getNextCursor(response.headers.get("link")); } while (cursor); - return { segments, truncatedBy, segmentsRead: segments.length, bytesRead }; + return { truncatedBy, segmentsRead, bytesRead }; } /** diff --git a/packages/mcp-core/src/api-client/replay-client.test.ts b/packages/mcp-core/src/api-client/replay-client.test.ts index 2a3a5cf13..295c8c9c0 100644 --- a/packages/mcp-core/src/api-client/replay-client.test.ts +++ b/packages/mcp-core/src/api-client/replay-client.test.ts @@ -124,6 +124,94 @@ describe("getReplayRecordingSegments", () => { }); }); +describe("streamReplayRecordingSegments", () => { + function streamSegments( + replayId: string, + onSegment: (segment: unknown, index: number) => "stop" | void, + overrides = {}, + ) { + return apiService.streamReplayRecordingSegments( + { + organizationSlug: "sentry-mcp-evals", + projectSlugOrId: String(replayDetailsFixture.project_id), + replayId, + ...overrides, + }, + onSegment, + ); + } + + it("delivers every segment in wire order without retaining them", async () => { + const seen: number[] = []; + const stats = await streamSegments(PAGED_REPLAY_ID, (_segment, index) => { + seen.push(index); + }); + + expect(seen).toEqual([0, 1, 2, 3, 4]); + expect(stats.segmentsRead).toBe(replayRecordingSegmentsPagedFixture.length); + expect(stats.truncatedBy).toBeNull(); + // The stats shape is the buffering read's minus the payload, so callers + // can report the same bounds without holding the recording. + expect(stats).not.toHaveProperty("segments"); + }); + + it("stops paging when the callback says stop", async () => { + // The point of the early exit: a read that has passed the moment it cares + // about should not fetch the rest of the session. + const requested: string[] = []; + mswServer.use( + http.get(SEGMENTS_URL(PAGED_REPLAY_ID), ({ request }) => { + requested.push(request.url); + return HttpResponse.json([replayRecordingSegmentsPagedFixture[0]], { + headers: { + Link: `<${SEGMENTS_URL(PAGED_REPLAY_ID)}?cursor=0:1:0>; rel="next"; results="true"; cursor="0:1:0"`, + }, + }); + }), + ); + + const stats = await streamSegments(PAGED_REPLAY_ID, () => "stop"); + + expect(requested).toHaveLength(1); + expect(stats.segmentsRead).toBe(1); + // Stopping by choice is not truncation; conflating them would make a + // deliberate early exit look like a lost tail. + expect(stats.truncatedBy).toBeNull(); + }); + + it("reports the segment budget the same way the buffering read does", async () => { + const stats = await streamSegments(PAGED_REPLAY_ID, () => {}, { + maxSegments: 3, + }); + + expect(stats.segmentsRead).toBe(3); + expect(stats.truncatedBy).toBe("segments"); + }); + + it("reports the byte budget the same way the buffering read does", async () => { + const stats = await streamSegments(PAGED_REPLAY_ID, () => {}, { + maxBytes: 1, + }); + + expect(stats.truncatedBy).toBe("bytes"); + expect(stats.bytesRead).toBeGreaterThan(0); + }); + + it("agrees with the buffering read on what it saw", async () => { + // The buffering read is now a thin wrapper over this one, so any drift + // between them is a bug in the wrapper. + const streamed: unknown[] = []; + const stats = await streamSegments(PAGED_REPLAY_ID, (segment) => { + streamed.push(segment); + }); + const buffered = await readSegments(PAGED_REPLAY_ID); + + expect(streamed).toEqual(buffered.segments); + expect(stats.segmentsRead).toBe(buffered.segmentsRead); + expect(stats.bytesRead).toBe(buffered.bytesRead); + }); +}); + describe("getReplayErrorEvents", () => { it("resolves a batch of error ids in one request", async () => { const requested: string[] = []; From 2dea16223c6b10517102d156172fe088f3e48655 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 10:18:16 -0700 Subject: [PATCH 16/26] feat(replays): Reconstruct DOM state from rrweb snapshots and mutations MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds the fold behind a DOM read: a flat node map built from a FullSnapshot (type 2) and advanced by IncrementalSnapshot (type 3) mutations up to a target time. Semantics come from rrweb's own type definitions rather than from the SDK, which does not re-export them. Input values are applied from source 5, not from attribute mutations. This is the failure worth naming: a source 0-only fold renders every form field at the value the page shipped with, so the tree looks current and reports stale content. The initial value still comes from the attribute, and the two are kept separate. A later FullSnapshot supersedes the map entirely rather than merging. Merging would resurrect nodes the page had already discarded, and it makes the one-snapshot and periodic-checkout cases behave identically without branching on which a recording uses. An add whose parentId is unknown is dropped and counted, never reparented to the root: fabricated structure is indistinguishable from recorded structure once rendered. Removes delete the subtree, so a later add cannot reattach orphans under a parent that no longer exists, and a relocation detaches before inserting so a moved node does not appear twice. Fidelity is reported, not assumed. Drops are counted by reason (unknown-parent, unknown-node, malformed, duplicate-id), a missing snapshot is distinguished from an empty page, and apply() returns past-target so a streaming caller stops paging at the moment it cares about. Subtree deletion is iterative — untrusted DOM depth should not reach the call stack. Five mutations of the risky paths each fail a test: dropping source 5, reparenting unknown parents, skipping subtree deletion, ignoring nextId ordering, and merging snapshots instead of superseding. Co-Authored-By: Claude Opus 5 --- .../mcp-core/src/internal/replay-dom.test.ts | 432 +++++++++++++ packages/mcp-core/src/internal/replay-dom.ts | 576 ++++++++++++++++++ 2 files changed, 1008 insertions(+) create mode 100644 packages/mcp-core/src/internal/replay-dom.test.ts create mode 100644 packages/mcp-core/src/internal/replay-dom.ts diff --git a/packages/mcp-core/src/internal/replay-dom.test.ts b/packages/mcp-core/src/internal/replay-dom.test.ts new file mode 100644 index 000000000..d752853ad --- /dev/null +++ b/packages/mcp-core/src/internal/replay-dom.test.ts @@ -0,0 +1,432 @@ +/** + * Reconstruction is the part of a DOM read that fails quietly: a tree built + * from a mishandled mutation stream is structurally valid and wrong, and + * nothing downstream can tell. These tests assert the resulting structure + * rather than that the code ran. + */ +import { describe, expect, it } from "vitest"; +import { + DomReconstructor, + NODE_TYPE_DOCUMENT, + NODE_TYPE_ELEMENT, + NODE_TYPE_TEXT, + countDropped, + type DomNode, +} from "./replay-dom.js"; +import type { ReplayRecordingEvent } from "../api-client"; + +const START_MS = 1_744_027_200_000; + +/** A serialized element node, as rrweb emits it inside a FullSnapshot. */ +function element( + id: number, + tagName: string, + attributes: Record = {}, + childNodes: unknown[] = [], +) { + return { id, type: NODE_TYPE_ELEMENT, tagName, attributes, childNodes }; +} + +function text(id: number, textContent: string) { + return { id, type: NODE_TYPE_TEXT, textContent }; +} + +/** + * A FullSnapshot wrapping the given body children. + * + * Mirrors the real shape: an id-less document wrapper containing `html`, + * containing `body`. The wrapper carries no id, which is why reconstruction has + * to descend through it rather than expect a root id. + */ +function snapshot(offsetMs: number, bodyChildren: unknown[]) { + return { + type: 2, + timestamp: START_MS + offsetMs, + data: { + node: { + type: NODE_TYPE_DOCUMENT, + childNodes: [ + element(1, "html", {}, [element(2, "body", {}, bodyChildren)]), + ], + }, + }, + } as unknown as ReplayRecordingEvent; +} + +function mutation( + offsetMs: number, + payload: { + adds?: unknown[]; + removes?: unknown[]; + attributes?: unknown[]; + texts?: unknown[]; + }, +) { + return { + type: 3, + timestamp: START_MS + offsetMs, + data: { + source: 0, + adds: payload.adds ?? [], + removes: payload.removes ?? [], + attributes: payload.attributes ?? [], + texts: payload.texts ?? [], + }, + } as unknown as ReplayRecordingEvent; +} + +function inputEvent( + offsetMs: number, + id: number, + fields: { text?: string; isChecked?: boolean }, +) { + return { + type: 3, + timestamp: START_MS + offsetMs, + data: { source: 5, id, ...fields }, + } as unknown as ReplayRecordingEvent; +} + +/** Feed events and finalize, as a streaming caller would. */ +function reconstruct(events: ReplayRecordingEvent[], atMs = START_MS + 10_000) { + const reconstructor = new DomReconstructor({ atMs }); + for (const event of events) { + if (reconstructor.apply(event) === "past-target") { + break; + } + } + return reconstructor.result(START_MS); +} + +function childTags(node: DomNode | undefined, nodes: Map) { + return (node?.childIds ?? []).map((id) => nodes.get(id)?.tagName ?? "?"); +} + +describe("FullSnapshot ingest", () => { + it("descends through the id-less document wrapper", () => { + const result = reconstruct([snapshot(400, [element(3, "div")])]); + + // The wrapper has no id, so `html` — the first identified node — is root. + expect(result.rootId).toBe(1); + expect(result.nodes.get(1)?.tagName).toBe("html"); + expect(result.missingSnapshot).toBe(false); + expect(result.snapshotOffsetMs).toBe(400); + }); + + it("records parent and child links in both directions", () => { + const result = reconstruct([snapshot(0, [element(3, "button")])]); + + expect(result.nodes.get(2)?.childIds).toEqual([3]); + expect(result.nodes.get(3)?.parentId).toBe(2); + }); + + it("keeps text as child nodes rather than element properties", () => { + // Element nodes carry no textContent in rrweb; a label is a child. + const result = reconstruct([ + snapshot(0, [element(3, "button", {}, [text(4, "Complete order")])]), + ]); + + expect(result.nodes.get(3)?.textContent).toBeUndefined(); + expect(result.nodes.get(4)?.textContent).toBe("Complete order"); + }); + + it("reports a missing snapshot rather than an empty tree", () => { + // Reading a window whose snapshot was never fetched must be + // distinguishable from reading a page that had no content. + const result = reconstruct([ + mutation(100, { texts: [{ id: 3, value: "x" }] }), + ]); + + expect(result.missingSnapshot).toBe(true); + expect(result.rootId).toBeNull(); + }); + + it("supersedes an earlier snapshot entirely", () => { + // A later FullSnapshot is a complete state. Merging into the previous one + // would resurrect nodes the page had already discarded. + const result = reconstruct([ + snapshot(0, [element(3, "div", { id: "old" })]), + mutation(100, { + attributes: [{ id: 3, attributes: { class: "stale" } }], + }), + snapshot(200, [element(9, "section", { id: "new" })]), + ]); + + expect(result.nodes.has(3)).toBe(false); + expect(result.nodes.get(9)?.attributes.id).toBe("new"); + expect(result.snapshotOffsetMs).toBe(200); + // Mutations applied before the newer snapshot no longer count toward it. + expect(result.mutationsApplied).toBe(0); + }); + + it("drops a duplicate id instead of overwriting a subtree", () => { + const result = reconstruct([ + snapshot(0, [ + element(3, "div", {}, [element(4, "span")]), + element(3, "aside"), + ]), + ]); + + expect(result.nodes.get(3)?.tagName).toBe("div"); + expect(result.dropped["duplicate-id"]).toBe(1); + }); +}); + +describe("mutations", () => { + it("inserts an added node before its nextId sibling", () => { + const result = reconstruct([ + snapshot(0, [element(3, "header"), element(4, "footer")]), + mutation(100, { + adds: [{ parentId: 2, nextId: 4, node: element(5, "main") }], + }), + ]); + + expect(childTags(result.nodes.get(2), result.nodes)).toEqual([ + "header", + "main", + "footer", + ]); + }); + + it("appends when nextId is null", () => { + const result = reconstruct([ + snapshot(0, [element(3, "header")]), + mutation(100, { + adds: [{ parentId: 2, nextId: null, node: element(5, "main") }], + }), + ]); + + expect(childTags(result.nodes.get(2), result.nodes)).toEqual([ + "header", + "main", + ]); + }); + + it("drops an add whose parent is unknown rather than reparenting it", () => { + // Reparenting to the root would fabricate structure that never existed, + // and the caller would have no way to know. + const result = reconstruct([ + snapshot(0, [element(3, "div")]), + mutation(100, { + adds: [{ parentId: 999, nextId: null, node: element(5, "span") }], + }), + ]); + + expect(result.nodes.has(5)).toBe(false); + expect(result.dropped["unknown-parent"]).toBe(1); + expect(countDropped(result.dropped)).toBe(1); + }); + + it("removes a node and its descendants", () => { + // Leaving descendants behind would let a later add reattach them under a + // parent that no longer exists. + const result = reconstruct([ + snapshot(0, [ + element(3, "div", {}, [element(4, "span", {}, [text(5, "hi")])]), + ]), + mutation(100, { removes: [{ parentId: 2, id: 3 }] }), + ]); + + expect(result.nodes.has(3)).toBe(false); + expect(result.nodes.has(4)).toBe(false); + expect(result.nodes.has(5)).toBe(false); + expect(result.nodes.get(2)?.childIds).toEqual([]); + }); + + it("relocates a node without duplicating its subtree", () => { + // rrweb emits a move as remove-plus-add. Handling the add alone would + // leave the node in two places. + const result = reconstruct([ + snapshot(0, [ + element(3, "div", {}, [element(4, "span")]), + element(6, "aside"), + ]), + mutation(100, { + adds: [{ parentId: 6, nextId: null, node: element(4, "span") }], + }), + ]); + + expect(result.nodes.get(3)?.childIds).toEqual([]); + expect(result.nodes.get(6)?.childIds).toEqual([4]); + expect(result.nodes.get(4)?.parentId).toBe(6); + }); + + it("applies attribute changes and removals", () => { + const result = reconstruct([ + snapshot(0, [element(3, "button", { disabled: true, class: "primary" })]), + mutation(100, { + attributes: [ + { id: 3, attributes: { disabled: null, class: "loading" } }, + ], + }), + ]); + + expect(result.nodes.get(3)?.attributes.disabled).toBeUndefined(); + expect(result.nodes.get(3)?.attributes.class).toBe("loading"); + }); + + it("records a style object change without reassembling the declaration", () => { + const result = reconstruct([ + snapshot(0, [element(3, "div")]), + mutation(100, { + attributes: [{ id: 3, attributes: { style: { display: "none" } } }], + }), + ]); + + expect(result.nodes.get(3)?.attributes.style).toBe("[style changed]"); + }); + + it("applies text changes", () => { + const result = reconstruct([ + snapshot(0, [element(3, "span", {}, [text(4, "Loading")])]), + mutation(100, { texts: [{ id: 4, value: "Failed" }] }), + ]); + + expect(result.nodes.get(4)?.textContent).toBe("Failed"); + }); + + it("counts a mutation against an unknown node", () => { + const result = reconstruct([ + snapshot(0, [element(3, "div")]), + mutation(100, { texts: [{ id: 999, value: "x" }] }), + ]); + + expect(result.dropped["unknown-node"]).toBe(1); + }); + + it("ignores sources that change nothing structural", () => { + // Mouse movement and scroll are the bulk of a real recording; treating + // them as mutations would inflate the fidelity report into noise. + const mouseMove = { + type: 3, + timestamp: START_MS + 100, + data: { source: 1, positions: [] }, + } as unknown as ReplayRecordingEvent; + + const result = reconstruct([snapshot(0, [element(3, "div")]), mouseMove]); + + expect(result.mutationsApplied).toBe(0); + expect(countDropped(result.dropped)).toBe(0); + }); +}); + +describe("input values", () => { + it("reads the starting value from the attribute", () => { + const result = reconstruct([ + snapshot(0, [element(3, "input", { value: "initial" })]), + ]); + + expect(result.nodes.get(3)?.inputValue).toBe("initial"); + }); + + it("updates from a source 5 input event, not an attribute mutation", () => { + // The failure this guards: applying only source 0 leaves every field at + // its initial value, producing a tree that looks current and is stale. + const result = reconstruct([ + snapshot(0, [element(3, "input", { value: "initial" })]), + inputEvent(100, 3, { text: "typed by the user" }), + ]); + + expect(result.nodes.get(3)?.inputValue).toBe("typed by the user"); + // The attribute is untouched — the two are deliberately separate, since + // the attribute records what the page shipped with. + expect(result.nodes.get(3)?.attributes.value).toBe("initial"); + }); + + it("tracks checkbox state", () => { + const result = reconstruct([ + snapshot(0, [element(3, "input", { type: "checkbox" })]), + inputEvent(100, 3, { isChecked: true }), + ]); + + expect(result.nodes.get(3)?.inputChecked).toBe(true); + }); + + it("counts an input event for an unknown node", () => { + const result = reconstruct([ + snapshot(0, [element(3, "div")]), + inputEvent(100, 999, { text: "x" }), + ]); + + expect(result.dropped["unknown-node"]).toBe(1); + }); +}); + +describe("the target time", () => { + it("reports past-target so a streaming caller can stop paging", () => { + const reconstructor = new DomReconstructor({ atMs: START_MS + 500 }); + + expect(reconstructor.apply(snapshot(0, [element(3, "div")]))).toBe( + "applied", + ); + expect(reconstructor.apply(mutation(400, { texts: [] }))).toBe("applied"); + expect(reconstructor.apply(mutation(600, { texts: [] }))).toBe( + "past-target", + ); + }); + + it("excludes mutations after the target", () => { + // The point of a point-in-time read: the page as it stood then, not as it + // ended up. + const result = reconstruct( + [ + snapshot(0, [element(3, "span", {}, [text(4, "before")])]), + mutation(1_000, { texts: [{ id: 4, value: "after" }] }), + ], + START_MS + 500, + ); + + expect(result.nodes.get(4)?.textContent).toBe("before"); + expect(result.mutationsApplied).toBe(0); + }); + + it("prefers the newest snapshot at or before the target", () => { + // Whether a recording carries one snapshot or many is not fixed, so the + // same rule has to serve both. + const result = reconstruct( + [ + snapshot(0, [element(3, "div", { id: "first" })]), + snapshot(200, [element(3, "div", { id: "second" })]), + snapshot(9_000, [element(3, "div", { id: "too-late" })]), + ], + START_MS + 1_000, + ); + + expect(result.nodes.get(3)?.attributes.id).toBe("second"); + expect(result.snapshotOffsetMs).toBe(200); + }); +}); + +describe("malformed payloads", () => { + it("counts a snapshot with no node", () => { + const broken = { + type: 2, + timestamp: START_MS, + data: {}, + } as unknown as ReplayRecordingEvent; + + const result = reconstruct([broken]); + + expect(result.missingSnapshot).toBe(true); + expect(result.dropped.malformed).toBe(1); + }); + + it("counts an add with no parentId", () => { + const result = reconstruct([ + snapshot(0, [element(3, "div")]), + mutation(100, { adds: [{ nextId: null, node: element(5, "span") }] }), + ]); + + expect(result.dropped.malformed).toBe(1); + }); + + it("survives an attribute payload that is not an object", () => { + const result = reconstruct([ + snapshot(0, [element(3, "div")]), + mutation(100, { attributes: [{ id: 3, attributes: "nonsense" }] }), + ]); + + expect(result.dropped.malformed).toBe(1); + expect(result.nodes.get(3)?.tagName).toBe("div"); + }); +}); diff --git a/packages/mcp-core/src/internal/replay-dom.ts b/packages/mcp-core/src/internal/replay-dom.ts new file mode 100644 index 000000000..bad50344f --- /dev/null +++ b/packages/mcp-core/src/internal/replay-dom.ts @@ -0,0 +1,576 @@ +/** + * DOM reconstruction from rrweb recording events. + * + * A replay recording is a stream of rrweb events. `replay-events.ts` reads the + * custom events (`type: 5`) that describe *what the user did*; this module + * reads the two that describe *what the page was*: + * + * - `type: 2` (`FullSnapshot`) — a complete serialized DOM. + * - `type: 3` (`IncrementalSnapshot`) — mutations against it. + * + * State at a moment is the last snapshot at or before that moment, with every + * intervening mutation applied in order. Upstream's `which()` ignores both + * types, so nothing here is a port — the semantics come from rrweb's own type + * definitions (`rrweb-io/rrweb`, `packages/types/src/index.ts`), against which + * the constants below were verified. + * + * Three details that are easy to get wrong, in descending order of how + * quietly they fail: + * + * - **Input values arrive as `source: 5`, not as attribute mutations.** A + * reconstruction that applies only `source: 0` shows every form field at its + * initial value. That is worse than showing nothing: it looks authoritative + * and is stale. + * - **Element nodes have no `textContent`.** A button's label is a child text + * node, not a property of the button. + * - **A recording may contain one snapshot or many.** rrweb re-snapshots on + * `checkoutEveryNms`, and Sentry's SDK treats a recording's first event as a + * checkout, so the count is not fixed. Keeping only the newest snapshot at or + * before the target handles both without branching. + * + * Fidelity is reported rather than assumed. A reconstruction can be complete, + * partial, or wrong, and those are indistinguishable from the resulting tree, + * so every dropped operation is counted and surfaced. + */ + +import type { ReplayRecordingEvent } from "../api-client"; +import { isPlainObject } from "./type-guards"; + +/** rrweb `EventType`. Only the two structural members are needed here. */ +const RRWEB_FULL_SNAPSHOT = 2; +const RRWEB_INCREMENTAL_SNAPSHOT = 3; + +/** rrweb `IncrementalSource`. */ +const SOURCE_MUTATION = 0; +const SOURCE_INPUT = 5; + +/** + * rrweb `NodeType`. + * + * Document and Element carry `childNodes`; Text, CDATA, and Comment carry + * `textContent`. DocumentType carries neither. + */ +export const NODE_TYPE_DOCUMENT = 0; +export const NODE_TYPE_DOCUMENT_TYPE = 1; +export const NODE_TYPE_ELEMENT = 2; +export const NODE_TYPE_TEXT = 3; +export const NODE_TYPE_CDATA = 4; +export const NODE_TYPE_COMMENT = 5; + +/** + * A node in the reconstructed tree. + * + * Children are held as ids rather than references so that a `remove` is a + * single splice and an `add` cannot create a cycle by aliasing. The node map + * owns every node; this is a flat store with parent and child links, not a + * nested object graph. + */ +export interface DomNode { + id: number; + nodeType: number; + /** Element tag name, lowercased by rrweb. Absent on non-elements. */ + tagName?: string; + /** Element attributes. Values may be strings, numbers, or `true`. */ + attributes: Record; + /** Text content, for text, CDATA, and comment nodes. */ + textContent?: string; + childIds: number[]; + parentId: number | null; + /** + * Current value, for form controls. + * + * Tracked separately from `attributes.value` because rrweb reports value + * changes through input events, not attribute mutations. The initial value + * arrives as an attribute; every change after that arrives here. + */ + inputValue?: string; + inputChecked?: boolean; +} + +/** Why a reconstruction dropped an operation. */ +export type DomDropReason = + | "unknown-parent" + | "unknown-node" + | "malformed" + | "duplicate-id"; + +/** + * What a reconstruction did, so a caller can say how much to trust it. + * + * A tree assembled from a snapshot that dropped a third of its mutations is + * not obviously different from a clean one, and the difference decides whether + * the answer is usable. + */ +export interface DomReconstruction { + /** The node map, keyed by rrweb node id. */ + nodes: Map; + /** Root node id — the document, or the first node with no parent. */ + rootId: number | null; + /** Offset of the snapshot this was built from, in ms from the replay start. */ + snapshotOffsetMs: number | null; + /** How many mutation events were applied. */ + mutationsApplied: number; + /** Dropped operations, by reason. */ + dropped: Record; + /** True when no `FullSnapshot` was found at or before the target. */ + missingSnapshot: boolean; +} + +export interface ReconstructOptions { + /** + * Target time, in epoch milliseconds. + * + * Events after this are ignored, so the result is the page as it stood at + * that moment rather than at the end of the recording. + */ + atMs: number; +} + +function emptyDropCounts(): Record { + return { + "unknown-parent": 0, + "unknown-node": 0, + malformed: 0, + "duplicate-id": 0, + }; +} + +/** + * Total dropped operations across all reasons. + */ +export function countDropped(dropped: Record): number { + return Object.values(dropped).reduce((sum, count) => sum + count, 0); +} + +/** + * Incrementally folds rrweb events into a DOM state. + * + * Segment-at-a-time rather than all-at-once, so a caller can stream a + * recording and never hold it. Feeding events out of order is not supported — + * rrweb mutations are only meaningful in sequence. + */ +export class DomReconstructor { + private nodes = new Map(); + private rootId: number | null = null; + private snapshotTimestampMs: number | null = null; + private mutationsApplied = 0; + private dropped = emptyDropCounts(); + private readonly atMs: number; + + constructor(options: ReconstructOptions) { + this.atMs = options.atMs; + } + + /** + * Applies one event. + * + * Returns `"past-target"` once an event's timestamp is beyond the target, + * which lets a streaming caller stop paging rather than read the rest of the + * session for events it will discard. + */ + apply(event: ReplayRecordingEvent): "applied" | "ignored" | "past-target" { + const timestamp = + typeof event.timestamp === "number" ? event.timestamp : null; + if (timestamp !== null && timestamp > this.atMs) { + return "past-target"; + } + + if (event.type === RRWEB_FULL_SNAPSHOT) { + // A later snapshot supersedes everything applied so far: it is a + // complete state, so replaying earlier mutations onto it would be wrong. + this.ingestSnapshot(event, timestamp); + return "applied"; + } + + if (event.type === RRWEB_INCREMENTAL_SNAPSHOT) { + return this.applyIncremental(event); + } + + return "ignored"; + } + + /** + * Finalizes and returns the reconstruction. + */ + result(replayStartedAtMs: number | null): DomReconstruction { + const snapshotOffsetMs = + this.snapshotTimestampMs !== null && replayStartedAtMs !== null + ? this.snapshotTimestampMs - replayStartedAtMs + : null; + + return { + nodes: this.nodes, + rootId: this.rootId, + snapshotOffsetMs, + mutationsApplied: this.mutationsApplied, + dropped: { ...this.dropped }, + missingSnapshot: this.snapshotTimestampMs === null, + }; + } + + private ingestSnapshot( + event: ReplayRecordingEvent, + timestamp: number | null, + ): void { + const data = event.data; + if (!isPlainObject(data)) { + this.dropped.malformed += 1; + return; + } + + const node = (data as { node?: unknown }).node; + if (!isPlainObject(node)) { + this.dropped.malformed += 1; + return; + } + + // Replace wholesale. Mutations applied before this snapshot described a + // state this snapshot already supersedes. + this.nodes = new Map(); + this.rootId = null; + this.mutationsApplied = 0; + this.snapshotTimestampMs = timestamp; + + const rootId = this.serializeInto(node, null); + this.rootId = rootId; + } + + /** + * Walks a serialized rrweb node into the flat node map. + * + * Returns the node's id, or null when it carried none — an unidentified node + * cannot be referenced by a later mutation, so it is dropped rather than + * given a synthetic id that nothing will match. + */ + private serializeInto(raw: unknown, parentId: number | null): number | null { + if (!isPlainObject(raw)) { + this.dropped.malformed += 1; + return null; + } + + const id = typeof raw.id === "number" ? raw.id : null; + const nodeType = typeof raw.type === "number" ? raw.type : null; + + // The document wrapper in a FullSnapshot carries childNodes but no id. + // Descend through it so the html element becomes the root. + if (id === null) { + const childNodes = Array.isArray(raw.childNodes) ? raw.childNodes : []; + let firstChildId: number | null = null; + for (const child of childNodes) { + const childId = this.serializeInto(child, parentId); + if (firstChildId === null) { + firstChildId = childId; + } + } + return firstChildId; + } + + if (this.nodes.has(id)) { + // rrweb ids are unique within a recording; a repeat means the payload is + // inconsistent, and overwriting would silently discard a subtree. + this.dropped["duplicate-id"] += 1; + return null; + } + + const node: DomNode = { + id, + nodeType: nodeType ?? NODE_TYPE_ELEMENT, + attributes: readAttributes(raw.attributes), + childIds: [], + parentId, + }; + + if (typeof raw.tagName === "string") { + node.tagName = raw.tagName; + } + if (typeof raw.textContent === "string") { + node.textContent = raw.textContent; + } + // A form control's starting value arrives as an attribute; input events + // take over from there. + const initialValue = node.attributes.value; + if (typeof initialValue === "string") { + node.inputValue = initialValue; + } + + this.nodes.set(id, node); + + const childNodes = Array.isArray(raw.childNodes) ? raw.childNodes : []; + for (const child of childNodes) { + const childId = this.serializeInto(child, id); + if (childId !== null) { + node.childIds.push(childId); + } + } + + return id; + } + + private applyIncremental(event: ReplayRecordingEvent): "applied" | "ignored" { + const data = event.data; + if (!isPlainObject(data)) { + return "ignored"; + } + + const source = (data as { source?: unknown }).source; + + if (source === SOURCE_MUTATION) { + this.applyMutation(data); + this.mutationsApplied += 1; + return "applied"; + } + + if (source === SOURCE_INPUT) { + this.applyInput(data); + this.mutationsApplied += 1; + return "applied"; + } + + // Every other source — mouse movement, scroll, viewport, canvas — changes + // nothing about the structure this module reports. + return "ignored"; + } + + /** + * Applies a `source: 0` mutation. + * + * Order matters: removes before adds, so a node moved within the tree does + * not collide with itself, and attributes and texts last so they apply to + * nodes this batch may have just added. + */ + private applyMutation(data: Record): void { + for (const removal of asArray(data.removes)) { + this.applyRemoval(removal); + } + for (const addition of asArray(data.adds)) { + this.applyAddition(addition); + } + for (const change of asArray(data.attributes)) { + this.applyAttributeChange(change); + } + for (const change of asArray(data.texts)) { + this.applyTextChange(change); + } + } + + private applyRemoval(raw: unknown): void { + if (!isPlainObject(raw) || typeof raw.id !== "number") { + this.dropped.malformed += 1; + return; + } + + const node = this.nodes.get(raw.id); + if (!node) { + this.dropped["unknown-node"] += 1; + return; + } + + this.detach(node); + // Drop the subtree with it. Leaving descendants in the map would let a + // later add reattach them under a parent that no longer exists. + this.deleteSubtree(node.id); + } + + private applyAddition(raw: unknown): void { + if (!isPlainObject(raw)) { + this.dropped.malformed += 1; + return; + } + + const parentId = typeof raw.parentId === "number" ? raw.parentId : null; + if (parentId === null) { + this.dropped.malformed += 1; + return; + } + + const parent = this.nodes.get(parentId); + if (!parent) { + // The parent was pruned, or arrived in a snapshot this read never saw. + // Reparenting to the root would fabricate structure that never existed, + // so drop and count instead. + this.dropped["unknown-parent"] += 1; + return; + } + + const existing = + isPlainObject(raw.node) && typeof raw.node.id === "number" + ? this.nodes.get(raw.node.id) + : undefined; + if (existing) { + // A move, not an insert: rrweb emits a remove plus an add for relocated + // nodes, but a stale duplicate would otherwise double the subtree. + this.detach(existing); + this.deleteSubtree(existing.id); + } + + const childId = this.serializeInto(raw.node, parentId); + if (childId === null) { + return; + } + + // `nextId` names the sibling this node precedes. `previousId` exists only + // for backward compatibility and is deliberately ignored. + const nextId = typeof raw.nextId === "number" ? raw.nextId : null; + const index = nextId !== null ? parent.childIds.indexOf(nextId) : -1; + if (index >= 0) { + parent.childIds.splice(index, 0, childId); + } else { + parent.childIds.push(childId); + } + } + + private applyAttributeChange(raw: unknown): void { + if (!isPlainObject(raw) || typeof raw.id !== "number") { + this.dropped.malformed += 1; + return; + } + + const node = this.nodes.get(raw.id); + if (!node) { + this.dropped["unknown-node"] += 1; + return; + } + + const attributes = isPlainObject(raw.attributes) ? raw.attributes : null; + if (!attributes) { + this.dropped.malformed += 1; + return; + } + + for (const [key, value] of Object.entries(attributes)) { + if (value === null) { + // rrweb signals attribute removal with null. + delete node.attributes[key]; + continue; + } + if ( + typeof value === "string" || + typeof value === "number" || + value === true + ) { + node.attributes[key] = value; + continue; + } + // A styleOMValue object. Recording that the style changed is honest; + // reassembling the declaration is not this module's job. + if (isPlainObject(value)) { + node.attributes[key] = "[style changed]"; + } + } + } + + private applyTextChange(raw: unknown): void { + if (!isPlainObject(raw) || typeof raw.id !== "number") { + this.dropped.malformed += 1; + return; + } + + const node = this.nodes.get(raw.id); + if (!node) { + this.dropped["unknown-node"] += 1; + return; + } + + node.textContent = typeof raw.value === "string" ? raw.value : ""; + } + + /** + * Applies a `source: 5` input event. + * + * This is the one a `source: 0`-only implementation misses, and missing it + * produces a tree that looks current and reports stale values. + */ + private applyInput(data: Record): void { + const id = typeof data.id === "number" ? data.id : null; + if (id === null) { + this.dropped.malformed += 1; + return; + } + + const node = this.nodes.get(id); + if (!node) { + this.dropped["unknown-node"] += 1; + return; + } + + if (typeof data.text === "string") { + node.inputValue = data.text; + } + if (typeof data.isChecked === "boolean") { + node.inputChecked = data.isChecked; + } + } + + /** Unlinks a node from its parent without deleting it. */ + private detach(node: DomNode): void { + if (node.parentId === null) { + return; + } + const parent = this.nodes.get(node.parentId); + if (!parent) { + return; + } + const index = parent.childIds.indexOf(node.id); + if (index >= 0) { + parent.childIds.splice(index, 1); + } + } + + /** Deletes a node and everything beneath it. */ + private deleteSubtree(id: number): void { + const node = this.nodes.get(id); + if (!node) { + return; + } + // Iterative rather than recursive: a deep DOM would otherwise risk the + // call stack on a path that is already handling untrusted depth. + const stack = [...node.childIds]; + this.nodes.delete(id); + while (stack.length > 0) { + const childId = stack.pop(); + if (childId === undefined) { + continue; + } + const child = this.nodes.get(childId); + if (!child) { + continue; + } + stack.push(...child.childIds); + this.nodes.delete(childId); + } + } +} + +function asArray(value: unknown): unknown[] { + return Array.isArray(value) ? value : []; +} + +/** + * Reads an rrweb attribute map. + * + * Values may be strings, numbers (scroll offsets, media positions), or `true` + * (a valueless attribute such as a checked radio). `_cssText` is dropped: it + * is a whole stylesheet, and no structural question needs it. + */ +function readAttributes( + raw: unknown, +): Record { + if (!isPlainObject(raw)) { + return {}; + } + + const attributes: Record = {}; + for (const [key, value] of Object.entries(raw)) { + if (key === "_cssText") { + continue; + } + if ( + typeof value === "string" || + typeof value === "number" || + value === true + ) { + attributes[key] = value; + } + } + return attributes; +} From f2ce062c64eb3b56af112c9a58829eded7961ffa Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 10:22:45 -0700 Subject: [PATCH 17/26] feat(replays): Render reconstructed DOM as an indented tree MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds the rendering half: selector, text, state attributes, and node id per element, with connectors showing structure. Two lenses, and deliberately not a third. `interactive` keeps what a user can act on — interactive tags, plus anything carrying role, href, onclick, or tabindex — and the ancestors needed to place them, so a button is never shown floating without a path. `full` keeps every element. There is no `visible` lens: visibility is a computed style, and a lens that guessed at the cascade would be confidently wrong about the thing a reader most wants to trust. Text comes from immediate text children only. rrweb element nodes carry no textContent, so a renderer that ignores children shows no labels at all, and one that walks descendants attributes a child's label to its container. Values render as recorded. SDK masking leaves no marker, so a masked field is indistinguishable from a real one, and claiming would assert something the recording cannot support — consistent with the rule the signal renderer already follows. The current input value is shown rather than the value attribute, since input events supersede it. A rootNodeId that is not in the reconstruction is reported rather than falling back to the document root, which would answer a question the caller did not ask. Mutation-verified. One mutation initially survived: the test for container text relied on directText only walking immediate children rather than on the node-type check. Added a case with an element child carrying textContent, which now fails without the check. Co-Authored-By: Claude Opus 5 --- .../mcp-core/src/internal/replay-dom.test.ts | 221 ++++++++++++++ packages/mcp-core/src/internal/replay-dom.ts | 285 ++++++++++++++++++ 2 files changed, 506 insertions(+) diff --git a/packages/mcp-core/src/internal/replay-dom.test.ts b/packages/mcp-core/src/internal/replay-dom.test.ts index d752853ad..a50b24f3a 100644 --- a/packages/mcp-core/src/internal/replay-dom.test.ts +++ b/packages/mcp-core/src/internal/replay-dom.test.ts @@ -11,6 +11,7 @@ import { NODE_TYPE_ELEMENT, NODE_TYPE_TEXT, countDropped, + renderDomTree, type DomNode, } from "./replay-dom.js"; import type { ReplayRecordingEvent } from "../api-client"; @@ -430,3 +431,223 @@ describe("malformed payloads", () => { expect(result.nodes.get(3)?.tagName).toBe("div"); }); }); + +describe("rendering", () => { + /** A checkout form: a container, two inputs, and a submit button. */ + function checkoutSnapshot() { + return snapshot(0, [ + element(10, "div", { class: "page wrapper" }, [ + element(11, "form", { id: "checkout-form" }, [ + element(12, "div", { class: "address-block" }, [ + element(13, "input", { id: "unit", value: "***" }), + element(14, "input", { id: "zip", value: "***" }), + ]), + element(15, "button", { id: "complete-order", disabled: true }, [ + text(16, "Complete order"), + ]), + ]), + ]), + ]); + } + + it("renders a tree with connectors and node ids", () => { + const rendered = renderDomTree(reconstruct([checkoutSnapshot()]), { + rootNodeId: 11, + }); + + expect(rendered.lines.join("\n")).toMatchInlineSnapshot(` + "form#checkout-form id=11 + ├─ div.address-block id=12 + │ ├─ input#unit [value="***"] id=13 + │ └─ input#zip [value="***"] id=14 + └─ button#complete-order "Complete order" [disabled] id=15" + `); + }); + + it("takes element text from child text nodes", () => { + // rrweb element nodes carry no textContent, so a label only appears if the + // renderer looks at children. + const rendered = renderDomTree(reconstruct([checkoutSnapshot()]), { + rootNodeId: 15, + }); + + expect(rendered.lines[0]).toContain('"Complete order"'); + }); + + it("does not let a container inherit its descendants' text", () => { + const rendered = renderDomTree(reconstruct([checkoutSnapshot()]), { + rootNodeId: 11, + }); + + expect(rendered.lines[0]).not.toContain("Complete order"); + }); + + it("takes text only from text children, not from nested elements", () => { + // An element child may carry its own text; treating it as this element's + // text would attribute a label to the wrong node. The distinguishing case + // needs an element child that has textContent set, which only happens for + // a malformed or unusual payload — hence asserting on node type rather + // than on the presence of the field. + const oddPayload = snapshot(0, [ + element(20, "div", { id: "container" }, [ + { + id: 21, + type: NODE_TYPE_ELEMENT, + tagName: "span", + attributes: {}, + childNodes: [], + textContent: "child element text", + }, + text(22, "own text"), + ]), + ]); + + const rendered = renderDomTree(reconstruct([oddPayload]), { + rootNodeId: 20, + lens: "full", + }); + + expect(rendered.lines[0]).toContain('"own text"'); + expect(rendered.lines[0]).not.toContain("child element text"); + }); + + it("roots at a node id, excluding everything above it", () => { + const rendered = renderDomTree(reconstruct([checkoutSnapshot()]), { + rootNodeId: 12, + }); + + expect(rendered.lines[0]).toContain("div.address-block"); + expect(rendered.lines.join("\n")).not.toContain("checkout-form"); + }); + + it("reports a root id that is not in the reconstruction", () => { + // Silently falling back to the document root would answer a question the + // caller did not ask. + const rendered = renderDomTree(reconstruct([checkoutSnapshot()]), { + rootNodeId: 9999, + }); + + expect(rendered.rootNotFound).toBe(true); + expect(rendered.lines).toEqual([]); + }); + + describe("the interactive lens", () => { + it("keeps interactive elements and the ancestors that place them", () => { + const rendered = renderDomTree(reconstruct([checkoutSnapshot()]), { + lens: "interactive", + }); + const output = rendered.lines.join("\n"); + + expect(output).toContain("button#complete-order"); + expect(output).toContain("input#unit"); + // div.address-block is inert but is on the path to the inputs. + expect(output).toContain("div.address-block"); + }); + + it("drops inert leaves", () => { + const withDecoration = snapshot(0, [ + element(10, "div", {}, [ + element(11, "span", { class: "decoration" }), + element(12, "button", { id: "go" }), + ]), + ]); + + const rendered = renderDomTree(reconstruct([withDecoration]), { + lens: "interactive", + }); + + expect(rendered.lines.join("\n")).toContain("button#go"); + expect(rendered.lines.join("\n")).not.toContain("decoration"); + }); + + it("keeps an element made interactive by an attribute alone", () => { + const withRole = snapshot(0, [ + element(10, "div", { role: "button", id: "fake-button" }), + ]); + + const rendered = renderDomTree(reconstruct([withRole]), { + lens: "interactive", + }); + + expect(rendered.lines.join("\n")).toContain("fake-button"); + }); + + it("keeps inert elements under the full lens", () => { + const withDecoration = snapshot(0, [ + element(10, "div", {}, [element(11, "span", { class: "decoration" })]), + ]); + + const rendered = renderDomTree(reconstruct([withDecoration]), { + lens: "full", + }); + + expect(rendered.lines.join("\n")).toContain("decoration"); + }); + }); + + it("renders the current input value, not the shipped attribute", () => { + const result = reconstruct([ + snapshot(0, [element(10, "input", { id: "email", value: "initial" })]), + inputEvent(100, 10, { text: "typed@example.com" }), + ]); + + const rendered = renderDomTree(result, { rootNodeId: 10 }); + + expect(rendered.lines[0]).toContain('[value="typed@example.com"]'); + expect(rendered.lines[0]).not.toContain("initial"); + }); + + it("renders masked values as delivered rather than claiming redaction", () => { + // SDK masking leaves no marker, so labeling this would assert + // something the recording cannot support. + const rendered = renderDomTree(reconstruct([checkoutSnapshot()]), { + rootNodeId: 13, + }); + + expect(rendered.lines[0]).toContain('[value="***"]'); + expect(rendered.lines[0]).not.toContain("redacted"); + }); + + it("stops at maxNodes and says so", () => { + const wide = snapshot(0, [ + element( + 10, + "div", + {}, + Array.from({ length: 20 }, (_unused, index) => + element(100 + index, "button", { id: `b${index}` }), + ), + ), + ]); + + const rendered = renderDomTree(reconstruct([wide]), { maxNodes: 5 }); + + expect(rendered.nodesRendered).toBe(5); + expect(rendered.truncated).toBe(true); + }); + + it("stops at maxDepth and says so", () => { + // Build a deep chain so depth, not breadth, is what cuts it off. + let deepest: unknown = element(60, "button", { id: "deep" }); + for (let id = 59; id >= 50; id -= 1) { + deepest = element(id, "div", {}, [deepest]); + } + + const rendered = renderDomTree(reconstruct([snapshot(0, [deepest])]), { + maxDepth: 3, + lens: "full", + }); + + expect(rendered.truncated).toBe(true); + expect(rendered.nodesRendered).toBeLessThan(10); + }); + + it("returns nothing for a reconstruction with no snapshot", () => { + const rendered = renderDomTree( + reconstruct([mutation(100, { texts: [{ id: 1, value: "x" }] })]), + ); + + expect(rendered.lines).toEqual([]); + expect(rendered.rootNotFound).toBe(false); + }); +}); diff --git a/packages/mcp-core/src/internal/replay-dom.ts b/packages/mcp-core/src/internal/replay-dom.ts index bad50344f..7eaeaf8a1 100644 --- a/packages/mcp-core/src/internal/replay-dom.ts +++ b/packages/mcp-core/src/internal/replay-dom.ts @@ -574,3 +574,288 @@ function readAttributes( } return attributes; } + +/** + * How much of the tree to render. + * + * `interactive` keeps what a user can act on — form controls, buttons, links, + * and anything carrying an interaction role — plus the ancestors needed to + * place them. `full` keeps every element. + * + * There is deliberately no `visible` lens. Visibility is a computed style, and + * resolving the cascade is a browser's job; a lens that guessed would be + * confidently wrong about the thing a reader most wants to trust. + */ +export type DomLens = "interactive" | "full"; + +/** Tag names that are interactive regardless of attributes. */ +const INTERACTIVE_TAGS = new Set([ + "a", + "button", + "input", + "select", + "textarea", + "option", + "label", + "form", + "details", + "summary", + "dialog", +]); + +/** Attributes that make an otherwise-inert element interactive. */ +const INTERACTIVE_ATTRIBUTES = ["role", "onclick", "tabindex", "href"]; + +/** Attributes worth rendering, in this order, when present. */ +const RENDERED_ATTRIBUTES = [ + "type", + "name", + "href", + "role", + "placeholder", + "aria-label", + "disabled", + "checked", + "required", + "readonly", +]; + +const MAX_TEXT_LENGTH = 80; + +export interface RenderTreeOptions { + lens?: DomLens; + /** Render only this node and its descendants. */ + rootNodeId?: number; + maxDepth?: number; + maxNodes?: number; +} + +export interface RenderedTree { + lines: string[]; + /** Nodes rendered, after lens filtering and limits. */ + nodesRendered: number; + /** True when `maxNodes` or `maxDepth` cut the output short. */ + truncated: boolean; + /** Set when `rootNodeId` named a node that is not in the reconstruction. */ + rootNotFound: boolean; +} + +/** + * Renders a reconstruction as an indented tree. + * + * Text is taken from child text nodes, since rrweb element nodes carry none. + * Values render exactly as recorded: SDK masking leaves no marker, so a masked + * field is indistinguishable from a real one and claiming redaction would + * assert something unknowable. Only Relay's marker supports that claim, and it + * does not appear in DOM payloads. + */ +export function renderDomTree( + reconstruction: DomReconstruction, + options: RenderTreeOptions = {}, +): RenderedTree { + const { + lens = "interactive", + rootNodeId, + maxDepth = 12, + maxNodes = 200, + } = options; + + const { nodes } = reconstruction; + const startId = rootNodeId ?? reconstruction.rootId; + + if (rootNodeId !== undefined && !nodes.has(rootNodeId)) { + return { + lines: [], + nodesRendered: 0, + truncated: false, + rootNotFound: true, + }; + } + + if (startId === null || startId === undefined || !nodes.has(startId)) { + return { + lines: [], + nodesRendered: 0, + truncated: false, + rootNotFound: false, + }; + } + + // Under the interactive lens, an element is kept when it is interactive or + // when it has a kept descendant — otherwise a button would be rendered with + // no indication of where it sits. + const keep = + lens === "full" ? null : collectInteractiveAncestry(nodes, startId); + + const lines: string[] = []; + let nodesRendered = 0; + let truncated = false; + + const walk = (id: number, depth: number, prefix: string, isLast: boolean) => { + if (truncated) { + return; + } + const node = nodes.get(id); + if (!node || node.nodeType !== NODE_TYPE_ELEMENT) { + return; + } + if (keep && !keep.has(id)) { + return; + } + if (nodesRendered >= maxNodes) { + truncated = true; + return; + } + if (depth > maxDepth) { + truncated = true; + return; + } + + const connector = depth === 0 ? "" : isLast ? "└─ " : "├─ "; + lines.push(`${prefix}${connector}${describeElement(node, nodes)}`); + nodesRendered += 1; + + const childPrefix = depth === 0 ? "" : `${prefix}${isLast ? " " : "│ "}`; + const renderable = node.childIds.filter((childId) => { + const child = nodes.get(childId); + if (!child || child.nodeType !== NODE_TYPE_ELEMENT) { + return false; + } + return !keep || keep.has(childId); + }); + + for (const [index, childId] of renderable.entries()) { + walk(childId, depth + 1, childPrefix, index === renderable.length - 1); + } + }; + + walk(startId, 0, "", true); + + return { lines, nodesRendered, truncated, rootNotFound: false }; +} + +/** + * Ids to keep under the interactive lens: interactive elements, plus every + * ancestor up to the render root so each one has a path. + */ +function collectInteractiveAncestry( + nodes: Map, + startId: number, +): Set { + const keep = new Set([startId]); + + const markAncestors = (id: number) => { + let current = nodes.get(id)?.parentId ?? null; + while (current !== null && !keep.has(current)) { + keep.add(current); + current = nodes.get(current)?.parentId ?? null; + } + }; + + // Walk the subtree under startId rather than the whole map, so a rooted + // render is not influenced by interactive elements elsewhere in the page. + const stack = [startId]; + while (stack.length > 0) { + const id = stack.pop(); + if (id === undefined) { + continue; + } + const node = nodes.get(id); + if (!node) { + continue; + } + if (isInteractive(node)) { + keep.add(id); + markAncestors(id); + } + stack.push(...node.childIds); + } + + return keep; +} + +function isInteractive(node: DomNode): boolean { + if (node.tagName && INTERACTIVE_TAGS.has(node.tagName.toLowerCase())) { + return true; + } + return INTERACTIVE_ATTRIBUTES.some((name) => name in node.attributes); +} + +/** + * One line for an element: a CSS-like selector, its text, and the attributes + * that carry state. + */ +function describeElement(node: DomNode, nodes: Map): string { + const parts = [describeSelector(node)]; + + const text = directText(node, nodes); + if (text) { + parts.push(`"${text}"`); + } + + // The current value, not the attribute: input events supersede it, and the + // attribute records only what the page shipped with. + if (node.inputValue !== undefined) { + parts.push(`[value=${JSON.stringify(truncateText(node.inputValue))}]`); + } + if (node.inputChecked !== undefined) { + parts.push(`[checked=${node.inputChecked}]`); + } + + const rendered = RENDERED_ATTRIBUTES.filter( + (name) => name in node.attributes && name !== "value", + ).map((name) => { + const value = node.attributes[name]; + return value === true ? `[${name}]` : `[${name}=${value}]`; + }); + parts.push(...rendered); + + parts.push(`id=${node.id}`); + + return parts.join(" "); +} + +/** + * `tag#id.class` for an element, matching the selector style used elsewhere in + * replay output. + */ +function describeSelector(node: DomNode): string { + const tag = node.tagName?.toLowerCase() ?? "node"; + const id = + typeof node.attributes.id === "string" ? `#${node.attributes.id}` : ""; + const className = + typeof node.attributes.class === "string" + ? node.attributes.class + .trim() + .split(/\s+/) + .filter(Boolean) + .slice(0, 2) + .map((name) => `.${name}`) + .join("") + : ""; + return `${tag}${id}${className}`; +} + +/** + * Text belonging directly to this element. + * + * Only immediate text children, so a container does not inherit the + * concatenated text of everything beneath it. + */ +function directText(node: DomNode, nodes: Map): string | null { + const parts: string[] = []; + for (const childId of node.childIds) { + const child = nodes.get(childId); + if (child?.nodeType === NODE_TYPE_TEXT && child.textContent) { + parts.push(child.textContent); + } + } + const joined = parts.join(" ").replace(/\s+/g, " ").trim(); + return joined ? truncateText(joined) : null; +} + +function truncateText(value: string): string { + return value.length > MAX_TEXT_LENGTH + ? `${value.slice(0, MAX_TEXT_LENGTH)}…` + : value; +} From 8baa26217e330efb462c2d2858d23e9d38f5f6c3 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 11:09:41 -0700 Subject: [PATCH 18/26] fix(replays): Render the real rrweb document shape MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Verified against the `@sentry-internal/rrweb-snapshot` build Sentry ships: `serializeNodeWithId` assigns an id to every node including the Document, so a snapshot's root is a `nodeType: 0` node rather than an element. The renderer walked from that root and bailed on the first non-element, so a real recording rendered nothing at all. Resolve the render root by descending to the first element instead, breadth-first so the doctype sibling cannot lead the search away from `html`. The unit fixtures had modelled an id-less document wrapper, which is what hid this. They now carry the real shape. Three rendering defects surfaced once the fixtures were honest, each confirmed against rrweb's own emission code rather than assumed: - `role` presence was treated as interactivity, so `role="alert"` pulled status banners into a tree labelled interactive. Match the WAI-ARIA widget roles by value instead. - `isChecked` is reported on every input event, not only for checkboxes and radios, so text fields rendered `[checked=false]` — a claim about a property they do not have. - A boolean HTML attribute serializes as the empty string, since `getAttribute("disabled")` returns `""`. `disabled` rendered as `[disabled=]`. Each fix is pinned by a test that fails without it. Co-Authored-By: Claude Opus 5 --- .../mcp-core/src/internal/replay-dom.test.ts | 89 ++++++++++++--- packages/mcp-core/src/internal/replay-dom.ts | 105 ++++++++++++++++-- 2 files changed, 170 insertions(+), 24 deletions(-) diff --git a/packages/mcp-core/src/internal/replay-dom.test.ts b/packages/mcp-core/src/internal/replay-dom.test.ts index a50b24f3a..ffa355307 100644 --- a/packages/mcp-core/src/internal/replay-dom.test.ts +++ b/packages/mcp-core/src/internal/replay-dom.test.ts @@ -32,12 +32,24 @@ function text(id: number, textContent: string) { return { id, type: NODE_TYPE_TEXT, textContent }; } +/** Ids for the document frame, kept clear of the small ids tests use. */ +const DOCUMENT_ID = 1; +const HTML_ID = 9001; +const BODY_ID = 9002; + /** * A FullSnapshot wrapping the given body children. * - * Mirrors the real shape: an id-less document wrapper containing `html`, - * containing `body`. The wrapper carries no id, which is why reconstruction has - * to descend through it rather than expect a root id. + * The document node carries an id, which is the detail that matters: verified + * against the `@sentry-internal/rrweb-snapshot` build Sentry ships, where + * `serializeNodeWithId` assigns an id to every node including the Document + * (`genId()` starts at 1, so it is normally id 1). A reconstruction's root is + * therefore a `nodeType: 0` node rather than an element, and a renderer that + * expects an element root produces nothing at all. + * + * `html` and `body` would really be 2 and 3. They are numbered out of the way + * here so each test can use small ids for the nodes it cares about — rrweb ids + * are opaque integers, and nothing under test depends on them being sequential. */ function snapshot(offsetMs: number, bodyChildren: unknown[]) { return { @@ -45,9 +57,12 @@ function snapshot(offsetMs: number, bodyChildren: unknown[]) { timestamp: START_MS + offsetMs, data: { node: { + id: DOCUMENT_ID, type: NODE_TYPE_DOCUMENT, childNodes: [ - element(1, "html", {}, [element(2, "body", {}, bodyChildren)]), + element(HTML_ID, "html", {}, [ + element(BODY_ID, "body", {}, bodyChildren), + ]), ], }, }, @@ -104,12 +119,14 @@ function childTags(node: DomNode | undefined, nodes: Map) { } describe("FullSnapshot ingest", () => { - it("descends through the id-less document wrapper", () => { + it("roots at the document node, which is not an element", () => { const result = reconstruct([snapshot(400, [element(3, "div")])]); - // The wrapper has no id, so `html` — the first identified node — is root. - expect(result.rootId).toBe(1); - expect(result.nodes.get(1)?.tagName).toBe("html"); + // rrweb assigns the Document an id like any other node, so the root of a + // reconstruction is a `nodeType: 0` node. Rendering has to descend past it. + expect(result.rootId).toBe(DOCUMENT_ID); + expect(result.nodes.get(DOCUMENT_ID)?.nodeType).toBe(NODE_TYPE_DOCUMENT); + expect(result.nodes.get(HTML_ID)?.tagName).toBe("html"); expect(result.missingSnapshot).toBe(false); expect(result.snapshotOffsetMs).toBe(400); }); @@ -117,8 +134,8 @@ describe("FullSnapshot ingest", () => { it("records parent and child links in both directions", () => { const result = reconstruct([snapshot(0, [element(3, "button")])]); - expect(result.nodes.get(2)?.childIds).toEqual([3]); - expect(result.nodes.get(3)?.parentId).toBe(2); + expect(result.nodes.get(BODY_ID)?.childIds).toEqual([3]); + expect(result.nodes.get(3)?.parentId).toBe(BODY_ID); }); it("keeps text as child nodes rather than element properties", () => { @@ -178,11 +195,11 @@ describe("mutations", () => { const result = reconstruct([ snapshot(0, [element(3, "header"), element(4, "footer")]), mutation(100, { - adds: [{ parentId: 2, nextId: 4, node: element(5, "main") }], + adds: [{ parentId: BODY_ID, nextId: 4, node: element(5, "main") }], }), ]); - expect(childTags(result.nodes.get(2), result.nodes)).toEqual([ + expect(childTags(result.nodes.get(BODY_ID), result.nodes)).toEqual([ "header", "main", "footer", @@ -193,11 +210,11 @@ describe("mutations", () => { const result = reconstruct([ snapshot(0, [element(3, "header")]), mutation(100, { - adds: [{ parentId: 2, nextId: null, node: element(5, "main") }], + adds: [{ parentId: BODY_ID, nextId: null, node: element(5, "main") }], }), ]); - expect(childTags(result.nodes.get(2), result.nodes)).toEqual([ + expect(childTags(result.nodes.get(BODY_ID), result.nodes)).toEqual([ "header", "main", ]); @@ -225,13 +242,13 @@ describe("mutations", () => { snapshot(0, [ element(3, "div", {}, [element(4, "span", {}, [text(5, "hi")])]), ]), - mutation(100, { removes: [{ parentId: 2, id: 3 }] }), + mutation(100, { removes: [{ parentId: BODY_ID, id: 3 }] }), ]); expect(result.nodes.has(3)).toBe(false); expect(result.nodes.has(4)).toBe(false); expect(result.nodes.has(5)).toBe(false); - expect(result.nodes.get(2)?.childIds).toEqual([]); + expect(result.nodes.get(BODY_ID)?.childIds).toEqual([]); }); it("relocates a node without duplicating its subtree", () => { @@ -450,6 +467,46 @@ describe("rendering", () => { ]); } + it("renders from the document root by descending to the first element", () => { + // The regression this guards: rrweb roots a snapshot at the Document node, + // which is not an element, and a renderer that walks from it directly emits + // nothing at all. Rendering unrooted must still produce a tree. + const rendered = renderDomTree(reconstruct([checkoutSnapshot()]), { + lens: "full", + }); + + expect(rendered.lines[0]).toBe("html id=9001"); + expect(rendered.rootNotFound).toBe(false); + expect(rendered.nodesRendered).toBeGreaterThan(1); + }); + + it("descends past a doctype sibling to reach html", () => { + // A real snapshot's document has the doctype as its first child, so a + // depth-first search for the first element would stop on the wrong node. + const withDoctype = { + type: 2, + timestamp: START_MS, + data: { + node: { + id: DOCUMENT_ID, + type: NODE_TYPE_DOCUMENT, + childNodes: [ + { id: 8000, type: 1, name: "html", publicId: "", systemId: "" }, + element(HTML_ID, "html", {}, [ + element(BODY_ID, "body", {}, [element(3, "button")]), + ]), + ], + }, + }, + } as unknown as ReplayRecordingEvent; + + const rendered = renderDomTree(reconstruct([withDoctype]), { + lens: "full", + }); + + expect(rendered.lines[0]).toBe("html id=9001"); + }); + it("renders a tree with connectors and node ids", () => { const rendered = renderDomTree(reconstruct([checkoutSnapshot()]), { rootNodeId: 11, diff --git a/packages/mcp-core/src/internal/replay-dom.ts b/packages/mcp-core/src/internal/replay-dom.ts index 7eaeaf8a1..a6d6604a8 100644 --- a/packages/mcp-core/src/internal/replay-dom.ts +++ b/packages/mcp-core/src/internal/replay-dom.ts @@ -603,8 +603,41 @@ const INTERACTIVE_TAGS = new Set([ "dialog", ]); -/** Attributes that make an otherwise-inert element interactive. */ -const INTERACTIVE_ATTRIBUTES = ["role", "onclick", "tabindex", "href"]; +/** + * Attributes whose mere presence makes an otherwise-inert element interactive. + * + * `role` is deliberately absent: it is matched by value below, because + * `role="alert"` is not something a user can act on and presence-matching would + * pull every status banner into a tree labelled interactive. + */ +const INTERACTIVE_ATTRIBUTES = ["onclick", "tabindex", "href"]; + +/** + * ARIA roles that describe a control, from the WAI-ARIA widget roles. + * + * Document-structure and live-region roles are excluded: they describe what an + * element *is*, not something to do to it. + */ +const INTERACTIVE_ROLES = new Set([ + "button", + "checkbox", + "combobox", + "gridcell", + "link", + "listbox", + "menuitem", + "menuitemcheckbox", + "menuitemradio", + "option", + "radio", + "searchbox", + "slider", + "spinbutton", + "switch", + "tab", + "textbox", + "treeitem", +]); /** Attributes worth rendering, in this order, when present. */ const RENDERED_ATTRIBUTES = [ @@ -681,11 +714,26 @@ export function renderDomTree( }; } + // The tree renders elements, and a recording's root is normally the Document + // node, which is not one. Descend to the first element beneath it — `html`, + // past any doctype — rather than render nothing. Only the root needs this: + // non-element children elsewhere are text, whose content is read into the + // parent's line. + const renderRootId = resolveRenderRoot(nodes, startId); + if (renderRootId === null) { + return { + lines: [], + nodesRendered: 0, + truncated: false, + rootNotFound: false, + }; + } + // Under the interactive lens, an element is kept when it is interactive or // when it has a kept descendant — otherwise a button would be rendered with // no indication of where it sits. const keep = - lens === "full" ? null : collectInteractiveAncestry(nodes, startId); + lens === "full" ? null : collectInteractiveAncestry(nodes, renderRootId); const lines: string[] = []; let nodesRendered = 0; @@ -729,11 +777,41 @@ export function renderDomTree( } }; - walk(startId, 0, "", true); + walk(renderRootId, 0, "", true); return { lines, nodesRendered, truncated, rootNotFound: false }; } +/** + * The first element at or beneath `startId`, breadth-first. + * + * rrweb assigns the Document node an id like any other, so a snapshot's root is + * a `nodeType: 0` node whose children are the doctype and `html`. Rendering + * from it directly produces an empty tree. Breadth-first so a doctype sibling + * cannot lead the search away from `html`. + */ +function resolveRenderRoot( + nodes: Map, + startId: number, +): number | null { + const queue = [startId]; + while (queue.length > 0) { + const id = queue.shift(); + if (id === undefined) { + continue; + } + const node = nodes.get(id); + if (!node) { + continue; + } + if (node.nodeType === NODE_TYPE_ELEMENT) { + return id; + } + queue.push(...node.childIds); + } + return null; +} + /** * Ids to keep under the interactive lens: interactive elements, plus every * ancestor up to the render root so each one has a path. @@ -778,7 +856,11 @@ function isInteractive(node: DomNode): boolean { if (node.tagName && INTERACTIVE_TAGS.has(node.tagName.toLowerCase())) { return true; } - return INTERACTIVE_ATTRIBUTES.some((name) => name in node.attributes); + if (INTERACTIVE_ATTRIBUTES.some((name) => name in node.attributes)) { + return true; + } + const role = node.attributes.role; + return typeof role === "string" && INTERACTIVE_ROLES.has(role.toLowerCase()); } /** @@ -798,15 +880,22 @@ function describeElement(node: DomNode, nodes: Map): string { if (node.inputValue !== undefined) { parts.push(`[value=${JSON.stringify(truncateText(node.inputValue))}]`); } - if (node.inputChecked !== undefined) { - parts.push(`[checked=${node.inputChecked}]`); + // rrweb reports `isChecked` on every input event, not only on checkboxes and + // radios, so a false value carries no information — and `[checked=false]` on + // a text field reads as a fact about it. An unchecked control is the absence + // of the attribute, exactly as in HTML. + if (node.inputChecked === true) { + parts.push("[checked]"); } const rendered = RENDERED_ATTRIBUTES.filter( (name) => name in node.attributes && name !== "value", ).map((name) => { const value = node.attributes[name]; - return value === true ? `[${name}]` : `[${name}=${value}]`; + // A boolean HTML attribute serializes as the empty string — + // `getAttribute("disabled")` returns `""` — so an empty value is presence, + // not an empty value, and `[disabled=]` would be nonsense. + return value === true || value === "" ? `[${name}]` : `[${name}=${value}]`; }); parts.push(...rendered); From 0ae43a3353dcff5106c5fe355daeed2e0ac1753b Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 11:09:53 -0700 Subject: [PATCH 19/26] feat(replays): Report the node id a DOM read can root at MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Click breadcrumbs carry `payload.data.node.id`, the rrweb node id, which is stable within a recording. It was parsed but never rendered, so nothing in the output named the element a structural read could focus on — "show me the DOM around what was rage-clicked" had no handle to pass along. Report it at detail grain. Also add the DOM events the recording fixture lacked. It carried only custom events (`type: 5`), so nothing exercised the two types that describe structure. The login and checkout pages are now snapshotted, and the checkout segment carries a `source: 5` input change, an ignored scroll, an added error banner, and a text-plus-attribute mutation that disables the submit button — the last two after the console error, so a read at the error must not show them. Co-Authored-By: Claude Opus 5 --- .../mcp-core/src/internal/replay-events.ts | 8 + .../tools/catalog/get-replay-activity.test.ts | 10 + .../fixtures/replay-recording-segments.json | 387 ++++++++++++++++++ 3 files changed, 405 insertions(+) diff --git a/packages/mcp-core/src/internal/replay-events.ts b/packages/mcp-core/src/internal/replay-events.ts index e4586ec06..012053f75 100644 --- a/packages/mcp-core/src/internal/replay-events.ts +++ b/packages/mcp-core/src/internal/replay-events.ts @@ -764,6 +764,14 @@ function describeClick( details.push(`clicks: ${clickCount}`); } + // The rrweb node id, which is the handle a DOM read roots at. Reported + // because it is otherwise unreachable: it is stable within a recording, but + // nothing else in this output names it, so "show me the DOM around what was + // clicked" would have no way to say which element. + if (typeof data?.node?.id === "number") { + details.push(`nodeId: ${data.node.id}`); + } + return { summary: `${verb} ${target}`, details, diff --git a/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts index 575dc80b5..f688c00d2 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts @@ -110,6 +110,16 @@ describe("get_replay_activity", () => { expect(result).toContain("response body: 87 bytes "); }); + it("reports the rrweb node id a DOM read can root at", async () => { + // The handoff to `get_replay_dom`: the id is stable within a recording + // and reaches us on every click breadcrumb, but nothing else in this + // output names it, so without this line "show me the DOM around what was + // rage-clicked" has no handle to pass along. + const result = await callTool({ grain: "detail", kinds: ["click"] }); + + expect(result).toContain("nodeId: 96"); + }); + it("omits payload lines at standard grain", async () => { const result = await callTool({ kinds: ["network"] }); diff --git a/packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json b/packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json index b43d514b3..984a871a3 100644 --- a/packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json +++ b/packages/mcp-server-mocks/src/fixtures/replay-recording-segments.json @@ -30,6 +30,125 @@ } } }, + { + "type": 2, + "timestamp": 1744027200200, + "data": { + "node": { + "type": 0, + "id": 1, + "childNodes": [ + { + "type": 1, + "id": 2, + "name": "html", + "publicId": "", + "systemId": "" + }, + { + "type": 2, + "id": 3, + "tagName": "html", + "attributes": {}, + "childNodes": [ + { + "type": 2, + "id": 4, + "tagName": "head", + "attributes": {}, + "childNodes": [ + { + "type": 2, + "id": 5, + "tagName": "title", + "attributes": {}, + "childNodes": [ + { + "type": 3, + "id": 6, + "textContent": "Sign in" + } + ] + } + ] + }, + { + "type": 2, + "id": 7, + "tagName": "body", + "attributes": {}, + "childNodes": [ + { + "type": 2, + "id": 8, + "tagName": "div", + "attributes": { + "id": "root" + }, + "childNodes": [ + { + "type": 2, + "id": 9, + "tagName": "form", + "attributes": { + "id": "login", + "action": "/session" + }, + "childNodes": [ + { + "type": 2, + "id": 40, + "tagName": "input", + "attributes": { + "id": "email", + "type": "email", + "value": "***" + }, + "childNodes": [] + }, + { + "type": 2, + "id": 41, + "tagName": "input", + "attributes": { + "id": "password", + "type": "password", + "value": "***" + }, + "childNodes": [] + }, + { + "type": 2, + "id": 42, + "tagName": "button", + "attributes": { + "id": "sign-in", + "type": "submit" + }, + "childNodes": [ + { + "type": 3, + "id": 43, + "textContent": "Sign in" + } + ] + } + ] + } + ] + } + ] + } + ] + } + ] + }, + "initialOffset": { + "top": 0, + "left": 0 + } + } + }, { "type": 5, "timestamp": 1744027200.5, @@ -167,6 +286,209 @@ } ], [ + { + "type": 2, + "timestamp": 1744027380000, + "data": { + "node": { + "type": 0, + "id": 1, + "childNodes": [ + { + "type": 1, + "id": 2, + "name": "html", + "publicId": "", + "systemId": "" + }, + { + "type": 2, + "id": 3, + "tagName": "html", + "attributes": {}, + "childNodes": [ + { + "type": 2, + "id": 4, + "tagName": "head", + "attributes": {}, + "childNodes": [ + { + "type": 2, + "id": 5, + "tagName": "title", + "attributes": {}, + "childNodes": [ + { + "type": 3, + "id": 6, + "textContent": "Checkout" + } + ] + } + ] + }, + { + "type": 2, + "id": 60, + "tagName": "body", + "attributes": {}, + "childNodes": [ + { + "type": 2, + "id": 61, + "tagName": "div", + "attributes": { + "id": "root" + }, + "childNodes": [ + { + "type": 2, + "id": 62, + "tagName": "main", + "attributes": {}, + "childNodes": [ + { + "type": 2, + "id": 63, + "tagName": "h1", + "attributes": {}, + "childNodes": [ + { + "type": 3, + "id": 64, + "textContent": "Checkout" + } + ] + }, + { + "type": 2, + "id": 80, + "tagName": "form", + "attributes": { + "id": "checkout-form" + }, + "childNodes": [ + { + "type": 2, + "id": 81, + "tagName": "div", + "attributes": { + "class": "address-block" + }, + "childNodes": [ + { + "type": 2, + "id": 82, + "tagName": "input", + "attributes": { + "id": "unit", + "name": "unit", + "value": "***" + }, + "childNodes": [] + }, + { + "type": 2, + "id": 83, + "tagName": "input", + "attributes": { + "id": "zip", + "name": "zip", + "value": "***" + }, + "childNodes": [] + } + ] + }, + { + "type": 2, + "id": 84, + "tagName": "label", + "attributes": { + "for": "quantity" + }, + "childNodes": [ + { + "type": 3, + "id": 85, + "textContent": "Quantity" + } + ] + }, + { + "type": 2, + "id": 90, + "tagName": "input", + "attributes": { + "id": "quantity", + "name": "quantity", + "type": "number", + "value": "1" + }, + "childNodes": [] + }, + { + "type": 2, + "id": 96, + "tagName": "button", + "attributes": { + "id": "complete-order", + "type": "button" + }, + "childNodes": [ + { + "type": 3, + "id": 97, + "textContent": "Complete order" + } + ] + } + ] + }, + { + "type": 2, + "id": 118, + "tagName": "a", + "attributes": { + "id": "download-receipt", + "href": "/receipts/latest" + }, + "childNodes": [ + { + "type": 3, + "id": 119, + "textContent": "Download receipt" + } + ] + } + ] + } + ] + } + ] + } + ] + } + ] + }, + "initialOffset": { + "top": 0, + "left": 0 + } + } + }, + { + "type": 3, + "timestamp": 1744027380300, + "data": { + "source": 5, + "id": 90, + "text": "3", + "isChecked": false, + "userTriggered": true + } + }, { "type": 5, "timestamp": 1744027380600, @@ -192,6 +514,16 @@ } } }, + { + "type": 3, + "timestamp": 1744027380800, + "data": { + "source": 3, + "id": 60, + "x": 0, + "y": 412 + } + }, { "type": 5, "timestamp": 1744027381, @@ -237,6 +569,61 @@ } } }, + { + "type": 3, + "timestamp": 1744027381400, + "data": { + "source": 0, + "texts": [], + "attributes": [], + "removes": [], + "adds": [ + { + "parentId": 62, + "nextId": null, + "node": { + "type": 2, + "id": 130, + "tagName": "div", + "attributes": { + "id": "order-error", + "role": "alert" + }, + "childNodes": [ + { + "type": 3, + "id": 131, + "textContent": "Payment failed. Please try again." + } + ] + } + } + ] + } + }, + { + "type": 3, + "timestamp": 1744027382000, + "data": { + "source": 0, + "texts": [ + { + "id": 97, + "value": "Processing\u2026" + } + ], + "attributes": [ + { + "id": 96, + "attributes": { + "disabled": "" + } + } + ], + "removes": [], + "adds": [] + } + }, { "type": 5, "timestamp": 1744027388400, From bd670672b89523f5c5110743e3bd6663ae7a010b Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 11:10:09 -0700 Subject: [PATCH 20/26] feat(replays): Add get_replay_dom for point-in-time structure MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Completes the three-step replay read: the map finds the failure, `get_replay_activity` names the element and reports its node id, and this explains what was around it. Each step hands the next a concrete handle. A separate catalog-only tool rather than a grain or a flag on `get_replay_activity`. A structural read has a different return shape, cost model, and failure mode from a windowed signal list, and it is the only surface where a read can refuse on budget grounds without complicating a contract agents call routinely. Catalog-only keeps the direct-surface budget unchanged at 5,598 tokens across 9 tools. `atMs` is required. Defaulting to either end of the session would answer a different question often enough that guessing is worse than asking. It is an offset from the replay's start, matching the other two tools, so a replay with no parseable `started_at` is refused rather than placed by guess. Refusal is a supported outcome. When the segment budget runs out before the target moment is reached, the tool says so and explains what would help. It does not return a partial tree, because a tree assembled from a truncated mutation history is indistinguishable from a complete one at the point of use. Reads stop paging as soon as an event passes the target, so cost is bounded by how far in the moment is rather than by session length. Fidelity is reported, not assumed: which snapshot it started from, how many mutations it applied, how many operations it dropped and why, and whether the recording ended before the moment asked for. Structure only. Text and input values arrive already masked by the SDK with no marker, so they render as delivered and are never labelled redacted — the tool cannot distinguish masked-at-capture from genuinely-this-text. It answers "was the button disabled", not "what did the user type". Co-Authored-By: Claude Opus 5 --- packages/mcp-core/src/skillDefinitions.json | 7 +- packages/mcp-core/src/toolDefinitions.json | 64 ++++ .../catalog-runtime/availability.test.ts | 12 +- .../src/tools/catalog/get-replay-dom.test.ts | 336 ++++++++++++++++++ .../src/tools/catalog/get-replay-dom.ts | 334 +++++++++++++++++ packages/mcp-core/src/tools/catalog/index.ts | 2 + 6 files changed, 750 insertions(+), 5 deletions(-) create mode 100644 packages/mcp-core/src/tools/catalog/get-replay-dom.test.ts create mode 100644 packages/mcp-core/src/tools/catalog/get-replay-dom.ts diff --git a/packages/mcp-core/src/skillDefinitions.json b/packages/mcp-core/src/skillDefinitions.json index 51c1e7c2d..1c8d1c360 100644 --- a/packages/mcp-core/src/skillDefinitions.json +++ b/packages/mcp-core/src/skillDefinitions.json @@ -5,7 +5,7 @@ "description": "Read-only access to core Sentry data: issues, events, traces, replays, releases, cron monitors, uptime monitors, profiles, documentation, and project metadata", "defaultEnabled": true, "order": 1, - "toolCount": 38, + "toolCount": 39, "tools": [ { "name": "find_alert_rules", @@ -137,6 +137,11 @@ "description": "Get high-level information about a specific Sentry replay by URL or replay ID.\n\nUSE THIS TOOL WHEN USERS:\n- Share a replay URL\n- Ask what happened in a specific replay\n- Want a concise replay summary plus the next issue or trace lookups to run\n\n\n### With replay URL\n```\nget_replay_details(replayUrl='https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/')\n```\n\n### With organization and replay ID\n```\nget_replay_details(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f')\n```\n", "requiredScopes": ["org:read", "project:read", "event:read"] }, + { + "name": "get_replay_dom", + "description": "Read the page structure of a Sentry replay at one moment in time.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what the page looked like when something failed\n- Need to know whether an element existed, was disabled, or was empty\n- Want the DOM around an element that was clicked or rage-clicked\n\nReturns structure only — what the page *was*, not what the user did.\nUse `get_replay_activity` for that, and to find the `nodeId` to root at.\n`atMs` is milliseconds from the start of the replay and is required.\n\nText and form values are masked by the SDK before upload, so this answers\nstructural questions, not what a user typed.\n\n\n### What the page looked like when the error fired\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)\n```\n\n### The subtree around a rage-clicked element\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')\n```\n", + "requiredScopes": ["org:read", "project:read", "event:read"] + }, { "name": "get_sentry_resource", "description": "Fetch a Sentry resource by URL, or by resourceType plus resourceId.\nPass a Sentry URL directly when possible; the resource type is auto-detected.\n\nSupports issues, events, traces, spans, AI conversations, replays, monitors, preprod snapshots, and snapshot images.\nTrace lookups return a condensed overview by default.\n\nAI Conversations: A conversation is a set of spans sharing the same gen_ai.conversation.id. Use resourceType='ai_conversation' with a conversation ID, or pass a Sentry conversation URL, to fetch the transcript/details. To discover or list conversations, use search_ai_conversations. Conversations are NOT issues — do not use search_issues for conversation queries.\n\nFor preprod snapshot URLs (matching 'sentry.io/preprod/snapshots/'):\n- Without ?selectedSnapshot=: returns the snapshot diff summary (changed, added, removed images)\n- With ?selectedSnapshot=: returns the image preview and metadata. Use the Sentry tool `get_snapshot_image` for full-resolution image bytes.\n\nResource IDs:\n- monitor: \n- snapshot: \n\n\nget_sentry_resource(url='https://sentry.io/issues/PROJECT-123/')\nget_sentry_resource(resourceType='issue', organizationSlug='my-org', resourceId='PROJECT-123')\nget_sentry_resource(resourceType='ai_conversation', organizationSlug='my-org', resourceId='conversation-123')\nget_sentry_resource(url='https://sentry.sentry.io/preprod/snapshots/123/')\nget_sentry_resource(url='https://sentry.sentry.io/preprod/snapshots/123/?selectedSnapshot=login_screen.png')\n", diff --git a/packages/mcp-core/src/toolDefinitions.json b/packages/mcp-core/src/toolDefinitions.json index bbcf60c85..3c623af9f 100644 --- a/packages/mcp-core/src/toolDefinitions.json +++ b/packages/mcp-core/src/toolDefinitions.json @@ -3603,6 +3603,70 @@ "skills": ["inspect"], "surface": "catalog" }, + { + "name": "get_replay_dom", + "description": "Read the page structure of a Sentry replay at one moment in time.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what the page looked like when something failed\n- Need to know whether an element existed, was disabled, or was empty\n- Want the DOM around an element that was clicked or rage-clicked\n\nReturns structure only — what the page *was*, not what the user did.\nUse `get_replay_activity` for that, and to find the `nodeId` to root at.\n`atMs` is milliseconds from the start of the replay and is required.\n\nText and form values are masked by the SDK before upload, so this answers\nstructural questions, not what a user typed.\n\n\n### What the page looked like when the error fired\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)\n```\n\n### The subtree around a rage-clicked element\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')\n```\n", + "inputSchema": { + "type": "object", + "properties": { + "replayUrl": { + "type": "string", + "format": "uri", + "description": "The URL of the replay. e.g. https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/" + }, + "organizationSlug": { + "type": "string", + "description": "The organization's slug. You can find a existing list of organizations you have access to using the `find_organizations()` tool." + }, + "replayId": { + "type": "string", + "description": "The replay ID. e.g. `7e07485f-12f9-416b-8b14-26260799b51f`" + }, + "regionUrl": { + "anyOf": [ + { + "type": "string", + "description": "The region URL for the organization you're querying, if known. For Sentry's Cloud Service (sentry.io), this is typically the region-specific URL like 'https://us.sentry.io'. For self-hosted Sentry installations, this parameter is usually not needed and should be omitted. You can find the correct regionUrl from the organization details using the `find_organizations()` tool." + }, + { + "type": "null" + } + ] + }, + "atMs": { + "type": "number", + "minimum": 0, + "description": "The moment to reconstruct, in milliseconds from the start of the replay. Required: a structural read has no sensible default moment." + }, + "rootNodeId": { + "description": "Render only this node and its descendants. Use the `nodeId` reported by a click signal in `get_replay_activity`.", + "type": "number" + }, + "lens": { + "default": "interactive", + "description": "`interactive` keeps elements a user can act on plus the ancestors that place them; `full` keeps every element.", + "type": "string", + "enum": ["interactive", "full"] + }, + "maxDepth": { + "default": 12, + "type": "number", + "minimum": 1, + "maximum": 50 + }, + "maxNodes": { + "default": 200, + "type": "number", + "minimum": 1, + "maximum": 1000 + } + }, + "required": ["atMs"] + }, + "requiredScopes": ["org:read", "project:read", "event:read"], + "skills": ["inspect"], + "surface": "catalog" + }, { "name": "get_sentry_resource", "description": "Fetch a Sentry resource by URL, or by resourceType plus resourceId.\nPass a Sentry URL directly when possible; the resource type is auto-detected.\n\nSupports issues, events, traces, spans, AI conversations, replays, monitors, preprod snapshots, and snapshot images.\nTrace lookups return a condensed overview by default.\n\nAI Conversations: A conversation is a set of spans sharing the same gen_ai.conversation.id. Use resourceType='ai_conversation' with a conversation ID, or pass a Sentry conversation URL, to fetch the transcript/details. To discover or list conversations, use search_ai_conversations. Conversations are NOT issues — do not use search_issues for conversation queries.\n\nFor preprod snapshot URLs (matching 'sentry.io/preprod/snapshots/'):\n- Without ?selectedSnapshot=: returns the snapshot diff summary (changed, added, removed images)\n- With ?selectedSnapshot=: returns the image preview and metadata. Use the Sentry tool `get_snapshot_image` for full-resolution image bytes.\n\nResource IDs:\n- monitor: \n- snapshot: \n\n\nget_sentry_resource(url='https://sentry.io/issues/PROJECT-123/')\nget_sentry_resource(resourceType='issue', organizationSlug='my-org', resourceId='PROJECT-123')\nget_sentry_resource(resourceType='ai_conversation', organizationSlug='my-org', resourceId='conversation-123')\nget_sentry_resource(url='https://sentry.sentry.io/preprod/snapshots/123/')\nget_sentry_resource(url='https://sentry.sentry.io/preprod/snapshots/123/?selectedSnapshot=login_screen.png')\n", diff --git a/packages/mcp-core/src/tools/catalog-runtime/availability.test.ts b/packages/mcp-core/src/tools/catalog-runtime/availability.test.ts index 4651e4c79..24b23f7a7 100644 --- a/packages/mcp-core/src/tools/catalog-runtime/availability.test.ts +++ b/packages/mcp-core/src/tools/catalog-runtime/availability.test.ts @@ -75,7 +75,11 @@ describe("catalog availability", () => { }); describe("replay capability gating", () => { - const REPLAY_TOOLS = ["get_replay_details", "get_replay_activity"]; + const REPLAY_TOOLS = [ + "get_replay_details", + "get_replay_activity", + "get_replay_dom", + ]; function getInspectContext( constraints: Partial = {}, @@ -86,7 +90,7 @@ describe("catalog availability", () => { }); } - it("offers both replay tools when the constrained project has replays", () => { + it("offers every replay tool when the constrained project has replays", () => { const names = getSearchableToolNames( getInspectContext({ organizationSlug: "my-org", @@ -98,7 +102,7 @@ describe("catalog availability", () => { expect(names).toEqual(expect.arrayContaining(REPLAY_TOOLS)); }); - it("hides both replay tools when the constrained project has no replays", () => { + it("hides every replay tool when the constrained project has no replays", () => { // Advertising a tool that can only fail costs a tool slot and invites a // call that returns nothing useful. const names = getSearchableToolNames( @@ -123,7 +127,7 @@ describe("catalog availability", () => { expect(names).toEqual(expect.arrayContaining(REPLAY_TOOLS)); }); - it("keeps both replay tools off the direct surface", () => { + it("keeps every replay tool off the direct surface", () => { // They are catalog-only by design: the direct surface is budgeted, and // replay review starts from a search or a pasted URL either way. const directToolNames = getToolsForMcpRegistration({ diff --git a/packages/mcp-core/src/tools/catalog/get-replay-dom.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-dom.test.ts new file mode 100644 index 000000000..a8fdfbb0e --- /dev/null +++ b/packages/mcp-core/src/tools/catalog/get-replay-dom.test.ts @@ -0,0 +1,336 @@ +import { afterEach, describe, expect, it } from "vitest"; +import { http, HttpResponse } from "msw"; +import { mswServer, replayDetailsFixture } from "@sentry/mcp-server-mocks"; +import getReplayDom from "./get-replay-dom.js"; +import { getServerContext } from "../../test-setup.js"; + +const REPLAY_URL = `https://us.sentry.io/api/0/organizations/sentry-mcp-evals/replays/${replayDetailsFixture.id}/`; +const SEGMENTS_URL = `https://us.sentry.io/api/0/projects/sentry-mcp-evals/${replayDetailsFixture.project_id}/replays/${replayDetailsFixture.id}/recording-segments/`; + +/** + * Offsets into the fixture recording, in ms from the replay start. + * + * The checkout snapshot lands at T+3m 0.0s. The console error is at T+3m 1.3s, + * and the error banner and disabled button arrive after it — so a read at the + * error must not show either, and a read at the end must show both. + */ +const AT_LOGIN = 12_400; +const AT_CHECKOUT_ERROR = 181_300; +const AT_END = 200_000; + +function callTool( + params: Record = {}, + context = getServerContext(), +) { + return getReplayDom.handler( + { + organizationSlug: "sentry-mcp-evals", + replayId: replayDetailsFixture.id, + regionUrl: "https://us.sentry.io", + atMs: AT_CHECKOUT_ERROR, + lens: "interactive", + maxDepth: 12, + maxNodes: 200, + ...params, + } as never, + context, + ); +} + +afterEach(() => { + mswServer.resetHandlers(); +}); + +describe("get_replay_dom", () => { + it("reconstructs the page as it stood at the requested moment", async () => { + const result = await callTool({ atMs: AT_CHECKOUT_ERROR }); + + expect(result).toMatchInlineSnapshot(` + "# Replay 7e07485f-12f9-416b-8b14-26260799b51f DOM at T+3m 1.3s + + Reconstructed from the snapshot at T+3m 0.0s, applying 1 mutation. + + \`\`\` + html id=3 + └─ body id=60 + └─ div#root id=61 + └─ main id=62 + ├─ form#checkout-form id=80 + │ ├─ div.address-block id=81 + │ │ ├─ input#unit [value="***"] [name=unit] id=82 + │ │ └─ input#zip [value="***"] [name=zip] id=83 + │ ├─ label "Quantity" id=84 + │ ├─ input#quantity [value="3"] [type=number] [name=quantity] id=90 + │ └─ button#complete-order "Complete order" [type=button] id=96 + └─ a#download-receipt "Download receipt" [href=/receipts/latest] id=118 + \`\`\` + + Showing interactive elements and their ancestors. Pass \`rootNodeId\` to focus a subtree, or \`lens: "full"\` for every element." + `); + }); + + it("takes an input value from the input event, not the shipped attribute", async () => { + // The quantity field ships with value="1" and is changed to "3" by a + // `source: 5` event. A reconstruction that applied only `source: 0` + // mutations would render "1" — plausible, authoritative-looking, and stale. + const result = await callTool({ atMs: AT_CHECKOUT_ERROR }); + + expect(result).toContain('input#quantity [value="3"]'); + expect(result).not.toContain('input#quantity [value="1"]'); + }); + + it("does not show structure that arrives after the requested moment", async () => { + // The error banner is added at T+3m 1.4s and the button is disabled at + // T+3m 2.0s, both after the moment asked for. Including either would make + // the tree describe a later page than the caller asked about. + const atError = await callTool({ atMs: AT_CHECKOUT_ERROR, lens: "full" }); + + expect(atError).not.toContain("order-error"); + expect(atError).not.toContain("[disabled]"); + + const atEnd = await callTool({ atMs: AT_END, lens: "full" }); + + expect(atEnd).toContain("div#order-error"); + expect(atEnd).toContain('"Payment failed. Please try again."'); + expect(atEnd).toContain("button#complete-order"); + expect(atEnd).toContain("[disabled]"); + }); + + it("applies a text mutation to the element that carries it", async () => { + // The button's label changes from "Complete order" to "Processing…" via a + // text mutation against the child text node, not the button. + const result = await callTool({ atMs: AT_END }); + + expect(result).toContain('button#complete-order "Processing…"'); + }); + + it("uses the snapshot in effect at the moment, not the first one", async () => { + // Segment 0 snapshots the login page; segment 1 re-snapshots checkout. A + // read at the login click must get the login page, which also proves a + // later snapshot is not merged into an earlier one. + const result = await callTool({ atMs: AT_LOGIN }); + + expect(result).toContain("form#login"); + expect(result).toContain("button#sign-in"); + expect(result).not.toContain("checkout-form"); + }); + + describe("rooting", () => { + it("renders only the named subtree", async () => { + // Node 80 is the checkout form. The receipt link is a sibling of the + // form, so it must not appear. + const result = await callTool({ + atMs: AT_CHECKOUT_ERROR, + rootNodeId: 80, + }); + + expect(result).toContain("form#checkout-form id=80"); + expect(result).not.toContain("download-receipt"); + expect(result).not.toContain("div#root"); + }); + + it("roots at a node id reported by a click signal", async () => { + // 96 is the id `get_replay_activity` reports for the rage-clicked + // button, which is the whole point of the handoff. + const result = await callTool({ + atMs: AT_CHECKOUT_ERROR, + rootNodeId: 96, + }); + + expect(result).toContain('button#complete-order "Complete order"'); + }); + + it("says so when the node does not exist at that moment", async () => { + // 130 is the error banner, which does not exist yet at this offset. + // Reporting an empty tree instead would read as "nothing was there". + const result = await callTool({ + atMs: AT_CHECKOUT_ERROR, + rootNodeId: 130, + }); + + expect(result).toContain("Node 130 does not exist in the DOM"); + expect(result).toContain( + "check the offset of the signal the id came from", + ); + }); + }); + + describe("lenses", () => { + it("drops inert containers under the interactive lens", async () => { + const interactive = await callTool({ atMs: AT_END }); + + // The h1 and the error banner are not interactive and have no + // interactive descendants. + expect(interactive).not.toContain("h1"); + expect(interactive).not.toContain("order-error"); + }); + + it("keeps every element under the full lens", async () => { + const full = await callTool({ atMs: AT_END, lens: "full" }); + + expect(full).toContain("h1"); + expect(full).toContain("div#order-error"); + }); + + it("reports truncation rather than silently cutting the tree", async () => { + const result = await callTool({ + atMs: AT_CHECKOUT_ERROR, + lens: "full", + maxNodes: 4, + }); + + expect(result).toContain("Tree truncated at 4 nodes"); + expect(result).toContain("Raise `maxNodes`"); + }); + }); + + describe("degradation", () => { + it("refuses rather than returning a partial tree when the budget is hit", async () => { + // A page large enough to trip the byte budget before the target moment + // is reached. A tree built from a truncated mutation history looks + // exactly like a complete one, so it must not be returned at all. + const padding = "x".repeat(11 * 1024 * 1024); + mswServer.use( + http.get(SEGMENTS_URL, () => + HttpResponse.json([ + [ + { + type: 5, + timestamp: 1744027200000, + data: { + tag: "breadcrumb", + payload: { category: "ui.click", message: padding }, + }, + }, + ], + ]), + ), + ); + + const result = await callTool({ atMs: AT_END }); + + expect(result).toContain("Cannot reconstruct the DOM"); + expect(result).toContain("A partial tree is not returned"); + expect(result).toContain("An earlier `atMs`"); + expect(result).not.toContain("```"); + }); + + it("says so when no snapshot exists at or before the moment", async () => { + // A recording whose only snapshot is later than the moment asked for. + mswServer.use( + http.get(SEGMENTS_URL, () => + HttpResponse.json([ + [ + { + type: 2, + timestamp: 1744027400000, + data: { node: { id: 1, type: 0, childNodes: [] } }, + }, + ], + ]), + ), + ); + + const result = await callTool({ atMs: 1_000 }); + + expect(result).toContain("No full DOM snapshot appears at or before"); + expect(result).toContain("Try a later `atMs`"); + }); + + it("reports when the recording ends before the requested moment", async () => { + // Not an error, but the tree is the last state on record rather than the + // state at the moment asked for, and those are different claims. + const result = await callTool({ atMs: 600_000 }); + + expect(result).toContain("The recording ends before T+10m 0.0s"); + }); + + it("reports an archived replay", async () => { + mswServer.use( + http.get(REPLAY_URL, () => + HttpResponse.json({ + data: { ...replayDetailsFixture, is_archived: true }, + }), + ), + ); + + const result = await callTool(); + + expect(result).toContain("archived"); + }); + + it("reports a replay with no recording segments", async () => { + mswServer.use( + http.get(REPLAY_URL, () => + HttpResponse.json({ + data: { ...replayDetailsFixture, count_segments: 0 }, + }), + ), + ); + + const result = await callTool(); + + expect(result).toContain("No recording segments are available"); + }); + + it("refuses when the replay has no usable start time", async () => { + // Offsets are relative to the replay's start; without one there is no way + // to place `atMs` on the recording, and either end of the session would + // answer a different question. + mswServer.use( + http.get(REPLAY_URL, () => + HttpResponse.json({ + data: { ...replayDetailsFixture, started_at: null }, + }), + ), + ); + + const result = await callTool(); + + expect(result).toContain("no usable start time"); + }); + + it("rejects a replay outside the active project constraint", async () => { + mswServer.use( + http.get( + "https://us.sentry.io/api/0/projects/sentry-mcp-evals/frontend/", + () => + HttpResponse.json({ + id: "9999999999999999", + slug: "frontend", + name: "frontend", + }), + ), + ); + + await expect( + callTool( + {}, + getServerContext({ constraints: { projectSlug: "frontend" } }), + ), + ).rejects.toThrow("outside the active project constraint"); + }); + }); + + describe("tool definition", () => { + it("is gated like the other replay tools", async () => { + expect(getReplayDom.requiredScopes).toEqual([ + "org:read", + "project:read", + "event:read", + ]); + expect(getReplayDom.requiredCapabilities).toEqual(["replays"]); + expect(getReplayDom.skills).toEqual(["inspect"]); + }); + + it("stays off the direct top-level surface", async () => { + const { isDefaultTopLevelToolName } = await import("../surfaces.js"); + expect(isDefaultTopLevelToolName("get_replay_dom")).toBe(false); + }); + + it("is reachable through the catalog", async () => { + const { default: catalog } = await import("./index.js"); + expect(catalog.get_replay_dom).toBe(getReplayDom); + }); + }); +}); diff --git a/packages/mcp-core/src/tools/catalog/get-replay-dom.ts b/packages/mcp-core/src/tools/catalog/get-replay-dom.ts new file mode 100644 index 000000000..8dfebec4d --- /dev/null +++ b/packages/mcp-core/src/tools/catalog/get-replay-dom.ts @@ -0,0 +1,334 @@ +import { getActiveSpan, setTag } from "@sentry/core"; +import { + MAX_REPLAY_SEGMENTS, + MAX_REPLAY_SEGMENT_BYTES, +} from "../../api-client"; +import type { ReplayRecordingSegmentsResult } from "../../api-client"; +import { formatReplayOffset } from "../../internal/replay-events"; +import type { DomLens, DomReconstruction } from "../../internal/replay-dom"; +import { + DomReconstructor, + countDropped, + renderDomTree, +} from "../../internal/replay-dom"; +import { defineTool } from "../../internal/tool-helpers/define"; +import { apiServiceFromContext } from "../../internal/tool-helpers/api"; +import { + assertReplayWithinProjectConstraint, + resolveReplayParams, +} from "../../internal/tool-helpers/replay"; +import { resolveRegionUrlForOrganization } from "../../internal/tool-helpers/resolve-region-url"; +import type { ServerContext } from "../../types"; +import { z } from "zod"; +import { + ParamOrganizationSlug, + ParamReplayId, + ParamRegionUrl, + ParamReplayUrl, +} from "../../schema"; + +export default defineTool({ + name: "get_replay_dom", + skills: ["inspect"], + requiredScopes: ["org:read", "project:read", "event:read"], + requiredCapabilities: ["replays"], + description: [ + "Read the page structure of a Sentry replay at one moment in time.", + "", + "USE THIS TOOL WHEN USERS:", + "- Ask what the page looked like when something failed", + "- Need to know whether an element existed, was disabled, or was empty", + "- Want the DOM around an element that was clicked or rage-clicked", + "", + "Returns structure only — what the page *was*, not what the user did.", + "Use `get_replay_activity` for that, and to find the `nodeId` to root at.", + "`atMs` is milliseconds from the start of the replay and is required.", + "", + "Text and form values are masked by the SDK before upload, so this answers", + "structural questions, not what a user typed.", + "", + "", + "### What the page looked like when the error fired", + "```", + "get_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)", + "```", + "", + "### The subtree around a rage-clicked element", + "```", + "get_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')", + "```", + "", + ].join("\n"), + inputSchema: { + replayUrl: ParamReplayUrl.optional(), + organizationSlug: ParamOrganizationSlug.optional(), + replayId: ParamReplayId.optional(), + regionUrl: ParamRegionUrl.nullable().optional(), + atMs: z + .number() + .min(0) + .describe( + "The moment to reconstruct, in milliseconds from the start of the replay. Required: a structural read has no sensible default moment.", + ), + rootNodeId: z + .number() + .optional() + .describe( + "Render only this node and its descendants. Use the `nodeId` reported by a click signal in `get_replay_activity`.", + ), + lens: z + .enum(["interactive", "full"]) + .default("interactive") + .describe( + "`interactive` keeps elements a user can act on plus the ancestors that place them; `full` keeps every element.", + ), + maxDepth: z.number().min(1).max(50).default(12), + maxNodes: z.number().min(1).max(1000).default(200), + }, + annotations: { + readOnlyHint: true, + destructiveHint: false, + openWorldHint: true, + }, + async handler(params, context: ServerContext) { + const resolved = resolveReplayParams(params); + const regionUrl = await resolveRegionUrlForOrganization({ + context, + organizationSlug: resolved.organizationSlug, + regionUrl: params.regionUrl, + }); + const apiService = apiServiceFromContext(context, { + regionUrl: regionUrl ?? undefined, + }); + + setTag("organization.slug", resolved.organizationSlug); + setTag("replay.id", resolved.replayId); + + const replay = await apiService.getReplayDetails({ + organizationSlug: resolved.organizationSlug, + replayId: resolved.replayId, + }); + await assertReplayWithinProjectConstraint({ + apiService, + organizationSlug: resolved.organizationSlug, + replay, + projectSlug: context.constraints.projectSlug, + }); + + const heading = `# Replay ${replay.id} DOM at ${formatReplayOffset(params.atMs)}`; + + if (replay.is_archived === true) { + return `${heading}\n\nRecording is archived and not available for playback.`; + } + + const projectId = + replay.project_id != null ? String(replay.project_id) : null; + if (!projectId || (replay.count_segments ?? 0) === 0) { + return `${heading}\n\nNo recording segments are available for this replay.`; + } + + // Offsets are relative to the replay's own start, matching + // `get_replay_activity`, but rrweb timestamps are absolute. Without a + // parseable start there is no way to place `atMs` on the recording, and + // picking either end of the session would answer a different question + // than the one asked. + const startedAtMs = replay.started_at ? Date.parse(replay.started_at) : NaN; + if (Number.isNaN(startedAtMs)) { + return `${heading}\n\nThis replay has no usable start time, so an offset cannot be placed on the recording.`; + } + + const reconstructor = new DomReconstructor({ + atMs: startedAtMs + params.atMs, + }); + + // Stop as soon as a segment carries an event past the target: everything + // after it would be discarded, and paging the rest of a long session to + // discard it is the difference between a bounded read and a whole-session + // one. + let reachedTarget = false; + const stats = await apiService.streamReplayRecordingSegments( + { + organizationSlug: resolved.organizationSlug, + projectSlugOrId: projectId, + replayId: resolved.replayId, + }, + (segment) => { + for (const event of segment) { + if (reconstructor.apply(event) === "past-target") { + reachedTarget = true; + return "stop"; + } + } + }, + ); + + const reconstruction = reconstructor.result(startedAtMs); + + const span = getActiveSpan(); + span?.setAttribute("replay.dom.at_ms", params.atMs); + span?.setAttribute("replay.dom.lens", params.lens); + span?.setAttribute("replay.dom.rooted", params.rootNodeId !== undefined); + span?.setAttribute("replay.dom.nodes", reconstruction.nodes.size); + span?.setAttribute("replay.dom.mutations", reconstruction.mutationsApplied); + span?.setAttribute( + "replay.dom.dropped", + countDropped(reconstruction.dropped), + ); + span?.setAttribute("replay.dom.segments_read", stats.segmentsRead); + + // A read that ran out of budget before reaching the target has an + // incomplete mutation history, and the resulting tree reads exactly like a + // complete one. Refuse instead, and say what would make the read fit. + if (stats.truncatedBy !== null) { + return [ + heading, + "", + refusalMessage(stats.truncatedBy, params.atMs, params.rootNodeId), + ].join("\n"); + } + + if (reconstruction.missingSnapshot) { + return [ + heading, + "", + `No full DOM snapshot appears at or before ${formatReplayOffset(params.atMs)}, so there is no structure to reconstruct from. Try a later \`atMs\`.`, + ].join("\n"); + } + + const tree = renderDomTree(reconstruction, { + lens: params.lens as DomLens, + rootNodeId: params.rootNodeId, + maxDepth: params.maxDepth, + maxNodes: params.maxNodes, + }); + + if (tree.rootNotFound) { + return [ + heading, + "", + `Node ${params.rootNodeId} does not exist in the DOM at ${formatReplayOffset(params.atMs)}. It may have been added later or removed earlier; check the offset of the signal the id came from.`, + ].join("\n"); + } + + span?.setAttribute("gen_ai.tool.call.result.count", tree.nodesRendered); + + return formatDomOutput({ + heading, + reconstruction, + tree, + lens: params.lens, + rootNodeId: params.rootNodeId, + reachedTarget, + atMs: params.atMs, + }); + }, +}); + +function refusalMessage( + truncatedBy: NonNullable, + atMs: number, + rootNodeId?: number, +): string { + const bound = + truncatedBy === "segments" + ? `the first ${MAX_REPLAY_SEGMENTS} segments` + : `${MAX_REPLAY_SEGMENT_BYTES / (1024 * 1024)}MB of recording data`; + + const lines = [ + `Cannot reconstruct the DOM at ${formatReplayOffset(atMs)}: the read hit ${bound} before reaching that moment.`, + "", + "A partial tree is not returned, because it would be indistinguishable from a complete one. What helps:", + `- An earlier \`atMs\`. Reconstruction cost grows with how far into the recording the moment is.`, + ]; + + if (rootNodeId === undefined) { + lines.push( + "- Nothing else, in this case: `rootNodeId` narrows what is rendered, not what must be read.", + ); + } + + lines.push( + '- `get_replay_activity` at `grain: "digest"` still works on this replay, and reports what happened without reconstructing structure.', + ); + + return lines.join("\n"); +} + +function formatDomOutput({ + heading, + reconstruction, + tree, + lens, + rootNodeId, + reachedTarget, + atMs, +}: { + heading: string; + reconstruction: DomReconstruction; + tree: ReturnType; + lens: DomLens; + rootNodeId?: number; + reachedTarget: boolean; + atMs: number; +}): string { + const lines: string[] = [heading, ""]; + + // Fidelity first. A tree assembled from a snapshot that dropped a third of + // its mutations looks identical to a clean one, and the difference decides + // whether the answer is usable. + const mutations = reconstruction.mutationsApplied; + lines.push( + `Reconstructed from the snapshot at ${formatReplayOffset(reconstruction.snapshotOffsetMs)}, applying ${mutations.toLocaleString("en-US")} mutation${mutations === 1 ? "" : "s"}.`, + ); + + const droppedTotal = countDropped(reconstruction.dropped); + if (droppedTotal > 0) { + const reasons = Object.entries(reconstruction.dropped) + .filter(([, count]) => count > 0) + .map(([reason, count]) => `${count} ${reason}`) + .join(", "); + lines.push( + `Dropped ${droppedTotal.toLocaleString("en-US")} operation${droppedTotal === 1 ? "" : "s"} (${reasons}); the structure below may be incomplete.`, + ); + } + + // The recording ending before `atMs` is not an error, but it does mean the + // tree is the last state on record rather than the state at the moment + // asked for. + if (!reachedTarget) { + lines.push( + `The recording ends before ${formatReplayOffset(atMs)}; this is its final state.`, + ); + } + + lines.push(""); + + if (tree.lines.length === 0) { + lines.push( + lens === "interactive" + ? 'No interactive elements are present here. Try `lens: "full"`.' + : "No elements are present here.", + ); + return lines.join("\n"); + } + + lines.push("```"); + lines.push(...tree.lines); + lines.push("```"); + + if (tree.truncated) { + lines.push(""); + lines.push( + `Tree truncated at ${tree.nodesRendered} nodes. Raise \`maxNodes\`/\`maxDepth\`, or pass a \`rootNodeId\` to narrow the read.`, + ); + } + + if (rootNodeId === undefined && lens === "interactive") { + lines.push(""); + lines.push( + 'Showing interactive elements and their ancestors. Pass `rootNodeId` to focus a subtree, or `lens: "full"` for every element.', + ); + } + + return lines.join("\n"); +} diff --git a/packages/mcp-core/src/tools/catalog/index.ts b/packages/mcp-core/src/tools/catalog/index.ts index 478bfda0f..73e4c212f 100644 --- a/packages/mcp-core/src/tools/catalog/index.ts +++ b/packages/mcp-core/src/tools/catalog/index.ts @@ -25,6 +25,7 @@ import getTraceDetails from "./get-trace-details"; import getSpanDetails from "./get-span-details"; import getReplayActivity from "./get-replay-activity"; import getReplayDetails from "./get-replay-details"; +import getReplayDom from "./get-replay-dom"; import getEventAttachment from "./get-event-attachment"; import updateIssue from "./update-issue"; import searchEvents from "./search-events"; @@ -90,6 +91,7 @@ const catalogTools = { get_span_details: getSpanDetails, get_replay_details: getReplayDetails, get_replay_activity: getReplayActivity, + get_replay_dom: getReplayDom, get_event_attachment: getEventAttachment, update_issue: updateIssue, search_events: searchEvents, From a41fe8c13d70861f06c4a6a74bdbd228f32f2556 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 11:14:10 -0700 Subject: [PATCH 21/26] docs(replays): Record get_replay_dom as built, not deferred MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Promote the DOM tree section from Future Work into Design, replace the proposed output with the shipped format, and add a requirement to the delta spec. The checkout-frequency question the section called "the single largest open question" is answered by design rather than measurement: superseding the node map on each snapshot makes the frequent-checkout and single-snapshot cases identical, so it no longer gates the interface. What remains unmeasured is whether the budget is generous enough on a long real replay — and if it is not, the tool refuses, which is a correct answer rather than a wrong tree. Also record the divergences: ARIA roles matched by value rather than `role` presence, node ids inline rather than in a trailing index, and two reported conditions the spec did not name (a recording that ends before `atMs`, and a replay with no parseable `started_at`). The testing list gained a case it was missing, added because it caused a real bug: a fixture's node shape must match what rrweb actually emits. A merely plausible fixture tests the implementation against itself. Co-Authored-By: Claude Opus 5 --- docs/specs/replay-review.md | 372 +++++++++++------- .../changes/improve-replay-review/proposal.md | 7 + .../specs/replay-review/spec.md | 53 +++ .../changes/improve-replay-review/tasks.md | 18 +- 4 files changed, 303 insertions(+), 147 deletions(-) diff --git a/docs/specs/replay-review.md b/docs/specs/replay-review.md index 2aa38f1ea..5fcc55fdc 100644 --- a/docs/specs/replay-review.md +++ b/docs/specs/replay-review.md @@ -113,12 +113,16 @@ Behavior worth porting rather than reinventing: ### Tool surface -Add one catalog-only tool. Keep the top-level surface unchanged. +Add catalog-only tools. Keep the top-level surface unchanged. ```text get_replay_activity +get_replay_dom ``` +`get_replay_dom` arrived later in the same change; see "`get_replay_dom` — the +structure" below for why it is a separate tool rather than a mode of the other. + Existing tools change behavior, not identity: - `get_replay_details` returns a map plus a suggested next call. @@ -264,136 +268,12 @@ Requirements: already-summarized replays unless the tool starts a task and accepts the LLM cost on every call. -## Fixes Outside the Tool Surface - -These are correctness bugs in shipped code, independent of the new tool: - -- Classify replay events by `payload.category`/`op`; port the noise filter and - the dead/rage click rules. -- Replace the magnitude-based timestamp heuristic with per-type units. -- Follow the `Link` header's `cursor` on the segments index; drop the no-op - `download=true`. This needs the raw-response request path, since - `requestJSON` does not expose response headers. Read at most 150 segments or - 10MB of raw segment JSON, whichever comes first, and say which bound stopped - the read. 150 matches the clamp Sentry's own summarize endpoint applies; the - byte ceiling is the real guard, since segment sizes vary by orders of - magnitude, parsed rrweb objects expand well beyond their JSON size, and - `mcp-cloudflare` runs under a 128MB Workers limit. Both are provisional and - should be measured against real replays during QA. -- Rebuild `replay-recording-segments.json` from real SDK shapes, re-baseline - inline snapshots, and add a replay eval. -- Report every truncation. Today `MAX_ACTIVITY_EVENTS`, `MAX_RELATED_ERRORS`, - and `MAX_RELATED_TRACES` stop silently; only the issue-details replay list - prints an "and N more" line. -- Reconcile replay sorting against Sentry's actual sort configuration. The - authority is `sort_config` in - `sentry/replays/usecases/query/configs/aggregate_sort.py`; both the scalar and - aggregated query paths order by it, and `_get_sort_column` raises a - `ParseError` for anything absent. `REPLAY_SORT_FIELDS` in - `packages/mcp-core/src/tools/support/search-events/replays.ts` is missing only - `count_screens` and the aliases `browser`, `os`, and `os_name`. - `count_traces`, `count_segments`, and `viewed_by_me` are **not** sortable and - must not be added. `device.model` is already correct; the mismatch is that - discovery advertises `device.model_id`, which is filterable but not sortable. - The real defect is on the discovery side: `REPLAY_FIELDS` in - `packages/mcp-core/src/internal/agents/tools/dataset-fields.ts` presents - filterable fields as if they were sortable, so an agent picks a sort from - discovery output and gets a `UserInputError`. Note that discovery carries no - sortability signal at all, so the fix is to add one rather than to correct an - existing claim — see the divergences section. -- Gate the `dataset="replays"` path in `search_events` on the `replays` - capability, so replay search and replay details agree about availability. - A tool-level `requiredCapabilities` will not do: `search_events` serves six - datasets, and per `tools/catalog-runtime/availability.ts` the check only - applies when a `projectSlug` constraint is set. This needs a runtime rejection - in the handler plus removal of `replays` from the advertised dataset options. -- Distinguish rate limiting from absence in `listReplayIdsForIssue`. The - `replay-count` endpoint enforces 20 req/s per IP, per user, **and per - organization**. `get_issue_details` calls it on every lookup, and the current - `.catch(() => undefined)` makes throttling indistinguishable from "no - replays" — so parallel issue triage silently loses the Session Replay - section. - -## Examples - -Orientation, then a targeted read: - -```text -get_sentry_resource(url='https://my-org.sentry.io/explore/replays/7e07485f…/') -→ map: 1,182 signals, 4 pages, 2 failed network, 2 rage clicks - error at T+311.8s → suggested window - -get_replay_activity(replayId='7e07485f…', startMs=306000, endMs=316000, grain='detail') -→ T+311.8s click button#complete-order "Complete order" - T+312.0s network POST /api/checkout → 500 (1240ms) - body: - T+312.1s console error TypeError: Cannot read 'id' of undefined - at submitOrder (checkout.tsx:214) - T+312.3s click rage ×5 on button#complete-order (7000ms, timeout) -``` - -Cheap shape check on a long session: - -```text -get_replay_activity(replayId='7e07485f…', grain='digest', kinds=['network','console']) -→ network ×58 (2 failed) · console ×4 (2 error) -``` - -## Implementation - -1. Port the event taxonomy and log-message semantics into a shared internal - module; unit-test against realistically shaped fixtures. -2. Fix segment pagination and the timestamp units in the API client. -3. Rebuild replay fixtures; re-baseline snapshots. -4. Restructure `get_replay_details` into map form with a suggested next call - derived from error timestamps resolved through `replays-events-meta`. -5. Add `get_replay_activity` as a catalog-only tool. -6. Wire the Seer summary as an optional chapters section, read once. -7. Apply the sort-field, capability-gating, and rate-limit fixes. -8. Add a replay eval, then run - `pnpm run --filter @sentry/mcp-core generate-definitions`. +### `get_replay_dom` — the structure -## Testing - -- Unit tests per event type using SDK-shaped events, including - seconds-versus-milliseconds timestamps and the dead/rage click boundaries - (`timeAfterClickMs` at 6999 vs 7000, `clickCount` at 4 vs 5). -- Snapshot tests for each grain and for a windowed slice. -- Degradation tests: summarize 403, summarize timeout, segment fetch 404, - archived replay, replay with zero segments, and a replay exceeding one page of - segments. -- Redaction tests asserting `` and `` rendering. -- A replay eval covering map-then-zoom navigation. -- `pnpm run tsc && pnpm run lint && pnpm run test` must pass, plus - `pnpm run measure-tokens` to confirm tool-definition overhead. - -## Migration - -`get_replay_details` output changes shape; its inputs and name do not. Callers -passing replay URLs through `get_sentry_resource` are unaffected. Existing -inline snapshots must be re-baselined as part of the taxonomy fix, not -separately. Tool count rises by one, well inside the 20-tool target. - -## Out of Scope - -Web visual snapshots. *Rendering* a DOM snapshot requires running rrweb playback -in a browser, which this server has no place to host. The image plumbing already -exists (`get_snapshot_image`, `createImagePreview` in -`packages/mcp-core/src/internal/blob-utils.ts`), so this is a hosting gap, not a -protocol one. - -Note the distinction from the DOM *tree*, which is a different problem with a -different answer. Rendering needs a layout engine; reading the structure does -not, and the structure is already in the segments we download. See "DOM tree -reads" under Future Work. - -Mobile replay video is tractable — replays are captured as video and -`/projects/{org}/{project}/replays/{replay_id}/videos/{segment_id}/` exists — -but it is deferred rather than half-built alongside the web path. - -## Future Work - -### DOM tree reads +**Implemented.** Originally deferred as Future Work; built in the same change +once the map and the zoom landed. The section below is the design as written, +with the as-built notes folded into "What QA no longer needs to decide" and the +Divergences section at the end. The largest capability gap against comparable tools. Peers expose a `review-snapshot`-style call returning a screenshot, a component tree, and @@ -466,6 +346,15 @@ QA should answer this before the interface is settled. Implementing for the frequent-checkout case and discovering the single-snapshot case in production means a tool that works on short replays and OOMs on the long ones that matter. +**As built, this question no longer gates the interface.** A later +`FullSnapshot` supersedes the node map wholesale rather than merging into it, so +the frequent-checkout and single-snapshot cases take the same code path with no +branch to get wrong. Reads also stop paging as soon as an event passes the +target, so cost is bounded by how far into the recording the moment is rather +than by session length. What remains unmeasured is whether the existing budget is +*generous enough* on a long real replay — if it is not, the tool refuses, which +is a correct answer rather than a wrong tree. + #### Fidelity must be reported, not assumed A reconstruction can be complete, partial, or wrong, and the three are @@ -564,23 +453,28 @@ implementation: `get_replay_activity`, so the two cannot drift into competing accounts of the same moment. -Expected output, rooted at a click's node id: +Output, rooted at a click's node id (as shipped): ```text -# Replay 7e07485f… DOM at T+3m 1.3s +# Replay 7e07485f-12f9-416b-8b14-26260799b51f DOM at T+3m 1.3s -Reconstructed from the snapshot at T+0.4s, applying 1,847 mutations. -Dropped 3 operations (unknown parentId). +Reconstructed from the snapshot at T+3m 0.0s, applying 1 mutation. -form#checkout-form - ├─ div.address-block - │ ├─ input#unit [value="***"] - │ └─ input#zip [value="***"] - └─ button#complete-order "Complete order" [disabled] - -Node ids: form#checkout-form=88, button#complete-order=96 +form#checkout-form id=80 +├─ div.address-block id=81 +│ ├─ input#unit [name=unit] [value="***"] id=82 +│ └─ input#zip [name=zip] [value="***"] id=83 +├─ label "Quantity" id=84 +├─ input#quantity [value="3"] [type=number] [name=quantity] id=90 +└─ button#complete-order "Complete order" [type=button] id=96 ``` +Two departures from the sketch above. Node ids are carried inline on each line +rather than gathered into a trailing index, because a trailing list is a second +thing to cross-reference and every line needs one anyway. And the dropped-operation +line appears only when something was dropped: a permanent `Dropped 0 operations` +line trains a reader to skip the place where the warning would appear. + #### Masking is not redaction `maskAllText` and `maskAllInputs` default to on, so a real tree arrives with @@ -612,17 +506,158 @@ Beyond the usual unit coverage, three cases carry the risk: top-level surface, reachable through the catalog. The existing `availability.test.ts` cases extend to cover it. +One case this list missed, added after it caused a real bug: **the fixture's node +shape must match what rrweb actually emits.** The unit fixtures modelled an +id-less document wrapper, but `serializeNodeWithId` assigns an id to every node +including the Document, so a real snapshot roots at a `nodeType: 0` node. Every +test passed while the renderer produced nothing on real input. A fixture that is +merely plausible tests the implementation against itself. + #### Open questions for QA -- Does a `FullSnapshot` appear only at segment zero, or periodically? This - decides whether reconstruction cost scales with session length or with - checkout spacing — see above. +- ~~Does a `FullSnapshot` appear only at segment zero, or periodically?~~ + Answered by design rather than measurement: a later snapshot supersedes the + node map wholesale, so both cases take the same path. What remains is whether + the budget is generous enough on a long real replay — and if it is not, the + tool refuses rather than answering wrongly. - Do real click breadcrumbs carry `payload.data.node.id` consistently, or only sometimes? This decides whether subtree rooting is a reliable entry point or - best-effort. + best-effort. The id is now reported at detail grain by + `get_replay_activity`, so the handoff exists; what is unmeasured is how often + it is populated in the wild. - How large is a real `FullSnapshot` in practice? The 10MB segment budget was set for signal extraction; a tree read has a different profile. + +## Fixes Outside the Tool Surface + +These are correctness bugs in shipped code, independent of the new tool: + +- Classify replay events by `payload.category`/`op`; port the noise filter and + the dead/rage click rules. +- Replace the magnitude-based timestamp heuristic with per-type units. +- Follow the `Link` header's `cursor` on the segments index; drop the no-op + `download=true`. This needs the raw-response request path, since + `requestJSON` does not expose response headers. Read at most 150 segments or + 10MB of raw segment JSON, whichever comes first, and say which bound stopped + the read. 150 matches the clamp Sentry's own summarize endpoint applies; the + byte ceiling is the real guard, since segment sizes vary by orders of + magnitude, parsed rrweb objects expand well beyond their JSON size, and + `mcp-cloudflare` runs under a 128MB Workers limit. Both are provisional and + should be measured against real replays during QA. +- Rebuild `replay-recording-segments.json` from real SDK shapes, re-baseline + inline snapshots, and add a replay eval. +- Report every truncation. Today `MAX_ACTIVITY_EVENTS`, `MAX_RELATED_ERRORS`, + and `MAX_RELATED_TRACES` stop silently; only the issue-details replay list + prints an "and N more" line. +- Reconcile replay sorting against Sentry's actual sort configuration. The + authority is `sort_config` in + `sentry/replays/usecases/query/configs/aggregate_sort.py`; both the scalar and + aggregated query paths order by it, and `_get_sort_column` raises a + `ParseError` for anything absent. `REPLAY_SORT_FIELDS` in + `packages/mcp-core/src/tools/support/search-events/replays.ts` is missing only + `count_screens` and the aliases `browser`, `os`, and `os_name`. + `count_traces`, `count_segments`, and `viewed_by_me` are **not** sortable and + must not be added. `device.model` is already correct; the mismatch is that + discovery advertises `device.model_id`, which is filterable but not sortable. + The real defect is on the discovery side: `REPLAY_FIELDS` in + `packages/mcp-core/src/internal/agents/tools/dataset-fields.ts` presents + filterable fields as if they were sortable, so an agent picks a sort from + discovery output and gets a `UserInputError`. Note that discovery carries no + sortability signal at all, so the fix is to add one rather than to correct an + existing claim — see the divergences section. +- Gate the `dataset="replays"` path in `search_events` on the `replays` + capability, so replay search and replay details agree about availability. + A tool-level `requiredCapabilities` will not do: `search_events` serves six + datasets, and per `tools/catalog-runtime/availability.ts` the check only + applies when a `projectSlug` constraint is set. This needs a runtime rejection + in the handler plus removal of `replays` from the advertised dataset options. +- Distinguish rate limiting from absence in `listReplayIdsForIssue`. The + `replay-count` endpoint enforces 20 req/s per IP, per user, **and per + organization**. `get_issue_details` calls it on every lookup, and the current + `.catch(() => undefined)` makes throttling indistinguishable from "no + replays" — so parallel issue triage silently loses the Session Replay + section. + +## Examples + +Orientation, then a targeted read: + +```text +get_sentry_resource(url='https://my-org.sentry.io/explore/replays/7e07485f…/') +→ map: 1,182 signals, 4 pages, 2 failed network, 2 rage clicks + error at T+311.8s → suggested window + +get_replay_activity(replayId='7e07485f…', startMs=306000, endMs=316000, grain='detail') +→ T+311.8s click button#complete-order "Complete order" + T+312.0s network POST /api/checkout → 500 (1240ms) + body: + T+312.1s console error TypeError: Cannot read 'id' of undefined + at submitOrder (checkout.tsx:214) + T+312.3s click rage ×5 on button#complete-order (7000ms, timeout) +``` + +Cheap shape check on a long session: + +```text +get_replay_activity(replayId='7e07485f…', grain='digest', kinds=['network','console']) +→ network ×58 (2 failed) · console ×4 (2 error) +``` + +## Implementation + +1. Port the event taxonomy and log-message semantics into a shared internal + module; unit-test against realistically shaped fixtures. +2. Fix segment pagination and the timestamp units in the API client. +3. Rebuild replay fixtures; re-baseline snapshots. +4. Restructure `get_replay_details` into map form with a suggested next call + derived from error timestamps resolved through `replays-events-meta`. +5. Add `get_replay_activity` as a catalog-only tool. +6. Wire the Seer summary as an optional chapters section, read once. +7. Apply the sort-field, capability-gating, and rate-limit fixes. +8. Add a replay eval, then run + `pnpm run --filter @sentry/mcp-core generate-definitions`. + +## Testing + +- Unit tests per event type using SDK-shaped events, including + seconds-versus-milliseconds timestamps and the dead/rage click boundaries + (`timeAfterClickMs` at 6999 vs 7000, `clickCount` at 4 vs 5). +- Snapshot tests for each grain and for a windowed slice. +- Degradation tests: summarize 403, summarize timeout, segment fetch 404, + archived replay, replay with zero segments, and a replay exceeding one page of + segments. +- Redaction tests asserting `` and `` rendering. +- A replay eval covering map-then-zoom navigation. +- `pnpm run tsc && pnpm run lint && pnpm run test` must pass, plus + `pnpm run measure-tokens` to confirm tool-definition overhead. + +## Migration + +`get_replay_details` output changes shape; its inputs and name do not. Callers +passing replay URLs through `get_sentry_resource` are unaffected. Existing +inline snapshots must be re-baselined as part of the taxonomy fix, not +separately. Tool count rises by one, well inside the 20-tool target. + +## Out of Scope + +Web visual snapshots. *Rendering* a DOM snapshot requires running rrweb playback +in a browser, which this server has no place to host. The image plumbing already +exists (`get_snapshot_image`, `createImagePreview` in +`packages/mcp-core/src/internal/blob-utils.ts`), so this is a hosting gap, not a +protocol one. + +Note the distinction from the DOM *tree*, which is a different problem with a +different answer. Rendering needs a layout engine; reading the structure does +not, and the structure is already in the segments we download. See +"`get_replay_dom` — the structure" above, which is built. + +Mobile replay video is tractable — replays are captured as video and +`/projects/{org}/{project}/replays/{replay_id}/videos/{segment_id}/` exists — +but it is deferred rather than half-built alongside the web path. + +## Future Work + ### Other - Friction analysis via `GET /organizations/{org}/replay-selectors/`, which @@ -729,16 +764,61 @@ Beyond the spec's list, the taxonomy port includes two upstream fixes: `resolveTimestampMs` returning `null` for a non-finite timestamp, so one scrubbed click cannot take out the events around it. +### The interactive lens matches ARIA roles by value, not `role` presence + +The spec listed `[role]` alongside `[onclick]` and `[tabindex]` as a marker of +interactivity. Presence-matching is wrong: `role="alert"` is a live region, not +a control, so every status banner was pulled into a tree labelled interactive. +The shipped lens matches the WAI-ARIA widget roles by value. `[onclick]`, +`[tabindex]`, and `[href]` remain presence-matched, since they have no +non-interactive spellings. + +### Node ids render inline rather than as a trailing index + +The proposed output gathered ids into a `Node ids: …` footer. Every rendered line +needs an id, so the footer duplicated the tree while adding a cross-reference +step. Ids are on the lines themselves. + +### The DOM tool reports two conditions the spec did not name + +Both are cases where the answer is available but does not mean what a tree would +imply, so silence would be a wrong answer rather than a missing one: + +- **The recording ends before `atMs`.** Not an error, but the tree is the last + state on record rather than the state at the moment asked for, and those are + different claims. +- **The replay has no parseable `started_at`.** Offsets are relative to the + replay's own start, matching `get_replay_activity`. Without one there is no way + to place `atMs` on the recording, and either end of the session would answer a + different question. + +### `get_replay_dom` shipped in the same change, not as follow-up work + +The spec filed DOM tree reads under Future Work, on the reasoning that the +checkout-frequency question had to be settled by QA before the interface could +be. That turned out to be avoidable: superseding the node map on each snapshot +makes both cases identical, so the question stopped gating the design. The map +and the zoom are also of limited use for structural questions without it — which +is what prompted building it immediately rather than later. + ### Still provisional The 150-segment and 10MB bounds remain unmeasured against real replays. Mocks cannot exercise them meaningfully — the fixtures are orders of magnitude smaller than a real recording — so they stand as reasoned defaults until QA against a -real organization confirms or moves them. +real organization confirms or moves them. `get_replay_dom` is the read most +exposed to this, since a DOM reconstruction must page to the requested moment +rather than sampling; it refuses when the budget runs out, so the failure mode is +a refusal rather than a wrong tree, but the frequency of that refusal on real +sessions is unknown. ## References - Implementation: `packages/mcp-core/src/tools/catalog/get-replay-details.ts` +- Activity tool: `packages/mcp-core/src/tools/catalog/get-replay-activity.ts` +- DOM tool: `packages/mcp-core/src/tools/catalog/get-replay-dom.ts` +- DOM reconstruction and rendering: `packages/mcp-core/src/internal/replay-dom.ts` +- Event classification: `packages/mcp-core/src/internal/replay-events.ts` - Replay search: `packages/mcp-core/src/tools/support/search-events/replays.ts` - API client: `packages/mcp-core/src/api-client/client.ts` - Schemas: `packages/mcp-core/src/api-client/schema.ts` diff --git a/openspec/changes/improve-replay-review/proposal.md b/openspec/changes/improve-replay-review/proposal.md index acc8a9d74..ee90aac11 100644 --- a/openspec/changes/improve-replay-review/proposal.md +++ b/openspec/changes/improve-replay-review/proposal.md @@ -31,6 +31,13 @@ cannot distinguish truncation from absence. - Add `get_replay_activity` as a catalog-only tool: a time window (`startMs`, `endMs`), a grain (`digest`, `standard`, `detail`), an optional `kinds` allow-list, and `limit`/`cursor` paging. +- Add `get_replay_dom` as a catalog-only tool: reconstruct the page structure at + a required `atMs` from the rrweb snapshot and mutation events already present + in the segments and discarded today, rooted at an optional `rootNodeId` taken + from a click signal. Originally filed as follow-up work; the map and the zoom + answer what happened but leave "what was the page" unanswerable, which is the + question a structural failure poses. Refuses rather than returning a partial + tree when the segment budget runs out. - Surface Sentry's experimental replay summary endpoint as an optional `## Chapters` section, read once per call — never started, never retried — strictly additive and degrading silently. diff --git a/openspec/changes/improve-replay-review/specs/replay-review/spec.md b/openspec/changes/improve-replay-review/specs/replay-review/spec.md index 02f8a1162..ff11fecaa 100644 --- a/openspec/changes/improve-replay-review/specs/replay-review/spec.md +++ b/openspec/changes/improve-replay-review/specs/replay-review/spec.md @@ -117,6 +117,59 @@ replay signals for a requested time window at a requested grain. - **WHEN** the MCP server registers tools - **THEN** `get_replay_activity` is catalog-only, requires the `inspect` skill, and requires the `replays` project capability +### Requirement: Replay DOM structure is readable at a point in time +The system SHALL provide a `get_replay_dom` catalog tool that reconstructs the +page structure of a replay at one moment from the recording's rrweb snapshot and +mutation events, and returns it as an indented tree. + +#### Scenario: Point-in-time reconstruction +- **WHEN** a caller supplies `atMs` +- **THEN** the tree reflects the last full snapshot at or before that moment with every intervening mutation applied, and excludes structure that arrives after it + +#### Scenario: Required moment +- **WHEN** a caller omits `atMs` +- **THEN** the request is rejected, because defaulting to either end of the session would answer a different question + +#### Scenario: Input values supersede shipped attributes +- **WHEN** a form control's value is changed by an rrweb input event rather than an attribute mutation +- **THEN** the tree renders the changed value, not the value the page shipped with + +#### Scenario: Later snapshot supersedes earlier state +- **WHEN** a recording contains more than one full snapshot at or before the requested moment +- **THEN** the newest one replaces prior state wholesale rather than being merged into it + +#### Scenario: Subtree rooting +- **WHEN** a caller supplies `rootNodeId` +- **THEN** only that node and its descendants are rendered + +#### Scenario: Rooting at a node that does not exist yet +- **WHEN** `rootNodeId` names a node absent from the DOM at the requested moment +- **THEN** the response says the node does not exist, rather than returning an empty tree + +#### Scenario: Interactive lens +- **WHEN** a caller requests the default `interactive` lens +- **THEN** elements a user can act on are kept along with the ancestors that place them, and inert leaves are dropped + +#### Scenario: Budget refusal +- **WHEN** the segment budget is exhausted before the requested moment is reached +- **THEN** the tool refuses and explains what would help, and does not return a partial tree + +#### Scenario: Fidelity reporting +- **WHEN** a reconstruction drops operations, or the recording ends before the requested moment +- **THEN** the response states it, so a partial reconstruction is not read as a complete one + +#### Scenario: Structural scope +- **WHEN** text or form values were masked by the SDK before upload +- **THEN** they render as delivered and are not labeled redacted, since client-side masking leaves no marker + +#### Scenario: Node id handoff +- **WHEN** a caller reads replay activity at `detail` grain +- **THEN** each click signal reports the rrweb node id that `get_replay_dom` accepts as `rootNodeId` + +#### Scenario: Tool availability +- **WHEN** the MCP server registers tools +- **THEN** `get_replay_dom` is catalog-only, requires the `inspect` skill, and requires the `replays` project capability + ### Requirement: Unavailable replay payload is labeled The system SHALL distinguish payload that was never captured from payload that was removed by masking. diff --git a/openspec/changes/improve-replay-review/tasks.md b/openspec/changes/improve-replay-review/tasks.md index b868276a9..f370ec7b4 100644 --- a/openspec/changes/improve-replay-review/tasks.md +++ b/openspec/changes/improve-replay-review/tasks.md @@ -81,4 +81,20 @@ baselines encode the same wrong event shape they do today. - [x] 9.3 Run `pnpm run lint`. 2 warnings, both pre-existing and unrelated (`mcp-cloudflare/index.html` noDescendingSpecificity, `oauth/helpers.test.ts` noUnusedImports). - [x] 9.4 Run `pnpm run test`. 8/8 tasks; mcp-core 1481 passed / 6 skipped. The `smoke-tests` package skips without a deployed `PREVIEW_URL`, which is expected locally and unrelated to replays. - [x] 9.5 Run `pnpm run measure-tokens` and confirm the added tool definition stays within budget. Unchanged at 5,598 tokens across 9 direct tools — `get_replay_activity` is catalog-only, so it costs no direct-surface budget. -- [ ] 9.6 QA against a real organization with the `mcp-qa` skill, since mocks cannot prove the SDK-shape fix. Confirm on a long real replay that the 150-segment and 10MB bounds hold under the Workers memory ceiling, and adjust them if not. While there, capture three observations for the deferred DOM tree work (see "DOM tree reads" in `docs/specs/replay-review.md`): whether `FullSnapshot` (`type: 2`) events appear periodically or only once at segment zero — this decides whether reconstruction cost scales with checkout spacing or with session length, and so decides the feature's shape; whether real click breadcrumbs carry `payload.data.node.id` consistently; and how large a real `FullSnapshot` is. None of these block this change. +- [ ] 9.6 QA against a real organization with the `mcp-qa` skill, since mocks cannot prove the SDK-shape fix. Confirm on a long real replay that the 150-segment and 10MB bounds hold under the Workers memory ceiling, and adjust them if not. `get_replay_dom` is the read most exposed to those bounds, since it must page to the requested moment rather than sampling; check how often a reconstruction refuses on real sessions. Two of the three DOM observations originally filed here no longer gate anything: snapshot frequency is handled by superseding the node map wholesale, so both cases take one path. Still worth capturing: whether real click breadcrumbs populate `payload.data.node.id` consistently, which decides whether rooting is a reliable entry point or best-effort, and how large a real `FullSnapshot` is. None of these block this change. + +## 10. DOM Tree Reads + +Promoted from Future Work and built in this change: the map and the zoom answer +what happened, but nothing answered what the page was, which is the question a +structural failure actually poses. + +- [x] 10.1 Add a streaming segment reader so a reconstruction never holds the raw recording alongside the derived state. Returning `"stop"` ends paging early and is reported as `truncatedBy: null`, since stopping by choice is not truncation. The buffering read is now a thin wrapper over it, with its existing tests untouched. +- [x] 10.2 Add `internal/replay-dom.ts`: fold `FullSnapshot` (`type: 2`) and `IncrementalSnapshot` (`type: 3`) into a flat node map. A later snapshot supersedes wholesale rather than merging, which makes checkout frequency irrelevant to correctness. Input values come from `source: 5`, not attribute mutations. Unknown-parent adds are dropped and counted, never reparented. Subtree deletion is iterative, since untrusted DOM depth should not reach the call stack. +- [x] 10.3 Render as an indented tree with `interactive` and `full` lenses and optional subtree rooting. No `visible` lens: visibility needs the cascade. Text comes from immediate text children only, since rrweb element nodes carry none. +- [x] 10.4 Report the rrweb node id at detail grain in `get_replay_activity`. It was parsed but never rendered, so nothing named the element a DOM read could root at — the handoff the design depends on did not exist. +- [x] 10.5 Add `get_replay_dom` as a catalog-only `inspect` tool with `requiredCapabilities: ["replays"]` and replay read scopes, reusing `internal/tool-helpers/replay.ts`. Required `atMs`; refusal on budget exhaustion rather than a partial tree; the recording ending early and an unparseable `started_at` are both reported rather than guessed past. Token budget unchanged at 5,598 across 9 direct tools. +- [x] 10.6 Add DOM events to the recording fixture: both pages snapshotted, plus a `source: 5`-only input change, an ignored scroll, an added error banner, and a text-plus-attribute mutation disabling the submit button after the console error. +- [x] 10.7 Test reconstruction, input-value precedence, target-time exclusion, rooting, lenses, truncation, budget refusal, and gating. Extend `availability.test.ts` to the third replay tool. +- [x] 10.8 Fix the render root. Verified against the `@sentry-internal/rrweb-snapshot` build Sentry ships: `serializeNodeWithId` assigns an id to every node including the Document, so a real snapshot roots at a `nodeType: 0` node and the renderer emitted nothing. The unit fixtures had modelled an id-less wrapper, which hid it — they now carry the real shape. Three rendering defects surfaced with them: `role` presence treated as interactivity, `isChecked` rendered for non-checkboxes, and empty-string boolean attributes rendered as `[disabled=]`. Each fix is pinned by a test that fails without it. +- [x] 10.9 Update `docs/specs/replay-review.md`: promote the DOM section from Future Work into Design, replace the proposed output with the shipped format, and record the divergences. From 878862ff842e75767700ad1fce021eb763d489ad Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 12:13:38 -0700 Subject: [PATCH 22/26] fix(replays): Prune deep DOM branches locally instead of aborting the walk MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found by pointing the tool at a real replay: the rendered tree stopped partway through `head` and never reached `body` at all. Two defects, both of which made a real page look far smaller than it was. Depth truncation aborted the whole traversal. `truncated` was a single flag and the walk returned as soon as it was set, so the first branch to exceed the depth limit hid every later sibling — and on a live page `head` is routinely deep enough to swallow `body` outright. Depth pruning is now local: the offending branch is dropped and counted, and its siblings still render. The node budget stays global, since once it is spent there is nothing left to spend. The default `maxDepth` of 12 was also too shallow to be useful. The real page measured 17 levels of elements with its interactive nodes below level 12, so the default clipped exactly what the interactive lens exists to surface. Raise it to 40 and document depth as a poor proxy for output size — component frameworks nest wrappers freely, so `maxNodes` is the real budget. Report the two limits separately. They call for different fixes, and the old single line named `maxNodes` even when depth was the cause, which sent a reader at the wrong dial. Each message now also names the way out, including that every rendered line carries the node id `rootNodeId` accepts. The first test written for this passed with the bug reintroduced: the fixture was not deep enough for one branch to hide another. Mutation testing caught that, and the replacement fails without the fix. Co-Authored-By: Claude Opus 5 --- .../mcp-core/src/internal/replay-dom.test.ts | 28 +++++++++ packages/mcp-core/src/internal/replay-dom.ts | 63 ++++++++++++++++--- packages/mcp-core/src/toolDefinitions.json | 8 ++- .../src/tools/catalog/get-replay-dom.test.ts | 34 +++++++++- .../src/tools/catalog/get-replay-dom.ts | 30 +++++++-- 5 files changed, 145 insertions(+), 18 deletions(-) diff --git a/packages/mcp-core/src/internal/replay-dom.test.ts b/packages/mcp-core/src/internal/replay-dom.test.ts index ffa355307..6ffab5b9a 100644 --- a/packages/mcp-core/src/internal/replay-dom.test.ts +++ b/packages/mcp-core/src/internal/replay-dom.test.ts @@ -467,6 +467,34 @@ describe("rendering", () => { ]); } + it("prunes a deep branch locally, leaving later siblings intact", () => { + // The real failure this reproduces: on a live page `head` is deep enough + // that aborting the walk at the depth limit hid `body` and everything under + // it. Depth pruning must drop only the offending branch. + const deepThenShallow = snapshot(0, [ + element(10, "div", { id: "deep" }, [ + element(11, "div", {}, [ + element(12, "div", {}, [element(13, "button", { id: "buried" })]), + ]), + ]), + element(20, "button", { id: "later-sibling" }), + ]); + + const rendered = renderDomTree(reconstruct([deepThenShallow]), { + lens: "full", + maxDepth: 3, + }); + const output = rendered.lines.join("\n"); + + // `div#deep` sits at the limit; its children are pruned. + expect(output).toContain("div#deep"); + expect(output).not.toContain("buried"); + // The sibling that comes after the pruned branch must still render. + expect(output).toContain("button#later-sibling"); + expect(rendered.depthLimitedSubtrees).toBeGreaterThan(0); + expect(rendered.nodeLimitReached).toBe(false); + }); + it("renders from the document root by descending to the first element", () => { // The regression this guards: rrweb roots a snapshot at the Document node, // which is not an element, and a renderer that walks from it directly emits diff --git a/packages/mcp-core/src/internal/replay-dom.ts b/packages/mcp-core/src/internal/replay-dom.ts index a6d6604a8..2c09dc528 100644 --- a/packages/mcp-core/src/internal/replay-dom.ts +++ b/packages/mcp-core/src/internal/replay-dom.ts @@ -655,6 +655,20 @@ const RENDERED_ATTRIBUTES = [ const MAX_TEXT_LENGTH = 80; +/** + * Depth allowed before a branch is pruned. + * + * Deliberately generous. Measured against a real Sentry replay, whose element + * tree is 17 levels deep with its interactive elements below level 12 — a + * shallower cap clipped exactly the nodes the interactive lens exists to show. + * Component frameworks nest wrapper elements freely, so depth is a poor proxy + * for output size; `maxNodes` is the real budget. + */ +const DEFAULT_MAX_DEPTH = 40; + +/** Elements rendered before the walk stops and says so. */ +const DEFAULT_MAX_NODES = 200; + export interface RenderTreeOptions { lens?: DomLens; /** Render only this node and its descendants. */ @@ -667,8 +681,22 @@ export interface RenderedTree { lines: string[]; /** Nodes rendered, after lens filtering and limits. */ nodesRendered: number; - /** True when `maxNodes` or `maxDepth` cut the output short. */ + /** + * True when either limit cut the output short. + * + * Kept as a single flag for callers that only need "is this the whole thing", + * but the two limits below say which, because they call for different fixes. + */ truncated: boolean; + /** + * Subtrees omitted for exceeding `maxDepth`. + * + * Depth pruning is local: the deep branch is dropped and its siblings still + * render, so a single deeply-nested branch cannot hide the rest of the page. + */ + depthLimitedSubtrees: number; + /** True when `maxNodes` was reached and the walk stopped. */ + nodeLimitReached: boolean; /** Set when `rootNodeId` named a node that is not in the reconstruction. */ rootNotFound: boolean; } @@ -689,8 +717,8 @@ export function renderDomTree( const { lens = "interactive", rootNodeId, - maxDepth = 12, - maxNodes = 200, + maxDepth = DEFAULT_MAX_DEPTH, + maxNodes = DEFAULT_MAX_NODES, } = options; const { nodes } = reconstruction; @@ -701,6 +729,8 @@ export function renderDomTree( lines: [], nodesRendered: 0, truncated: false, + depthLimitedSubtrees: 0, + nodeLimitReached: false, rootNotFound: true, }; } @@ -710,6 +740,8 @@ export function renderDomTree( lines: [], nodesRendered: 0, truncated: false, + depthLimitedSubtrees: 0, + nodeLimitReached: false, rootNotFound: false, }; } @@ -725,6 +757,8 @@ export function renderDomTree( lines: [], nodesRendered: 0, truncated: false, + depthLimitedSubtrees: 0, + nodeLimitReached: false, rootNotFound: false, }; } @@ -737,10 +771,12 @@ export function renderDomTree( const lines: string[] = []; let nodesRendered = 0; - let truncated = false; + let nodeLimitReached = false; + let depthLimitedSubtrees = 0; const walk = (id: number, depth: number, prefix: string, isLast: boolean) => { - if (truncated) { + // The node budget is global: once spent, nothing further can render. + if (nodeLimitReached) { return; } const node = nodes.get(id); @@ -751,11 +787,15 @@ export function renderDomTree( return; } if (nodesRendered >= maxNodes) { - truncated = true; + nodeLimitReached = true; return; } + // The depth budget is local: prune this branch and count it, but keep + // walking its siblings. Aborting the traversal here would let one deeply + // nested branch hide the entire rest of the page — `head` is often deep + // enough to swallow `body`. if (depth > maxDepth) { - truncated = true; + depthLimitedSubtrees += 1; return; } @@ -779,7 +819,14 @@ export function renderDomTree( walk(renderRootId, 0, "", true); - return { lines, nodesRendered, truncated, rootNotFound: false }; + return { + lines, + nodesRendered, + truncated: nodeLimitReached || depthLimitedSubtrees > 0, + depthLimitedSubtrees, + nodeLimitReached, + rootNotFound: false, + }; } /** diff --git a/packages/mcp-core/src/toolDefinitions.json b/packages/mcp-core/src/toolDefinitions.json index 3c623af9f..6661bfcc8 100644 --- a/packages/mcp-core/src/toolDefinitions.json +++ b/packages/mcp-core/src/toolDefinitions.json @@ -3649,16 +3649,18 @@ "enum": ["interactive", "full"] }, "maxDepth": { - "default": 12, + "default": 40, + "description": "How deep to descend. Branches below this are pruned and counted, not silently dropped; their siblings still render. Real pages nest deeply, so lower this only to skim.", "type": "number", "minimum": 1, - "maximum": 50 + "maximum": 200 }, "maxNodes": { "default": 200, + "description": "How many elements to render before stopping. This is the real budget on output size; raise it, or pass `rootNodeId`, to see more.", "type": "number", "minimum": 1, - "maximum": 1000 + "maximum": 2000 } }, "required": ["atMs"] diff --git a/packages/mcp-core/src/tools/catalog/get-replay-dom.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-dom.test.ts index a8fdfbb0e..474f3a5ba 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-dom.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-dom.test.ts @@ -172,15 +172,43 @@ describe("get_replay_dom", () => { expect(full).toContain("div#order-error"); }); - it("reports truncation rather than silently cutting the tree", async () => { + it("reports hitting the node budget, and how to see more", async () => { const result = await callTool({ atMs: AT_CHECKOUT_ERROR, lens: "full", maxNodes: 4, }); - expect(result).toContain("Tree truncated at 4 nodes"); - expect(result).toContain("Raise `maxNodes`"); + expect(result).toContain("Stopped after 4 elements (`maxNodes`)"); + expect(result).toContain("raise `maxNodes`"); + expect(result).toContain("`rootNodeId`"); + }); + + it("prunes a deep branch without dropping its siblings", async () => { + // The bug this guards: depth truncation used to abort the whole walk, so + // one deep branch hid every later sibling — on a real page `head` is + // routinely deep enough to swallow `body` entirely. + const result = await callTool({ + atMs: AT_CHECKOUT_ERROR, + lens: "full", + maxDepth: 2, + }); + + // `head` and `body` are both at depth 1, so both must appear even though + // everything below them is pruned. + expect(result).toContain("head"); + expect(result).toContain("body"); + expect(result).toContain("deeper than `maxDepth`"); + expect(result).toContain("not shown"); + }); + + it("renders a deeply nested real-world page at the default depth", async () => { + // The default was 12, which clipped a real Sentry page whose element tree + // is 17 levels deep — hiding exactly the interactive elements the lens + // exists to surface. + const result = await callTool({ atMs: AT_CHECKOUT_ERROR }); + + expect(result).not.toContain("deeper than `maxDepth`"); }); }); diff --git a/packages/mcp-core/src/tools/catalog/get-replay-dom.ts b/packages/mcp-core/src/tools/catalog/get-replay-dom.ts index 8dfebec4d..14974793d 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-dom.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-dom.ts @@ -82,8 +82,22 @@ export default defineTool({ .describe( "`interactive` keeps elements a user can act on plus the ancestors that place them; `full` keeps every element.", ), - maxDepth: z.number().min(1).max(50).default(12), - maxNodes: z.number().min(1).max(1000).default(200), + maxDepth: z + .number() + .min(1) + .max(200) + .default(40) + .describe( + "How deep to descend. Branches below this are pruned and counted, not silently dropped; their siblings still render. Real pages nest deeply, so lower this only to skim.", + ), + maxNodes: z + .number() + .min(1) + .max(2000) + .default(200) + .describe( + "How many elements to render before stopping. This is the real budget on output size; raise it, or pass `rootNodeId`, to see more.", + ), }, annotations: { readOnlyHint: true, @@ -316,10 +330,18 @@ function formatDomOutput({ lines.push(...tree.lines); lines.push("```"); - if (tree.truncated) { + // The two limits call for different fixes, so they are reported separately + // rather than as one "truncated" line the reader has to guess at. + if (tree.nodeLimitReached) { + lines.push(""); + lines.push( + `Stopped after ${tree.nodesRendered} elements (\`maxNodes\`). To see more: raise \`maxNodes\`, or pass \`rootNodeId\` to render one subtree in full — every line above carries the id to use.`, + ); + } + if (tree.depthLimitedSubtrees > 0) { lines.push(""); lines.push( - `Tree truncated at ${tree.nodesRendered} nodes. Raise \`maxNodes\`/\`maxDepth\`, or pass a \`rootNodeId\` to narrow the read.`, + `${tree.depthLimitedSubtrees} subtree${tree.depthLimitedSubtrees === 1 ? "" : "s"} ${tree.depthLimitedSubtrees === 1 ? "was" : "were"} deeper than \`maxDepth\` and ${tree.depthLimitedSubtrees === 1 ? "is" : "are"} not shown. Raise \`maxDepth\`, or root at the deepest node shown to continue from there.`, ); } From 6a8e1ea4073689457161e10b3c205a8941499d90 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 13:11:36 -0700 Subject: [PATCH 23/26] feat(replays): Steer replay investigation toward structure over metrics MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Agents investigating a replay over-rotate onto web vitals and console warnings instead of the DOM. Three causes, all in what the output emphasizes rather than in the model. The chain dead-ended. `get_replay_details` prints an explicit `get_replay_activity(...)` call to run next, but activity pointed nowhere, so the structural read was reachable only by knowing it existed and searching the catalog — while console and vital signals sat in the output already in hand. Activity now prints a rooted `get_replay_dom(...)` call, so the three steps read end to end. The suggestion is conditional. It appears for a rage click, a dead click, or a hydration error — cases where the user acted and the page did not answer, and the explanation is the element rather than a response. It stays silent after an ordinary click or a failed request, because a suggestion on every response is one nobody reads, and silent when the DOM tool is absent from the session, because naming an unreachable tool sends the reader after nothing. A web vital that met its threshold no longer renders as a signal. `good`-rated LCP beside a 500 and a rage click invites investigation of a metric that is already fine; kind counts still include it. The rrweb node id is now carried on the signal rather than only in its rendered detail line, so the handoff does not depend on parsing prose it also produces. Descriptions say structural questions exist and which tool answers them, and that vitals and warnings are context rather than cause. Verified against a real replay: of 16 signals, 8 carry a node id, and this recording has no rage clicks, dead clicks, or hydration errors — so the suggestion correctly stays silent. That means prod has not yet exercised the trigger; the gating is covered by tests, including one that fails if the availability guard is removed. Co-Authored-By: Claude Opus 5 --- .../src/internal/replay-events.test.ts | 36 ++++++ .../mcp-core/src/internal/replay-events.ts | 25 +++- packages/mcp-core/src/skillDefinitions.json | 4 +- packages/mcp-core/src/toolDefinitions.json | 4 +- .../tools/catalog/get-replay-activity.test.ts | 45 ++++++- .../src/tools/catalog/get-replay-activity.ts | 120 ++++++++++++++++++ .../src/tools/catalog/get-replay-dom.ts | 5 + 7 files changed, 231 insertions(+), 8 deletions(-) diff --git a/packages/mcp-core/src/internal/replay-events.test.ts b/packages/mcp-core/src/internal/replay-events.test.ts index 1d0f78198..bee722d29 100644 --- a/packages/mcp-core/src/internal/replay-events.test.ts +++ b/packages/mcp-core/src/internal/replay-events.test.ts @@ -126,6 +126,42 @@ describe("classifyReplayEvent", () => { } }); + it("renders a web vital only when it missed its threshold", () => { + // A paint that met its threshold is not a finding. Rendering it beside a + // failed request invites investigation of a metric that is already fine, + // which is the behaviour this drops. + const good = signalsFrom([ + span("web-vital", { + description: "largest-contentful-paint", + data: { size: 1200, rating: "good" }, + }), + ]); + expect(good).toEqual([]); + + const poor = signalsFrom([ + span("web-vital", { + description: "largest-contentful-paint", + data: { size: 4800, rating: "poor" }, + }), + ]); + expect(poor).toHaveLength(1); + expect(poor[0].summary).toBe("Largest contentful paint: 4800ms (poor)"); + }); + + it("carries the rrweb node id as data, not only as rendered text", () => { + // The handoff to a structural read depends on this being a field. Parsing + // it back out of the detail line would break the first time that line is + // reworded. + const signals = signalsFrom([ + breadcrumb("ui.click", { + message: "button#pay", + data: { node: { id: 4242, tagName: "button" } }, + }), + ]); + + expect(signals[0].nodeId).toBe(4242); + }); + it("splits web vitals by description", () => { expect( classifyReplayEvent( diff --git a/packages/mcp-core/src/internal/replay-events.ts b/packages/mcp-core/src/internal/replay-events.ts index 012053f75..200bf0aa7 100644 --- a/packages/mcp-core/src/internal/replay-events.ts +++ b/packages/mcp-core/src/internal/replay-events.ts @@ -118,6 +118,13 @@ export interface ReplaySignal { details: string[]; /** True when this signal indicates a failure (error log, failed request). */ isError: boolean; + /** + * rrweb node id of the element this signal is about, when it names one. + * + * Kept as data rather than only as a rendered detail line so a caller can + * hand it to a structural read without parsing it back out of prose. + */ + nodeId?: number; } /** Rendered when the SDK never captured a value. */ @@ -579,7 +586,10 @@ function isErrorEvent( return type === "hydration-error"; } -type SummarizedEvent = Pick; +type SummarizedEvent = Pick< + ReplaySignal, + "summary" | "details" | "isError" | "nodeId" +>; /** * Describe an event, or return null when it should not be rendered. @@ -657,6 +667,13 @@ function summarizeEvent( case "lcp": { const size = data?.size; const rating = data?.rating; + // A paint that met its threshold is not an event worth a line beside a + // failed request or a rage click. Rendering it as a peer invites an agent + // to investigate a metric that is already fine; the count remains + // available through `countReplayKinds`. + if (rating === "good") { + return null; + } return { summary: size != null && rating != null @@ -768,8 +785,9 @@ function describeClick( // because it is otherwise unreachable: it is stable within a recording, but // nothing else in this output names it, so "show me the DOM around what was // clicked" would have no way to say which element. - if (typeof data?.node?.id === "number") { - details.push(`nodeId: ${data.node.id}`); + const nodeId = typeof data?.node?.id === "number" ? data.node.id : undefined; + if (nodeId !== undefined) { + details.push(`nodeId: ${nodeId}`); } return { @@ -777,6 +795,7 @@ function describeClick( details, // A dead or rage click is a failure of the page, not of the request. isError: verb !== "Clicked", + nodeId, }; } diff --git a/packages/mcp-core/src/skillDefinitions.json b/packages/mcp-core/src/skillDefinitions.json index 1c8d1c360..74dfda3f2 100644 --- a/packages/mcp-core/src/skillDefinitions.json +++ b/packages/mcp-core/src/skillDefinitions.json @@ -129,7 +129,7 @@ }, { "name": "get_replay_activity", - "description": "Read what happened during a window of a Sentry replay session.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what happened around a specific moment in a replay\n- Need the requests, clicks, or console output behind a replay failure\n- Want more or less detail than the replay map provides\n\nCall `get_replay_details` first for the session map; it suggests a window.\nOffsets are milliseconds from the start of the replay.\n\n\n### Zoom into a failure\n```\nget_replay_activity(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', startMs=306000, endMs=316000, grain='detail')\n```\n\n### Cheap shape check on a long session\n```\nget_replay_activity(replayUrl='https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/', grain='digest', kinds=['network','console'])\n```\n", + "description": "Read what happened during a window of a Sentry replay session.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what happened around a specific moment in a replay\n- Need the requests, clicks, or console output behind a replay failure\n- Want more or less detail than the replay map provides\n\nCall `get_replay_details` first for the session map; it suggests a window.\nOffsets are milliseconds from the start of the replay.\n\nThis answers what *happened*, not what the page *was*. When a click went\nunanswered or an element looks wrong, `get_replay_dom` reads the structure\nat that moment — this tool reports the `nodeId` to root it at.\nWeb vitals and console warnings are context, rarely the cause; prefer the\nfailed request, the unanswered click, or the DOM around it.\n\n\n### Zoom into a failure\n```\nget_replay_activity(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', startMs=306000, endMs=316000, grain='detail')\n```\n\n### Cheap shape check on a long session\n```\nget_replay_activity(replayUrl='https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/', grain='digest', kinds=['network','console'])\n```\n", "requiredScopes": ["org:read", "project:read", "event:read"] }, { @@ -139,7 +139,7 @@ }, { "name": "get_replay_dom", - "description": "Read the page structure of a Sentry replay at one moment in time.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what the page looked like when something failed\n- Need to know whether an element existed, was disabled, or was empty\n- Want the DOM around an element that was clicked or rage-clicked\n\nReturns structure only — what the page *was*, not what the user did.\nUse `get_replay_activity` for that, and to find the `nodeId` to root at.\n`atMs` is milliseconds from the start of the replay and is required.\n\nText and form values are masked by the SDK before upload, so this answers\nstructural questions, not what a user typed.\n\n\n### What the page looked like when the error fired\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)\n```\n\n### The subtree around a rage-clicked element\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')\n```\n", + "description": "Read the page structure of a Sentry replay at one moment in time.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what the page looked like when something failed\n- Need to know whether an element existed, was disabled, or was empty\n- Want the DOM around an element that was clicked or rage-clicked\n\nReturns structure only — what the page *was*, not what the user did.\nUse `get_replay_activity` for that, and to find the `nodeId` to root at.\n\nReach for this when a rage or dead click, a hydration error, or a missing\nelement needs explaining: those are page-state questions, and no signal\nlist can answer them. It is the DOM, not a slow paint or a console warning,\nthat explains why a click did nothing.\n`atMs` is milliseconds from the start of the replay and is required.\n\nText and form values are masked by the SDK before upload, so this answers\nstructural questions, not what a user typed.\n\n\n### What the page looked like when the error fired\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)\n```\n\n### The subtree around a rage-clicked element\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')\n```\n", "requiredScopes": ["org:read", "project:read", "event:read"] }, { diff --git a/packages/mcp-core/src/toolDefinitions.json b/packages/mcp-core/src/toolDefinitions.json index 6661bfcc8..eea07257b 100644 --- a/packages/mcp-core/src/toolDefinitions.json +++ b/packages/mcp-core/src/toolDefinitions.json @@ -3483,7 +3483,7 @@ }, { "name": "get_replay_activity", - "description": "Read what happened during a window of a Sentry replay session.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what happened around a specific moment in a replay\n- Need the requests, clicks, or console output behind a replay failure\n- Want more or less detail than the replay map provides\n\nCall `get_replay_details` first for the session map; it suggests a window.\nOffsets are milliseconds from the start of the replay.\n\n\n### Zoom into a failure\n```\nget_replay_activity(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', startMs=306000, endMs=316000, grain='detail')\n```\n\n### Cheap shape check on a long session\n```\nget_replay_activity(replayUrl='https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/', grain='digest', kinds=['network','console'])\n```\n", + "description": "Read what happened during a window of a Sentry replay session.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what happened around a specific moment in a replay\n- Need the requests, clicks, or console output behind a replay failure\n- Want more or less detail than the replay map provides\n\nCall `get_replay_details` first for the session map; it suggests a window.\nOffsets are milliseconds from the start of the replay.\n\nThis answers what *happened*, not what the page *was*. When a click went\nunanswered or an element looks wrong, `get_replay_dom` reads the structure\nat that moment — this tool reports the `nodeId` to root it at.\nWeb vitals and console warnings are context, rarely the cause; prefer the\nfailed request, the unanswered click, or the DOM around it.\n\n\n### Zoom into a failure\n```\nget_replay_activity(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', startMs=306000, endMs=316000, grain='detail')\n```\n\n### Cheap shape check on a long session\n```\nget_replay_activity(replayUrl='https://my-organization.sentry.io/explore/replays/7e07485f-12f9-416b-8b14-26260799b51f/', grain='digest', kinds=['network','console'])\n```\n", "inputSchema": { "type": "object", "properties": { @@ -3605,7 +3605,7 @@ }, { "name": "get_replay_dom", - "description": "Read the page structure of a Sentry replay at one moment in time.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what the page looked like when something failed\n- Need to know whether an element existed, was disabled, or was empty\n- Want the DOM around an element that was clicked or rage-clicked\n\nReturns structure only — what the page *was*, not what the user did.\nUse `get_replay_activity` for that, and to find the `nodeId` to root at.\n`atMs` is milliseconds from the start of the replay and is required.\n\nText and form values are masked by the SDK before upload, so this answers\nstructural questions, not what a user typed.\n\n\n### What the page looked like when the error fired\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)\n```\n\n### The subtree around a rage-clicked element\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')\n```\n", + "description": "Read the page structure of a Sentry replay at one moment in time.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what the page looked like when something failed\n- Need to know whether an element existed, was disabled, or was empty\n- Want the DOM around an element that was clicked or rage-clicked\n\nReturns structure only — what the page *was*, not what the user did.\nUse `get_replay_activity` for that, and to find the `nodeId` to root at.\n\nReach for this when a rage or dead click, a hydration error, or a missing\nelement needs explaining: those are page-state questions, and no signal\nlist can answer them. It is the DOM, not a slow paint or a console warning,\nthat explains why a click did nothing.\n`atMs` is milliseconds from the start of the replay and is required.\n\nText and form values are masked by the SDK before upload, so this answers\nstructural questions, not what a user typed.\n\n\n### What the page looked like when the error fired\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)\n```\n\n### The subtree around a rage-clicked element\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')\n```\n", "inputSchema": { "type": "object", "properties": { diff --git a/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts index f688c00d2..d11225335 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts @@ -53,7 +53,10 @@ describe("get_replay_activity", () => { T+3m 1.0s network Fetch POST example.com/api/checkout failed with 500 T+3m 1.3s console Console error: TypeError: Cannot read properties of undefined (reading 'id') T+3m 8.4s rage-click Rage click on body > div#root > main > button#complete-order - T+3m 41.7s dead-click Dead click — no response from body > div#root > main > a#download-receipt" + T+3m 41.7s dead-click Dead click — no response from body > div#root > main > a#download-receipt + + A click the page did not answer is usually explained by the element itself. Use the Sentry tool \`get_replay_dom\` to see the page structure at that moment: + get_replay_dom(organizationSlug='sentry-mcp-evals', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=188400, rootNodeId=96)" `); }); @@ -315,6 +318,46 @@ describe("get_replay_activity", () => { }); }); + describe("structural read handoff", () => { + it("prints a rooted get_replay_dom call after an unanswered click", async () => { + // Without a printed call the third step of the chain is only reachable by + // knowing it exists and searching the catalog. The map hands this tool its + // next call the same way. + const result = await callTool(); + + expect(result).toContain( + "get_replay_dom(organizationSlug='sentry-mcp-evals'", + ); + // Rooted at the rage-clicked button, at the moment of the rage click. + expect(result).toContain("atMs=188400"); + expect(result).toContain("rootNodeId=96"); + }); + + it("does not suggest a structural read when nothing structural happened", async () => { + // A suggestion on every response is one nobody reads. A failed request is + // explained by its response, not by the DOM. + const result = await callTool({ kinds: ["network", "console"] }); + + expect(result).not.toContain("get_replay_dom"); + }); + + it("stays silent when the DOM tool is not in this session", async () => { + // Naming an unreachable tool sends the reader after something that cannot + // be called. + const result = await callTool( + {}, + getServerContext({ + availableToolNames: new Set([ + "get_replay_activity", + "get_replay_details", + ]), + }), + ); + + expect(result).not.toContain("get_replay_dom"); + }); + }); + describe("tool definition", () => { it("is gated like get_replay_details", async () => { expect(getReplayActivity.requiredScopes).toEqual([ diff --git a/packages/mcp-core/src/tools/catalog/get-replay-activity.ts b/packages/mcp-core/src/tools/catalog/get-replay-activity.ts index de02c8e71..83dbc8395 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-activity.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-activity.ts @@ -18,6 +18,10 @@ import { resolveReplayParams, } from "../../internal/tool-helpers/replay"; import { resolveRegionUrlForOrganization } from "../../internal/tool-helpers/resolve-region-url"; +import { + formatToolCall, + formatToolCallInstruction, +} from "../../internal/tool-helpers/tool-call-formatting"; import { UserInputError } from "../../errors"; import type { ServerContext } from "../../types"; import { z } from "zod"; @@ -62,6 +66,12 @@ export default defineTool({ "Call `get_replay_details` first for the session map; it suggests a window.", "Offsets are milliseconds from the start of the replay.", "", + "This answers what *happened*, not what the page *was*. When a click went", + "unanswered or an element looks wrong, `get_replay_dom` reads the structure", + "at that moment — this tool reports the `nodeId` to root it at.", + "Web vitals and console warnings are context, rarely the cause; prefer the", + "failed request, the unanswered click, or the DOM around it.", + "", "", "### Zoom into a failure", "```", @@ -204,10 +214,49 @@ export default defineTool({ endMs, kinds, truncatedBy: recording.truncatedBy, + organizationSlug: resolved.organizationSlug, + context, }); }, }); +/** + * Signals that raise a question about the page, not about a request. + * + * A rage or dead click means the user acted and the page did not respond, and a + * hydration error means the DOM the server sent and the one the client built + * disagreed. In all three the useful next question is what the element looked + * like, which the signal list cannot answer. Ordinary clicks and network + * failures are deliberately excluded: a 500 is explained by the response, and + * suggesting a structural read after every click would train a reader to ignore + * the suggestion. + */ +const STRUCTURAL_SIGNAL_TYPES = new Set([ + "rage-click", + "dead-click", + "hydration-error", +]); + +/** + * Pick the signal whose structure is worth looking at. + * + * Prefers one that names a node, since rooting a read at the element beats + * rendering the whole page. Falls back to the first structural signal so the + * offset is still offered when no id came through — real recordings populate + * `node.id` on most but not all click breadcrumbs. + */ +function findStructuralSignal(signals: ReplaySignal[]): ReplaySignal | null { + const candidates = signals.filter( + (signal) => + STRUCTURAL_SIGNAL_TYPES.has(signal.type) && signal.offsetMs !== null, + ); + return ( + candidates.find((signal) => signal.nodeId !== undefined) ?? + candidates[0] ?? + null + ); +} + function matchesQuery( signal: ReplaySignal, { @@ -251,6 +300,8 @@ function formatActivityOutput({ endMs, kinds, truncatedBy, + organizationSlug, + context, }: { replayId: string; signals: ReplaySignal[]; @@ -261,6 +312,8 @@ function formatActivityOutput({ startMs?: number; endMs?: number; kinds?: string[]; + organizationSlug: string; + context: ServerContext; truncatedBy: ReplayRecordingSegmentsResult["truncatedBy"]; }): string { const lines: string[] = []; @@ -312,9 +365,76 @@ function formatActivityOutput({ ); } + lines.push( + ...suggestStructuralRead({ replayId, signals, organizationSlug, context }), + ); + return lines.join("\n"); } +/** + * Point at a structural read when a signal raises a question this tool cannot + * answer. + * + * The signal list explains what happened; it cannot say what the page was. When + * a click went unanswered or hydration disagreed, that second question is the + * one that matters, and without a printed call the reader has to know a third + * tool exists and go looking for it. The map already hands this tool its next + * call the same way, so the chain reads end to end. + * + * Deliberately conditional. A suggestion on every response is a suggestion + * nobody reads, so it appears only for signals whose explanation is structural. + */ +function suggestStructuralRead({ + replayId, + signals, + organizationSlug, + context, +}: { + replayId: string; + signals: ReplaySignal[]; + organizationSlug: string; + context: ServerContext; +}): string[] { + const signal = findStructuralSignal(signals); + if (!signal || signal.offsetMs === null) { + return []; + } + + const instruction = formatToolCallInstruction({ + toolName: "get_replay_dom", + experimentalMode: context.experimentalMode ?? false, + availableToolNames: context.availableToolNames, + directToolNames: context.directToolNames, + // Silence beats a dangling pointer: if the tool is not in this session, + // naming it would send the reader after something unreachable. + fallbackInstruction: "", + purpose: "to see the page structure at that moment", + }); + if (!instruction) { + return []; + } + + const reason = + signal.type === "hydration-error" + ? "A hydration error means the server and client DOM disagreed" + : "A click the page did not answer is usually explained by the element itself"; + + return [ + "", + `${reason}. ${instruction}:`, + formatToolCall({ + toolName: "get_replay_dom", + arguments: { + organizationSlug, + replayId, + atMs: signal.offsetMs, + ...(signal.nodeId !== undefined ? { rootNodeId: signal.nodeId } : {}), + }, + }), + ]; +} + function encodeCursor(cursor: ActivityCursor): string { return Buffer.from(JSON.stringify(cursor), "utf8").toString("base64url"); } diff --git a/packages/mcp-core/src/tools/catalog/get-replay-dom.ts b/packages/mcp-core/src/tools/catalog/get-replay-dom.ts index 14974793d..403c40e7c 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-dom.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-dom.ts @@ -42,6 +42,11 @@ export default defineTool({ "", "Returns structure only — what the page *was*, not what the user did.", "Use `get_replay_activity` for that, and to find the `nodeId` to root at.", + "", + "Reach for this when a rage or dead click, a hydration error, or a missing", + "element needs explaining: those are page-state questions, and no signal", + "list can answer them. It is the DOM, not a slow paint or a console warning,", + "that explains why a click did nothing.", "`atMs` is milliseconds from the start of the replay and is required.", "", "Text and form values are masked by the SDK before upload, so this answers", From caaddb4d411fd4ebbad64d98739a7bea39d7365b Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 13:54:19 -0700 Subject: [PATCH 24/26] fix(replays): Report network duration, partial rollups, and structural reads MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three findings from manual use. Network requests never reported a duration. The code read `data.duration`, which the SDK does not set — `NetworkRequestData` in `@sentry-internal/replay` has no such field, and real spans carry timing in `startTimestamp`/`endTimestamp` instead. The line was dead in every recording. Derive the elapsed time from the span's own bounds and put it in the summary rather than the detail lines: whether a failure was rejected or timed out is the first question asked of it, so it belongs on the line every grain shows. An absent bound reports nothing rather than 0ms, which would claim the request was instant. A truncated digest was actively misleading. `navigation x2 / click x1` reads as a whole-session rollup while the rage click, dead click, failed request and console error sit on later pages — the counts look authoritative and describe a healthier session than the real one. Say so above the numbers, since the numbers are what gets believed. Continuation is now a callable `get_replay_activity(...)` rather than a bare `cursor='...'` the reader has to rebuild a call around, and it names raising `limit` or windowing as the cheaper alternatives to paging. The structural read went unused on client-side state questions. The handoff only fired for a rage click, dead click, or hydration error, but a message that flashed or a control that never rendered produces no such signal — and an agent with no pointer goes looking for the explanation in application source instead of reconstructing the page. State the capability whenever there is a timeline to aim into, anchored on a failure rather than the first signal, since a leading navigation is a useless moment to reconstruct. A specific signal still earns a specific rooted call; this is the fallback, not a replacement. Two tests here passed against shapes that do not occur: one asserted the nonexistent `data.duration`, and one could not tell a failure-anchored offset from a first-signal one. Both were caught by mutation testing and now fail without the behaviour they cover. Co-Authored-By: Claude Opus 5 --- .../src/internal/replay-events.test.ts | 49 +++++++++- .../mcp-core/src/internal/replay-events.ts | 46 ++++++++- .../tools/catalog/get-replay-activity.test.ts | 79 +++++++++++++++- .../src/tools/catalog/get-replay-activity.ts | 94 ++++++++++++++----- 4 files changed, 235 insertions(+), 33 deletions(-) diff --git a/packages/mcp-core/src/internal/replay-events.test.ts b/packages/mcp-core/src/internal/replay-events.test.ts index bee722d29..ab5f470a9 100644 --- a/packages/mcp-core/src/internal/replay-events.test.ts +++ b/packages/mcp-core/src/internal/replay-events.test.ts @@ -542,7 +542,12 @@ describe("rendering grains", () => { "resource.fetch", { description: "https://example.com/api/checkout", - data: { method: "POST", statusCode: 500, duration: 1240 }, + data: { method: "POST", statusCode: 500 }, + // Span frames carry timing here, in seconds. There is no `duration` + // field on `NetworkRequestData`; asserting one meant asserting a shape + // that never arrives. + startTimestamp: (SESSION_START_MS + 181_000) / 1000, + endTimestamp: (SESSION_START_MS + 182_240) / 1000, }, (SESSION_START_MS + 181_000) / 1000, ), @@ -564,17 +569,53 @@ describe("rendering grains", () => { it("renders one line per signal at standard grain", () => { expect(renderReplaySignals(signalsFrom(events), "standard")).toEqual([ "T+3m 0.6s click Clicked button#complete-order", - "T+3m 1.0s network Fetch POST example.com/api/checkout failed with 500", + "T+3m 1.0s network Fetch POST example.com/api/checkout failed with 500 in 1.2s", "T+3m 1.3s console Console error: TypeError: Cannot read 'id' of undefined", ]); }); it("adds payload lines at detail grain", () => { const lines = renderReplaySignals(signalsFrom(events), "detail"); - expect(lines).toContain(" duration: 1240ms"); expect(lines).toContain(` request body: ${NOT_CAPTURED}`); }); + it("reports how long a failing request took, in the summary", () => { + // Duration answers "rejected or timed out", which is the first question + // asked of a failure, so it belongs on the line every grain shows rather + // than in detail-only payload. + const [signal] = signalsFrom([ + span( + "resource.fetch", + { + description: "https://example.com/api/slow", + data: { method: "GET", statusCode: 504 }, + startTimestamp: SESSION_START_MS / 1000, + endTimestamp: SESSION_START_MS / 1000 + 30, + }, + SESSION_START_MS / 1000, + ), + ]); + + expect(signal.summary).toBe( + "Fetch GET example.com/api/slow failed with 504 in 30.0s", + ); + }); + + it("omits the duration when the span carries no bounds", () => { + // Reporting an unknown duration as 0ms would assert the request was + // instant, which is a stronger claim than saying nothing. + const [signal] = signalsFrom([ + span("resource.fetch", { + description: "https://example.com/api/unknown", + data: { method: "GET", statusCode: 500 }, + }), + ]); + + expect(signal.summary).toBe( + "Fetch GET example.com/api/unknown failed with 500", + ); + }); + it("merges repeated signals at standard grain", () => { const repeated = Array.from({ length: 3 }, (_, index) => breadcrumb( @@ -609,7 +650,7 @@ describe("against the recorded fixture", () => { "Clicked body > div#root > form#login > button#sign-in", "Navigated to example.com/checkout", "Clicked body > div#root > main > button#complete-order", - "Fetch POST example.com/api/checkout failed with 500", + "Fetch POST example.com/api/checkout failed with 500 in 1.2s", "Console error: TypeError: Cannot read properties of undefined (reading 'id')", "Rage click on body > div#root > main > button#complete-order", "Dead click — no response from body > div#root > main > a#download-receipt", diff --git a/packages/mcp-core/src/internal/replay-events.ts b/packages/mcp-core/src/internal/replay-events.ts index 200bf0aa7..6ef3dfcad 100644 --- a/packages/mcp-core/src/internal/replay-events.ts +++ b/packages/mcp-core/src/internal/replay-events.ts @@ -823,10 +823,15 @@ function describeNetworkRequest( const request = method ? `${method} ${url}` : url; const status = statusCode != null ? String(statusCode) : "no response"; + // Duration belongs in the summary, not the detail lines: how long a failing + // request took is often the difference between a rejected call and a timeout, + // and it is the first thing asked about a slow page. It was previously read + // from `data.duration`, which the SDK never sets — `NetworkRequestData` has no + // such field — so the line silently never rendered. + const durationMs = spanDurationMs(payload); + const timing = durationMs !== null ? ` in ${formatDuration(durationMs)}` : ""; + const details: string[] = []; - if (typeof data?.duration === "number") { - details.push(`duration: ${data.duration}ms`); - } details.push( `request body: ${describeBody(data?.request, data?.requestBodySize)}`, ); @@ -835,12 +840,45 @@ function describeNetworkRequest( ); return { - summary: `${label} ${request} failed with ${status}`, + summary: `${label} ${request} failed with ${status}${timing}`, details, isError: true, }; } +/** + * Elapsed time of a span frame, in milliseconds. + * + * Span frames carry `startTimestamp`/`endTimestamp` in seconds — verified + * against `ReplayBaseSpanFrame` in `@sentry-internal/replay` — and no duration + * field of their own. Returns null rather than zero when either bound is + * missing, so an unknown duration is not reported as instant. + */ +function spanDurationMs( + payload: ReplayRecordingPayload | undefined, +): number | null { + const start = payload?.startTimestamp; + const end = payload?.endTimestamp; + if (typeof start !== "number" || typeof end !== "number") { + return null; + } + const elapsed = (end - start) * 1000; + return elapsed >= 0 ? elapsed : null; +} + +/** + * Render a millisecond duration at a precision that stays readable. + * + * Sub-second timings are the common case and matter to the millisecond; + * anything longer is about magnitude, not precision. + */ +function formatDuration(ms: number): string { + if (ms < 1000) { + return `${Math.round(ms)}ms`; + } + return `${(ms / 1000).toFixed(1)}s`; +} + /** * Describe a captured request or response body. * diff --git a/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts index d11225335..1b37ebdc9 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-activity.test.ts @@ -50,7 +50,7 @@ describe("get_replay_activity", () => { T+12.4s click Clicked body > div#root > form#login > button#sign-in T+14.2s navigation Navigated to example.com/checkout T+3m 0.6s click Clicked body > div#root > main > button#complete-order - T+3m 1.0s network Fetch POST example.com/api/checkout failed with 500 + T+3m 1.0s network Fetch POST example.com/api/checkout failed with 500 in 1.2s T+3m 1.3s console Console error: TypeError: Cannot read properties of undefined (reading 'id') T+3m 8.4s rage-click Rage click on body > div#root > main > button#complete-order T+3m 41.7s dead-click Dead click — no response from body > div#root > main > a#download-receipt @@ -220,6 +220,39 @@ describe("get_replay_activity", () => { expect(result).toContain("cursor='"); }); + it("prints a callable continuation, not a bare cursor", async () => { + // A bare `cursor='…'` leaves the reader to rebuild the call around it, + // including the org and replay it belongs to. + const result = await callTool({ limit: 3 }); + + expect(result).toContain( + "get_replay_activity(organizationSlug='sentry-mcp-evals'", + ); + expect(result).toContain("(`limit` was 3)"); + expect(result).toContain("Or raise `limit`"); + }); + + it("warns above the numbers when a digest rollup is partial", async () => { + // The failure mode this addresses: a truncated digest reads as a complete + // session rollup. Here the rage click, dead click, network failure and + // console error all fall on later pages, so the counts shown describe a + // healthier session than the real one. + const result = await callTool({ grain: "digest", limit: 3 }); + + expect(result).toContain("**Partial rollup.**"); + expect(result).toContain("only signals 1–3 of 8"); + // The warning must precede the counts it qualifies. + expect(result.indexOf("Partial rollup")).toBeLessThan( + result.indexOf("navigation ×"), + ); + }); + + it("does not warn when a digest covers everything that matched", async () => { + const result = await callTool({ grain: "digest" }); + + expect(result).not.toContain("Partial rollup"); + }); + it("continues from the cursor without repeating or skipping signals", async () => { const first = await callTool({ limit: 3 }); const second = await callTool({ cursor: cursorFrom(first), limit: 3 }); @@ -333,11 +366,49 @@ describe("get_replay_activity", () => { expect(result).toContain("rootNodeId=96"); }); - it("does not suggest a structural read when nothing structural happened", async () => { - // A suggestion on every response is one nobody reads. A failed request is - // explained by its response, not by the DOM. + it("still names the capability when no signal points at an element", async () => { + // Plenty of replay questions are about page state with no signal behind + // them — a message that flashed, a control that was missing. Without a + // pointer an agent goes looking for the explanation in application source + // instead, so the capability is stated even with no moment to aim at. const result = await callTool({ kinds: ["network", "console"] }); + expect(result).toContain("get_replay_dom"); + expect(result).toContain("what the page showed"); + // No element to root at, so none is invented. + expect(result).not.toContain("rootNodeId"); + // Anchored on the failure rather than the first signal in the page: a + // leading navigation explains nothing, and the example offset is the one + // a reader is most likely to reuse. + expect(result).toContain("atMs=181000"); + }); + + it("prefers a rooted call over the general pointer when both apply", async () => { + // A specific signal earns a specific call; the general note would be + // weaker advice in the same space. + const result = await callTool(); + + expect(result).toContain("rootNodeId=96"); + expect(result).not.toContain("what the page showed"); + }); + + it("anchors the example offset on a failure, not the first signal", async () => { + // With navigations and clicks in the page, the first signal is a + // navigation at T+0.5s — a useless moment to reconstruct. The failure is + // the offset a reader will actually reuse. + const result = await callTool({ + kinds: ["navigation", "click", "network"], + }); + + expect(result).toContain("atMs=181000"); + expect(result).not.toContain("atMs=500"); + }); + + it("says nothing when no signal has a resolvable offset", async () => { + // With no timeline there is no moment to pass, and a call the reader + // cannot complete is worse than no call. + const result = await callTool({ kinds: ["web-vital"] }); + expect(result).not.toContain("get_replay_dom"); }); diff --git a/packages/mcp-core/src/tools/catalog/get-replay-activity.ts b/packages/mcp-core/src/tools/catalog/get-replay-activity.ts index 83dbc8395..50638b5bf 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-activity.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-activity.ts @@ -329,6 +329,20 @@ function formatActivityOutput({ ); lines.push(""); + const nextOffset = offset + signals.length; + const isTruncated = nextOffset < matchedCount; + + // A digest is a rollup, and a rollup of one page reads exactly like a rollup + // of the session — the counts look authoritative while the failures that + // decide the answer sit on a later page. Say so above the numbers, not only + // below them, because the numbers are what gets believed. + if (isTruncated && grain === "digest" && signals.length > 0) { + lines.push( + `**Partial rollup.** These counts cover only signals ${offset + 1}\u2013${nextOffset} of ${matchedCount}, not the whole window. Raise \`limit\` for a complete rollup before drawing conclusions from them.`, + ); + lines.push(""); + } + if (signals.length === 0) { lines.push( matchedCount === 0 @@ -339,15 +353,27 @@ function formatActivityOutput({ lines.push(...renderReplaySignals(signals, grain)); } - // Truncation is always stated, and always with the means to continue. - const nextOffset = offset + signals.length; - if (nextOffset < matchedCount) { + // Truncation is always stated, and always with the means to continue — as a + // callable tool call rather than a bare cursor, so continuing does not require + // reconstructing the call around it. + if (isTruncated) { lines.push(""); lines.push( - `Showing ${offset + 1}–${nextOffset} of ${matchedCount} matching signals. Continue with:`, + `Showing ${offset + 1}\u2013${nextOffset} of ${matchedCount} matching signals${offset === 0 ? ` (\`limit\` was ${limit})` : ""}. To continue:`, + ); + lines.push( + formatToolCall({ + toolName: "get_replay_activity", + arguments: { + organizationSlug, + replayId, + cursor: encodeCursor({ startMs, endMs, kinds, offset: nextOffset }), + grain, + }, + }), ); lines.push( - `cursor='${encodeCursor({ startMs, endMs, kinds, offset: nextOffset })}'`, + "Or raise `limit` (max 200) to see more at once; a windowed read with `startMs`/`endMs` is cheaper than paging a whole session.", ); } @@ -396,11 +422,6 @@ function suggestStructuralRead({ organizationSlug: string; context: ServerContext; }): string[] { - const signal = findStructuralSignal(signals); - if (!signal || signal.offsetMs === null) { - return []; - } - const instruction = formatToolCallInstruction({ toolName: "get_replay_dom", experimentalMode: context.experimentalMode ?? false, @@ -415,22 +436,53 @@ function suggestStructuralRead({ return []; } - const reason = - signal.type === "hydration-error" - ? "A hydration error means the server and client DOM disagreed" - : "A click the page did not answer is usually explained by the element itself"; + const signal = findStructuralSignal(signals); + + // A specific signal earns a specific, rooted call. + if (signal && signal.offsetMs !== null) { + const reason = + signal.type === "hydration-error" + ? "A hydration error means the server and client DOM disagreed" + : "A click the page did not answer is usually explained by the element itself"; + + return [ + "", + `${reason}. ${instruction}:`, + formatToolCall({ + toolName: "get_replay_dom", + arguments: { + organizationSlug, + replayId, + atMs: signal.offsetMs, + ...(signal.nodeId !== undefined ? { rootNodeId: signal.nodeId } : {}), + }, + }), + ]; + } + + // Nothing here names an element, but plenty of replay questions are about + // page state with no signal at all behind them: a message that flashed, a + // control that was missing, a spinner that never resolved. The signal list + // cannot answer those, and an agent that does not know a structural read + // exists will go looking for an explanation in application source instead. + // So the capability is stated once, without a specific moment to aim at, + // whenever there is a timeline to aim into. + // Prefer a failure as the example offset: it is the moment a reader is most + // likely to ask about, and the first signal in the page is usually a + // navigation that explains nothing. Falls back to any placed signal so the + // call is always completable. + const placed = signals.filter((candidate) => candidate.offsetMs !== null); + const anchor = placed.find((candidate) => candidate.isError) ?? placed[0]; + if (!anchor || anchor.offsetMs === null) { + return []; + } return [ "", - `${reason}. ${instruction}:`, + `If the question is about what the page showed — a message that appeared, a control that was missing or disabled, an element that never rendered — the signals above cannot answer it. ${instruction}, passing the moment you care about:`, formatToolCall({ toolName: "get_replay_dom", - arguments: { - organizationSlug, - replayId, - atMs: signal.offsetMs, - ...(signal.nodeId !== undefined ? { rootNodeId: signal.nodeId } : {}), - }, + arguments: { organizationSlug, replayId, atMs: anchor.offsetMs }, }), ]; } From c834adb75ee1479e3b33857845e4845fcf5337e1 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Tue, 18 Aug 2026 14:38:26 -0700 Subject: [PATCH 25/26] fix(replays): Name the DOM read from the map and stop understating text MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An agent asked to confirm the wording shown during checkout read the map, concluded "the replay tooling only exposes metadata/breadcrumbs — not rendered DOM text", and went to read application source at a release SHA. Three separate causes, none of them discovery: `get_replay_dom` already ranks first or second in tool search for phrasings like "rendered DOM text replay". The map named only `get_replay_activity`. The chain runs map to activity to DOM, and the handoff added for the third step lived in activity — which this agent never called, because the map looked sufficient. The map is where a reader stops, so it now names the structural read too, anchored on the same moment it suggests zooming into. The description understated what a tree contains, in the exact direction that mattered. It claimed text and form values are masked, so the tool "answers structural questions, not what a user typed" — which reads as "rendered text is unavailable" and is wrong. `maskAllText` targets user-entered content: on a real recording, 22 of 24 text nodes are intact, including product copy like "Session replay" and "Watch real user sessions to see what went wrong". Only the user's own details were masked. Reading the wording a user was shown is now stated as a primary use, with its own example, and the masking note scoped to what it actually covers. The agent's conclusion was a positive claim about capability, and nothing in the map contradicted it. That is the failure mode being fixed: silence about a capability reads as its absence. Verified against the SDK rather than assumed — `maskAllText` defaults to true at the integration level, and the masking function targets text nodes by selector, which is why static copy survives. Co-Authored-By: Claude Opus 5 --- packages/mcp-core/src/skillDefinitions.json | 2 +- packages/mcp-core/src/toolDefinitions.json | 2 +- .../tools/catalog/get-replay-details.test.ts | 49 ++++++++++++++- .../src/tools/catalog/get-replay-details.ts | 63 +++++++++++++++++++ .../src/tools/catalog/get-replay-dom.ts | 31 ++++++--- 5 files changed, 134 insertions(+), 13 deletions(-) diff --git a/packages/mcp-core/src/skillDefinitions.json b/packages/mcp-core/src/skillDefinitions.json index 74dfda3f2..326cdaedd 100644 --- a/packages/mcp-core/src/skillDefinitions.json +++ b/packages/mcp-core/src/skillDefinitions.json @@ -139,7 +139,7 @@ }, { "name": "get_replay_dom", - "description": "Read the page structure of a Sentry replay at one moment in time.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what the page looked like when something failed\n- Need to know whether an element existed, was disabled, or was empty\n- Want the DOM around an element that was clicked or rage-clicked\n\nReturns structure only — what the page *was*, not what the user did.\nUse `get_replay_activity` for that, and to find the `nodeId` to root at.\n\nReach for this when a rage or dead click, a hydration error, or a missing\nelement needs explaining: those are page-state questions, and no signal\nlist can answer them. It is the DOM, not a slow paint or a console warning,\nthat explains why a click did nothing.\n`atMs` is milliseconds from the start of the replay and is required.\n\nText and form values are masked by the SDK before upload, so this answers\nstructural questions, not what a user typed.\n\n\n### What the page looked like when the error fired\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)\n```\n\n### The subtree around a rage-clicked element\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')\n```\n", + "description": "Read the page a Sentry replay recorded at one moment: structure and text.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what wording, label, or message was shown to a user\n- Ask what the page looked like when something failed\n- Need to know whether an element existed, was disabled, or was empty\n- Want the DOM around an element that was clicked or rage-clicked\n\nReturns the page itself — elements, attributes, and the text they rendered.\nUse `get_replay_activity` for what the user *did*, and to find a `nodeId`.\n`atMs` is milliseconds from the start of the replay and is required.\n\nUSE THIS FOR RENDERED TEXT. Static UI copy — labels, headings, button text,\nerror banners, option descriptions — is recorded verbatim and readable here.\nThis is how to confirm the exact wording a user saw, rather than inferring it\nfrom application source at some release.\n\nAlso reach for this when a rage or dead click, a hydration error, or a\nmissing element needs explaining: those are page-state questions, and no\nsignal list can answer them.\n\nThe SDK's `maskAllText` targets user-entered content, so names, emails, and\nform values arrive as `***`. Product copy is not masked. Values render as\nrecorded and are never labeled redacted, since masking leaves no marker.\n\n\n### Confirm the exact wording a user was shown\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, lens='full')\n```\n\n### What the page looked like when the error fired\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)\n```\n\n### The subtree around a rage-clicked element\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')\n```\n", "requiredScopes": ["org:read", "project:read", "event:read"] }, { diff --git a/packages/mcp-core/src/toolDefinitions.json b/packages/mcp-core/src/toolDefinitions.json index eea07257b..8cbc0cc21 100644 --- a/packages/mcp-core/src/toolDefinitions.json +++ b/packages/mcp-core/src/toolDefinitions.json @@ -3605,7 +3605,7 @@ }, { "name": "get_replay_dom", - "description": "Read the page structure of a Sentry replay at one moment in time.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what the page looked like when something failed\n- Need to know whether an element existed, was disabled, or was empty\n- Want the DOM around an element that was clicked or rage-clicked\n\nReturns structure only — what the page *was*, not what the user did.\nUse `get_replay_activity` for that, and to find the `nodeId` to root at.\n\nReach for this when a rage or dead click, a hydration error, or a missing\nelement needs explaining: those are page-state questions, and no signal\nlist can answer them. It is the DOM, not a slow paint or a console warning,\nthat explains why a click did nothing.\n`atMs` is milliseconds from the start of the replay and is required.\n\nText and form values are masked by the SDK before upload, so this answers\nstructural questions, not what a user typed.\n\n\n### What the page looked like when the error fired\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)\n```\n\n### The subtree around a rage-clicked element\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')\n```\n", + "description": "Read the page a Sentry replay recorded at one moment: structure and text.\n\nUSE THIS TOOL WHEN USERS:\n- Ask what wording, label, or message was shown to a user\n- Ask what the page looked like when something failed\n- Need to know whether an element existed, was disabled, or was empty\n- Want the DOM around an element that was clicked or rage-clicked\n\nReturns the page itself — elements, attributes, and the text they rendered.\nUse `get_replay_activity` for what the user *did*, and to find a `nodeId`.\n`atMs` is milliseconds from the start of the replay and is required.\n\nUSE THIS FOR RENDERED TEXT. Static UI copy — labels, headings, button text,\nerror banners, option descriptions — is recorded verbatim and readable here.\nThis is how to confirm the exact wording a user saw, rather than inferring it\nfrom application source at some release.\n\nAlso reach for this when a rage or dead click, a hydration error, or a\nmissing element needs explaining: those are page-state questions, and no\nsignal list can answer them.\n\nThe SDK's `maskAllText` targets user-entered content, so names, emails, and\nform values arrive as `***`. Product copy is not masked. Values render as\nrecorded and are never labeled redacted, since masking leaves no marker.\n\n\n### Confirm the exact wording a user was shown\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, lens='full')\n```\n\n### What the page looked like when the error fired\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)\n```\n\n### The subtree around a rage-clicked element\n```\nget_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, rootNodeId=96, lens='full')\n```\n", "inputSchema": { "type": "object", "properties": { diff --git a/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts b/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts index af16d0a7a..f8b3c46de 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-details.test.ts @@ -70,7 +70,10 @@ describe("get_replay_details", () => { ## Next Error CLOUDFLARE-MCP-41 occurred at T+3m 1.3s. Use the Sentry tool \`get_replay_activity\` to read the signals in a time window: - get_replay_activity(organizationSlug='sentry-mcp-evals', replayId='7e07485f-12f9-416b-8b14-26260799b51f', startMs=176300, endMs=186300, grain='detail')" + get_replay_activity(organizationSlug='sentry-mcp-evals', replayId='7e07485f-12f9-416b-8b14-26260799b51f', startMs=176300, endMs=186300, grain='detail') + + For what the page itself showed — the wording of a label or message, whether a control was present or disabled — the signals above cannot answer it. Use the Sentry tool \`get_replay_dom\` to read the page as it was rendered: + get_replay_dom(organizationSlug='sentry-mcp-evals', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, lens='full')" `); }); @@ -752,3 +755,47 @@ describe("get_replay_details", () => { }); }); }); + +describe("structural read pointer", () => { + afterEach(() => { + mswServer.resetHandlers(); + }); + + function readMap(context = getServerContext()) { + return getReplayDetails.handler( + { + organizationSlug: "sentry-mcp-evals", + replayId: replayDetailsFixture.id, + }, + context, + ); + } + + it("names the DOM read from the map, not only from activity", async () => { + // Observed failure: an agent read the map, concluded the replay tooling + // exposes "metadata/breadcrumbs — not rendered DOM text", and went to read + // application source at a release SHA instead. The map is where a reader + // stops, so the third tool has to be named there too. + const result = await readMap(); + + expect(result).toContain( + "get_replay_dom(organizationSlug='sentry-mcp-evals'", + ); + expect(result).toContain("the wording of a label or message"); + expect(result).toContain("lens='full'"); + }); + + it("stays silent when the DOM tool is absent from the session", async () => { + // Naming a tool this session cannot call sends the reader after nothing. + const result = await readMap( + getServerContext({ + availableToolNames: new Set([ + "get_replay_details", + "get_replay_activity", + ]), + }), + ); + + expect(result).not.toContain("get_replay_dom"); + }); +}); diff --git a/packages/mcp-core/src/tools/catalog/get-replay-details.ts b/packages/mcp-core/src/tools/catalog/get-replay-details.ts index 07b7b6ee1..a87af5ba6 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-details.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-details.ts @@ -723,6 +723,14 @@ function buildNextStepLines({ toolName: "get_replay_activity", arguments: { organizationSlug, replayId, grain: "digest" }, }), + ...structuralReadLines({ + organizationSlug, + replayId, + // No failure to anchor on, so offer the start of the session as a + // starting point rather than omit the call. + atMs: 0, + context, + }), ]; } @@ -741,6 +749,61 @@ function buildNextStepLines({ grain: "detail", }, }), + ...structuralReadLines({ + organizationSlug, + replayId, + atMs: anchor.offsetMs, + context, + }), + ]; +} + +/** + * Name the structural read from the map itself. + * + * The map is where a reader stops when the map looks sufficient, and a chain + * that only names its immediate next step leaves the third tool unreachable. + * Observed failure: an agent read the map, concluded "the replay tooling only + * exposes metadata and breadcrumbs, not rendered DOM text", and went looking for + * the answer in application source at a release SHA. That conclusion is wrong, + * and nothing in the map contradicted it. + * + * Text is called out specifically because it is the non-obvious capability. The + * SDK masks user-entered values but not product copy, so the rendered wording of + * a label or message is recoverable — which is exactly the question source code + * at a release cannot answer reliably. + */ +function structuralReadLines({ + organizationSlug, + replayId, + atMs, + context, +}: { + organizationSlug: string; + replayId: string; + atMs: number; + context: ServerContext; +}): string[] { + const instruction = formatToolCallInstruction({ + toolName: "get_replay_dom", + experimentalMode: context.experimentalMode ?? false, + availableToolNames: context.availableToolNames, + directToolNames: context.directToolNames, + // Silence beats a dangling pointer to a tool this session cannot call. + fallbackInstruction: "", + purpose: "to read the page as it was rendered", + }); + if (!instruction) { + return []; + } + + return [ + "", + `For what the page itself showed — the wording of a label or message, whether a control was present or disabled — the signals above cannot answer it. ${instruction}:`, + formatToolCall({ + toolName: "get_replay_dom", + arguments: { organizationSlug, replayId, atMs, lens: "full" }, + }), ]; } diff --git a/packages/mcp-core/src/tools/catalog/get-replay-dom.ts b/packages/mcp-core/src/tools/catalog/get-replay-dom.ts index 403c40e7c..b6b1aad71 100644 --- a/packages/mcp-core/src/tools/catalog/get-replay-dom.ts +++ b/packages/mcp-core/src/tools/catalog/get-replay-dom.ts @@ -33,26 +33,37 @@ export default defineTool({ requiredScopes: ["org:read", "project:read", "event:read"], requiredCapabilities: ["replays"], description: [ - "Read the page structure of a Sentry replay at one moment in time.", + "Read the page a Sentry replay recorded at one moment: structure and text.", "", "USE THIS TOOL WHEN USERS:", + "- Ask what wording, label, or message was shown to a user", "- Ask what the page looked like when something failed", "- Need to know whether an element existed, was disabled, or was empty", "- Want the DOM around an element that was clicked or rage-clicked", "", - "Returns structure only — what the page *was*, not what the user did.", - "Use `get_replay_activity` for that, and to find the `nodeId` to root at.", - "", - "Reach for this when a rage or dead click, a hydration error, or a missing", - "element needs explaining: those are page-state questions, and no signal", - "list can answer them. It is the DOM, not a slow paint or a console warning,", - "that explains why a click did nothing.", + "Returns the page itself — elements, attributes, and the text they rendered.", + "Use `get_replay_activity` for what the user *did*, and to find a `nodeId`.", "`atMs` is milliseconds from the start of the replay and is required.", "", - "Text and form values are masked by the SDK before upload, so this answers", - "structural questions, not what a user typed.", + "USE THIS FOR RENDERED TEXT. Static UI copy — labels, headings, button text,", + "error banners, option descriptions — is recorded verbatim and readable here.", + "This is how to confirm the exact wording a user saw, rather than inferring it", + "from application source at some release.", + "", + "Also reach for this when a rage or dead click, a hydration error, or a", + "missing element needs explaining: those are page-state questions, and no", + "signal list can answer them.", + "", + "The SDK's `maskAllText` targets user-entered content, so names, emails, and", + "form values arrive as `***`. Product copy is not masked. Values render as", + "recorded and are never labeled redacted, since masking leaves no marker.", "", "", + "### Confirm the exact wording a user was shown", + "```", + "get_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300, lens='full')", + "```", + "", "### What the page looked like when the error fired", "```", "get_replay_dom(organizationSlug='my-organization', replayId='7e07485f-12f9-416b-8b14-26260799b51f', atMs=181300)", From 242ab6cab67f4f76ee19bfb4df53f2736bf33f17 Mon Sep 17 00:00:00 2001 From: Matt Duncan Date: Wed, 19 Aug 2026 11:54:45 -0700 Subject: [PATCH 26/26] fix(replays): Raise the DOM text cap above real product copy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 80-character cap was deliberate when the tree was a structural artifact and text was only a label to identify an element by. It became wrong once confirming the exact wording a user was shown was a documented use of the tool, and nothing flagged the contradiction. Measured against a real onboarding page: product copy runs 60 to 100 characters, so 80 cuts through the middle of the distribution. The longest line there — "Catch breaking changes, automatically root cause issues in production, and fix what you missed." at 95 characters — lost its last two words, which is the worst available outcome. It looks complete enough to quote and is not. Rendered text now allows 400 characters. The node budget is what bounds response size; this only bounds a single pathological node, so it can afford to be generous. Form values keep a tighter 80. Values are usually masked to `***`, and an unmasked one long enough to hit the cap is a payload — a token or a pasted blob — where the length is the informative part rather than the content. The cap had no test at all, which is why the number survived a change in what the tool is for. Both limits are now covered, each verified by a mutation that fails the new tests. Co-Authored-By: Claude Opus 5 --- .../mcp-core/src/internal/replay-dom.test.ts | 49 +++++++++++++++++++ packages/mcp-core/src/internal/replay-dom.ts | 42 +++++++++++++--- 2 files changed, 85 insertions(+), 6 deletions(-) diff --git a/packages/mcp-core/src/internal/replay-dom.test.ts b/packages/mcp-core/src/internal/replay-dom.test.ts index 6ffab5b9a..8bf03de6e 100644 --- a/packages/mcp-core/src/internal/replay-dom.test.ts +++ b/packages/mcp-core/src/internal/replay-dom.test.ts @@ -467,6 +467,55 @@ describe("rendering", () => { ]); } + it("keeps real product copy intact rather than cutting mid-sentence", () => { + // Measured from a real onboarding page: product copy runs 60-100 chars, so + // the previous 80-char cap severed the longest line two words from its end + // — complete-looking enough to quote, and wrong. Confirming exact wording is + // a primary use of this tool, so the cap must clear real copy. + const copy = + "Catch breaking changes, automatically root cause issues in production, and fix what you missed."; + const rendered = renderDomTree( + reconstruct([snapshot(0, [element(10, "p", {}, [text(11, copy)])])]), + { lens: "full" }, + ); + + expect(rendered.lines.join("\n")).toContain(`"${copy}"`); + expect(rendered.lines.join("\n")).not.toContain("…"); + }); + + it("elides text that is genuinely unbounded, and marks it", () => { + // A single pathological node should not become the whole response. The + // ellipsis matters: without it the fragment reads as the entire value. + const rendered = renderDomTree( + reconstruct([ + snapshot(0, [element(10, "p", {}, [text(11, "x".repeat(1000))])]), + ]), + { lens: "full" }, + ); + const line = rendered.lines.join("\n"); + + expect(line).toContain("…"); + expect(line.length).toBeLessThan(600); + }); + + it("holds form values to a tighter cap than rendered copy", () => { + // Values are usually masked; an unmasked one this long is a payload, where + // the length is the informative part rather than the content. + const rendered = renderDomTree( + reconstruct([ + snapshot(0, [ + element(10, "input", { id: "token", value: "v".repeat(300) }), + ]), + ]), + { lens: "full" }, + ); + const line = rendered.lines.join("\n"); + + expect(line).toContain("…"); + // 80-char cap, not the 400 allowed for rendered text. + expect(line.length).toBeLessThan(200); + }); + it("prunes a deep branch locally, leaving later siblings intact", () => { // The real failure this reproduces: on a live page `head` is deep enough // that aborting the walk at the depth limit hid `body` and everything under diff --git a/packages/mcp-core/src/internal/replay-dom.ts b/packages/mcp-core/src/internal/replay-dom.ts index 2c09dc528..2ff79f791 100644 --- a/packages/mcp-core/src/internal/replay-dom.ts +++ b/packages/mcp-core/src/internal/replay-dom.ts @@ -653,7 +653,31 @@ const RENDERED_ATTRIBUTES = [ "readonly", ]; -const MAX_TEXT_LENGTH = 80; +/** + * Rendered text kept per element before eliding. + * + * Confirming the exact wording a user was shown is a primary use of this tool, + * and a cap that lands mid-sentence defeats it. Measured against real product + * copy, which clusters between 60 and 100 characters: at 80 the longest line on + * a real onboarding page ("Catch breaking changes, automatically root cause + * issues in production, and fix what you missed.", 95 chars) was cut two words + * from the end, which is the worst possible outcome — it looks complete enough + * to quote and is not. + * + * The node budget is what bounds output size; this only bounds a single + * pathological node, so it can afford to be generous. + */ +const MAX_TEXT_LENGTH = 400; + +/** + * Form values kept before eliding. + * + * Shorter than rendered text on purpose. Values are usually masked to `***`, + * and an unmasked one long enough to hit this is a payload rather than + * something a reader is reading — a serialized token or a pasted blob. Their + * length is the informative part, not their content. + */ +const MAX_VALUE_LENGTH = 80; /** * Depth allowed before a branch is pruned. @@ -925,7 +949,9 @@ function describeElement(node: DomNode, nodes: Map): string { // The current value, not the attribute: input events supersede it, and the // attribute records only what the page shipped with. if (node.inputValue !== undefined) { - parts.push(`[value=${JSON.stringify(truncateText(node.inputValue))}]`); + parts.push( + `[value=${JSON.stringify(truncateText(node.inputValue, MAX_VALUE_LENGTH))}]`, + ); } // rrweb reports `isChecked` on every input event, not only on checkboxes and // radios, so a false value carries no information — and `[checked=false]` on @@ -990,8 +1016,12 @@ function directText(node: DomNode, nodes: Map): string | null { return joined ? truncateText(joined) : null; } -function truncateText(value: string): string { - return value.length > MAX_TEXT_LENGTH - ? `${value.slice(0, MAX_TEXT_LENGTH)}…` - : value; +/** + * Shorten a string, marking that it was shortened. + * + * The ellipsis is load-bearing: without it a truncated string reads as the whole + * value, and the entire point of rendering text here is that it can be quoted. + */ +function truncateText(value: string, limit = MAX_TEXT_LENGTH): string { + return value.length > limit ? `${value.slice(0, limit)}…` : value; }