Skip to content

Commit 1bd2754

Browse files
skaastencodex
andauthored
test(evals): add representative natural-language search coverage (#1315)
## Summary Add compact natural-language search eval coverage across errors, logs, spans, metrics, and issues. ### Key Changes - Add one representative scenario for each supported search dataset - Validate translated datasets, filters, fields, sorting, and time ranges - Assert attribute discovery for the metrics scenario ### Breaking Changes - None Co-authored-by: Codex CLI Agent <noreply@openai.com>
1 parent a63a28e commit 1bd2754

1 file changed

Lines changed: 139 additions & 0 deletions

File tree

Lines changed: 139 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,139 @@
1+
import { SentryApiService } from "@sentry/mcp-core/api-client";
2+
import { searchEventsAgent } from "@sentry/mcp-core/tools/search-events/agent";
3+
import { searchIssuesAgent } from "@sentry/mcp-core/tools/search-issues/agent";
4+
import { describeEval, ToolCallScorer } from "vitest-evals";
5+
import "../setup-env";
6+
import { StructuredOutputScorer } from "./utils/structuredOutputScorer";
7+
8+
function toToolArguments(args: unknown): Record<string, unknown> {
9+
return args && typeof args === "object" && !Array.isArray(args)
10+
? (args as Record<string, unknown>)
11+
: {};
12+
}
13+
14+
describeEval("natural-language-search-events", {
15+
data: async () => [
16+
{
17+
input:
18+
"Count unhandled error events by environment over the last 24 hours",
19+
expectedTools: [],
20+
expected: {
21+
dataset: "errors",
22+
query: /error\.handled:false|error\.unhandled:true/,
23+
fields: (value: unknown) =>
24+
Array.isArray(value) &&
25+
value.includes("environment") &&
26+
value.includes("count()"),
27+
sort: "-count()",
28+
timeRange: { statsPeriod: "24h" },
29+
},
30+
},
31+
{
32+
input: "Show warning logs from production over the last 7 days",
33+
expectedTools: [],
34+
expected: {
35+
dataset: "logs",
36+
query: (value: unknown) =>
37+
typeof value === "string" &&
38+
/(?:severity|level):warn(?:ing)?/.test(value) &&
39+
value.includes("environment:production"),
40+
sort: "-timestamp",
41+
timeRange: { statsPeriod: "7d" },
42+
},
43+
},
44+
{
45+
input:
46+
"In spans, show p95 span duration grouped by span.op over the last 7 days",
47+
expectedTools: [],
48+
expected: {
49+
dataset: "spans",
50+
query: "",
51+
fields: (value: unknown) =>
52+
Array.isArray(value) &&
53+
value.includes("span.op") &&
54+
value.includes("p95(span.duration)"),
55+
sort: "-p95(span.duration)",
56+
timeRange: { statsPeriod: "7d" },
57+
},
58+
},
59+
{
60+
input:
61+
"In metrics, show p95 http.request.duration grouped by environment over the last 24 hours",
62+
expectedTools: [
63+
{
64+
name: "datasetAttributes",
65+
},
66+
],
67+
expected: {
68+
dataset: "metrics",
69+
query: (value: unknown) =>
70+
typeof value === "string" &&
71+
value.includes("metric.name:http.request.duration") &&
72+
value.includes("metric.type:distribution"),
73+
fields: (value: unknown) =>
74+
Array.isArray(value) &&
75+
value.includes("environment") &&
76+
value.includes(
77+
"p95(value,http.request.duration,distribution,millisecond)",
78+
),
79+
sort: "-p95(value,http.request.duration,distribution,millisecond)",
80+
timeRange: { statsPeriod: "24h" },
81+
},
82+
},
83+
],
84+
task: async (input) => {
85+
const apiService = new SentryApiService({ accessToken: "test-token" });
86+
const agentResult = await searchEventsAgent({
87+
query: input,
88+
organizationSlug: "sentry-mcp-evals",
89+
apiService,
90+
environmentNames: [],
91+
});
92+
93+
return {
94+
result: JSON.stringify(agentResult.result),
95+
toolCalls: agentResult.toolCalls.map((call) => ({
96+
name: call.toolName,
97+
arguments: toToolArguments(call.args),
98+
})),
99+
};
100+
},
101+
scorers: [ToolCallScorer(), StructuredOutputScorer({ match: "fuzzy" })],
102+
threshold: 0.6,
103+
timeout: 30000,
104+
});
105+
106+
describeEval("natural-language-search-issues", {
107+
data: async () => [
108+
{
109+
input: "Show resolved high-priority issues, newest first",
110+
expectedTools: [],
111+
expected: {
112+
query: (value: unknown) =>
113+
typeof value === "string" &&
114+
value.includes("is:resolved") &&
115+
value.includes("issue.priority:high"),
116+
sort: "new",
117+
},
118+
},
119+
],
120+
task: async (input) => {
121+
const apiService = new SentryApiService({ accessToken: "test-token" });
122+
const agentResult = await searchIssuesAgent({
123+
query: input,
124+
organizationSlug: "sentry-mcp-evals",
125+
apiService,
126+
});
127+
128+
return {
129+
result: JSON.stringify(agentResult.result),
130+
toolCalls: agentResult.toolCalls.map((call) => ({
131+
name: call.toolName,
132+
arguments: toToolArguments(call.args),
133+
})),
134+
};
135+
},
136+
scorers: [ToolCallScorer(), StructuredOutputScorer({ match: "fuzzy" })],
137+
threshold: 0.6,
138+
timeout: 30000,
139+
});

0 commit comments

Comments
 (0)