Skip to content

Commit 30f9fe2

Browse files
ci(evals): run eval workflow on OpenRouter
Switch the GitHub Actions eval job and eval harness defaults from OPENAI_API_KEY/gpt-4o to OPENROUTER_API_KEY so CI no longer depends on a direct OpenAI key. Co-Authored-By: Daniel Griesser <dgriesser@sentry.io>
1 parent 6b1e970 commit 30f9fe2

9 files changed

Lines changed: 23 additions & 17 deletions

File tree

‎.github/workflows/eval.yml‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -60,7 +60,8 @@ jobs:
6060
run: pnpm eval:ci evals
6161
continue-on-error: true
6262
env:
63-
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
63+
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
64+
EMBEDDED_AGENT_PROVIDER: openrouter
6465

6566
- name: Create eval status check
6667
uses: actions/github-script@v7

‎docs/testing/overview.md‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -279,7 +279,7 @@ describeEval("tool-name", {
279279
### Running Evals
280280

281281
```bash
282-
# Requires OPENAI_API_KEY in .env
282+
# Requires OPENROUTER_API_KEY in .env
283283
pnpm eval
284284

285285
# Run specific eval

‎packages/mcp-server-evals/src/bin/start-mock-stdio.ts‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -8,7 +8,10 @@ import type { ServerContext } from "@sentry/mcp-core/types";
88

99
mswServer.listen({
1010
onUnhandledRequest: (req, print) => {
11-
if (req.url.startsWith("https://api.openai.com/")) {
11+
if (
12+
req.url.startsWith("https://api.openai.com/") ||
13+
req.url.startsWith("https://openrouter.ai/")
14+
) {
1215
return;
1316
}
1417

‎packages/mcp-server-evals/src/evals/search-events.eval.ts‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,7 +1,7 @@
11
import { describeEval } from "vitest-evals";
22
import { FIXTURES, NoOpTaskRunner, ToolPredictionScorer } from "./utils";
33

4-
// Note: This eval requires OPENAI_API_KEY to be set in the environment
4+
// Note: This eval requires OPENROUTER_API_KEY to be set in the environment
55
// The search_events tool uses the AI SDK to translate natural language queries
66
describeEval("search-events", {
77
data: async () => {

‎packages/mcp-server-evals/src/evals/search-issue-events.eval.ts‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,7 +1,7 @@
11
import { describeEval } from "vitest-evals";
22
import { FIXTURES, NoOpTaskRunner, ToolPredictionScorer } from "./utils";
33

4-
// Note: This eval requires OPENAI_API_KEY to be set in the environment
4+
// Note: This eval requires OPENROUTER_API_KEY to be set in the environment
55
// The search_issue_events tool uses the AI SDK to translate natural language queries
66
describeEval("search-issue-events", {
77
data: async () => {

‎packages/mcp-server-evals/src/evals/search-issues.eval.ts‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,7 +1,7 @@
11
import { describeEval } from "vitest-evals";
22
import { FIXTURES, NoOpTaskRunner, ToolPredictionScorer } from "./utils";
33

4-
// Note: This eval requires OPENAI_API_KEY to be set in the environment
4+
// Note: This eval requires OPENROUTER_API_KEY to be set in the environment
55
// The search_issues tool uses the AI SDK to translate natural language queries
66
describeEval("search-issues", {
77
data: async () => {

‎packages/mcp-server-evals/src/evals/utils/mcpToolCallRunner.ts‎

Lines changed: 2 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -1,10 +1,8 @@
11
import { experimental_createMCPClient } from "@ai-sdk/mcp";
22
import { Experimental_StdioMCPTransport } from "@ai-sdk/mcp/mcp-stdio";
3-
import { openai } from "@ai-sdk/openai";
3+
import { getOpenRouterModel } from "@sentry/mcp-core/internal/agents/openrouter-provider";
44
import { generateText, stepCountIs, type LanguageModel } from "ai";
55

6-
const defaultModel = openai("gpt-4o");
7-
86
function toToolCall(call: { toolName: string; input: unknown }) {
97
const input =
108
call.input && typeof call.input === "object" && !Array.isArray(call.input)
@@ -18,7 +16,7 @@ function toToolCall(call: { toolName: string; input: unknown }) {
1816
}
1917

2018
export function McpToolCallTaskRunner(
21-
model: LanguageModel = defaultModel,
19+
model: LanguageModel = getOpenRouterModel(),
2220
maxSteps = 6,
2321
) {
2422
return async function McpToolCallTaskRunner(input: string) {

‎packages/mcp-server-evals/src/evals/utils/toolPredictionScorer.ts‎

Lines changed: 5 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -1,8 +1,8 @@
1-
import { openai } from "@ai-sdk/openai";
21
import { generateObject, type LanguageModel } from "ai";
32
import { z } from "zod";
43
import { experimental_createMCPClient } from "@ai-sdk/mcp";
54
import { Experimental_StdioMCPTransport } from "@ai-sdk/mcp/mcp-stdio";
5+
import { getOpenRouterModel } from "@sentry/mcp-core/internal/agents/openrouter-provider";
66

77
// Cache for available tools to avoid reconnecting for each test
88
let cachedTools: string[] | null = null;
@@ -64,8 +64,6 @@ interface ToolPredictionScorerOptions {
6464
result?: any;
6565
}
6666

67-
const defaultModel = openai("gpt-4o");
68-
6967
const predictionSchema = z.object({
7068
score: z.number().min(0).max(1).describe("Score from 0 to 1"),
7169
rationale: z.string().describe("Explanation of the score"),
@@ -129,7 +127,7 @@ CRITICAL: The expected tools represent the actual realistic behavior for this sp
129127
* A scorer that uses AI to predict what tools would be called without executing them.
130128
* This is much faster than actually running the tools and checking what was called.
131129
*
132-
* @param model - Optional language model to use for predictions (defaults to gpt-4o)
130+
* @param model - Optional language model to use for predictions (defaults to OpenRouter)
133131
* @returns A scorer function that compares predicted vs expected tool calls
134132
*
135133
* @example
@@ -170,7 +168,9 @@ CRITICAL: The expected tools represent the actual realistic behavior for this sp
170168
* If `expectedTools` is not provided in test data, the scorer is automatically skipped
171169
* and returns `{ score: null }` to allow other scorers to run without interference.
172170
*/
173-
export function ToolPredictionScorer(model: LanguageModel = defaultModel) {
171+
export function ToolPredictionScorer(
172+
model: LanguageModel = getOpenRouterModel(),
173+
) {
174174
return async function ToolPredictionScorer(
175175
opts: ToolPredictionScorerOptions,
176176
) {

‎packages/mcp-server-mocks/src/utils.ts‎

Lines changed: 6 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -17,8 +17,12 @@ export function startMockServer(options?: {
1717

1818
mswServer.listen({
1919
onUnhandledRequest: (req: any, print: any) => {
20-
// Ignore OpenAI requests if specified (default behavior for AI agent tests)
21-
if (ignoreOpenAI && req.url.startsWith("https://api.openai.com/")) {
20+
// Ignore LLM provider requests if specified (default behavior for AI agent tests)
21+
if (
22+
ignoreOpenAI &&
23+
(req.url.startsWith("https://api.openai.com/") ||
24+
req.url.startsWith("https://openrouter.ai/"))
25+
) {
2226
return;
2327
}
2428

0 commit comments

Comments
 (0)