Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions src/api/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@ import {
retiredProviderIdentifiers,
type ProviderSettings,
type ModelInfo,
type ReasoningEffortExtended,
} from "@roo-code/types"

import { getRouterRemovalMessage } from "../core/config/routerRemoval"
Expand Down Expand Up @@ -115,6 +116,14 @@ export interface ApiHandlerCreateMessageMetadata {
* when the user clicks stop, preventing wasted API tokens/compute on the provider side.
*/
abortSignal?: AbortSignal
/**
* Per-request thinking effort override (DTE series 2/5).
* When defined, takes precedence over the settings-derived `reasoningEffort`
* wherever the effective effort is resolved (see `resolveEffectiveReasoningEffort`).
* Task-scoped and transient: it applies to this request only (the next request
* after being set — no mid-stream effect) and is never persisted to settings.
*/
reasoningEffort?: ReasoningEffortExtended
}

export interface ApiHandler {
Expand Down
297 changes: 297 additions & 0 deletions src/api/providers/__tests__/anthropic-adaptive-effort.spec.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,297 @@
// npx vitest run src/api/providers/__tests__/anthropic-adaptive-effort.spec.ts
//
// DTE series 2/5 — per-request adaptive thinking effort envelope
// (output_config.effort) on the main Anthropic handler.
//
// Kept in a dedicated file (rather than anthropic.spec.ts) so the DTE series PRs
// stay mergeable while other series PRs extend the shared spec file.

import { AnthropicHandler } from "../anthropic"
import type { ApiHandlerOptions } from "../../../shared/api"
import type { ReasoningEffortExtended } from "@roo-code/types"
import { asyncStreamFrom, collectStream } from "../../../test-utils/stream"
import { clearAllMocks } from "../../../test-utils/reset"
import type { ApiHandlerCreateMessageMetadata } from "../../../api"

// Mock TelemetryService
vitest.mock("@roo-code/telemetry", () => ({
TelemetryService: {
instance: {
captureException: vitest.fn(),
},
},
}))

const mockCreate = vitest.fn()

// Same SDK mock pattern as anthropic.spec.ts: createMessage resolves to a short
// finite stream so the handler's for-await loop terminates cleanly.
vitest.mock("@anthropic-ai/sdk", () => {
const mockAnthropicConstructor = vitest.fn().mockImplementation(function () {
return {
messages: {
create: mockCreate.mockImplementation(async (options: { stream?: boolean; model?: string }) => {
if (!options.stream) {
return {
id: "test-completion",
content: [{ type: "text", text: "Test response" }],
role: "assistant",
model: options.model,
usage: { input_tokens: 10, output_tokens: 5 },
}
}
return asyncStreamFrom([
{
type: "message_start",
message: {
usage: {
input_tokens: 100,
output_tokens: 50,
cache_creation_input_tokens: 20,
cache_read_input_tokens: 10,
},
},
},
{
type: "content_block_start",
index: 0,
content_block: { type: "text", text: "Hello" },
},
{
type: "content_block_delta",
delta: { type: "text_delta", text: " world" },
},
])
}),
},
}
})

return {
Anthropic: mockAnthropicConstructor,
}
})

const userMessage = {
role: "user" as const,
content: [{ type: "text" as const, text: "Hi" }],
}

/** Runs createMessage to completion and returns the request params sent to the SDK. */
async function sentRequestParams(
handler: AnthropicHandler,
metadata?: ApiHandlerCreateMessageMetadata,
): Promise<Record<string, unknown>> {
const stream = handler.createMessage("system prompt", [userMessage], metadata)
await collectStream(stream)
const call = mockCreate.mock.calls.at(-1)
if (!call) {
throw new Error("Expected the SDK messages.create to have been called")
}
return call[0] as Record<string, unknown>
}

function makeHandler(options: {
apiModelId?: string
enableReasoningEffort?: boolean
reasoningEffort?: ApiHandlerOptions["reasoningEffort"]
}): AnthropicHandler {
return new AnthropicHandler({
apiKey: "test-api-key",
apiModelId: options.apiModelId ?? "claude-opus-4-7",
enableReasoningEffort: options.enableReasoningEffort,
reasoningEffort: options.reasoningEffort,
})
}

describe("AnthropicHandler adaptive effort envelope (DTE series 2/5)", () => {
beforeEach(() => {
clearAllMocks()
})

describe("output_config.effort on adaptive-thinking requests", () => {
const inRangeEfforts: ReasoningEffortExtended[] = ["low", "medium", "high", "xhigh", "max"]

it.each(inRangeEfforts)(
"sends the settings effort %s as output_config.effort for an adaptive model",
async (effort) => {
const handler = makeHandler({ enableReasoningEffort: true, reasoningEffort: effort })

const params = await sentRequestParams(handler)

expect(params.thinking).toEqual({ type: "adaptive" })
expect(params.output_config).toEqual({ effort })
},
)

it("sends the envelope from the first (cache-control) requestParams branch", async () => {
// claude-opus-4-8 takes the first (cache-control) requestParams branch;
// the default branch is covered below via an unknown model id.
const handler = makeHandler({
apiModelId: "claude-opus-4-8",
enableReasoningEffort: true,
reasoningEffort: "xhigh",
})

const params = await sentRequestParams(handler)

expect(params.thinking).toEqual({ type: "adaptive" })
expect(params.output_config).toEqual({ effort: "xhigh" })
})

it("sends the envelope from the default requestParams branch", async () => {
// Unknown model id -> falls through to the default switch branch, while the
// guessed model info (claude-opus-4-7 substring) is adaptive-capable.
const handler = makeHandler({
apiModelId: "claude-opus-4-7-custom",
enableReasoningEffort: true,
reasoningEffort: "high",
})

const params = await sentRequestParams(handler)

expect(params.model).toBe("claude-opus-4-7-custom")
expect(params.thinking).toEqual({ type: "adaptive" })
expect(params.output_config).toEqual({ effort: "high" })
})
})

describe("envelope omission (out-of-range or non-adaptive)", () => {
const settingsEfforts: ApiHandlerOptions["reasoningEffort"][] = ["none", "minimal", "disable"]

it.each(settingsEfforts)(
"omits output_config when the settings effort is %s on an adaptive model",
async (effort) => {
const handler = makeHandler({ enableReasoningEffort: true, reasoningEffort: effort })

const params = await sentRequestParams(handler)

// Adaptive thinking is still requested, but no envelope is sent so the
// API applies its own default effort.
expect(params.thinking).toEqual({ type: "adaptive" })
expect(params).not.toHaveProperty("output_config")
},
)

it("omits output_config when no effort is set anywhere on an adaptive model", async () => {
const handler = makeHandler({ enableReasoningEffort: true })

const params = await sentRequestParams(handler)

expect(params.thinking).toEqual({ type: "adaptive" })
expect(params).not.toHaveProperty("output_config")
})

it("omits output_config for a non-adaptive model even with an in-range effort", async () => {
// Budget-based extended thinking (type: "enabled") never carries the
// adaptive envelope.
const handler = makeHandler({
apiModelId: "claude-sonnet-4-5",
enableReasoningEffort: true,
reasoningEffort: "xhigh",
})

const params = await sentRequestParams(handler)

expect(params.thinking).toMatchObject({ type: "enabled" })
expect(params).not.toHaveProperty("output_config")
})

it("omits output_config when adaptive thinking itself is not requested", async () => {
// enableReasoningEffort=false -> thinking is undefined -> no envelope even
// with an in-range settings effort.
const handler = makeHandler({ enableReasoningEffort: false, reasoningEffort: "xhigh" })

const params = await sentRequestParams(handler)

expect(params.thinking).toBeUndefined()
expect(params).not.toHaveProperty("output_config")
})

it("keeps the pre-DTE request shape for a plain model with no reasoning settings", async () => {
// Guard: no reasoning settings and no metadata -> no output_config.
const handler = makeHandler({ apiModelId: "claude-3-5-haiku-20241022" })

const params = await sentRequestParams(handler)

expect(params.thinking).toBeUndefined()
expect(params).not.toHaveProperty("output_config")
})
})

describe("per-request override (metadata.reasoningEffort) precedence", () => {
const baseOptions: {
apiModelId?: string
enableReasoningEffort?: boolean
reasoningEffort?: ApiHandlerOptions["reasoningEffort"]
} = {
apiModelId: "claude-opus-4-7",
enableReasoningEffort: true,
}

it("lets metadata.reasoningEffort override the settings value", async () => {
const handler = makeHandler({ ...baseOptions, reasoningEffort: "low" })

const params = await sentRequestParams(handler, {
taskId: "task-1",
reasoningEffort: "xhigh",
})

expect(params.output_config).toEqual({ effort: "xhigh" })
})

it("suppresses the envelope when the metadata override is out-of-range", async () => {
// Settings would send "high"; the override wins and is out-of-range, so
// the envelope is omitted entirely.
const handler = makeHandler({ ...baseOptions, reasoningEffort: "high" })

const params = await sentRequestParams(handler, {
taskId: "task-1",
reasoningEffort: "minimal",
})

expect(params.thinking).toEqual({ type: "adaptive" })
expect(params).not.toHaveProperty("output_config")
})

const overrideEfforts: ReasoningEffortExtended[] = ["none", "minimal"]

it.each(overrideEfforts)(
"suppresses the envelope for metadata override %s even with an in-range settings value",
async (effort) => {
const handler = makeHandler({ ...baseOptions, reasoningEffort: "max" })

const params = await sentRequestParams(handler, {
taskId: "task-1",
reasoningEffort: effort,
})

expect(params).not.toHaveProperty("output_config")
},
)

it("applies the settings value when metadata carries no override", async () => {
const handler = makeHandler({ ...baseOptions, reasoningEffort: "medium" })

const params = await sentRequestParams(handler, { taskId: "task-1" })

expect(params.output_config).toEqual({ effort: "medium" })
})

it("keeps non-adaptive requests envelope-free even with a metadata override", async () => {
const handler = makeHandler({
apiModelId: "claude-sonnet-4-5",
enableReasoningEffort: true,
reasoningEffort: "low",
})

const params = await sentRequestParams(handler, {
taskId: "task-1",
reasoningEffort: "xhigh",
})

expect(params.thinking).toMatchObject({ type: "enabled" })
expect(params).not.toHaveProperty("output_config")
})
})
})
44 changes: 43 additions & 1 deletion src/api/providers/anthropic.ts
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,11 @@ import type { ApiHandlerOptions } from "../../shared/api"
import { ApiStream } from "../transform/stream"
import { getModelParams } from "../transform/model-params"
import { filterNonAnthropicBlocks } from "../transform/anthropic-filter"
import { getAnthropicProviderReasoning } from "../transform/reasoning"
import {
ADAPTIVE_OUTPUT_CONFIG_EFFORTS,
getAnthropicProviderReasoning,
resolveEffectiveReasoningEffort,
} from "../transform/reasoning"
import { handleProviderError } from "./utils/error-handler"

import { BaseProvider } from "./base-provider"
Expand Down Expand Up @@ -58,6 +62,21 @@ export class AnthropicHandler extends BaseProvider implements SingleCompletionHa
})
}

/**
* Creates a streaming Anthropic message for the current model.
*
* Resolves the effective thinking effort for this request through the shared
* `resolveEffectiveReasoningEffort` point (per-request override → settings →
* model default). For adaptive-thinking models, when the resolved effort is one
* of `ADAPTIVE_OUTPUT_CONFIG_EFFORTS` (low|medium|high|xhigh|max), the request
* carries `output_config: { effort }` (DTE series 2/5); out-of-range or unset
* efforts omit it so the API default applies.
*
* @param systemPrompt - The system prompt for the request.
* @param messages - The message history to send.
* @param metadata - Per-request metadata (carries the task-local effort override).
* @returns An async iterator of parsed Anthropic stream events.
*/
async *createMessage(
systemPrompt: string,
messages: Anthropic.Messages.MessageParam[],
Expand All @@ -79,6 +98,25 @@ export class AnthropicHandler extends BaseProvider implements SingleCompletionHa
settings: this.options,
})

// DTE series 2/5: per-request adaptive effort envelope (output_config.effort).
// The task-local per-request override (metadata.reasoningEffort) takes
// precedence over the settings-derived value (shared resolution in
// resolveEffectiveReasoningEffort). Only adaptive-thinking requests whose
// effective effort is in-range get the envelope; everything else (unset,
// "disable", "none", "minimal") omits it and lets the API apply its default.
const effectiveReasoningEffort = resolveEffectiveReasoningEffort({
override: metadata?.reasoningEffort,
settingsReasoningEffort: this.options.reasoningEffort,
modelDefaultEffort: info.reasoningEffort,
})
const adaptiveEffort =
thinking?.type === "adaptive" &&
effectiveReasoningEffort !== undefined &&
effectiveReasoningEffort !== "disable" &&
ADAPTIVE_OUTPUT_CONFIG_EFFORTS.includes(effectiveReasoningEffort)
? effectiveReasoningEffort
: undefined

// Filter out non-Anthropic blocks (reasoning, thoughtSignature, etc.) before sending to the API
const sanitizedMessages = filterNonAnthropicBlocks(messages)

Expand Down Expand Up @@ -141,6 +179,8 @@ export class AnthropicHandler extends BaseProvider implements SingleCompletionHa
max_tokens: maxTokens ?? ANTHROPIC_DEFAULT_MAX_TOKENS,
temperature,
thinking,
// DTE series 2/5: adaptive effort envelope (omitted unless in-range).
...(adaptiveEffort !== undefined ? { output_config: { effort: adaptiveEffort } } : {}),
// Setting cache breakpoint for system prompt so new tasks can reuse it.
system: [{ text: systemPrompt, type: "text", cache_control: cacheControl }],
messages: sanitizedMessages.map((message, index) => {
Expand Down Expand Up @@ -216,6 +256,8 @@ export class AnthropicHandler extends BaseProvider implements SingleCompletionHa
max_tokens: maxTokens ?? ANTHROPIC_DEFAULT_MAX_TOKENS,
temperature,
thinking,
// DTE series 2/5: adaptive effort envelope (omitted unless in-range).
...(adaptiveEffort !== undefined ? { output_config: { effort: adaptiveEffort } } : {}),
system: [{ text: systemPrompt, type: "text" }],
messages: sanitizedMessages,
stream: true,
Expand Down
Loading
Loading