Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .changeset/sonnet-55-done-tool.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
"@browserbasehq/stagehand": patch
---

Use automatic tool choice for the final agent `done` call with Claude Sonnet 5.5, Opus 5.5, and Fable 5.1.
2 changes: 1 addition & 1 deletion packages/core/lib/v3/agent/utils/handleDoneToolCall.ts
Original file line number Diff line number Diff line change
Expand Up @@ -150,7 +150,7 @@ Call the "done" tool with:
const modelProvider = typeof model === "string" ? undefined : model.provider;
const fallbacks = anthropicFallbacksOptions(modelId);

// Models whose always-on thinking rejects forced tool use go straight to
// Models that reject forced tool use go straight to
// "auto" — the prompt already instructs calling "done", and the
// no-tool-call case below handles a plain-text answer.
const result = await generateText({
Expand Down
14 changes: 9 additions & 5 deletions packages/core/lib/v3/llm/anthropicOptions.ts
Original file line number Diff line number Diff line change
Expand Up @@ -60,13 +60,17 @@ export function isAnthropicFable5Model(modelId: string): boolean {
}

/**
* True for models that reject forced tool use
* (`tool_choice: { type: "tool" }`). Forced tool choice is incompatible with
* active extended thinking, and on Fable 5 thinking is always on — so the
* rejection is a certainty there, not a transient quirk.
* Models that require automatic tool choice for the final agent assessment.
*/
const MODELS_WITHOUT_FORCED_TOOL_USE = new Set<string>([
"claude-fable-5",
"claude-fable-5-1",
"claude-opus-5-5",
"claude-sonnet-5-5",
]);

export function rejectsForcedToolUse(modelId: string): boolean {
return isAnthropicFable5Model(modelId);
return MODELS_WITHOUT_FORCED_TOOL_USE.has(stripModelProvider(modelId));
}

const VALID_EFFORTS: ReadonlySet<string> = new Set([
Expand Down
85 changes: 85 additions & 0 deletions packages/core/tests/unit/agent-done-tool-choice.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,85 @@
import { describe, expect, it } from "vitest";
import type { LanguageModelV2 } from "@ai-sdk/provider";

import { handleDoneToolCall } from "../../lib/v3/agent/utils/handleDoneToolCall.js";

const automaticToolChoiceModels = [
"claude-sonnet-5-5",
"claude-opus-5-5",
"claude-fable-5-1",
].flatMap((modelId) => [modelId, `anthropic/${modelId}`]);
Comment on lines +6 to +10

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P3: automaticToolChoiceModels omits claude-fable-5, one of the four entries in MODELS_WITHOUT_FORCED_TOOL_USE (anthropicOptions.ts), so the handleDoneToolCall auto-tool-choice path for the base Fable model is never exercised here. Add it (or export the implementation set) so the test and source lists cannot drift.

Prompt for AI agents
Check if this issue is valid — if so, understand the root cause and fix it. When an issue isn't valid or won't be fixed in this PR, reply in its thread with the reason and then resolve the thread. At packages/core/tests/unit/agent-done-tool-choice.test.ts, line 6:

<comment>`automaticToolChoiceModels` omits `claude-fable-5`, one of the four entries in `MODELS_WITHOUT_FORCED_TOOL_USE` (anthropicOptions.ts), so the `handleDoneToolCall` auto-tool-choice path for the base Fable model is never exercised here. Add it (or export the implementation set) so the test and source lists cannot drift.</comment>

<file context>
@@ -3,7 +3,13 @@ import type { LanguageModelV2 } from "@ai-sdk/provider";
 import { handleDoneToolCall } from "../../lib/v3/agent/utils/handleDoneToolCall.js";
 
-function sonnetMock(modelId: string, response: "done" | "text") {
+const automaticToolChoiceModels = [
+  "claude-sonnet-5-5",
+  "claude-opus-5-5",
</file context>
Suggested change
const automaticToolChoiceModels = [
"claude-sonnet-5-5",
"claude-opus-5-5",
"claude-fable-5-1",
].flatMap((modelId) => [modelId, `anthropic/${modelId}`]);
const automaticToolChoiceModels = [
"claude-fable-5",
"claude-fable-5-1",
"claude-opus-5-5",
"claude-sonnet-5-5",
].flatMap((modelId) => [modelId, `anthropic/${modelId}`]);


function modelMock(modelId: string, response: "done" | "text") {
const toolChoices: unknown[] = [];
const model = {
specificationVersion: "v2",
provider: "anthropic",
modelId,
supportedUrls: {},
async doGenerate(options: Parameters<LanguageModelV2["doGenerate"]>[0]) {
toolChoices.push(options.toolChoice);
if (
options.toolChoice?.type === "tool" ||
options.toolChoice?.type === "required"
) {
throw new Error(`${modelId} rejects forced tool use`);
}
return {
finishReason: "stop" as const,
usage: { inputTokens: 1, outputTokens: 1, totalTokens: 2 },
content:
response === "done"
? [
{
type: "tool-call" as const,
toolCallId: "done-1",
toolName: "done",
input: JSON.stringify({
reasoning: "The task is complete.",
taskComplete: true,
}),
},
]
: [{ type: "text" as const, text: "The task needs review." }],
warnings: [] as [],
};
},
} as unknown as LanguageModelV2;
return { model, toolChoices };
}

describe("v3 agent done tool choice", () => {
it.each(automaticToolChoiceModels)(
"lets %s call done without forcing tool use",
async (modelId) => {
const { model, toolChoices } = modelMock(modelId, "done");
const result = await handleDoneToolCall({
model,
inputMessages: [{ role: "user", content: "The task is complete." }],
instruction: "Complete the task",
logger: () => {},
});

expect(toolChoices).toEqual([{ type: "auto" }]);
expect(result.taskComplete).toBe(true);
expect(result.reasoning).toBe("The task is complete.");
},
);

it.each(automaticToolChoiceModels)(
"keeps a text reply from %s as an incomplete result",
async (modelId) => {
const { model, toolChoices } = modelMock(modelId, "text");
const result = await handleDoneToolCall({
model,
inputMessages: [{ role: "user", content: "The task is incomplete." }],
instruction: "Complete the task",
logger: () => {},
});

expect(toolChoices).toEqual([{ type: "auto" }]);
expect(result.taskComplete).toBe(false);
expect(result.reasoning).toBe("The task needs review.");
},
);
});
8 changes: 8 additions & 0 deletions packages/core/tests/unit/anthropic-options.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -90,6 +90,14 @@ describe("rejectsForcedToolUse", () => {
expect(rejectsForcedToolUse("anthropic/claude-fable-5")).toBe(true);
});

it.each(["claude-sonnet-5-5", "claude-opus-5-5", "claude-fable-5-1"])(
"is true for %s with or without a provider prefix",
(modelId) => {
expect(rejectsForcedToolUse(modelId)).toBe(true);
expect(rejectsForcedToolUse(`anthropic/${modelId}`)).toBe(true);
},
);

it("is false for models that accept forced tool use", () => {
expect(rejectsForcedToolUse("claude-opus-4-8")).toBe(false);
expect(rejectsForcedToolUse("claude-haiku-4-5-20251001")).toBe(false);
Expand Down
Loading