Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
205 changes: 205 additions & 0 deletions packages/evals/framework/browseCliMcpBridge.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,205 @@
/**
* browse_cli as an MCP mount.
*
* Harnesses that drive a shell (claude_code, codex) expose browse_cli by
* putting a pinned `browse` wrapper on PATH and letting the agent call it from
* Bash. Harnesses whose agent process is not a shell host — Deep Agents runs as
* a separate Python process whose only tool channel is stdio MCP — cannot use
* that path at all, which is why browse_cli was unavailable to them.
*
* This bridge closes that gap without changing the surface the model sees: a
* single `browse` tool whose one argument is the exact command line the browse
* skill documents. The command allowlist is the same one the Claude Code
* adapter enforces in `canUseTool`, so neither harness can run anything but
* `browse`, and one tool call still maps to one browse command.
*/
import { execFile } from "node:child_process";
import { z } from "zod/v4";
import { EvalsError } from "../errors.js";
import type { EvalLogger } from "../logger.js";
import { isAllowedBrowseCommand } from "./claudeCodeToolAdapter.js";
import { clipToolResult } from "./codeExposure.js";
import { startMcpLoopbackBridge, type McpLoopbackBridge } from "./mcpLoopbackBridge.js";

/** MCP server name; tool names reach the trajectory as `${server}.${tool}`. */
export const BROWSE_CLI_MCP_SERVER_NAME = "browse";

/** The single tool the bridge exposes. */
export const BROWSE_CLI_MCP_TOOL_NAME = "browse";

const BROWSE_CLI_MCP_TOOL_DESCRIPTION =
"Run one browse CLI command against the browser session pinned to this eval. " +
"Pass the full command line, including the leading `browse` (for example " +
"`browse open https://example.com`). There is no shell: exactly one command " +
"per call, no pipes, redirection, or chaining. Environment and session flags " +
"are appended by the harness — never pass --local, --remote, or --session.";

const BROWSE_CLI_MCP_COMMAND_DESCRIPTION =
"A single browse command line, e.g. `browse snapshot` or `browse click @e12`.";

export interface BrowseCliMcpBridgeInput {
/** Pinned `browse` wrapper created by `prepareBrowseCliHarnessAdapter`. */
wrapperPath: string;
cwd: string;
env: Record<string, string>;
logger: EvalLogger;
logCategory: string;
}

export type BrowseCliMcpBridge = McpLoopbackBridge;

/**
* Splits a validated browse command line into argv.
*
* `isAllowedBrowseCommand` has already rejected shell metacharacters, so the
* only shell syntax left to honor is quoting: anything else would silently
* change the argument the model meant to pass.
*/
export function tokenizeBrowseCommand(command: string): string[] {
const tokens: string[] = [];
let current = "";
let started = false;
let quote: '"' | "'" | undefined;

for (let index = 0; index < command.length; index += 1) {
const char = command[index];
if (char === "\\" && quote !== "'" && index + 1 < command.length) {
current += command[index + 1];
started = true;
index += 1;
continue;
}
if (quote) {
if (char === quote) quote = undefined;
else current += char;
continue;
}
if (char === '"' || char === "'") {
quote = char;
started = true;
continue;
}
if (/\s/.test(char)) {
if (started) tokens.push(current);
current = "";
started = false;
continue;
}
current += char;
started = true;
}
if (quote) throw new EvalsError(`Unterminated ${quote} quote in browse command.`);
if (started) tokens.push(current);

if (tokens[0] !== "browse") {
throw new EvalsError("Only browse commands are allowed for this eval harness.");
}
return tokens.slice(1);
}

export async function startBrowseCliMcpBridge(
input: BrowseCliMcpBridgeInput,
): Promise<BrowseCliMcpBridge> {
return startMcpLoopbackBridge({
serverName: "stagehand-evals-browse-cli",
logger: input.logger,
logCategory: input.logCategory,
register: (server) => {
server.registerTool(
BROWSE_CLI_MCP_TOOL_NAME,
{
description: BROWSE_CLI_MCP_TOOL_DESCRIPTION,
inputSchema: { command: z.string().describe(BROWSE_CLI_MCP_COMMAND_DESCRIPTION) },
},
({ command }) => runBrowseCommand(command, input),
);
},
});
}

async function runBrowseCommand(
command: string,
input: BrowseCliMcpBridgeInput,
): Promise<{ content: Array<{ type: "text"; text: string }>; isError?: boolean }> {
if (!isAllowedBrowseCommand(command)) {
return toolError("Only browse commands are allowed for this eval harness.");
}

let args: string[];
try {
args = tokenizeBrowseCommand(command.trim());
} catch (error) {
return toolError(describeError(error));
}

input.logger.log({
category: input.logCategory,
message: `browse tool: ${command.trim()}`,
level: 2,
});

try {
const { stdout, stderr } = await execFileAsync(input.wrapperPath, args, {
cwd: input.cwd,
env: input.env,
maxBuffer: 10 * 1024 * 1024,
timeout: readPositiveIntEnv("EVAL_BROWSE_CLI_TOOL_TIMEOUT_MS", 120_000),
});
const text = stdout.trim() || stderr.trim() || "(no output)";
input.logger.log({
category: input.logCategory,
message: `browse tool completed: ${clipToolResult(text, 500)}`,
level: 2,
});
return { content: [{ type: "text", text }] };
} catch (error) {
const detail = execFailureDetail(error) || describeError(error);
input.logger.warn({
category: input.logCategory,
message: `browse tool failed (${command.trim()}): ${detail}`,
level: 1,
});
return toolError(detail);
}
}

const execFileAsync = (
file: string,
args: string[],
options: { cwd: string; env: Record<string, string>; maxBuffer: number; timeout: number },
): Promise<{ stdout: string; stderr: string }> =>
new Promise((resolve, reject) => {
execFile(file, args, { ...options, encoding: "utf8" }, (error, stdout, stderr) => {
if (error) {
reject(Object.assign(error, { stdout, stderr }));
return;
}
resolve({ stdout, stderr });
});
});

function execFailureDetail(error: unknown): string {
if (typeof error !== "object" || error === null) return "";
const { stderr, stdout } = error as { stderr?: unknown; stdout?: unknown };
const detail = [stderr, stdout]
.filter((part): part is string => typeof part === "string")
.map((part) => part.trim())
.find((part) => part.length > 0);
return detail ?? "";
}

function toolError(message: string): {
content: Array<{ type: "text"; text: string }>;
isError: true;
} {
return { content: [{ type: "text", text: message }], isError: true };
}

function describeError(error: unknown): string {
return error instanceof Error ? error.message : String(error);
}

function readPositiveIntEnv(key: string, fallback: number): number {
const parsed = Number.parseInt(process.env[key] ?? "", 10);
return Number.isFinite(parsed) && parsed > 0 ? parsed : fallback;
}
91 changes: 33 additions & 58 deletions packages/evals/framework/claudeCodeToolAdapter.ts
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@ import type { ProbeEvidence } from "stagehand-v3";
import { startAgentToolRuntime } from "./agentToolRuntime.js";
import type { BrowserSessionInfo } from "./browserSession.js";
import type { ExternalHarnessTaskPlan } from "./externalHarnessPlan.js";
import { executeCodeExposureSnippet } from "./codeExposure.js";
import { ObservationRecorder, type StepObservation } from "./observationRecorder.js";
import { resolveStartupProfile, resolveToolSurface } from "./harnesses/toolSurfaceResolution.js";

Expand Down Expand Up @@ -74,6 +75,8 @@ export interface PreparedBrowseCliHarnessAdapter {
startupProfile: StartupProfile;
cwd: string;
env: Record<string, string>;
/** Pinned `browse` wrapper inside `cwd`; also first on the adapter's PATH. */
wrapperPath: string;
promptInstructions: string;
/** browse_cli owns its daemon; the Browserbase session id is not reported. */
browserSession: BrowserSessionInfo;
Expand Down Expand Up @@ -107,7 +110,13 @@ export const CLAUDE_CODE_TOOL_SURFACES: ToolSurface[] = [
// conflicting examples in the body — at install time, so the harness ships
// one source of truth (the real, maintained browse skill) instead of a
// second copy that drifts.
const EVAL_HARNESS_ADDENDUM = `
//
// Only the invocation rule differs per harness: shell harnesses run `browse`
// from Bash, while harnesses without a shell reach the same CLI through the
// browse_cli MCP bridge. Everything else is identical, so the caller supplies
// just that paragraph.
function browseEvalHarnessAddendum(invocationRules: string): string {
return `
## Eval Harness Addendum

This skill is installed by the Stagehand eval harness, which overrides some of
Expand All @@ -118,9 +127,7 @@ the guidance below:
install/upgrade it. Never pass \`--local\`, \`--remote\`, or \`--session\` —
the harness's wrapper appends the correct environment and session flags to
every command automatically.
- Run exactly one \`browse ...\` command per Bash tool call. Shell operators
(\`|\`, \`&&\`, \`;\`, backticks, \`$()\`, and redirection) are rejected by the
harness, so chained or piped commands will fail.
${invocationRules}
- Ignore the sections below about installing \`browse\`, Browse.sh skill
discovery/installation (\`browse skills ...\`), Browserbase cloud/session/
context/extension management (\`browse cloud ...\`), Functions
Expand All @@ -132,6 +139,13 @@ the guidance below:
- When finished, report the result in the exact \`EVAL_RESULT\` format
requested by the harness prompt.
`;
}

/** Shell harnesses (claude_code, codex) invoke the pinned wrapper from Bash. */
const BASH_INVOCATION_RULES = `- Run exactly one \`browse ...\` command per Bash tool call. Shell operators
(\`|\`, \`&&\`, \`;\`, backticks, \`$()\`, and redirection) are rejected by the
harness, so chained or piped commands will fail.`;

const ALLOW_UNSANDBOXED_LOCAL_ENV = "EVAL_CLAUDE_CODE_ALLOW_UNSANDBOXED_LOCAL";
const RUN_TOOL_SERVER = AGENT_RUN_TOOL_SERVER;
const RUN_TOOL_NAME = AGENT_RUN_TOOL_NAME;
Expand Down Expand Up @@ -330,6 +344,7 @@ export async function prepareBrowseCliHarnessAdapter(
startupProfile: input.startupProfile,
cwd,
env,
wrapperPath,
promptInstructions: buildBrowseCliPromptInstructions(input.plan),
browserSession: { provider: input.environment === "BROWSERBASE" ? "browserbase" : "local" },
metadata: getBrowseCliToolMetadata(),
Expand Down Expand Up @@ -551,7 +566,7 @@ async function executeCodeExposureRunTool(input: {
}): Promise<ClaudeToolResult> {
try {
const result = await withTimeout(
executeCodeExposureSnippet(input),
executeCodeExposureSnippet({ ...input, logCategory: "claude_code" }),
readPositiveIntEnv("EVAL_CLAUDE_CODE_RUN_TOOL_TIMEOUT_MS", 60_000),
);
const text = stringifyToolResult(result);
Expand All @@ -577,54 +592,6 @@ async function executeCodeExposureRunTool(input: {
}
}

async function executeCodeExposureSnippet(input: {
code: string;
handles: Record<string, unknown>;
runToolSpec: AgentRunToolSpec;
plan: ExternalHarnessTaskPlan;
logger: EvalLogger;
}): Promise<unknown> {
const AsyncFunction = Object.getPrototypeOf(async function () {}).constructor as new (
...args: string[]
) => (...values: unknown[]) => Promise<unknown>;
// Snippet scope = the exposure's handle names plus startUrl/task/console.
// Object.keys/Object.values over the same object are guaranteed to align,
// so names — not positions — bind the values.
const fn = new AsyncFunction(
...Object.keys(input.handles),
"startUrl",
"task",
"console",
input.code,
);
return fn(
...Object.values(input.handles),
input.plan.startUrl,
{
dataset: input.plan.dataset,
id: input.plan.taskId,
startUrl: input.plan.startUrl,
instruction: input.plan.instruction,
},
buildRunToolConsole(input.logger),
);
}

function buildRunToolConsole(logger: EvalLogger): Pick<Console, "log" | "warn" | "error"> {
const write = (level: "log" | "warn" | "error", values: unknown[]) => {
logger.log({
category: "claude_code",
message: `run console.${level}: ${values.map(stringifyToolResult).join(" ")}`,
level: 1,
});
};
return {
log: (...values: unknown[]) => write("log", values),
warn: (...values: unknown[]) => write("warn", values),
error: (...values: unknown[]) => write("error", values),
};
}

function buildBrowseCliPromptInstructions(plan: ExternalHarnessTaskPlan): string {
void plan;
return [
Expand All @@ -636,14 +603,22 @@ function buildBrowseCliPromptInstructions(plan: ExternalHarnessTaskPlan): string
].join("\n");
}

/**
* The shipped browse skill with the eval-harness addendum spliced in. Shell
* harnesses write it to disk for their Skill tool; harnesses without one
* inline the same text into the agent prompt.
*/
export async function buildBrowseSkillDocument(
invocationRules: string = BASH_INVOCATION_RULES,
): Promise<string> {
const cliSkill = await fsp.readFile(BROWSE_SKILL_SOURCE, "utf8");
return insertAfterFrontmatter(cliSkill, browseEvalHarnessAddendum(invocationRules));
}

export async function installBrowseSkill(cwd: string): Promise<void> {
const targetDir = path.join(cwd, ".claude", "skills", "browse");
await fsp.mkdir(targetDir, { recursive: true });
const cliSkill = await fsp.readFile(BROWSE_SKILL_SOURCE, "utf8");
await fsp.writeFile(
path.join(targetDir, "SKILL.md"),
insertAfterFrontmatter(cliSkill, EVAL_HARNESS_ADDENDUM),
);
await fsp.writeFile(path.join(targetDir, "SKILL.md"), await buildBrowseSkillDocument());
}

// Inserts `addition` immediately after the skill's YAML frontmatter (so
Expand Down
Loading
Loading