Skip to content

Commit 04fa17a

Browse files
andreiborzaclaude
andauthored
feat(server-utils): Instrument Vercel AI experimental_evaluate (#24694)
## What Vercel AI `experimental_evaluate` calls now create a `gen_ai.evaluate` span with the state and questions as input messages and the answers as output messages. ## Why AI SDK 7 added evaluation (for example TypeSafe's Jev through AI Gateway), and we did not capture these calls. Closes: #24692 --------- Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
1 parent 2a58c97 commit 04fa17a

4 files changed

Lines changed: 214 additions & 8 deletions

File tree

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,11 @@
1+
import * as Sentry from '@sentry/node';
2+
import { loggingTransport } from '@sentry-internal/node-integration-tests';
3+
4+
Sentry.init({
5+
dsn: 'https://public@dsn.ingest.sentry.io/1337',
6+
release: '1.0',
7+
tracesSampleRate: 1.0,
8+
// `NO_RECORDING` turns off recording of inputs and outputs for the privacy test.
9+
dataCollection: process.env.NO_RECORDING ? { genAI: { inputs: false, outputs: false } } : {},
10+
transport: loggingTransport,
11+
});
Lines changed: 43 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,43 @@
1+
import * as Sentry from '@sentry/node';
2+
import { experimental_evaluate } from 'ai';
3+
import { Experimental_EvaluationMockModelV4 } from 'ai/test';
4+
5+
async function run() {
6+
await Sentry.startSpan({ op: 'function', name: 'main' }, async () => {
7+
await experimental_evaluate({
8+
model: new Experimental_EvaluationMockModelV4({
9+
provider: 'gateway',
10+
modelId: 'typesafe-ai/jev',
11+
doEvaluate: async () => ({
12+
answers: {
13+
authIssue: { type: 'boolean', probability: 0.97 },
14+
department: {
15+
type: 'choice',
16+
choice: 'billing',
17+
probabilities: { billing: 0.64, technical: 0.36 },
18+
},
19+
wantsRefund: { type: 'boolean', probability: 0.99 },
20+
urgency: { type: 'score', score: 1.8, probabilities: { 0: 0, 1: 0.2, 2: 0.8 } },
21+
},
22+
usage: { inputTokens: 275, outputTokens: 20 },
23+
// What the AI SDK TypeSafe provider returns: answer confidence moves into provider metadata.
24+
providerMetadata: { typesafe: { confidence: { department: 0.28 } } },
25+
warnings: [],
26+
}),
27+
}),
28+
state: 'I cannot log in, and I also want a refund for last month.',
29+
questions: {
30+
authIssue: { type: 'boolean', instructions: 'Is there a login problem?' },
31+
department: {
32+
type: 'choice',
33+
instructions: 'Which team should handle this?',
34+
criteria: { billing: 'Charges and refunds', technical: 'Bugs and outages' },
35+
},
36+
wantsRefund: { type: 'boolean', instructions: 'Is a refund requested?' },
37+
urgency: { type: 'score', instructions: 'How urgent is this ticket?', criteria: ['low', 'medium', 'high'] },
38+
},
39+
});
40+
});
41+
}
42+
43+
run();

‎dev-packages/node-integration-tests/suites/tracing/vercelai/v6_v7/test.ts‎

Lines changed: 103 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -4,6 +4,7 @@ import {
44
GEN_AI_CONVERSATION_ID,
55
GEN_AI_EMBEDDINGS_INPUT,
66
GEN_AI_INPUT_MESSAGES,
7+
GEN_AI_OPERATION_NAME,
78
GEN_AI_OUTPUT_MESSAGES,
89
GEN_AI_PROVIDER_NAME,
910
GEN_AI_REQUEST_MODEL,
@@ -997,3 +998,105 @@ describe.each(matrix)('Vercel AI integration (version %s)', (version, vercelAiVe
997998
},
998999
);
9991000
});
1001+
1002+
describe('Vercel AI integration experimental_evaluate', () => {
1003+
afterAll(() => {
1004+
cleanupChildProcesses();
1005+
});
1006+
1007+
createEsmTests(
1008+
__dirname,
1009+
'scenario-evaluate.mjs',
1010+
'instrument-evaluate.mjs',
1011+
(createRunner, test) => {
1012+
test('creates an evaluate span', async () => {
1013+
await createRunner()
1014+
.unordered()
1015+
.expect({
1016+
span: container => {
1017+
const evaluateSpan = container.items.find(
1018+
span => span.attributes['sentry.op']?.value === 'gen_ai.evaluate',
1019+
)!;
1020+
expect(evaluateSpan).toBeDefined();
1021+
expect(evaluateSpan.name).toBe('evaluate typesafe-ai/jev');
1022+
expect(evaluateSpan.status).toBe('ok');
1023+
expect(evaluateSpan.attributes['sentry.origin']?.value).toBe('auto.vercelai.channel');
1024+
expect(evaluateSpan.attributes[GEN_AI_OPERATION_NAME]?.value).toBe('evaluate');
1025+
expect(evaluateSpan.attributes[GEN_AI_PROVIDER_NAME]?.value).toBe('gateway');
1026+
expect(evaluateSpan.attributes[GEN_AI_REQUEST_MODEL]?.value).toBe('typesafe-ai/jev');
1027+
expect(evaluateSpan.attributes[GEN_AI_RESPONSE_MODEL]?.value).toBe('typesafe-ai/jev');
1028+
expect(evaluateSpan.attributes[GEN_AI_USAGE_INPUT_TOKENS]?.value).toBe(275);
1029+
expect(evaluateSpan.attributes[GEN_AI_USAGE_OUTPUT_TOKENS]?.value).toBe(20);
1030+
expect(evaluateSpan.attributes[GEN_AI_USAGE_TOTAL_TOKENS]?.value).toBe(295);
1031+
expect(JSON.parse(evaluateSpan.attributes[GEN_AI_INPUT_MESSAGES]?.value as string)).toEqual([
1032+
{
1033+
type: 'evaluation',
1034+
state: 'I cannot log in, and I also want a refund for last month.',
1035+
questions: {
1036+
authIssue: { type: 'boolean', instructions: 'Is there a login problem?' },
1037+
department: {
1038+
type: 'choice',
1039+
instructions: 'Which team should handle this?',
1040+
criteria: { billing: 'Charges and refunds', technical: 'Bugs and outages' },
1041+
},
1042+
wantsRefund: { type: 'boolean', instructions: 'Is a refund requested?' },
1043+
urgency: {
1044+
type: 'score',
1045+
instructions: 'How urgent is this ticket?',
1046+
criteria: ['low', 'medium', 'high'],
1047+
},
1048+
},
1049+
},
1050+
]);
1051+
expect(JSON.parse(evaluateSpan.attributes[GEN_AI_OUTPUT_MESSAGES]?.value as string)).toEqual([
1052+
{
1053+
type: 'evaluation',
1054+
answers: {
1055+
authIssue: { type: 'boolean', probability: 0.97 },
1056+
department: {
1057+
type: 'choice',
1058+
choice: 'billing',
1059+
probabilities: { billing: 0.64, technical: 0.36 },
1060+
confidence: 0.28,
1061+
},
1062+
wantsRefund: { type: 'boolean', probability: 0.99 },
1063+
urgency: { type: 'score', score: 1.8, probabilities: { 0: 0, 1: 0.2, 2: 0.8 } },
1064+
},
1065+
},
1066+
]);
1067+
},
1068+
})
1069+
.start()
1070+
.completed();
1071+
});
1072+
1073+
test('does not record inputs or outputs when recording is off', async () => {
1074+
await createRunner()
1075+
.withEnv({ NO_RECORDING: 'true' })
1076+
.unordered()
1077+
.expect({
1078+
span: container => {
1079+
const evaluateSpan = container.items.find(
1080+
span => span.attributes['sentry.op']?.value === 'gen_ai.evaluate',
1081+
)!;
1082+
expect(evaluateSpan).toBeDefined();
1083+
expect(evaluateSpan.attributes[GEN_AI_INPUT_MESSAGES]).toBeUndefined();
1084+
expect(evaluateSpan.attributes[GEN_AI_OUTPUT_MESSAGES]).toBeUndefined();
1085+
// State, questions and answers must not come back through another attribute. Only the attributes
1086+
// are checked (timestamps could match a number), and `probabilities` only occurs in answers.
1087+
expect(JSON.stringify(evaluateSpan.attributes)).not.toMatch(
1088+
/cannot log in|Charges and refunds|probabilities/,
1089+
);
1090+
},
1091+
})
1092+
.start()
1093+
.completed();
1094+
});
1095+
},
1096+
{
1097+
additionalDependencies: {
1098+
ai: '^7.0.111',
1099+
},
1100+
},
1101+
);
1102+
});

‎packages/server-utils/src/integrations/vercel-ai/vercel-ai-dc-subscriber.ts‎

Lines changed: 57 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -65,10 +65,14 @@ const AI_SDK_TELEMETRY_TRACING_CHANNEL = 'ai:telemetry';
6565

6666
const ORIGIN = 'auto.vercelai.channel';
6767

68+
// Not yet in `@sentry/conventions`.
69+
const GEN_AI_EVALUATE = 'gen_ai.evaluate';
70+
6871
// `gen_ai.operation.name` values, keyed to the span op they map to.
6972
const GEN_AI_OPERATION_SPAN_OPS = {
7073
embeddings: GEN_AI_EMBEDDINGS,
7174
rerank: GEN_AI_RERANK,
75+
evaluate: GEN_AI_EVALUATE,
7276
invoke_agent: GEN_AI_INVOKE_AGENT,
7377
execute_tool: GEN_AI_EXECUTE_TOOL,
7478
// The model-call op matches the Vercel AI OTel integration (`gen_ai.generate_content`) rather than
@@ -211,7 +215,8 @@ export type ChannelEventType =
211215
| 'executeTool'
212216
| 'embed'
213217
| 'embedMany'
214-
| 'rerank';
218+
| 'rerank'
219+
| 'experimental_evaluate';
215220

216221
/**
217222
* The context object the AI SDK passes through one tracing-channel call. It is the same object
@@ -444,6 +449,17 @@ export function createSpanFromMessage(
444449
}
445450
case 'rerank':
446451
return startGenAiSpan('rerank', modelId, baseAttributes);
452+
case 'experimental_evaluate':
453+
return startGenAiSpan('evaluate', modelId, {
454+
...baseAttributes,
455+
...(recordInputs
456+
? {
457+
[GEN_AI_INPUT_MESSAGES]: stringify([
458+
{ type: 'evaluation', state: event.state, questions: event.questions },
459+
]),
460+
}
461+
: {}),
462+
});
447463
default:
448464
// Unknown event type: opt out rather than open a span we can't shape correctly.
449465
return undefined;
@@ -608,19 +624,52 @@ export function enrichSpanOnEnd(
608624
span.setAttributes(providerAttributes);
609625

610626
if (recordOutputs) {
611-
// `languageModelCall` exposes the response as a `content` parts array; top-level results expose
612-
// `text` + `toolCalls`. Both normalize into the OTel `gen_ai.output.messages` assistant message.
613-
const parts =
614-
type === 'languageModelCall' && Array.isArray(result.content)
615-
? partsFromContent(result.content)
616-
: partsFromTextAndToolCalls(result.text, result.toolCalls);
617-
const outputMessages = buildOutputMessages(parts, finishReason);
627+
const outputMessages = getOutputMessages(type, result, finishReason);
618628
if (outputMessages) {
619629
span.setAttribute(GEN_AI_OUTPUT_MESSAGES, outputMessages);
620630
}
621631
}
622632
}
623633

634+
function getOutputMessages(
635+
type: ChannelEventType,
636+
result: Record<string, unknown>,
637+
finishReason: string | undefined,
638+
): string | undefined {
639+
if (type === 'experimental_evaluate') {
640+
return stringify([{ type: 'evaluation', answers: withProviderConfidence(result) }]);
641+
}
642+
// `languageModelCall` exposes the response as a `content` parts array; top-level results expose
643+
// `text` + `toolCalls`. Both normalize into the OTel `gen_ai.output.messages` assistant message.
644+
const parts =
645+
type === 'languageModelCall' && Array.isArray(result.content)
646+
? partsFromContent(result.content)
647+
: partsFromTextAndToolCalls(result.text, result.toolCalls);
648+
return buildOutputMessages(parts, finishReason);
649+
}
650+
651+
/**
652+
* The AI SDK TypeSafe provider moves each answer's `confidence` out of the answers into
653+
* `providerMetadata.typesafe.confidence` (keyed by question id). Put it back so evaluate answers keep it.
654+
*/
655+
function withProviderConfidence(result: Record<string, unknown>): unknown {
656+
const { answers, providerMetadata } = result;
657+
const typesafe = isObjectLike(providerMetadata) ? providerMetadata.typesafe : undefined;
658+
const confidence = isObjectLike(typesafe) && isObjectLike(typesafe.confidence) ? typesafe.confidence : undefined;
659+
if (!confidence || !isObjectLike(answers)) {
660+
return answers;
661+
}
662+
663+
return Object.fromEntries(
664+
Object.entries(answers).map(([id, answer]) => [
665+
id,
666+
isObjectLike(answer) && typeof confidence[id] === 'number' && answer.confidence === undefined
667+
? { ...answer, confidence: confidence[id] }
668+
: answer,
669+
]),
670+
);
671+
}
672+
624673
/** Maps a Vercel AI finish reason to the OTel `gen_ai.output.messages` form (`tool-calls` → `tool_call`). */
625674
function normalizeFinishReason(finishReason: string | undefined): string {
626675
return finishReason === 'tool-calls' ? 'tool_call' : (finishReason ?? 'stop');

0 commit comments

Comments
 (0)