Skip to content

Commit 43e293e

Browse files
nicohrubeccodex
andauthored
fix(server-utils): Preserve raw Groq and Together response bodies (#24906)
Groq and Together automatic instrumentation could consume `.asResponse()` bodies before application code read them. Reuse the shared API-promise observer to preserve raw chat, streaming, and embeddings responses while completing their spans. Stacked on #24902. Part of #24882. Co-authored-by: GPT-6 <codex@openai.com>
1 parent d2d1dcf commit 43e293e

6 files changed

Lines changed: 146 additions & 9 deletions

File tree

‎dev-packages/node-integration-tests/suites/tracing/groq/scenario.mjs‎

Lines changed: 25 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -130,6 +130,31 @@ async function run() {
130130
model: 'nomic-embed-text-v1_5',
131131
input: 'Embedding test!',
132132
});
133+
134+
const rawChatResponse = await client.chat.completions
135+
.create({
136+
model: 'llama-3.3-70b-versatile',
137+
messages: [{ role: 'user', content: 'Raw response test!' }],
138+
})
139+
.asResponse();
140+
await rawChatResponse.json();
141+
142+
const rawStreamResponse = await client.chat.completions
143+
.create({
144+
model: 'llama-3.1-8b-instant',
145+
messages: [{ role: 'user', content: 'Raw stream test!' }],
146+
stream: true,
147+
})
148+
.asResponse();
149+
await rawStreamResponse.text();
150+
151+
const rawEmbeddingsResponse = await client.embeddings
152+
.create({
153+
model: 'nomic-embed-text-v1_5',
154+
input: 'Raw embedding test!',
155+
})
156+
.asResponse();
157+
await rawEmbeddingsResponse.json();
133158
});
134159

135160
await Sentry.flush(2000);

‎dev-packages/node-integration-tests/suites/tracing/groq/test.ts‎

Lines changed: 39 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -74,6 +74,25 @@ describe('Groq integration', () => {
7474
expect(embeddingsSpan!.attributes[GEN_AI_PROVIDER_NAME]?.value).toBe(PROVIDER);
7575
expect(embeddingsSpan!.attributes[GEN_AI_USAGE_INPUT_TOKENS]?.value).toBe(8);
7676
expect(embeddingsSpan!.attributes[GEN_AI_EMBEDDINGS_INPUT]).toBeUndefined();
77+
78+
const rawSpans = container.items.filter(
79+
s =>
80+
s.attributes[GEN_AI_PROVIDER_NAME]?.value === PROVIDER &&
81+
s.status === 'ok' &&
82+
s.attributes[GEN_AI_RESPONSE_ID] === undefined,
83+
);
84+
expect(rawSpans).toHaveLength(3);
85+
expect(rawSpans.map(s => s.name)).toEqual([
86+
'chat llama-3.3-70b-versatile',
87+
'chat llama-3.1-8b-instant',
88+
'embeddings nomic-embed-text-v1_5',
89+
]);
90+
for (const span of rawSpans) {
91+
expect(span.attributes[SEMANTIC_ATTRIBUTE_SENTRY_ORIGIN]?.value).toBe(ORIGIN);
92+
expect(span.attributes[GEN_AI_RESPONSE_MODEL]).toBeUndefined();
93+
expect(span.attributes[GEN_AI_USAGE_TOTAL_TOKENS]).toBeUndefined();
94+
expect(span.attributes[GEN_AI_RESPONSE_TEXT]).toBeUndefined();
95+
}
7796
},
7897
})
7998
.start()
@@ -100,6 +119,26 @@ describe('Groq integration', () => {
100119
);
101120
expect(embeddingsSpan).toBeDefined();
102121
expect(embeddingsSpan!.attributes[GEN_AI_EMBEDDINGS_INPUT]?.value).toContain('Embedding test!');
122+
123+
const rawChatSpan = container.items.find(
124+
s =>
125+
s.attributes[GEN_AI_REQUEST_MODEL]?.value === 'llama-3.3-70b-versatile' &&
126+
s.attributes[GEN_AI_RESPONSE_ID] === undefined,
127+
);
128+
expect(rawChatSpan).toBeDefined();
129+
expect(rawChatSpan!.attributes[GEN_AI_INPUT_MESSAGES]?.value).toBe(
130+
'[{"role":"user","content":"Raw response test!"}]',
131+
);
132+
expect(rawChatSpan!.attributes[GEN_AI_RESPONSE_TEXT]).toBeUndefined();
133+
134+
const rawEmbeddingsSpan = container.items.find(
135+
s =>
136+
s.attributes[GEN_AI_REQUEST_MODEL]?.value === 'nomic-embed-text-v1_5' &&
137+
s.attributes[GEN_AI_RESPONSE_ID] === undefined,
138+
);
139+
expect(rawEmbeddingsSpan).toBeDefined();
140+
expect(rawEmbeddingsSpan!.attributes[GEN_AI_EMBEDDINGS_INPUT]?.value).toContain('Raw embedding test!');
141+
expect(rawEmbeddingsSpan!.attributes[GEN_AI_USAGE_INPUT_TOKENS]).toBeUndefined();
103142
},
104143
})
105144
.start()

‎dev-packages/node-integration-tests/suites/tracing/together-ai/scenario.mjs‎

Lines changed: 25 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -129,6 +129,31 @@ async function run() {
129129
model: 'togethercomputer/m2-bert-80M-8k-retrieval',
130130
input: 'Embedding test!',
131131
});
132+
133+
const rawChatResponse = await client.chat.completions
134+
.create({
135+
model: 'meta-llama/Llama-3.3-70B-Instruct-Turbo',
136+
messages: [{ role: 'user', content: 'Raw response test!' }],
137+
})
138+
.asResponse();
139+
await rawChatResponse.json();
140+
141+
const rawStreamResponse = await client.chat.completions
142+
.create({
143+
model: 'meta-llama/Llama-3.1-8B-Instruct-Turbo',
144+
messages: [{ role: 'user', content: 'Raw stream test!' }],
145+
stream: true,
146+
})
147+
.asResponse();
148+
await rawStreamResponse.text();
149+
150+
const rawEmbeddingsResponse = await client.embeddings
151+
.create({
152+
model: 'togethercomputer/m2-bert-80M-8k-retrieval',
153+
input: 'Raw embedding test!',
154+
})
155+
.asResponse();
156+
await rawEmbeddingsResponse.json();
132157
});
133158

134159
await Sentry.flush(2000);

‎dev-packages/node-integration-tests/suites/tracing/together-ai/test.ts‎

Lines changed: 39 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -74,6 +74,25 @@ describe('Together integration', () => {
7474
expect(embeddingsSpan!.attributes[GEN_AI_PROVIDER_NAME]?.value).toBe(PROVIDER);
7575
expect(embeddingsSpan!.attributes[GEN_AI_USAGE_INPUT_TOKENS]?.value).toBe(8);
7676
expect(embeddingsSpan!.attributes[GEN_AI_EMBEDDINGS_INPUT]).toBeUndefined();
77+
78+
const rawSpans = container.items.filter(
79+
s =>
80+
s.attributes[GEN_AI_PROVIDER_NAME]?.value === PROVIDER &&
81+
s.status === 'ok' &&
82+
s.attributes[GEN_AI_RESPONSE_ID] === undefined,
83+
);
84+
expect(rawSpans).toHaveLength(3);
85+
expect(rawSpans.map(s => s.name)).toEqual([
86+
'chat meta-llama/Llama-3.3-70B-Instruct-Turbo',
87+
'chat meta-llama/Llama-3.1-8B-Instruct-Turbo',
88+
'embeddings togethercomputer/m2-bert-80M-8k-retrieval',
89+
]);
90+
for (const span of rawSpans) {
91+
expect(span.attributes[SEMANTIC_ATTRIBUTE_SENTRY_ORIGIN]?.value).toBe(ORIGIN);
92+
expect(span.attributes[GEN_AI_RESPONSE_MODEL]).toBeUndefined();
93+
expect(span.attributes[GEN_AI_USAGE_TOTAL_TOKENS]).toBeUndefined();
94+
expect(span.attributes[GEN_AI_RESPONSE_TEXT]).toBeUndefined();
95+
}
7796
},
7897
})
7998
.start()
@@ -100,6 +119,26 @@ describe('Together integration', () => {
100119
);
101120
expect(embeddingsSpan).toBeDefined();
102121
expect(embeddingsSpan!.attributes[GEN_AI_EMBEDDINGS_INPUT]?.value).toContain('Embedding test!');
122+
123+
const rawChatSpan = container.items.find(
124+
s =>
125+
s.attributes[GEN_AI_REQUEST_MODEL]?.value === 'meta-llama/Llama-3.3-70B-Instruct-Turbo' &&
126+
s.attributes[GEN_AI_RESPONSE_ID] === undefined,
127+
);
128+
expect(rawChatSpan).toBeDefined();
129+
expect(rawChatSpan!.attributes[GEN_AI_INPUT_MESSAGES]?.value).toBe(
130+
'[{"role":"user","content":"Raw response test!"}]',
131+
);
132+
expect(rawChatSpan!.attributes[GEN_AI_RESPONSE_TEXT]).toBeUndefined();
133+
134+
const rawEmbeddingsSpan = container.items.find(
135+
s =>
136+
s.attributes[GEN_AI_REQUEST_MODEL]?.value === 'togethercomputer/m2-bert-80M-8k-retrieval' &&
137+
s.attributes[GEN_AI_RESPONSE_ID] === undefined,
138+
);
139+
expect(rawEmbeddingsSpan).toBeDefined();
140+
expect(rawEmbeddingsSpan!.attributes[GEN_AI_EMBEDDINGS_INPUT]?.value).toContain('Raw embedding test!');
141+
expect(rawEmbeddingsSpan!.attributes[GEN_AI_USAGE_INPUT_TOKENS]).toBeUndefined();
103142
},
104143
})
105144
.start()

‎packages/server-utils/src/integrations/openai-compatible.ts‎

Lines changed: 14 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -9,6 +9,7 @@ import {
99
} from '@sentry/core';
1010
import { GEN_AI_PROVIDER_NAME } from '@sentry/conventions/attributes';
1111
import { getGenAiSpanOp, resolveAIRecordingOptions } from '../ai/core/utils';
12+
import { onApiPromiseResponse } from '../ai/core/apiPromise';
1213
import { addRequestAttributes, extractRequestAttributes } from '../ai/openai';
1314
import { instrumentStream } from '../ai/openai/streaming';
1415
import type { OpenAiOptions } from '../ai/openai/types';
@@ -35,8 +36,7 @@ export interface OpenAiCompatibleProvider {
3536

3637
/**
3738
* The context orchestrion shares across the tracing-channel lifecycle hooks: `arguments` is the live args
38-
* array passed to `Completions.create(body, options)`, and Node's `tracingChannel` attaches `result` when
39-
* the returned promise settles.
39+
* array passed to `Completions.create(body, options)`, and `result` holds its returned `APIPromise`.
4040
*/
4141
interface OpenAiCompatibleChannelContext {
4242
arguments: unknown[];
@@ -67,8 +67,17 @@ export function createOpenAiCompatibleIntegration<T extends OpenAiCompatibleProv
6767
beforeSpanEnd: (span, data) => {
6868
addResponseAttributes(span, data.result, resolveAIRecordingOptions(options).recordOutputs);
6969
},
70-
// Streaming: the result is a `Stream` consumed later, so instrument it and let it end the span.
71-
deferSpanEnd: ({ span, data }) => wrapStreamResult(span, data, options),
70+
deferSpanEnd: ({ span, data, end }) =>
71+
onApiPromiseResponse(
72+
data.result,
73+
response => {
74+
data.result = response;
75+
if (!wrapStreamResult(span, data, options)) {
76+
end();
77+
}
78+
},
79+
end,
80+
) || wrapStreamResult(span, data, options),
7281
},
7382
);
7483
}
@@ -135,7 +144,7 @@ function isAsyncIterable(value: unknown): value is AsyncIterableStream {
135144
/**
136145
* For a streaming `create({ stream: true })` the result is a `Stream` the caller consumes later. We can't
137146
* swap what `create` returns, but the `Stream` in `data.result` is the same instance the caller holds and
138-
* `asyncEnd` fires before the caller iterates — so we patch its async iterator in place to run through
147+
* we observe it before the caller iterates — so we patch its async iterator in place to run through
139148
* `instrumentStream`, which accumulates the streamed attributes and ends the span when iteration finishes.
140149
* Only a streaming call resolves to an async-iterable, so that check alone distinguishes it. Returns `true`
141150
* to hand span-ending ownership to `instrumentStream`; `false` for non-streaming/errored results, which end

‎packages/server-utils/src/orchestrion/config/openai-compatible.ts‎

Lines changed: 4 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -4,8 +4,8 @@ import type { InstrumentationConfig } from '../apmTypes';
44
* Many providers (Groq, Together, ...) ship a Stainless-generated SDK that mirrors the classic `openai`
55
* file layout: `resources/chat/completions.{js,mjs}` (class `Completions`) and
66
* `resources/embeddings.{js,mjs}` (class `Embeddings`), each with a `create(body, options)` that returns
7-
* a thenable `APIPromise` — so `kind: 'Auto'` resolves to `wrapPromise`, and a streaming call resolves to
8-
* the same OpenAI-style async-iterable `Stream` the openai integration already knows how to consume. These
7+
* a thenable `APIPromise`. Sync tracing preserves its lazy body parsing; Auto would trigger it via `.then()`.
8+
* A streaming call resolves to the same OpenAI-style async-iterable `Stream` the openai integration consumes. These
99
* SDKs ship dual CJS/ESM and the matcher compares `filePath` exactly, hence one entry per built file.
1010
*
1111
* Because the wire format is OpenAI-compatible, the span building, streaming and response parsing are all
@@ -16,12 +16,12 @@ export function openAiCompatibleConfig(module: { name: string; versionRange: str
1616
...['resources/chat/completions.js', 'resources/chat/completions.mjs'].map(filePath => ({
1717
channelName: 'chat',
1818
module: { ...module, filePath },
19-
functionQuery: { className: 'Completions', methodName: 'create', kind: 'Auto' as const },
19+
functionQuery: { className: 'Completions', methodName: 'create', kind: 'Sync' as const },
2020
})),
2121
...['resources/embeddings.js', 'resources/embeddings.mjs'].map(filePath => ({
2222
channelName: 'embeddings',
2323
module: { ...module, filePath },
24-
functionQuery: { className: 'Embeddings', methodName: 'create', kind: 'Auto' as const },
24+
functionQuery: { className: 'Embeddings', methodName: 'create', kind: 'Sync' as const },
2525
})),
2626
];
2727
}

0 commit comments

Comments
 (0)