From 79196170678db521cf2dc8de63b0ae11f8111680 Mon Sep 17 00:00:00 2001 From: RulaKhaled Date: Mon, 14 Sep 2026 22:52:12 +0300 Subject: [PATCH] test(e2e): Add a Mistral E2E app driving a live model Follows `node-eve` and `node-mastra`: the Mistral SDK talks to a real endpoint through `E2E_OPENROUTER_API_KEY` rather than a mock, and the app is marked `sentryTest.optional` so it lands on the job where that secret is set. The suite runs twice the way `node-mastra` splits dev and prod, so both instrumentation paths are covered by the same assertions: production - `dist/app.cjs`, whose Mistral, dataloader and express copies were transformed at build time by `sentryEsbuildPlugin` development - unbundled ESM behind the runtime `--import` hook What it covers: - gen_ai spans for streaming and non-streaming calls, and the shapes of `gen_ai.response.text` and `gen_ai.output.messages` - a rejected model id reaches Sentry as an error, the gen_ai span is marked errored and records nothing from a response, and the error shares its trace - a manual span nests under the request span, and the gen_ai span directly under the manual one, for both streaming and not - streams drained through `tee()` and `pipeThrough()`, which take their reader from internal slots and so produced no span before the accompanying fix - dataloader spans in the same trace as gen_ai spans, which also covers a CommonJS and an ESM-only module through the same transform in one process Assertions avoid anything a live model decides. Token counts are checked as positive numbers and response text for shape, while origin, provider, operation name, the `chat {model}` naming rule, the stream flags and every parent/child relationship are exact. `Sentry.init` registers the runtime hook unless `enableRuntimeChannelInjection` is false, which the bundled mode sets, so the production run also asserts the injected channel names are present in the built file. Without that a passing production run would not distinguish build-time instrumentation from a silent runtime fallback. OpenRouter serves an OpenAI-compatible `/v1/chat/completions`, which is what `chat.complete` and `chat.stream` post to, and the SDK's response schemas accept it: `usage` carries a `catchall` and `finish_reason` is an open enum. Co-Authored-By: Claude Opus 5 --- .../test-applications/node-mistral/.gitignore | 1 + .../test-applications/node-mistral/build.mjs | 40 ++++ .../node-mistral/package.json | 36 ++++ .../node-mistral/playwright.config.mjs | 15 ++ .../node-mistral/src/app.mjs | 180 ++++++++++++++++++ .../node-mistral/src/instrument.mjs | 24 +++ .../node-mistral/start-event-proxy.mjs | 6 + .../node-mistral/tests/ai-spans.test.ts | 65 +++++++ .../tests/co-instrumentation.test.ts | 33 ++++ .../node-mistral/tests/drain-paths.test.ts | 57 ++++++ .../node-mistral/tests/errors.test.ts | 51 +++++ .../tests/instrumentation-path.test.ts | 17 ++ .../node-mistral/tests/nesting.test.ts | 60 ++++++ .../node-mistral/tests/utils.ts | 53 ++++++ 14 files changed, 638 insertions(+) create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/.gitignore create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/build.mjs create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/package.json create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/playwright.config.mjs create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/src/app.mjs create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/src/instrument.mjs create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/start-event-proxy.mjs create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/tests/ai-spans.test.ts create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/tests/co-instrumentation.test.ts create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/tests/drain-paths.test.ts create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/tests/errors.test.ts create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/tests/instrumentation-path.test.ts create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/tests/nesting.test.ts create mode 100644 dev-packages/e2e-tests/test-applications/node-mistral/tests/utils.ts diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/.gitignore b/dev-packages/e2e-tests/test-applications/node-mistral/.gitignore new file mode 100644 index 000000000000..1521c8b7652b --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/.gitignore @@ -0,0 +1 @@ +dist diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/build.mjs b/dev-packages/e2e-tests/test-applications/node-mistral/build.mjs new file mode 100644 index 000000000000..acb282f9c706 --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/build.mjs @@ -0,0 +1,40 @@ +// Produces the prod-mode artifact: a single bundle whose `@mistralai/mistralai`, `dataloader` and +// `express` copies were transformed at build time by `sentryEsbuildPlugin`. Nothing is left for a +// runtime hook to do, which is what `enableRuntimeChannelInjection: false` in `instrument.mjs` +// asserts. +// +// `@sentry/node` stays external: the SDK is the subscriber, not a transform target, and inlining it +// would force its CommonJS `require('node:async_hooks')` through esbuild's ESM interop for no gain. +// CJS output for the same reason the `node-esbuild` app uses it. Left unminified so the injected +// snippet keeps its identifiers. +import { rmSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { sentryEsbuildPlugin } from '@sentry/node/esbuild'; +import { build } from 'esbuild'; + +const __dirname = dirname(fileURLToPath(import.meta.url)); + +rmSync(join(__dirname, 'dist'), { recursive: true, force: true }); + +await build({ + entryPoints: [join(__dirname, 'src', 'app.mjs')], + outfile: join(__dirname, 'dist', 'app.cjs'), + bundle: true, + platform: 'node', + format: 'cjs', + target: 'node18', + external: ['@sentry/node'], + minify: false, + logLevel: 'info', + plugins: [ + sentryEsbuildPlugin({ + telemetry: false, + sourcemaps: { disable: true }, + release: { create: false, finalize: false, inject: false }, + }), + ], +}); + +// eslint-disable-next-line no-console +console.log('built dist/app.cjs with sentryEsbuildPlugin'); diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/package.json b/dev-packages/e2e-tests/test-applications/node-mistral/package.json new file mode 100644 index 000000000000..16b53b297d36 --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/package.json @@ -0,0 +1,36 @@ +{ + "name": "node-mistral", + "description": "Mistral AI gen_ai spans, errors, span nesting and co-instrumented dataloader spans, exercised through both the runtime loader (dev) and a bundler-instrumented build (prod)", + "version": "1.0.0", + "private": true, + "type": "module", + "scripts": { + "start": "node --import ./src/instrument.mjs src/app.mjs", + "start:bundled": "node dist/app.cjs", + "build": "node build.mjs", + "clean": "npx rimraf node_modules dist pnpm-lock.yaml", + "test:build": "pnpm install && pnpm build", + "test:assert": "pnpm test:prod && pnpm test:dev", + "test:prod": "TEST_ENV=production playwright test", + "test:dev": "TEST_ENV=development playwright test" + }, + "dependencies": { + "@mistralai/mistralai": "^2.6.4", + "@sentry/node": "file:../../packed/sentry-node-packed.tgz", + "dataloader": "^2.2.2", + "express": "^4.21.2" + }, + "devDependencies": { + "@playwright/test": "~1.56.0", + "@sentry-internal/test-utils": "link:../../../test-utils", + "@sentry/bundler-plugins": "file:../../packed/sentry-bundler-plugins-packed.tgz", + "@sentry/core": "file:../../packed/sentry-core-packed.tgz", + "esbuild": "0.28.2" + }, + "sentryTest": { + "optional": true + }, + "volta": { + "extends": "../../package.json" + } +} diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/playwright.config.mjs b/dev-packages/e2e-tests/test-applications/node-mistral/playwright.config.mjs new file mode 100644 index 000000000000..39daff08107f --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/playwright.config.mjs @@ -0,0 +1,15 @@ +import { getPlaywrightConfig } from '@sentry-internal/test-utils'; + +// The suite runs twice, once per instrumentation path, the way `node-mastra` splits dev and prod: +// +// production - `dist/app.cjs`, whose Mistral, dataloader and express copies were transformed at +// build time by `sentryEsbuildPlugin`. `instrument.mjs` turns runtime injection off +// there, so the bundler plugin is the only thing that can have instrumented them. +// development - unbundled ESM behind the runtime `--import` hook. +const isDev = process.env.TEST_ENV === 'development'; + +const config = getPlaywrightConfig({ + startCommand: isDev ? 'pnpm start' : 'pnpm start:bundled', +}); + +export default config; diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/src/app.mjs b/dev-packages/e2e-tests/test-applications/node-mistral/src/app.mjs new file mode 100644 index 000000000000..f9e408f057b4 --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/src/app.mjs @@ -0,0 +1,180 @@ +// `instrument.mjs` is imported for its side effect in the prod bundle; in dev `--import` has already +// run it, and a second import is a no-op because ES modules are evaluated once. +import './instrument.mjs'; + +import { Mistral } from '@mistralai/mistralai'; +import * as Sentry from '@sentry/node'; +import DataLoader from 'dataloader'; +import express from 'express'; + +const apiKey = process.env.E2E_OPENROUTER_API_KEY; +if (!apiKey) { + throw new Error('E2E_OPENROUTER_API_KEY is not set'); +} + +// The Mistral SDK talks to OpenRouter rather than api.mistral.ai, so the suite needs only the one +// OpenRouter key the other AI e2e apps already use. OpenRouter serves an OpenAI-compatible +// `/v1/chat/completions`, which is the endpoint `chat.complete` and `chat.stream` post to, and the +// SDK's response schemas are lenient enough to accept it (`usage` has a `catchall`, `finish_reason` +// is an open enum). What is under test is the SDK's own code path, which is what Sentry instruments. +const client = new Mistral({ apiKey, serverURL: 'https://openrouter.ai/api' }); + +// Same model the eve and mastra apps drive through this key. The model is incidental here; the +// Mistral SDK request/response path is the thing being instrumented. +const MODEL = 'openai/gpt-4o-mini'; + +// Kept short so a live model stays cheap and quick, and so streamed responses still arrive in more +// than one chunk. +const SHORT_ANSWER = 'Answer in at most five words.'; + +const userLoader = new DataLoader(async keys => keys.map(key => ({ id: key, name: `user-${key}` }))); + +async function main() { + const port = Number(process.env.PORT ?? 3030); + const app = express(); + + app.get('/chat', async (req, res) => { + // A manual span wrapping the SDK call: the gen_ai span has to nest inside this one, and this one + // has to nest inside the auto-instrumented request span. + const answer = await Sentry.startSpan({ name: 'ai-workflow', op: 'function' }, async () => { + const completion = await client.chat.complete({ + model: MODEL, + messages: [ + { role: 'system', content: 'You are a helpful assistant used by an automated test.' }, + { role: 'user', content: `What is the capital of France? ${SHORT_ANSWER}` }, + ], + temperature: 0, + maxTokens: 32, + }); + + // A manual sibling of the gen_ai span, so the assertions can tell "child of the manual span" + // apart from "child of whatever ran last". + return Sentry.startSpan( + { name: 'post-process', op: 'function' }, + () => completion.choices?.[0]?.message?.content ?? '', + ); + }); + + res.send({ answer }); + }); + + app.get('/chat-stream', async (req, res) => { + const chunks = []; + + await Sentry.startSpan({ name: 'ai-stream-workflow', op: 'function' }, async () => { + const stream = await client.chat.stream({ + model: MODEL, + messages: [{ role: 'user', content: `Name three colours. ${SHORT_ANSWER}` }], + temperature: 0, + maxTokens: 32, + }); + + for await (const event of stream) { + const content = event.data?.choices?.[0]?.delta?.content; + if (typeof content === 'string') { + chunks.push(content); + } + } + }); + + res.send({ answer: chunks.join('') }); + }); + + // `tee()` acquires its reader through internal slots rather than the public `getReader`, so it is + // the drain path most likely to escape instrumentation. Both branches are drained so the response + // only comes back once the stream is finished. + app.get('/chat-stream-tee', async (req, res) => { + const branches = await Sentry.startSpan({ name: 'ai-tee-workflow', op: 'function' }, async () => { + const stream = await client.chat.stream({ + model: MODEL, + messages: [{ role: 'user', content: `Name three colours. ${SHORT_ANSWER}` }], + temperature: 0, + maxTokens: 32, + }); + + const [left, right] = stream.tee(); + + const drain = async branch => { + const parts = []; + for await (const event of branch) { + const content = event.data?.choices?.[0]?.delta?.content; + if (typeof content === 'string') { + parts.push(content); + } + } + return parts.join(''); + }; + + return Promise.all([drain(left), drain(right)]); + }); + + res.send({ left: branches[0], right: branches[1] }); + }); + + // Relays the stream through a transform, the shape an edge handler would use to forward tokens. + app.get('/chat-stream-pipe', async (req, res) => { + const answer = await Sentry.startSpan({ name: 'ai-pipe-workflow', op: 'function' }, async () => { + const stream = await client.chat.stream({ + model: MODEL, + messages: [{ role: 'user', content: `Name three colours. ${SHORT_ANSWER}` }], + temperature: 0, + maxTokens: 32, + }); + + const relayed = stream.pipeThrough( + new TransformStream({ + transform(event, controller) { + controller.enqueue(event.data?.choices?.[0]?.delta?.content ?? ''); + }, + }), + ); + + const parts = []; + for await (const part of relayed) { + parts.push(part); + } + return parts.join(''); + }); + + res.send({ answer }); + }); + + // A model id the upstream will reject, so the failure is a real API error rather than a simulated + // one. The caller-supplied id makes each request identifiable in the spans it produces. + app.get('/chat-error', async (req, res, next) => { + const model = `no-such-model/${req.query.id ?? 'default'}`; + + try { + await client.chat.complete({ model, messages: [{ role: 'user', content: 'This will fail' }] }); + res.send({ ok: true }); + } catch (error) { + // Rethrown through the express error handler so the SDK captures it the way a real app would. + next(new Error(`Mistral call failed for ${model}: ${error.message}`)); + } + }); + + // A dataloader (orchestrion-instrumented, like Mistral) and a Mistral call in one request, so the + // assertions can prove both sets of spans land in the same trace. + app.get('/dataloader-and-chat', async (req, res) => { + const user = await userLoader.load(`${req.query.id ?? '1'}`); + + const completion = await client.chat.complete({ + model: MODEL, + messages: [{ role: 'user', content: `Say hello to ${user.name}. ${SHORT_ANSWER}` }], + temperature: 0, + maxTokens: 32, + }); + + res.send({ user, answer: completion.choices?.[0]?.message?.content ?? '' }); + }); + + Sentry.setupExpressErrorHandler(app); + + app.use((error, req, res, _next) => { + res.status(500).send({ message: error.message }); + }); + + app.listen(port); +} + +void main(); diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/src/instrument.mjs b/dev-packages/e2e-tests/test-applications/node-mistral/src/instrument.mjs new file mode 100644 index 000000000000..30abaa918aa1 --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/src/instrument.mjs @@ -0,0 +1,24 @@ +// Shared Sentry bootstrap for both modes. +// +// dev - loaded through `node --import`, so the runtime channel-injection hook transforms +// `@mistralai/mistralai`, `dataloader` and `express` as they load. +// prod - bundled into `dist/app.cjs` by `build.mjs`, where `sentryEsbuildPlugin` applies the same +// transforms at build time. Runtime injection is switched off there so the bundler plugin is +// the only possible injector and a passing prod test really proves the build-time path. +import * as Sentry from '@sentry/node'; + +// `production` is the bundled build, where `sentryEsbuildPlugin` already injected the channels. +const isDev = process.env.TEST_ENV === 'development'; + +Sentry.init({ + environment: 'qa', + dsn: process.env.E2E_TEST_DSN, + debug: !!process.env.DEBUG, + tunnel: 'http://localhost:3031/', + tracesSampleRate: 1, + traceLifecycle: 'stream', + enableRuntimeChannelInjection: isDev, + integrations: [Sentry.spanStreamingIntegration()], +}); + +Sentry.setTag('e2e.mode', isDev ? 'development' : 'production'); diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/start-event-proxy.mjs b/dev-packages/e2e-tests/test-applications/node-mistral/start-event-proxy.mjs new file mode 100644 index 000000000000..2c8fdc947553 --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/start-event-proxy.mjs @@ -0,0 +1,6 @@ +import { startEventProxyServer } from '@sentry-internal/test-utils'; + +startEventProxyServer({ + port: 3031, + proxyServerName: 'node-mistral', +}); diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/tests/ai-spans.test.ts b/dev-packages/e2e-tests/test-applications/node-mistral/tests/ai-spans.test.ts new file mode 100644 index 000000000000..aabe2af66f5c --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/tests/ai-spans.test.ts @@ -0,0 +1,65 @@ +import { expect, test } from '@playwright/test'; +import { collectStreamedSpansUntilSegment } from '@sentry-internal/test-utils'; +import { APP, attr, expectCommonChatAttributes, isChatSpan } from './utils'; + +test('emits a gen_ai.chat span for a non-streaming call', async ({ baseURL, request }) => { + const spansPromise = collectStreamedSpansUntilSegment(APP, 'GET /chat'); + + const response = await request.get(`${baseURL}/chat`); + expect(response.status()).toBe(200); + expect((await response.json()).answer).toBeTruthy(); + + const spans = await spansPromise; + const chatSpan = spans.find(isChatSpan); + + expect(chatSpan).toBeDefined(); + expectCommonChatAttributes(chatSpan!); + expect(attr(chatSpan!, 'gen_ai.request.stream')).toBe(false); + expect(attr(chatSpan!, 'gen_ai.request.temperature')).toBe(0); + expect(attr(chatSpan!, 'gen_ai.request.max_tokens')).toBe(32); +}); + +test('emits a gen_ai.chat span for a streaming call', async ({ baseURL, request }) => { + const spansPromise = collectStreamedSpansUntilSegment(APP, 'GET /chat-stream'); + + const response = await request.get(`${baseURL}/chat-stream`); + expect(response.status()).toBe(200); + expect((await response.json()).answer).toBeTruthy(); + + const spans = await spansPromise; + const streamSpan = spans.find(isChatSpan); + + expect(streamSpan).toBeDefined(); + expectCommonChatAttributes(streamSpan!); + // Set from the called method: v2's `stream` request field is optional and the app never passes it. + expect(attr(streamSpan!, 'gen_ai.request.stream')).toBe(true); + expect(attr(streamSpan!, 'gen_ai.response.streaming')).toBe(true); +}); + +test('records inputs and outputs in the shape the gen_ai conventions specify', async ({ baseURL, request }) => { + const spansPromise = collectStreamedSpansUntilSegment(APP, 'GET /chat'); + + await request.get(`${baseURL}/chat`); + + const spans = await spansPromise; + const chatSpan = spans.find(isChatSpan)!; + + // The system message is split out from the rest of the prompt. + expect(attr(chatSpan, 'gen_ai.system_instructions')).toContain('automated test'); + expect(attr(chatSpan, 'gen_ai.input.messages')).toContain('capital of France'); + + // A stringified array of messages, not one concatenated string. + const responseText = JSON.parse(attr(chatSpan, 'gen_ai.response.text') as string); + expect(Array.isArray(responseText)).toBe(true); + expect(responseText).toHaveLength(1); + expect(typeof responseText[0]).toBe('string'); + + const outputMessages = JSON.parse(attr(chatSpan, 'gen_ai.output.messages') as string); + expect(outputMessages).toEqual([ + { + role: 'assistant', + parts: [{ type: 'text', content: expect.any(String) }], + finish_reason: expect.any(String), + }, + ]); +}); diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/tests/co-instrumentation.test.ts b/dev-packages/e2e-tests/test-applications/node-mistral/tests/co-instrumentation.test.ts new file mode 100644 index 000000000000..9888e154f7cf --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/tests/co-instrumentation.test.ts @@ -0,0 +1,33 @@ +import { expect, test } from '@playwright/test'; +import { collectStreamedSpansUntilSegment, getSpanOp } from '@sentry-internal/test-utils'; +import { APP, attr, isChatSpan } from './utils'; + +// Mistral and dataloader are both instrumented through orchestrion, so one request that touches +// both proves the Mistral channels coexist with the rest of the injected set rather than displacing +// them. dataloader is also CommonJS where Mistral is ESM-only, so this covers both module formats +// going through the same transform in one process. +test('emits dataloader spans alongside gen_ai spans in one trace', async ({ baseURL, request }) => { + const spansPromise = collectStreamedSpansUntilSegment(APP, 'GET /dataloader-and-chat'); + + const response = await request.get(`${baseURL}/dataloader-and-chat?id=7`); + expect(response.status()).toBe(200); + expect((await response.json()).user).toEqual({ id: '7', name: 'user-7' }); + + const spans = await spansPromise; + const segment = spans.find(span => span.is_segment && span.name === 'GET /dataloader-and-chat')!; + + const chatSpan = spans.find(isChatSpan); + const dataloaderSpans = spans.filter(span => attr(span, 'sentry.origin') === 'auto.db.dataloader'); + + expect(chatSpan).toBeDefined(); + expect(dataloaderSpans.length).toBeGreaterThan(0); + + // `load` is recorded as a cache read. + expect(dataloaderSpans.some(span => getSpanOp(span) === 'cache.get')).toBe(true); + + // Both instrumentations contribute to the same trace, under the same request. + expect(chatSpan!.trace_id).toBe(segment.trace_id); + for (const span of dataloaderSpans) { + expect(span.trace_id).toBe(segment.trace_id); + } +}); diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/tests/drain-paths.test.ts b/dev-packages/e2e-tests/test-applications/node-mistral/tests/drain-paths.test.ts new file mode 100644 index 000000000000..40563d376a4d --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/tests/drain-paths.test.ts @@ -0,0 +1,57 @@ +import { expect, test } from '@playwright/test'; +import { collectStreamedSpansUntilSegment } from '@sentry-internal/test-utils'; +import { APP, attr, byName, expectCommonChatAttributes, isChatSpan } from './utils'; + +// `tee`, `pipeTo` and `pipeThrough` take their reader from internal slots rather than the public +// `getReader`, so they bypass a stream instrumented only through `getReader` and the async iterator. +// These cover the two an app is realistically built on: teeing to relay and persist at once, and +// piping through a transform to forward tokens to a client. + +test('records a gen_ai span for a teed stream, once', async ({ baseURL, request }) => { + const spansPromise = collectStreamedSpansUntilSegment(APP, 'GET /chat-stream-tee'); + + const response = await request.get(`${baseURL}/chat-stream-tee`); + expect(response.status()).toBe(200); + + // Both branches receive the same stream. + const { left, right } = await response.json(); + expect(left).toBeTruthy(); + expect(left).toBe(right); + + const spans = await spansPromise; + const chatSpans = spans.filter(isChatSpan); + + // One span, not one per tee branch. + expect(chatSpans).toHaveLength(1); + const chatSpan = chatSpans[0]!; + + expectCommonChatAttributes(chatSpan); + expect(attr(chatSpan, 'gen_ai.request.stream')).toBe(true); + expect(attr(chatSpan, 'gen_ai.response.streaming')).toBe(true); + + // Nesting still holds on this drain path. + const segment = spans.find(span => span.is_segment && span.name === 'GET /chat-stream-tee')!; + const workflow = byName(spans, 'ai-tee-workflow'); + expect(chatSpan.parent_span_id).toBe(workflow.span_id); + expect(chatSpan.trace_id).toBe(segment.trace_id); +}); + +test('records a gen_ai span for a stream relayed through a transform', async ({ baseURL, request }) => { + const spansPromise = collectStreamedSpansUntilSegment(APP, 'GET /chat-stream-pipe'); + + const response = await request.get(`${baseURL}/chat-stream-pipe`); + expect(response.status()).toBe(200); + expect((await response.json()).answer).toBeTruthy(); + + const spans = await spansPromise; + const chatSpan = spans.find(isChatSpan); + + expect(chatSpan).toBeDefined(); + expectCommonChatAttributes(chatSpan!); + expect(attr(chatSpan!, 'gen_ai.response.streaming')).toBe(true); + + const segment = spans.find(span => span.is_segment && span.name === 'GET /chat-stream-pipe')!; + const workflow = byName(spans, 'ai-pipe-workflow'); + expect(chatSpan!.parent_span_id).toBe(workflow.span_id); + expect(chatSpan!.trace_id).toBe(segment.trace_id); +}); diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/tests/errors.test.ts b/dev-packages/e2e-tests/test-applications/node-mistral/tests/errors.test.ts new file mode 100644 index 000000000000..c2dc4067d9d5 --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/tests/errors.test.ts @@ -0,0 +1,51 @@ +import { expect, test } from '@playwright/test'; +import { collectStreamedSpans, waitForError } from '@sentry-internal/test-utils'; +import { APP, attr, isChatSpan } from './utils'; + +test('captures an error thrown by a failed Mistral call', async ({ baseURL, request }) => { + const model = 'no-such-model/capture'; + const errorPromise = waitForError( + APP, + event => !event.type && !!event.exception?.values?.[0]?.value?.includes(model), + ); + + const response = await request.get(`${baseURL}/chat-error?id=capture`); + expect(response.status()).toBe(500); + + const errorEvent = await errorPromise; + + expect(errorEvent.exception?.values?.[0]?.value).toContain('Mistral call failed'); + expect(errorEvent.transaction).toBe('GET /chat-error'); + expect(errorEvent.contexts?.trace?.trace_id).toMatch(/[a-f0-9]{32}/); +}); + +test('marks the gen_ai span errored and ties it to the captured error', async ({ baseURL, request }) => { + const id = 'linked'; + const model = `no-such-model/${id}`; + + const errorPromise = waitForError( + APP, + event => !event.type && !!event.exception?.values?.[0]?.value?.includes(model), + ); + // Every request to this route produces an equivalent-looking trace, so the predicate names the + // per-request model rather than the route. + const spansPromise = collectStreamedSpans( + APP, + spansOfTrace => + spansOfTrace.some(span => span.is_segment && span.name === 'GET /chat-error') && + spansOfTrace.some(span => attr(span, 'gen_ai.request.model') === model), + ); + + await request.get(`${baseURL}/chat-error?id=${id}`); + + const [errorEvent, spans] = await Promise.all([errorPromise, spansPromise]); + const chatSpan = spans.find(isChatSpan)!; + + expect(chatSpan).toBeDefined(); + expect(chatSpan.status).not.toBe('ok'); + // No response was produced, so nothing should have been recorded from one. + expect(attr(chatSpan, 'gen_ai.response.text')).toBeUndefined(); + expect(attr(chatSpan, 'gen_ai.output.messages')).toBeUndefined(); + + expect(chatSpan.trace_id).toBe(errorEvent.contexts?.trace?.trace_id); +}); diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/tests/instrumentation-path.test.ts b/dev-packages/e2e-tests/test-applications/node-mistral/tests/instrumentation-path.test.ts new file mode 100644 index 000000000000..1c53b9d332ae --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/tests/instrumentation-path.test.ts @@ -0,0 +1,17 @@ +import { readFileSync } from 'node:fs'; +import { expect, test } from '@playwright/test'; + +// Guards the premise the prod run rests on. `Sentry.init` registers the runtime injection hook +// unless `enableRuntimeChannelInjection` is false, which `instrument.mjs` sets outside dev. With +// that off and no `--import` on the bundled start command, the bundler plugin is the only thing +// that can have injected these channels, so finding them in the built file is what makes a passing +// production run mean build-time instrumentation rather than a silent fallback. +test('the bundle carries build-time injected channels', () => { + test.skip(process.env.TEST_ENV === 'development', 'the dev run is instrumented by the runtime hook'); + + const bundle = readFileSync('dist/app.cjs', 'utf8'); + + expect(bundle).toContain('orchestrion:@mistralai/mistralai:chat'); + expect(bundle).toContain('orchestrion:@mistralai/mistralai:chat-stream'); + expect(bundle).toContain('orchestrion:dataloader:load'); +}); diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/tests/nesting.test.ts b/dev-packages/e2e-tests/test-applications/node-mistral/tests/nesting.test.ts new file mode 100644 index 000000000000..8cb72ffa1728 --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/tests/nesting.test.ts @@ -0,0 +1,60 @@ +import { expect, test } from '@playwright/test'; +import { collectStreamedSpansUntilSegment } from '@sentry-internal/test-utils'; +import { ancestorIds, APP, byName, describeTree, isChatSpan } from './utils'; + +test('nests the manual span under the request span and the gen_ai span under the manual span', async ({ + baseURL, + request, +}) => { + const spansPromise = collectStreamedSpansUntilSegment(APP, 'GET /chat'); + + await request.get(`${baseURL}/chat`); + + const spans = await spansPromise; + const tree = describeTree(spans); + + const segment = spans.find(span => span.is_segment && span.name === 'GET /chat')!; + const workflow = byName(spans, 'ai-workflow'); + const postProcess = byName(spans, 'post-process'); + const chatSpan = spans.find(isChatSpan)!; + + // Manual span inside the generated request span. Express contributes its own middleware and + // request-handler spans in between, so this is an ancestry check, not a direct-parent one. + expect(ancestorIds(spans, workflow), `ai-workflow is not under the request span:\n${tree}`).toContain( + segment.span_id, + ); + + // Generated span directly inside the manual one: nothing should slip between them. + expect(chatSpan.parent_span_id, `gen_ai span is not a child of ai-workflow:\n${tree}`).toBe(workflow.span_id); + + // A second manual span, sibling of the gen_ai span rather than its child. + expect(postProcess.parent_span_id, `post-process is not a child of ai-workflow:\n${tree}`).toBe(workflow.span_id); + + for (const span of [workflow, postProcess, chatSpan]) { + expect(span.trace_id).toBe(segment.trace_id); + } +}); + +test('nests the streaming gen_ai span under its manual parent', async ({ baseURL, request }) => { + const spansPromise = collectStreamedSpansUntilSegment(APP, 'GET /chat-stream'); + + await request.get(`${baseURL}/chat-stream`); + + const spans = await spansPromise; + const tree = describeTree(spans); + + const segment = spans.find(span => span.is_segment && span.name === 'GET /chat-stream')!; + const workflow = byName(spans, 'ai-stream-workflow'); + const streamSpan = spans.find(isChatSpan)!; + + expect(ancestorIds(spans, workflow), `ai-stream-workflow is not under the request span:\n${tree}`).toContain( + segment.span_id, + ); + + // The stream is drained inside the manual span, so the gen_ai span has to close under it rather + // than escaping to the request root. + expect(streamSpan.parent_span_id, `streamed gen_ai span is not a child of ai-stream-workflow:\n${tree}`).toBe( + workflow.span_id, + ); + expect(streamSpan.trace_id).toBe(segment.trace_id); +}); diff --git a/dev-packages/e2e-tests/test-applications/node-mistral/tests/utils.ts b/dev-packages/e2e-tests/test-applications/node-mistral/tests/utils.ts new file mode 100644 index 000000000000..5172a74beb54 --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-mistral/tests/utils.ts @@ -0,0 +1,53 @@ +import { expect } from '@playwright/test'; +import type { SerializedStreamedSpan } from '@sentry-internal/test-utils'; +import { getSpanOp } from '@sentry-internal/test-utils'; + +export const APP = 'node-mistral'; + +export const attr = (span: SerializedStreamedSpan, key: string): unknown => span.attributes?.[key]?.value; + +export const isChatSpan = (span: SerializedStreamedSpan): boolean => getSpanOp(span) === 'gen_ai.chat'; + +/** A readable span tree, used as a failure message so a broken assertion is diagnosable. */ +export function describeTree(spans: SerializedStreamedSpan[]): string { + return spans + .map(span => `${span.name} [${getSpanOp(span) ?? '-'}] id=${span.span_id} parent=${span.parent_span_id ?? '-'}`) + .join('\n'); +} + +/** Walk to the trace root, so assertions can allow auto-instrumented spans in between. */ +export function ancestorIds(spans: SerializedStreamedSpan[], span: SerializedStreamedSpan): string[] { + const byId = new Map(spans.map(candidate => [candidate.span_id, candidate])); + const ids: string[] = []; + + let current: SerializedStreamedSpan | undefined = span; + while (current?.parent_span_id) { + ids.push(current.parent_span_id); + current = byId.get(current.parent_span_id); + } + + return ids; +} + +export function byName(spans: SerializedStreamedSpan[], name: string): SerializedStreamedSpan { + const span = spans.find(candidate => candidate.name === name); + expect(span, `expected a span named "${name}" in:\n${describeTree(spans)}`).toBeDefined(); + return span!; +} + +/** + * Attributes every successful gen_ai span carries, whatever the model happens to answer. Values that + * depend on the model (token counts, response text) are checked for shape and not for content. + */ +export function expectCommonChatAttributes(span: SerializedStreamedSpan): void { + expect(attr(span, 'sentry.origin')).toBe('auto.ai.mistralai'); + expect(attr(span, 'gen_ai.provider.name')).toBe('mistralai'); + expect(attr(span, 'gen_ai.operation.name')).toBe('chat'); + expect(span.name).toBe(`chat ${attr(span, 'gen_ai.request.model')}`); + expect(span.status).toBe('ok'); + + expect(typeof attr(span, 'gen_ai.response.model')).toBe('string'); + expect(attr(span, 'gen_ai.usage.input_tokens')).toBeGreaterThan(0); + expect(attr(span, 'gen_ai.usage.output_tokens')).toBeGreaterThan(0); + expect(attr(span, 'gen_ai.usage.total_tokens')).toBeGreaterThan(0); +}