11import { expect , test } from '@playwright/test' ;
2+ import {
3+ GEN_AI_CONVERSATION_ID ,
4+ GEN_AI_PROVIDER_NAME ,
5+ GEN_AI_REQUEST_MODEL ,
6+ GEN_AI_RESPONSE_FINISH_REASONS ,
7+ GEN_AI_TOOL_CALL_ARGUMENTS ,
8+ GEN_AI_TOOL_CALL_RESULT ,
9+ GEN_AI_TOOL_NAME ,
10+ GEN_AI_USAGE_INPUT_TOKENS ,
11+ GEN_AI_USAGE_OUTPUT_TOKENS ,
12+ SENTRY_ORIGIN ,
13+ SERVER_ADDRESS ,
14+ } from '@sentry/conventions/attributes' ;
15+ import { DB_QUERY , GEN_AI_CHAT , GEN_AI_INVOKE_AGENT , HTTP_CLIENT , HTTP_SERVER } from '@sentry/conventions/op' ;
216import { collectStreamedSpans , getSpanOp , waitForError } from '@sentry-internal/test-utils' ;
317import { newAgentId , prompt , submitPrompt , waitForOperation } from './utils' ;
418
@@ -9,8 +23,8 @@ test('traces a PiHarness prompt as invoke_agent with chat, execute_tool and prov
923 const spansPromise = collectStreamedSpans (
1024 APP ,
1125 spansOfTrace =>
12- spansOfTrace . some ( span => span . is_segment && getSpanOp ( span ) === 'gen_ai.invoke_agent' ) &&
13- spansOfTrace . some ( span => span . attributes [ 'gen_ai.tool.name' ] ?. value === 'get_weather' ) ,
26+ spansOfTrace . some ( span => span . is_segment && getSpanOp ( span ) === GEN_AI_INVOKE_AGENT ) &&
27+ spansOfTrace . some ( span => span . attributes [ GEN_AI_TOOL_NAME ] ?. value === 'get_weather' ) ,
1428 ) ;
1529
1630 const answer = await prompt (
@@ -22,52 +36,52 @@ test('traces a PiHarness prompt as invoke_agent with chat, execute_tool and prov
2236
2337 const spans = await spansPromise ;
2438 const agent = spans . find ( span => span . is_segment ) ! ;
25- const chats = spans . filter ( span => getSpanOp ( span ) === 'gen_ai.chat' ) ;
39+ const chats = spans . filter ( span => getSpanOp ( span ) === GEN_AI_CHAT ) ;
2640 // The manual span of the tool is the one way to pick the call that ran, should the model call twice.
2741 const manualSpan = spans . find ( span => span . name === 'resolve-weather' ) ! ;
2842 const tool = spans . find ( span => span . span_id === manualSpan . parent_span_id ) ! ;
29- const providerCalls = spans . filter ( span => getSpanOp ( span ) === 'http.client' ) ;
43+ const providerCalls = spans . filter ( span => getSpanOp ( span ) === HTTP_CLIENT ) ;
3044
31- expect ( agent . attributes [ 'sentry.origin' ] ?. value ) . toBe ( 'auto.ai.pi_durable' ) ;
45+ expect ( agent . attributes [ SENTRY_ORIGIN ] ?. value ) . toBe ( 'auto.ai.pi_durable' ) ;
3246 // The PiHarness root session is pi conversation 1, in a Harness of its own per Durable Object.
33- expect ( agent . attributes [ 'gen_ai.conversation.id' ] ?. value ) . toMatch ( / ^ [ 0 - 9 a - f ] { 32 } : 1 $ / ) ;
47+ expect ( agent . attributes [ GEN_AI_CONVERSATION_ID ] ?. value ) . toMatch ( / ^ [ 0 - 9 a - f ] { 32 } : 1 $ / ) ;
3448 // The request that submitted the prompt is a trace of its own: the scheduler runs the work.
35- expect ( spans . some ( span => getSpanOp ( span ) === 'http.server' ) ) . toBe ( false ) ;
49+ expect ( spans . some ( span => getSpanOp ( span ) === HTTP_SERVER ) ) . toBe ( false ) ;
3650 // pi-durable keeps its state in the `pi_` tables of `PiHarness`, whose statements start no span.
37- expect ( spans . filter ( span => getSpanOp ( span ) === 'db.query' && / \b p i _ / . test ( String ( span . name ) ) ) ) . toEqual ( [ ] ) ;
51+ expect ( spans . filter ( span => getSpanOp ( span ) === DB_QUERY && / \b p i _ / . test ( String ( span . name ) ) ) ) . toEqual ( [ ] ) ;
3852 // pi-ai sends the requests through `@anthropic-ai/sdk`, whose own integration must stay out so
3953 // each request is reported once.
40- expect ( spans . filter ( span => String ( span . attributes [ 'sentry.origin' ] ?. value ) . startsWith ( 'auto.ai.' ) ) ) . toEqual (
41- spans . filter ( span => span . attributes [ 'sentry.origin' ] ?. value === 'auto.ai.pi_durable' ) ,
54+ expect ( spans . filter ( span => String ( span . attributes [ SENTRY_ORIGIN ] ?. value ) . startsWith ( 'auto.ai.' ) ) ) . toEqual (
55+ spans . filter ( span => span . attributes [ SENTRY_ORIGIN ] ?. value === 'auto.ai.pi_durable' ) ,
4256 ) ;
4357
4458 // One tool-calling response, then the answer.
4559 expect ( chats . length ) . toBeGreaterThanOrEqual ( 2 ) ;
4660 for ( const chat of chats ) {
4761 expect ( chat . parent_span_id ) . toBe ( agent . span_id ) ;
48- expect ( chat . attributes [ 'gen_ai.provider.name' ] ?. value ) . toBe ( 'openrouter' ) ;
49- expect ( chat . attributes [ 'gen_ai.request.model' ] ?. value ) . toBe ( 'anthropic/claude-haiku-4.5' ) ;
50- expect ( chat . attributes [ 'gen_ai.conversation.id' ] ?. value ) . toBe ( agent . attributes [ 'gen_ai.conversation.id' ] ?. value ) ;
62+ expect ( chat . attributes [ GEN_AI_PROVIDER_NAME ] ?. value ) . toBe ( 'openrouter' ) ;
63+ expect ( chat . attributes [ GEN_AI_REQUEST_MODEL ] ?. value ) . toBe ( 'anthropic/claude-haiku-4.5' ) ;
64+ expect ( chat . attributes [ GEN_AI_CONVERSATION_ID ] ?. value ) . toBe ( agent . attributes [ GEN_AI_CONVERSATION_ID ] ?. value ) ;
5165 // pi-durable retries a provider error inside the run; such a request has no usage and no HTTP span
5266 // of its own to assert on.
5367 if ( chat . status === 'ok' ) {
54- expect ( typeof chat . attributes [ 'gen_ai.usage.input_tokens' ] ?. value ) . toBe ( 'number' ) ;
55- expect ( typeof chat . attributes [ 'gen_ai.usage.output_tokens' ] ?. value ) . toBe ( 'number' ) ;
68+ expect ( typeof chat . attributes [ GEN_AI_USAGE_INPUT_TOKENS ] ?. value ) . toBe ( 'number' ) ;
69+ expect ( typeof chat . attributes [ GEN_AI_USAGE_OUTPUT_TOKENS ] ?. value ) . toBe ( 'number' ) ;
5670 expect ( providerCalls . some ( providerCall => providerCall . parent_span_id === chat . span_id ) ) . toBe ( true ) ;
5771 }
5872 }
5973
60- expect ( tool . attributes [ 'gen_ai.tool.name' ] ?. value ) . toBe ( 'get_weather' ) ;
74+ expect ( tool . attributes [ GEN_AI_TOOL_NAME ] ?. value ) . toBe ( 'get_weather' ) ;
6175 expect ( tool . parent_span_id ) . toBe ( agent . span_id ) ;
6276 expect ( tool . status ) . toBe ( 'ok' ) ;
63- expect ( tool . attributes [ 'gen_ai.tool.call.arguments' ] ?. value ) . toContain ( 'Vienna' ) ;
64- expect ( tool . attributes [ 'gen_ai.tool.call.result' ] ?. value ) . toContain ( '21 degrees and sunny in' ) ;
77+ expect ( tool . attributes [ GEN_AI_TOOL_CALL_ARGUMENTS ] ?. value ) . toContain ( 'Vienna' ) ;
78+ expect ( tool . attributes [ GEN_AI_TOOL_CALL_RESULT ] ?. value ) . toContain ( '21 degrees and sunny in' ) ;
6579
6680 // The provider's HTTP calls nest inside the `chat` span that sent them.
6781 expect ( providerCalls . length ) . toBeGreaterThan ( 0 ) ;
6882 for ( const providerCall of providerCalls ) {
6983 expect ( chats . map ( chat => chat . span_id ) ) . toContain ( providerCall . parent_span_id ) ;
70- expect ( providerCall . attributes [ 'server.address' ] ?. value ) . toBe ( 'openrouter.ai' ) ;
84+ expect ( providerCall . attributes [ SERVER_ADDRESS ] ?. value ) . toBe ( 'openrouter.ai' ) ;
7185 }
7286} ) ;
7387
@@ -79,8 +93,8 @@ test('reports a throwing tool as an error on its execute_tool span', async ({ ba
7993 const spansPromise = collectStreamedSpans (
8094 APP ,
8195 spansOfTrace =>
82- spansOfTrace . some ( span => span . is_segment && getSpanOp ( span ) === 'gen_ai.invoke_agent' ) &&
83- spansOfTrace . some ( span => span . attributes [ 'gen_ai.tool.name' ] ?. value === 'fail_now' ) ,
96+ spansOfTrace . some ( span => span . is_segment && getSpanOp ( span ) === GEN_AI_INVOKE_AGENT ) &&
97+ spansOfTrace . some ( span => span . attributes [ GEN_AI_TOOL_NAME ] ?. value === 'fail_now' ) ,
8498 ) ;
8599
86100 const answer = await prompt ( baseURL ! , newAgentId ( 'fail' ) , 'Call the fail_now tool, then tell me what happened.' ) ;
@@ -90,7 +104,7 @@ test('reports a throwing tool as an error on its execute_tool span', async ({ ba
90104 // The error is reported on the span of the call that threw.
91105 const tool = spans . find ( span => span . span_id === error . contexts ?. trace ?. span_id ) ! ;
92106
93- expect ( tool . attributes [ 'gen_ai.tool.name' ] ?. value ) . toBe ( 'fail_now' ) ;
107+ expect ( tool . attributes [ GEN_AI_TOOL_NAME ] ?. value ) . toBe ( 'fail_now' ) ;
94108 expect ( tool . status ) . toBe ( 'error' ) ;
95109 expect ( tool . trace_id ) . toBe ( error . contexts ?. trace ?. trace_id ) ;
96110 expect ( error . exception ?. values ?. [ 0 ] ?. mechanism ) . toEqual ( { type : 'auto.ai.pi_durable' , handled : false } ) ;
@@ -104,8 +118,8 @@ test('resumes a run in a new trace after the Durable Object resets during a tool
104118 const spansPromise = collectStreamedSpans (
105119 APP ,
106120 spansOfTrace =>
107- spansOfTrace . some ( span => span . is_segment && getSpanOp ( span ) === 'gen_ai.invoke_agent' ) &&
108- spansOfTrace . some ( span => span . attributes [ 'gen_ai.tool.name' ] ?. value === 'crash_once' && span . status === 'ok' ) ,
121+ spansOfTrace . some ( span => span . is_segment && getSpanOp ( span ) === GEN_AI_INVOKE_AGENT ) &&
122+ spansOfTrace . some ( span => span . attributes [ GEN_AI_TOOL_NAME ] ?. value === 'crash_once' && span . status === 'ok' ) ,
109123 ) ;
110124
111125 // The reset fails the request that submitted the prompt, but pi has stored the input already.
@@ -120,11 +134,11 @@ test('resumes a run in a new trace after the Durable Object resets during a tool
120134
121135 const spans = await spansPromise ;
122136 const agent = spans . find ( span => span . is_segment ) ! ;
123- const tool = spans . find ( span => span . attributes [ 'gen_ai.tool.name' ] ?. value === 'crash_once' ) ! ;
124- const chats = spans . filter ( span => getSpanOp ( span ) === 'gen_ai.chat' ) ;
125- const finalAnswer = chats . find ( span => span . attributes [ 'gen_ai.response.finish_reasons' ] ?. value === '["stop"]' ) ;
137+ const tool = spans . find ( span => span . attributes [ GEN_AI_TOOL_NAME ] ?. value === 'crash_once' ) ! ;
138+ const chats = spans . filter ( span => getSpanOp ( span ) === GEN_AI_CHAT ) ;
139+ const finalAnswer = chats . find ( span => span . attributes [ GEN_AI_RESPONSE_FINISH_REASONS ] ?. value === '["stop"]' ) ;
126140
127- expect ( getSpanOp ( agent ) ) . toBe ( 'gen_ai.invoke_agent' ) ;
141+ expect ( getSpanOp ( agent ) ) . toBe ( GEN_AI_INVOKE_AGENT ) ;
128142 expect ( tool . parent_span_id ) . toBe ( agent . span_id ) ;
129143 expect ( finalAnswer ?. parent_span_id ) . toBe ( agent . span_id ) ;
130144 // The request that called the tool ran before the reset, so this trace starts with the rerun of the
0 commit comments