7979 SpanEvents ,
8080 add_span_event ,
8181 anonymize_value ,
82+ llm_inference_span_attributes ,
8283 record_exception ,
84+ root_span_turn_attributes ,
8385 set_span_attributes ,
8486)
8587from utils .prompts import get_system_prompt
@@ -158,18 +160,18 @@ def _count_request_attachments(response_input: ResponseInput) -> int:
158160def _finalize_responses_root_span (
159161 root_span : trace .Span ,
160162 turn_summary : TurnSummary ,
163+ compacted : bool = False ,
161164) -> None :
162165 """Set final root-span attributes and completion events for /responses.
163166
164167 Args:
165168 root_span: OpenTelemetry root span for the request.
166- turn_summary: Completed turn summary with output text.
169+ turn_summary: Completed turn summary with the LLM response text.
170+ compacted: Whether the turn used compacted conversation context.
167171 """
168172 set_span_attributes (
169173 root_span ,
170- {
171- SpanAttributes .OUTPUT : turn_summary .llm_response ,
172- },
174+ root_span_turn_attributes (turn_summary , compacted = compacted ),
173175 )
174176 add_span_event (root_span , SpanEvents .LLM_RESPONSE_COMPLETED )
175177
@@ -206,33 +208,29 @@ def _start_llm_inference_span(
206208def _complete_llm_inference_span (
207209 span : trace .Span ,
208210 turn_summary : TurnSummary ,
211+ model : str ,
212+ inference_time : float ,
209213) -> None :
210- """Record usage/tool attrs and completion event, then end an inference span.
214+ """Record turn-summary attrs and completion event, then end an inference span.
211215
212216 Args:
213217 span: The ``llm.inference`` span to finalize.
214- turn_summary: Completed turn summary providing token usage and tool calls.
218+ turn_summary: Completed turn summary with tools, RAG, and tokens.
219+ model: Composite model identifier in ``provider/model`` format.
220+ inference_time: Inference duration in seconds.
215221 """
222+ provider_id , bare_model_id = extract_provider_and_model_from_model_id (model )
216223 set_span_attributes (
217224 span ,
218- {
219- SpanAttributes .LLM_USAGE_INPUT_TOKENS : (
220- turn_summary .token_usage .input_tokens
221- ),
222- SpanAttributes .LLM_USAGE_OUTPUT_TOKENS : (
223- turn_summary .token_usage .output_tokens
224- ),
225- },
225+ llm_inference_span_attributes (
226+ turn_summary ,
227+ bare_model_id ,
228+ provider_id ,
229+ inference_time ,
230+ ),
226231 )
227- tool_names = [tc .name for tc in turn_summary .tool_calls ]
228- if tool_names :
229- set_span_attributes (
230- span ,
231- {
232- SpanAttributes .TOOL_CALLS_COUNT : len (tool_names ),
233- SpanAttributes .TOOL_CALLS_NAMES : tool_names ,
234- },
235- )
232+ if turn_summary .tool_calls :
233+ tool_names = [tc .name for tc in turn_summary .tool_calls ]
236234 add_span_event (
237235 span ,
238236 SpanEvents .TOOL_EXECUTION_COMPLETED ,
@@ -570,7 +568,7 @@ async def handle_responses_with_tracing( # pylint: disable=too-many-locals
570568 )
571569 attachments_count = _count_request_attachments (original_request .input )
572570
573- span_attributes : dict [str , Any ] = {
571+ span_attributes : dict [SpanAttributes , Any ] = {
574572 SpanAttributes .USER_ID : anonymize_value (user_id ),
575573 SpanAttributes .INPUT : input_text ,
576574 SpanAttributes .REQUEST_ATTACHMENTS_COUNT : attachments_count ,
@@ -1231,7 +1229,8 @@ async def response_generator(
12311229 )
12321230 raise
12331231
1234- # Populate tools before closing llm.inference so tool attrs land on that span.
1232+ # Extract response metadata from final response object before closing
1233+ # the inference span so tool/RAG attrs can be recorded on it.
12351234 if latest_response_object :
12361235 _populate_turn_summary (
12371236 latest_response_object ,
@@ -1243,6 +1242,8 @@ async def response_generator(
12431242 _complete_llm_inference_span (
12441243 inference_span ,
12451244 turn_summary ,
1245+ api_params .model ,
1246+ time .monotonic () - inference_start_time ,
12461247 )
12471248
12481249 # Explicitly append the turn to conversation if context passed by previous response
@@ -1302,7 +1303,11 @@ async def generate_response(
13021303 completed_at ,
13031304 turn_summary .llm_response ,
13041305 )
1305- _finalize_responses_root_span (root_span , turn_summary )
1306+ _finalize_responses_root_span (
1307+ root_span ,
1308+ turn_summary ,
1309+ context .compacted_original_input is not None ,
1310+ )
13061311 # Persist conversation state before clients can close the stream.
13071312 yield "data: [DONE]\n \n "
13081313 finally :
@@ -1325,9 +1330,10 @@ async def handle_non_streaming_response(
13251330 """
13261331 root_span = context .root_span
13271332 user_id = context .auth [0 ]
1333+ inference_span : Optional [trace .Span ] = None
1334+ inference_start_time : Optional [float ] = None
13281335
13291336 # Fork: Get response object (blocked vs normal)
1330- inference_span : Optional [trace .Span ] = None
13311337 if context .moderation_result .decision == "blocked" :
13321338 output_text = context .moderation_result .message
13331339 api_response = OpenAIResponseObject .model_construct (
@@ -1367,6 +1373,8 @@ async def handle_non_streaming_response(
13671373 token_usage = extract_token_usage (
13681374 api_response .usage , api_params .model , context .endpoint_path
13691375 )
1376+ # Keep inference span open until turn_summary is built below so
1377+ # tool/RAG attributes can be recorded on llm.inference.
13701378 logger .info ("Consuming tokens" )
13711379 consume_query_tokens (
13721380 user_id = user_id ,
@@ -1411,11 +1419,13 @@ async def handle_non_streaming_response(
14111419 )
14121420 turn_summary .rag_chunks .extend (context .inline_rag_context .rag_chunks )
14131421
1414- # Close llm.inference after tools are known so usage + tool attrs share one span.
1415- if inference_span is not None :
1422+ # Close llm.inference after tools/RAG are known so usage + eval attrs share one span.
1423+ if inference_span is not None and inference_start_time is not None :
14161424 _complete_llm_inference_span (
14171425 inference_span ,
14181426 turn_summary ,
1427+ api_params .model ,
1428+ time .monotonic () - inference_start_time ,
14191429 )
14201430
14211431 # Get available quotas
@@ -1446,7 +1456,11 @@ async def handle_non_streaming_response(
14461456 completed_at ,
14471457 output_text ,
14481458 )
1449- _finalize_responses_root_span (root_span , turn_summary )
1459+ _finalize_responses_root_span (
1460+ root_span ,
1461+ turn_summary ,
1462+ context .compacted_original_input is not None ,
1463+ )
14501464 configured_mcp_labels = {s .name for s in configuration .mcp_servers }
14511465 response_dict = (
14521466 api_response .model_dump (exclude_none = True )
0 commit comments