progress-observability 1.1.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,804 @@
1
+ """Model Fix Processor
2
+
3
+ Custom span processor to fix missing gen_ai attributes in AI model spans.
4
+ This processor addresses bugs where AI model instrumentations fail to set
5
+ critical gen_ai attributes like gen_ai.provider.name, gen_ai.request.model, etc.
6
+
7
+ IMPORTANT: Model identifiers are preserved in their full form (including regional
8
+ prefixes, dates, and version suffixes) to maintain accurate pricing information
9
+ and version specificity.
10
+
11
+ SUPPORTED CASES & FIXES:
12
+
13
+ 1. OpenAI API Spans ("openai.chat", "ChatOpenAI.chat"):
14
+ - Fixes gen_ai.provider.name (and gen_ai.system for backwards compatibility): "OpenAI" (default) vs "Azure" (when azure in endpoint) vs "OpenRouter" (when openrouter in endpoint OR model has provider prefix like google/, anthropic/, etc.) vs "Ollama" (when :11434 or ollama in endpoint)
15
+ - Fixes missing gen_ai.request.model from gen_ai.response.model
16
+ - Fixes missing/incorrect gen_ai.response.model
17
+ - Preserves full model identifiers (e.g., gpt-4o-2024-11-20, google/gemini-2.0-flash-001)
18
+
19
+ 2. Azure OpenAI LlamaIndex Spans ("AzureOpenAI.workflow"):
20
+ - Forces gen_ai.provider.name to "Azure" (and gen_ai.system for backwards compatibility)
21
+ - Aggressively fixes gen_ai.request.model when it differs from response model
22
+ - Handles cases like request="gpt-35-turbo" vs response="gpt-4o-2024-11-20"
23
+ - Preserves full model identifiers
24
+
25
+ 3. Ollama Direct Spans ("ChatOllama.chat", "Ollama.workflow"):
26
+ - Forces gen_ai.provider.name to "Ollama" (and gen_ai.system for backwards compatibility)
27
+ - Extracts model name from traceloop.association.properties.ls_model_name
28
+ - Derives token usage from Ollama-specific attributes:
29
+ * prompt_eval_count → gen_ai.usage.input_tokens (and gen_ai.usage.prompt_tokens for backwards compatibility)
30
+ * eval_count → gen_ai.usage.output_tokens (and gen_ai.usage.completion_tokens for backwards compatibility)
31
+ * prompt_eval_count + eval_count → llm.usage.total_tokens
32
+ - Parses JSON in traceloop.entity.output for token counts when direct attributes missing
33
+
34
+ 4. AWS Bedrock Spans ("bedrock.converse", "BedrockConverse.workflow"):
35
+ - Forces gen_ai.provider.name to "AWS" (and gen_ai.system for backwards compatibility)
36
+ - Fixes missing gen_ai.response.model from gen_ai.request.model
37
+ - Preserves full Bedrock model IDs (e.g., us.anthropic.claude-sonnet-4-5-20250929-v1:0)
38
+ - Regional prefixes (us., eu., au., jp., global.) preserved for accurate pricing
39
+ - Derives token usage from traceloop.entity.output JSON:
40
+ * raw.usage.inputTokens → gen_ai.usage.input_tokens (and gen_ai.usage.prompt_tokens for backwards compatibility)
41
+ * raw.usage.outputTokens → gen_ai.usage.output_tokens (and gen_ai.usage.completion_tokens for backwards compatibility)
42
+ * raw.usage.totalTokens → llm.usage.total_tokens
43
+ * alternative_kwargs.prompt_tokens/completion_tokens/total_tokens (fallback)
44
+
45
+ 5. Provider Detection Logic:
46
+ - Azure: "azure" keyword in any endpoint/URL attribute
47
+ - OpenRouter: "openrouter" keyword in any endpoint/URL attribute OR model name with provider prefix (google/, anthropic/, meta-llama/, etc.)
48
+ - Ollama: ":11434" port or "ollama" keyword in endpoint
49
+ - AWS: "AWS" in gen_ai.provider.name/gen_ai.system attribute or bedrock span names
50
+ - OpenAI: Default fallback for openai.chat spans
51
+
52
+ """
53
+
54
+ import os
55
+ import re
56
+ import logging
57
+ from collections import deque
58
+ from typing import Tuple, Optional, Any, Set
59
+ from opentelemetry.sdk.trace import SpanProcessor, ReadableSpan
60
+
61
+ logger = logging.getLogger(__name__)
62
+
63
+ # ---- Constants & simple helpers -------------------------------------------------
64
+
65
+ # System names
66
+ SYSTEM_AZURE = "Azure"
67
+ SYSTEM_OPENAI = "OpenAI"
68
+ SYSTEM_OPENROUTER = "OpenRouter"
69
+ SYSTEM_OLLAMA = "Ollama"
70
+ SYSTEM_AWS = "AWS"
71
+
72
+ # Invalid/empty model values
73
+ INVALID_MODEL_VALUES = (None, "", "unknown")
74
+
75
+ # Keywords for endpoint detection
76
+ AZURE_KEYWORD = "azure"
77
+ OPENROUTER_KEYWORD = "openrouter"
78
+ OLLAMA_KEYWORD = "ollama"
79
+
80
+ # OpenRouter model name prefixes (these indicate OpenRouter is being used)
81
+ OPENROUTER_MODEL_PREFIXES = (
82
+ "google/", "anthropic/", "openai/", "meta-llama/", "meta/", "microsoft/",
83
+ "mistralai/", "cohere/", "ai21/", "huggingfaceh4/", "teknium/",
84
+ "nousresearch/", "openchat/", "codellama/", "phind/", "wizardlm/",
85
+ "upstage/", "01-ai/", "alpindale/", "austism/", "cognitivecomputations/",
86
+ "databricks/", "deepseek/", "gryphe/", "intel/", "jondurbin/",
87
+ "lizpreciator/", "migtissera/", "neversleep/", "undi95/", "xwin-lm/"
88
+ )
89
+
90
+ OLLAMA_DEFAULT_PORT = 11434
91
+ OLLAMA_SPAN_NAMES = ("ChatOllama.chat", "Ollama.workflow")
92
+ BEDROCK_SPAN_NAMES = ("bedrock.converse", "BedrockConverse.workflow")
93
+ ANTHROPIC_SPAN_NAMES = ("anthropic.converse", "Anthropic.workflow")
94
+
95
+ # Span names we explicitly care about (OpenAI, Azure OpenAI via LlamaIndex, Ollama, Bedrock)
96
+ SUPPORTED_SPAN_NAMES = {
97
+ "openai.chat",
98
+ "openai.response",
99
+ "ChatOpenAI.chat",
100
+ "AzureOpenAI.workflow",
101
+ "Agent Workflow",
102
+ *OLLAMA_SPAN_NAMES,
103
+ *BEDROCK_SPAN_NAMES,
104
+ *ANTHROPIC_SPAN_NAMES
105
+ }
106
+
107
+ def _is_ollama_span(span_name: str) -> bool:
108
+ return span_name in OLLAMA_SPAN_NAMES
109
+
110
+ def _is_bedrock_span(span_name: str) -> bool:
111
+ return span_name in BEDROCK_SPAN_NAMES
112
+
113
+ def _is_openrouter_model(model_name: Optional[str]) -> bool:
114
+ """Check if a model name indicates OpenRouter usage."""
115
+ if not model_name:
116
+ return False
117
+ model_name_lower = model_name.lower()
118
+ return any(model_name_lower.startswith(prefix) for prefix in OPENROUTER_MODEL_PREFIXES)
119
+
120
+ def _is_invalid_model(value: Optional[str]) -> bool:
121
+ """Check if a model value is invalid/empty."""
122
+ return value in INVALID_MODEL_VALUES
123
+
124
+ def _extract_endpoint_info(attributes: dict) -> tuple[str, bool, bool, bool]:
125
+ """
126
+ Extract and analyze endpoint information from span attributes.
127
+
128
+ Args:
129
+ attributes: Dictionary of span attributes
130
+
131
+ Returns:
132
+ Tuple of (endpoint_string, is_azure, is_openrouter, is_ollama)
133
+ """
134
+ api_base = attributes.get("gen_ai.openai.api_base", "")
135
+ endpoint = attributes.get("server.address") or attributes.get("http.url") or api_base
136
+ endpoint_str = str(endpoint or "").lower()
137
+
138
+ is_azure = AZURE_KEYWORD in endpoint_str
139
+ is_openrouter = OPENROUTER_KEYWORD in endpoint_str
140
+ is_ollama = f":{OLLAMA_DEFAULT_PORT}" in endpoint_str or OLLAMA_KEYWORD in endpoint_str
141
+
142
+ return endpoint_str, is_azure, is_openrouter, is_ollama
143
+
144
+
145
+ class ModelFixProcessor(SpanProcessor):
146
+ """
147
+ Custom span processor to fix missing gen_ai attributes in AI model spans.
148
+
149
+ This processor addresses bugs where AI model instrumentations fail to set
150
+ critical gen_ai attributes like gen_ai.system, gen_ai.request.model, etc.
151
+
152
+ Since we cannot modify span attributes after the span ends, we use a different
153
+ approach: we monitor spans and try to set missing attributes during the span
154
+ lifecycle by setting them as early as possible.
155
+
156
+ Currently fixes:
157
+ - gen_ai.provider.name (e.g., "Azure", "Ollama", "AWS") and gen_ai.system (backwards compatibility)
158
+ - gen_ai.request.model (when missing but response.model is present)
159
+ - gen_ai.response.model (when missing or incorrect)
160
+ - gen_ai.usage.input_tokens and gen_ai.usage.prompt_tokens (for Ollama/Bedrock spans from provider-specific attributes)
161
+ - gen_ai.usage.output_tokens and gen_ai.usage.completion_tokens (for Ollama/Bedrock spans from provider-specific attributes)
162
+ - llm.usage.total_tokens (for Ollama/Bedrock spans from provider-specific attributes)
163
+
164
+ Supported span types:
165
+ - OpenAI models (openai.chat spans) - including Azure OpenAI
166
+ - Azure OpenAI via LlamaIndex (AzureOpenAI.workflow spans)
167
+ - Ollama models (ChatOllama.chat spans and Ollama.workflow spans)
168
+ - AWS Bedrock models (bedrock.converse spans and BedrockConverse.workflow spans)
169
+
170
+ Args:
171
+ debug: Enable verbose debugging output. Can also be controlled via
172
+ OBSERVABILITY_DEBUG environment variable. Defaults to False.
173
+ max_processed_spans: Maximum number of span IDs to track for double-processing
174
+ prevention. Uses LRU eviction. Defaults to 10000.
175
+ """
176
+
177
+ # Backwards compatibility: retain attribute for external inspection if needed
178
+ SUPPORTED_SPAN_NAMES = SUPPORTED_SPAN_NAMES
179
+
180
+ def __init__(self, debug: bool = False, max_processed_spans: int = 10000):
181
+ # Track processed spans with LRU eviction to prevent unbounded memory growth
182
+ # Using deque for O(1) append and automatic size limiting
183
+ self._processed_spans_deque: deque = deque(maxlen=max_processed_spans)
184
+ self._processed_spans_set: Set[Tuple[int, int]] = set()
185
+
186
+ # Debug flag - check env var or use parameter
187
+ self.debug_enabled = debug or os.getenv('OBSERVABILITY_DEBUG', '').lower() in ('1', 'true', 'yes')
188
+
189
+ if self.debug_enabled:
190
+ logger.info(f"ModelFixProcessor initialized with debug mode enabled, max_processed_spans={max_processed_spans}")
191
+
192
+ def on_start(self, span: Any, parent_context=None) -> None:
193
+ """
194
+ Called when a span is started (SpanProcessor interface method).
195
+
196
+ This implementation intentionally does nothing as all processing happens in on_end()
197
+ via the span_postprocess_callback mechanism.
198
+
199
+ Args:
200
+ span: The span being started
201
+ parent_context: Optional parent context
202
+ """
203
+ pass
204
+
205
+ def on_end(self, readable_span: ReadableSpan) -> None:
206
+ """
207
+ Called when a span is ended. Attempts to fix missing gen_ai attributes.
208
+
209
+ This method is invoked via the span_postprocess_callback mechanism, allowing
210
+ us to inspect and modify span attributes before they are exported.
211
+
212
+ Args:
213
+ readable_span: The span that has ended, containing attributes to potentially fix
214
+ """
215
+ # Skip unsupported span types
216
+ if readable_span.name not in self.SUPPORTED_SPAN_NAMES:
217
+ return
218
+
219
+ # Prevent double-processing with LRU eviction
220
+ span_context_id = (readable_span.context.span_id, readable_span.context.trace_id)
221
+
222
+ if span_context_id in self._processed_spans_set:
223
+ if self.debug_enabled:
224
+ print(f"[ModelFixProcessor.on_end] Span {readable_span.name} already processed, skipping")
225
+ return
226
+
227
+ # Mark as processed - add to both deque and set
228
+ self._processed_spans_deque.append(span_context_id)
229
+ self._processed_spans_set.add(span_context_id)
230
+
231
+ # Clean up set when deque evicts old items (deque auto-evicts at maxlen)
232
+ if len(self._processed_spans_set) > len(self._processed_spans_deque):
233
+ # Rebuild set from deque to match current LRU state
234
+ self._processed_spans_set = set(self._processed_spans_deque)
235
+
236
+ if self.debug_enabled:
237
+ print(f"\n{'='*80}")
238
+ print(f"[ModelFixProcessor.on_end] Processing span: {readable_span.name}")
239
+ print(f"{'='*80}\n")
240
+
241
+ logger.debug(f"on_end called for span: {readable_span.name}")
242
+
243
+ attributes = dict(readable_span.attributes or {})
244
+ request_model_attr = attributes.get("gen_ai.request.model")
245
+ response_model_attr = attributes.get("gen_ai.response.model")
246
+ system_attr = attributes.get("gen_ai.provider.name") if "gen_ai.provider.name" in attributes else attributes.get("gen_ai.system")
247
+
248
+ if self.debug_enabled:
249
+ print(f"[ModelFixProcessor] BEFORE: request={request_model_attr}, response={response_model_attr}, system={system_attr}")
250
+
251
+ logger.debug(f"Attributes before fixes: request_model={request_model_attr}, response_model={response_model_attr}, system={system_attr}")
252
+
253
+ # Get correct values from various sources
254
+ correct_model_name = self._get_correct_model_name(attributes, readable_span.name)
255
+ correct_system = self._get_correct_system(attributes, readable_span.name)
256
+
257
+ # Apply fixes
258
+ self._apply_attribute_fixes(
259
+ readable_span=readable_span,
260
+ span_name=readable_span.name,
261
+ system_attr=system_attr,
262
+ correct_system=correct_system,
263
+ request_model_attr=request_model_attr,
264
+ response_model_attr=response_model_attr,
265
+ correct_model_name=correct_model_name,
266
+ )
267
+
268
+ if self.debug_enabled:
269
+ final_attrs = self._get_target_attributes(readable_span)
270
+ if final_attrs:
271
+ print(f"[ModelFixProcessor] AFTER: request={final_attrs.get('gen_ai.request.model')}, response={final_attrs.get('gen_ai.response.model')}, system={final_attrs.get('gen_ai.system')}")
272
+ print(f"{'='*80}\n")
273
+
274
+ def _get_correct_model_name(self, attributes: dict, span_name: str) -> Optional[str]:
275
+ """
276
+ Get the correct model name from various attribute sources.
277
+
278
+ Returns the full model identifier to preserve regional and version information.
279
+
280
+ Args:
281
+ attributes: Dictionary of span attributes
282
+ span_name: Name of the span
283
+
284
+ Returns:
285
+ The correct model name if found, None otherwise
286
+ """
287
+ # For Bedrock spans, prioritize request model since response model is often missing
288
+ if _is_bedrock_span(span_name):
289
+ request_model = attributes.get("gen_ai.request.model")
290
+ if request_model and not _is_invalid_model(request_model):
291
+ return request_model
292
+
293
+ # For other spans, try to get from standard response model if it's valid
294
+ response_model = attributes.get("gen_ai.response.model")
295
+ if response_model and not _is_invalid_model(response_model):
296
+ return response_model
297
+
298
+ # For Ollama spans, check the traceloop association properties
299
+ if _is_ollama_span(span_name):
300
+ ls_model_name = attributes.get("traceloop.association.properties.ls_model_name")
301
+ if ls_model_name:
302
+ return ls_model_name
303
+
304
+ return None
305
+
306
+
307
+ def _get_correct_system(self, attributes: dict, span_name: str) -> Optional[str]:
308
+ """
309
+ Get the correct gen_ai.system value based on span characteristics.
310
+
311
+ Args:
312
+ attributes: Dictionary of span attributes
313
+ span_name: Name of the span
314
+
315
+ Returns:
316
+ The correct system value if determinable, None otherwise
317
+ """
318
+ # For Azure OpenAI workflow spans (LlamaIndex Azure OpenAI)
319
+ if span_name == "AzureOpenAI.workflow":
320
+ return SYSTEM_AZURE
321
+
322
+ # For Bedrock spans
323
+ if _is_bedrock_span(span_name):
324
+ return SYSTEM_AWS
325
+
326
+ # For Ollama spans
327
+ if _is_ollama_span(span_name):
328
+ return SYSTEM_OLLAMA
329
+
330
+ # For OpenAI-compatible spans, use unified detection logic
331
+ if span_name in ("openai.chat", "openai.response", "ChatOpenAI.chat", "Agent Workflow"):
332
+ return self._detect_openai_compatible_system(attributes)
333
+
334
+ # For other spans, try to infer from existing attributes
335
+ return self._detect_openai_compatible_system(attributes)
336
+
337
+ def _detect_openai_compatible_system(self, attributes: dict) -> Optional[str]:
338
+ """
339
+ Detect the correct system for OpenAI-compatible spans using endpoint and model name patterns.
340
+
341
+ Args:
342
+ attributes: Dictionary of span attributes
343
+
344
+ Returns:
345
+ The correct system value if determinable, None otherwise
346
+ """
347
+ # Try endpoint detection first
348
+ _, is_azure, is_openrouter, is_ollama = _extract_endpoint_info(attributes)
349
+
350
+ if is_azure:
351
+ return SYSTEM_AZURE
352
+ if is_openrouter:
353
+ return SYSTEM_OPENROUTER
354
+ if is_ollama:
355
+ return SYSTEM_OLLAMA
356
+
357
+ # If endpoint detection fails, try model name detection for OpenRouter
358
+ request_model = attributes.get("gen_ai.request.model")
359
+ response_model = attributes.get("gen_ai.response.model")
360
+ llm_model_name = attributes.get("traceloop.association.properties.ls_model_name")
361
+
362
+ if _is_openrouter_model(request_model) or _is_openrouter_model(response_model) or _is_openrouter_model(llm_model_name):
363
+ return SYSTEM_OPENROUTER
364
+
365
+ # Default to OpenAI for OpenAI-compatible spans
366
+ return SYSTEM_OPENAI
367
+
368
+ def _should_fix_request_model(self, span_name: str, request_model: Optional[str], response_model: Optional[str]) -> bool:
369
+ """
370
+ Determine whether we should fix the gen_ai.request.model attribute.
371
+
372
+ Args:
373
+ span_name: Name of the span
374
+ request_model: Current gen_ai.request.model value
375
+ response_model: Current gen_ai.response.model value
376
+
377
+ Returns:
378
+ True if we should fix the request model, False otherwise
379
+ """
380
+ # Always fix if request model is missing or unknown
381
+ if _is_invalid_model(request_model):
382
+ return True
383
+
384
+ # For AzureOpenAI.workflow spans, be more aggressive about fixing request model
385
+ # since the response model tends to be more accurate
386
+ if span_name in ("AzureOpenAI.workflow", "Agent Workflow") and response_model:
387
+ if response_model and request_model != response_model:
388
+ return True
389
+
390
+ return False
391
+
392
+ def _fix_ollama_usage_attributes(self, attributes_dict: dict) -> None:
393
+ """
394
+ Fix missing Ollama token usage attributes by extracting from raw response attributes.
395
+
396
+ Ollama responses include token counts in different attribute names:
397
+ - prompt_eval_count → gen_ai.usage.input_tokens (and gen_ai.usage.prompt_tokens for backwards compatibility)
398
+ - eval_count → gen_ai.usage.output_tokens (and gen_ai.usage.completion_tokens for backwards compatibility)
399
+ - prompt_eval_count + eval_count → llm.usage.total_tokens
400
+
401
+ Args:
402
+ attributes_dict: Dictionary of span attributes to modify
403
+ """
404
+ try:
405
+ # Prefer new attribute names; legacy names used only as fallback/presence indicator
406
+ new_prompt_val = attributes_dict.get("gen_ai.usage.input_tokens")
407
+ new_completion_val = attributes_dict.get("gen_ai.usage.output_tokens")
408
+ legacy_prompt_present = "gen_ai.usage.prompt_tokens" in attributes_dict
409
+ legacy_completion_present = "gen_ai.usage.completion_tokens" in attributes_dict
410
+ has_total_tokens = attributes_dict.get("llm.usage.total_tokens") not in (None, "", 0)
411
+
412
+ # Look for Ollama-specific attributes
413
+ prompt_eval_count = attributes_dict.get("prompt_eval_count")
414
+ eval_count = attributes_dict.get("eval_count")
415
+
416
+ # Try to extract from traceloop entity output (JSON response data)
417
+ if prompt_eval_count is None or eval_count is None:
418
+ entity_output = attributes_dict.get("traceloop.entity.output")
419
+ if entity_output and isinstance(entity_output, str):
420
+ prompt_eval_count, eval_count = self._extract_ollama_tokens_from_json(entity_output)
421
+
422
+ if prompt_eval_count is not None and eval_count is not None:
423
+ # Convert to integers
424
+ prompt_eval_count = int(prompt_eval_count) if isinstance(prompt_eval_count, (str, int, float)) else 0
425
+ eval_count = int(eval_count) if isinstance(eval_count, (str, int, float)) else 0
426
+
427
+ # Fix missing standard attributes (set both old and new names for backwards compatibility)
428
+ # Set new attributes when missing or falsy; only write legacy keys if they already exist
429
+ if not new_prompt_val and prompt_eval_count > 0:
430
+ attributes_dict["gen_ai.usage.input_tokens"] = prompt_eval_count
431
+ if legacy_prompt_present:
432
+ # update legacy key only if it already existed on the span
433
+ attributes_dict["gen_ai.usage.prompt_tokens"] = prompt_eval_count
434
+
435
+ if not new_completion_val and eval_count > 0:
436
+ attributes_dict["gen_ai.usage.output_tokens"] = eval_count
437
+ if legacy_completion_present:
438
+ attributes_dict["gen_ai.usage.completion_tokens"] = eval_count
439
+
440
+ if not has_total_tokens and (prompt_eval_count > 0 or eval_count > 0):
441
+ attributes_dict["llm.usage.total_tokens"] = prompt_eval_count + eval_count
442
+
443
+ except (ValueError, TypeError, AttributeError):
444
+ pass
445
+
446
+ def _fix_bedrock_usage_attributes(self, attributes_dict: dict) -> None:
447
+ """
448
+ Fix missing Bedrock token usage attributes by extracting from traceloop.entity.output.
449
+
450
+ Bedrock responses include token counts in the raw.usage structure:
451
+ - inputTokens → gen_ai.usage.input_tokens (and gen_ai.usage.prompt_tokens for backwards compatibility)
452
+ - outputTokens → gen_ai.usage.output_tokens (and gen_ai.usage.completion_tokens for backwards compatibility)
453
+ - totalTokens → llm.usage.total_tokens
454
+
455
+ Args:
456
+ attributes_dict: Dictionary of span attributes to modify
457
+ """
458
+ try:
459
+ # Prefer new attribute names; legacy names used only as fallback/presence indicator
460
+ new_prompt_val = attributes_dict.get("gen_ai.usage.input_tokens")
461
+ new_completion_val = attributes_dict.get("gen_ai.usage.output_tokens")
462
+ legacy_prompt_val = attributes_dict.get("gen_ai.usage.prompt_tokens")
463
+ legacy_completion_val = attributes_dict.get("gen_ai.usage.completion_tokens")
464
+ has_total_tokens = attributes_dict.get("llm.usage.total_tokens") not in (None, "", 0)
465
+
466
+ # If all attributes are already present with valid values, no need to extract
467
+ if (new_prompt_val or legacy_prompt_val) and (new_completion_val or legacy_completion_val) and has_total_tokens:
468
+ return
469
+
470
+ # Try to extract from traceloop entity output (JSON response data)
471
+ entity_output = attributes_dict.get("traceloop.entity.output")
472
+ if entity_output and isinstance(entity_output, str):
473
+ prompt_tokens, completion_tokens, total_tokens = self._extract_bedrock_tokens_from_json(entity_output)
474
+
475
+ if prompt_tokens is not None and completion_tokens is not None:
476
+ # Fix missing standard attributes: set new names first; set legacy keys only if they existed
477
+ if not new_prompt_val and prompt_tokens > 0:
478
+ attributes_dict["gen_ai.usage.input_tokens"] = prompt_tokens
479
+ if "gen_ai.usage.prompt_tokens" in attributes_dict:
480
+ attributes_dict["gen_ai.usage.prompt_tokens"] = prompt_tokens
481
+
482
+ if not new_completion_val and completion_tokens > 0:
483
+ attributes_dict["gen_ai.usage.output_tokens"] = completion_tokens
484
+ if "gen_ai.usage.completion_tokens" in attributes_dict:
485
+ attributes_dict["gen_ai.usage.completion_tokens"] = completion_tokens
486
+
487
+ if not has_total_tokens:
488
+ if total_tokens is not None and total_tokens > 0:
489
+ attributes_dict["llm.usage.total_tokens"] = total_tokens
490
+ elif prompt_tokens > 0 or completion_tokens > 0:
491
+ attributes_dict["llm.usage.total_tokens"] = prompt_tokens + completion_tokens
492
+
493
+ except (ValueError, TypeError, AttributeError):
494
+ pass
495
+
496
+
497
+ def _extract_ollama_tokens_from_json(self, json_output: str) -> Tuple[Optional[int], Optional[int]]:
498
+ """
499
+ Extract prompt_eval_count and eval_count from Ollama JSON response in traceloop.entity.output.
500
+
501
+ The JSON structure is expected to be:
502
+ {"message": {...}, "raw": {"prompt_eval_count": X, "eval_count": Y, ...}}
503
+
504
+ Args:
505
+ json_output: JSON string from traceloop.entity.output
506
+
507
+ Returns:
508
+ Tuple of (prompt_eval_count, eval_count) or (None, None) if not found
509
+ """
510
+ try:
511
+ import json
512
+ data = json.loads(json_output)
513
+
514
+ # Handle double-encoded JSON (data might be a JSON string)
515
+ if isinstance(data, str):
516
+ data = json.loads(data)
517
+
518
+ # Use defensive extraction
519
+ prompt_eval_count = self._safe_extract_nested(data, 'raw', 'prompt_eval_count')
520
+ eval_count = self._safe_extract_nested(data, 'raw', 'eval_count')
521
+
522
+ if prompt_eval_count is not None and eval_count is not None:
523
+ return int(prompt_eval_count), int(eval_count)
524
+
525
+ except (json.JSONDecodeError, ValueError, TypeError) as e:
526
+ if self.debug_enabled:
527
+ logger.warning(f"Failed to extract Ollama tokens from JSON: {e}")
528
+
529
+ return None, None
530
+
531
+ def _extract_bedrock_tokens_from_json(self, json_output: str) -> Tuple[Optional[int], Optional[int], Optional[int]]:
532
+ """
533
+ Extract token usage from Bedrock JSON response in traceloop.entity.output.
534
+
535
+ The JSON structure is expected to be:
536
+ {"raw": {"usage": {"inputTokens": X, "outputTokens": Y, "totalTokens": Z, ...}}}
537
+ or
538
+ {"additional_kwargs": {"prompt_tokens": X, "completion_tokens": Y, "total_tokens": Z}}
539
+
540
+ Args:
541
+ json_output: JSON string from traceloop.entity.output
542
+
543
+ Returns:
544
+ Tuple of (prompt_tokens, completion_tokens, total_tokens) or (None, None, None) if not found
545
+ """
546
+ try:
547
+ import json
548
+ data = json.loads(json_output)
549
+
550
+ # Handle double-encoded JSON (data might be a JSON string)
551
+ if isinstance(data, str):
552
+ data = json.loads(data)
553
+
554
+ # Try to extract from raw.usage (Bedrock native format)
555
+ input_tokens = self._safe_extract_nested(data, 'raw', 'usage', 'inputTokens')
556
+ output_tokens = self._safe_extract_nested(data, 'raw', 'usage', 'outputTokens')
557
+ total_tokens = self._safe_extract_nested(data, 'raw', 'usage', 'totalTokens')
558
+
559
+ if input_tokens is not None and output_tokens is not None:
560
+ return (
561
+ int(input_tokens),
562
+ int(output_tokens),
563
+ int(total_tokens) if total_tokens is not None else None
564
+ )
565
+
566
+ # Try to extract from additional_kwargs (alternative format)
567
+ prompt_tokens = self._safe_extract_nested(data, 'additional_kwargs', 'prompt_tokens')
568
+ completion_tokens = self._safe_extract_nested(data, 'additional_kwargs', 'completion_tokens')
569
+ total_tokens_alt = self._safe_extract_nested(data, 'additional_kwargs', 'total_tokens')
570
+
571
+ if prompt_tokens is not None and completion_tokens is not None:
572
+ return (
573
+ int(prompt_tokens),
574
+ int(completion_tokens),
575
+ int(total_tokens_alt) if total_tokens_alt is not None else None
576
+ )
577
+
578
+ except (json.JSONDecodeError, ValueError, TypeError) as e:
579
+ if self.debug_enabled:
580
+ logger.warning(f"Failed to extract Bedrock tokens from JSON: {e}")
581
+
582
+ return None, None, None
583
+
584
+ # ---- Internal helper methods -------------------------------------------------
585
+
586
+ def _safe_extract_nested(self, data: Any, *keys: str, default: Any = None) -> Any:
587
+ """
588
+ Safely extract a nested dictionary value using a sequence of keys.
589
+
590
+ This helper provides defensive access to nested dictionary structures,
591
+ commonly found in JSON responses from AI providers. It handles:
592
+ - Missing keys gracefully
593
+ - Non-dict intermediate values
594
+ - Empty dict results
595
+
596
+ Args:
597
+ data: The root dictionary or data structure to extract from
598
+ *keys: Variable number of keys to traverse (e.g., 'raw', 'usage', 'inputTokens')
599
+ default: Value to return if extraction fails or result is empty. Defaults to None.
600
+
601
+ Returns:
602
+ The extracted value if found, otherwise the default value.
603
+
604
+ Example:
605
+ >>> data = {"raw": {"usage": {"inputTokens": 100}}}
606
+ >>> self._safe_extract_nested(data, 'raw', 'usage', 'inputTokens')
607
+ 100
608
+ >>> self._safe_extract_nested(data, 'raw', 'missing', 'key', default=0)
609
+ 0
610
+ """
611
+ result = data
612
+ for key in keys:
613
+ if not isinstance(result, dict):
614
+ if self.debug_enabled:
615
+ logger.debug(f"_safe_extract_nested: Expected dict at key '{key}', got {type(result).__name__}")
616
+ return default
617
+ result = result.get(key)
618
+ if result is None:
619
+ if self.debug_enabled:
620
+ logger.debug(f"_safe_extract_nested: Key '{key}' not found in path {keys}")
621
+ return default
622
+
623
+ # Return default if result is empty dict
624
+ if result == {}:
625
+ return default
626
+
627
+ return result
628
+
629
+ def _get_target_attributes(self, readable_span: ReadableSpan) -> Optional[dict]:
630
+ """
631
+ Get the target attributes dictionary for modification.
632
+
633
+ Args:
634
+ readable_span: The span whose attributes we want to access
635
+
636
+ Returns:
637
+ Writable attributes dictionary if available, None otherwise
638
+ """
639
+ if hasattr(readable_span, '_attributes'):
640
+ readable_span._attributes = readable_span._attributes or {}
641
+ logger.debug(f"Got target attributes via _attributes, type={type(readable_span._attributes)}")
642
+ return readable_span._attributes
643
+ elif hasattr(readable_span, 'attributes') and isinstance(readable_span.attributes, dict):
644
+ logger.warning("Using attributes property directly (may not persist changes)")
645
+ return readable_span.attributes
646
+
647
+ logger.error(f"Could not find writable attributes on span '{readable_span.name}'!")
648
+ return None
649
+
650
+ def _fix_system_attribute(self, target_attrs: dict, system_attr: Optional[str], correct_system: Optional[str]) -> None:
651
+ """
652
+ Fix the gen_ai.system and gen_ai.provider.name attributes if needed.
653
+
654
+ Args:
655
+ target_attrs: Dictionary of span attributes to modify
656
+ system_attr: Current gen_ai.system or gen_ai.provider.name value
657
+ correct_system: Correct system value to set
658
+ """
659
+ if correct_system and system_attr != correct_system:
660
+ if self.debug_enabled:
661
+ logger.debug(f"Fixing gen_ai.system/gen_ai.provider.name: '{system_attr}' -> '{correct_system}'")
662
+ target_attrs["gen_ai.provider.name"] = correct_system
663
+ if "gen_ai.system" in target_attrs:
664
+ target_attrs["gen_ai.system"] = correct_system # Backwards compatibility
665
+ target_attrs["traceloop.association.properties.ls_provider"] = correct_system
666
+
667
+ def _fix_model_attributes(
668
+ self,
669
+ target_attrs: dict,
670
+ span_name: str,
671
+ request_model_attr: Optional[str],
672
+ response_model_attr: Optional[str],
673
+ correct_model_name: Optional[str]
674
+ ) -> None:
675
+ """
676
+ Fix the gen_ai.request.model and gen_ai.response.model attributes if needed.
677
+
678
+ This method implements different strategies for different span types:
679
+ - Bedrock spans: Use request model as source of truth (response often missing)
680
+ - Other spans: Prefer response model, fall back to request or correct_model_name
681
+
682
+ Args:
683
+ target_attrs: Dictionary of span attributes to modify
684
+ span_name: Name of the span being processed
685
+ request_model_attr: Current gen_ai.request.model value
686
+ response_model_attr: Current gen_ai.response.model value
687
+ correct_model_name: Correct model name from alternative sources
688
+ """
689
+
690
+ # For Bedrock spans, always use request model as the source of truth
691
+ if _is_bedrock_span(span_name):
692
+ if request_model_attr and not _is_invalid_model(request_model_attr):
693
+ # Keep the full model name to preserve regional and version information
694
+ logger.debug(f"Bedrock span: using full model name {request_model_attr}")
695
+
696
+ # Set both request and response model to the full model name
697
+ target_attrs["gen_ai.request.model"] = request_model_attr
698
+ target_attrs["gen_ai.response.model"] = request_model_attr
699
+ else:
700
+ logger.warning(f"Cannot set Bedrock model: request_model={request_model_attr}")
701
+ return
702
+
703
+ # For non-Bedrock spans, use existing logic
704
+ final_request_model = None
705
+ final_response_model = None
706
+
707
+ # Check if we have valid models
708
+ has_valid_request = not _is_invalid_model(request_model_attr)
709
+ has_valid_response = not _is_invalid_model(response_model_attr)
710
+ has_correct_model = correct_model_name is not None
711
+
712
+ if has_valid_response and has_valid_request:
713
+ # Both valid - prefer response model
714
+ final_request_model = response_model_attr
715
+ final_response_model = response_model_attr
716
+ elif has_valid_response:
717
+ # Only response is valid
718
+ final_request_model = response_model_attr
719
+ final_response_model = response_model_attr
720
+ elif has_valid_request:
721
+ # Only request is valid
722
+ final_request_model = request_model_attr
723
+ final_response_model = request_model_attr
724
+ elif has_correct_model:
725
+ # Neither is valid but we have correct_model_name
726
+ final_request_model = correct_model_name
727
+ final_response_model = correct_model_name
728
+
729
+ # Apply fixes only if we determined new values and they differ from current
730
+ if final_request_model and final_request_model != request_model_attr:
731
+ if self._should_fix_request_model(span_name, request_model_attr, response_model_attr):
732
+ if self.debug_enabled:
733
+ logger.debug(f"Fixing gen_ai.request.model: '{request_model_attr}' -> '{final_request_model}'")
734
+ target_attrs["gen_ai.request.model"] = final_request_model
735
+
736
+ if final_response_model and final_response_model != response_model_attr:
737
+ if self.debug_enabled:
738
+ logger.debug(f"Fixing gen_ai.response.model: '{response_model_attr}' -> '{final_response_model}'")
739
+ target_attrs["gen_ai.response.model"] = final_response_model
740
+ def _apply_attribute_fixes(
741
+ self,
742
+ *,
743
+ readable_span: ReadableSpan,
744
+ span_name: str,
745
+ system_attr: Optional[str],
746
+ correct_system: Optional[str],
747
+ request_model_attr: Optional[str],
748
+ response_model_attr: Optional[str],
749
+ correct_model_name: Optional[str],
750
+ ) -> None:
751
+ """
752
+ Apply system, model and token usage fixes to the span attributes.
753
+
754
+ Args:
755
+ readable_span: The span to fix
756
+ span_name: Name of the span being processed
757
+ system_attr: Current gen_ai.system value
758
+ correct_system: Correct system value
759
+ request_model_attr: Current request model value
760
+ response_model_attr: Current response model value
761
+ correct_model_name: Correct model name value
762
+ """
763
+ try:
764
+ target_attrs = self._get_target_attributes(readable_span)
765
+ if target_attrs is None:
766
+ logger.warning(f"Cannot apply fixes to span '{span_name}': no writable attributes found")
767
+ return
768
+
769
+ self._fix_system_attribute(target_attrs, system_attr, correct_system)
770
+ self._fix_model_attributes(target_attrs, span_name, request_model_attr, response_model_attr, correct_model_name)
771
+
772
+ # Ollama usage metrics
773
+ if _is_ollama_span(span_name):
774
+ self._fix_ollama_usage_attributes(target_attrs)
775
+
776
+ # Bedrock usage metrics
777
+ if _is_bedrock_span(span_name):
778
+ self._fix_bedrock_usage_attributes(target_attrs)
779
+ except Exception as e:
780
+ logger.error(f"ModelFixProcessor._apply_attribute_fixes failed for span '{span_name}': {e}", exc_info=self.debug_enabled)
781
+
782
+ def shutdown(self) -> None:
783
+ """
784
+ Called when the processor is shut down.
785
+
786
+ This is part of the SpanProcessor interface. Currently no cleanup is needed
787
+ as the processor doesn't hold any resources that require explicit cleanup.
788
+ """
789
+ pass
790
+
791
+ def force_flush(self, timeout_millis: int = 30000) -> bool:
792
+ """
793
+ Called to force flush any buffered spans.
794
+
795
+ This is part of the SpanProcessor interface. Since this processor doesn't
796
+ buffer spans (it processes them synchronously in on_end), this always returns True.
797
+
798
+ Args:
799
+ timeout_millis: Maximum time to wait for flush in milliseconds (unused)
800
+
801
+ Returns:
802
+ True indicating successful flush (no-op in this processor)
803
+ """
804
+ return True