langfuse-haystack 3.2.1__tar.gz → 3.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (22) hide show
  1. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/CHANGELOG.md +17 -0
  2. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/PKG-INFO +1 -1
  3. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/pyproject.toml +2 -2
  4. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/src/haystack_integrations/components/connectors/langfuse/langfuse_connector.py +5 -5
  5. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/src/haystack_integrations/tracing/langfuse/tracer.py +122 -82
  6. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/tests/test_tracer.py +268 -11
  7. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/tests/test_tracing.py +8 -8
  8. langfuse_haystack-3.2.1/pydoc/config.yml +0 -30
  9. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/.gitignore +0 -0
  10. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/LICENSE.txt +0 -0
  11. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/README.md +0 -0
  12. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/example/basic_rag.py +0 -0
  13. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/example/chat.py +0 -0
  14. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/example/requirements.txt +0 -0
  15. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/pydoc/config_docusaurus.yml +0 -0
  16. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/src/haystack_integrations/components/connectors/__init__.py +0 -0
  17. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/src/haystack_integrations/components/connectors/langfuse/__init__.py +0 -0
  18. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/src/haystack_integrations/components/connectors/py.typed +0 -0
  19. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/src/haystack_integrations/tracing/langfuse/__init__.py +0 -0
  20. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/src/haystack_integrations/tracing/py.typed +0 -0
  21. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/tests/__init__.py +0 -0
  22. {langfuse_haystack-3.2.1 → langfuse_haystack-3.3.1}/tests/test_langfuse_connector.py +0 -0
@@ -1,5 +1,22 @@
1
1
  # Changelog
2
2
 
3
+ ## [integrations/langfuse-v3.3.0] - 2025-11-21
4
+
5
+ ### 🚀 Features
6
+
7
+ - *(langfuse)* Embedder, retriever and generator as obs. type (#2497)
8
+
9
+ ### 🌀 Miscellaneous
10
+
11
+ - Enhancement: Adopt PEP 585 type hinting (part 4) (#2527)
12
+ - *(langfuse)* Log levels (#2522)
13
+
14
+ ## [integrations/langfuse-v3.2.1] - 2025-11-07
15
+
16
+ ### 🌀 Miscellaneous
17
+
18
+ - Chore: Upgrade langfuse dep, observation types require version>=3.3.1 (#2493)
19
+
3
20
  ## [integrations/langfuse-v3.2.0] - 2025-11-07
4
21
 
5
22
  ### 🐛 Bug Fixes
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: langfuse-haystack
3
- Version: 3.2.1
3
+ Version: 3.3.1
4
4
  Summary: Langfuse integration for Haystack
5
5
  Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/langfuse#readme
6
6
  Project-URL: Issues, https://github.com/deepset-ai/haystack-core-integrations/issues
@@ -46,7 +46,7 @@ installer = "uv"
46
46
  dependencies = ["haystack-pydoc-tools", "ruff"]
47
47
 
48
48
  [tool.hatch.envs.default.scripts]
49
- docs = ["pydoc-markdown pydoc/config.yml"]
49
+ docs = ["pydoc-markdown pydoc/config_docusaurus.yml"]
50
50
  fmt = "ruff check --fix {args} && ruff format {args}"
51
51
  fmt-check = "ruff check {args} && ruff format --check {args}"
52
52
 
@@ -82,7 +82,7 @@ allow-direct-references = true
82
82
 
83
83
 
84
84
  [tool.ruff]
85
- target-version = "py38"
85
+ target-version = "py39"
86
86
  line-length = 120
87
87
 
88
88
  [tool.ruff.lint]
@@ -2,7 +2,7 @@
2
2
  #
3
3
  # SPDX-License-Identifier: Apache-2.0
4
4
 
5
- from typing import Any, Dict, Optional
5
+ from typing import Any, Optional
6
6
 
7
7
  import httpx
8
8
  from haystack import component, default_from_dict, default_to_dict, logging, tracing
@@ -124,7 +124,7 @@ class LangfuseConnector:
124
124
  span_handler: Optional[SpanHandler] = None,
125
125
  *,
126
126
  host: Optional[str] = None,
127
- langfuse_client_kwargs: Optional[Dict[str, Any]] = None,
127
+ langfuse_client_kwargs: Optional[dict[str, Any]] = None,
128
128
  ) -> None:
129
129
  """
130
130
  Initialize the LangfuseConnector component.
@@ -172,7 +172,7 @@ class LangfuseConnector:
172
172
  tracing.enable_tracing(self.tracer)
173
173
 
174
174
  @component.output_types(name=str, trace_url=str, trace_id=str)
175
- def run(self, invocation_context: Optional[Dict[str, Any]] = None) -> Dict[str, str]:
175
+ def run(self, invocation_context: Optional[dict[str, Any]] = None) -> dict[str, str]:
176
176
  """
177
177
  Runs the LangfuseConnector component.
178
178
 
@@ -191,7 +191,7 @@ class LangfuseConnector:
191
191
  )
192
192
  return {"name": self.name, "trace_url": self.tracer.get_trace_url(), "trace_id": self.tracer.get_trace_id()}
193
193
 
194
- def to_dict(self) -> Dict[str, Any]:
194
+ def to_dict(self) -> dict[str, Any]:
195
195
  """
196
196
  Serialize this component to a dictionary.
197
197
 
@@ -218,7 +218,7 @@ class LangfuseConnector:
218
218
  )
219
219
 
220
220
  @classmethod
221
- def from_dict(cls, data: Dict[str, Any]) -> "LangfuseConnector":
221
+ def from_dict(cls, data: dict[str, Any]) -> "LangfuseConnector":
222
222
  """
223
223
  Deserialize this component from a dictionary.
224
224
 
@@ -4,13 +4,15 @@
4
4
 
5
5
  import contextlib
6
6
  import os
7
+ import sys
7
8
  from abc import ABC, abstractmethod
8
9
  from collections import Counter
10
+ from collections.abc import Iterator
9
11
  from contextlib import AbstractContextManager
10
12
  from contextvars import ContextVar
11
13
  from dataclasses import dataclass
12
14
  from datetime import datetime
13
- from typing import Any, Dict, Iterator, List, Literal, Optional
15
+ from typing import Any, Literal, Optional, cast
14
16
 
15
17
  from haystack import default_from_dict, default_to_dict, logging
16
18
  from haystack.dataclasses import ChatMessage
@@ -25,27 +27,6 @@ from langfuse.types import TraceMetadata
25
27
  logger = logging.getLogger(__name__)
26
28
 
27
29
  HAYSTACK_LANGFUSE_ENFORCE_FLUSH_ENV_VAR = "HAYSTACK_LANGFUSE_ENFORCE_FLUSH"
28
- _SUPPORTED_GENERATORS = [
29
- "AzureOpenAIGenerator",
30
- "OpenAIGenerator",
31
- "AnthropicGenerator",
32
- "HuggingFaceAPIGenerator",
33
- "HuggingFaceLocalGenerator",
34
- "CohereGenerator",
35
- "OllamaGenerator",
36
- ]
37
- _SUPPORTED_CHAT_GENERATORS = [
38
- "AmazonBedrockChatGenerator",
39
- "AzureOpenAIChatGenerator",
40
- "OpenAIChatGenerator",
41
- "AnthropicChatGenerator",
42
- "HuggingFaceAPIChatGenerator",
43
- "HuggingFaceLocalChatGenerator",
44
- "CohereChatGenerator",
45
- "OllamaChatGenerator",
46
- "GoogleGenAIChatGenerator",
47
- ]
48
- _ALL_SUPPORTED_GENERATORS = _SUPPORTED_GENERATORS + _SUPPORTED_CHAT_GENERATORS
49
30
 
50
31
  # These are the keys used by Haystack for traces and span.
51
32
  # We keep them here to avoid making typos when using them.
@@ -58,13 +39,16 @@ _COMPONENT_TYPE_KEY = "haystack.component.type"
58
39
  _COMPONENT_OUTPUT_KEY = "haystack.component.output"
59
40
  _COMPONENT_INPUT_KEY = "haystack.component.input"
60
41
 
42
+ # Type alias for observation span types
43
+ ObservationSpanType = Literal["tool", "agent", "retriever", "embedding", "generation"]
44
+
61
45
  # External session metadata for trace correlation (Haystack system)
62
46
  # Stores trace_id, user_id, session_id, tags, version for root trace creation
63
- tracing_context_var: ContextVar[Dict[Any, Any]] = ContextVar("tracing_context")
47
+ tracing_context_var: ContextVar[dict[Any, Any]] = ContextVar("tracing_context")
64
48
 
65
49
  # Internal span execution hierarchy for our tracer
66
50
  # Manages parent-child relationships and prevents cross-request span interleaving
67
- span_stack_var: ContextVar[Optional[List["LangfuseSpan"]]] = ContextVar("span_stack", default=None)
51
+ span_stack_var: ContextVar[Optional[list["LangfuseSpan"]]] = ContextVar("span_stack", default=None)
68
52
 
69
53
 
70
54
  class LangfuseSpan(Span):
@@ -81,7 +65,7 @@ class LangfuseSpan(Span):
81
65
  `langfuse.get_client().start_as_current_observation`.
82
66
  """
83
67
  self._span = context_manager.__enter__()
84
- self._data: Dict[str, Any] = {}
68
+ self._data: dict[str, Any] = {}
85
69
  self._context_manager = context_manager
86
70
 
87
71
  def set_tag(self, key: str, value: Any) -> None:
@@ -132,7 +116,7 @@ class LangfuseSpan(Span):
132
116
  """
133
117
  return self._span
134
118
 
135
- def get_data(self) -> Dict[str, Any]:
119
+ def get_data(self) -> dict[str, Any]:
136
120
  """
137
121
  Return the data associated with the span.
138
122
 
@@ -140,7 +124,7 @@ class LangfuseSpan(Span):
140
124
  """
141
125
  return self._data
142
126
 
143
- def get_correlation_data_for_logs(self) -> Dict[str, Any]:
127
+ def get_correlation_data_for_logs(self) -> dict[str, Any]:
144
128
  return {}
145
129
 
146
130
 
@@ -167,7 +151,7 @@ class SpanContext:
167
151
  name: str
168
152
  operation_name: str
169
153
  component_type: Optional[str]
170
- tags: Dict[str, Any]
154
+ tags: dict[str, Any]
171
155
  parent_span: Optional[Span]
172
156
  trace_name: str = "Haystack"
173
157
  public: bool = False
@@ -249,50 +233,39 @@ class SpanHandler(ABC):
249
233
  pass
250
234
 
251
235
  @classmethod
252
- def from_dict(cls, data: Dict[str, Any]) -> "SpanHandler":
236
+ def from_dict(cls, data: dict[str, Any]) -> "SpanHandler":
253
237
  return default_from_dict(cls, data)
254
238
 
255
- def to_dict(self) -> Dict[str, Any]:
239
+ def to_dict(self) -> dict[str, Any]:
256
240
  return default_to_dict(self)
257
241
 
258
242
 
259
- def _sanitize_usage_data(usage: Dict[str, Any]) -> Dict[str, Any]:
243
+ def _sanitize_usage_data(usage: dict[str, Any]) -> dict[str, Any]:
260
244
  """
261
- Sanitize usage data for Langfuse by flattening to a single-level dictionary.
245
+ Sanitize usage data for Langfuse by converting provider-specific keys to Langfuse standard keys.
262
246
 
263
- Langfuse's usage_details must be a flat dictionary with only numeric values. This function:
264
- - Flattens nested dictionaries using dot notation (e.g., cache_creation.input_tokens)
265
- - Keeps int and float values
266
- - Skips None, boolean, string, and other non-numeric types
247
+ Langfuse expects usage_details with standard keys: input_tokens, output_tokens, and total_tokens.
248
+ This function converts provider-specific keys to Langfuse's expected format:
249
+ - prompt_tokens -> input_tokens
250
+ - completion_tokens -> output_tokens
251
+ - total_tokens -> total_tokens (preserved as-is)
267
252
 
268
253
  :param usage: Raw usage dictionary from the provider.
269
- :returns: Flat dictionary with only numeric values (int or float).
254
+ :returns: Dictionary with Langfuse standard keys (input_tokens, output_tokens, total_tokens).
270
255
  """
271
256
  if not isinstance(usage, dict):
272
257
  return {}
273
258
 
274
- sanitized: Dict[str, Any] = {}
275
-
276
- def _flatten(data: Dict[str, Any], prefix: str = "") -> None:
277
- """Recursively flatten nested dictionaries."""
278
- for key, value in data.items():
279
- full_key = f"{prefix}.{key}" if prefix else key
280
-
281
- if value is None:
282
- # Skip None values (e.g., Anthropic's server_tool_use)
283
- continue
284
- elif isinstance(value, bool):
285
- # Skip boolean values
286
- continue
287
- elif isinstance(value, (int, float)):
288
- # Keep numeric values
289
- sanitized[full_key] = value
290
- elif isinstance(value, dict):
291
- # Recursively flatten nested dicts
292
- _flatten(value, full_key)
293
- # Skip strings and other non-numeric types (e.g., Anthropic's service_tier)
294
-
295
- _flatten(usage)
259
+ # Start with Langfuse standard keys from usage if present
260
+ sanitized: dict[str, Any] = {
261
+ k: v for k, v in usage.items() if k in ("input_tokens", "output_tokens", "total_tokens")
262
+ }
263
+ # Convert provider format to Langfuse standard keys if not already present
264
+ if "input_tokens" not in sanitized and "prompt_tokens" in usage:
265
+ sanitized["input_tokens"] = usage["prompt_tokens"]
266
+ if "output_tokens" not in sanitized and "completion_tokens" in usage:
267
+ sanitized["output_tokens"] = usage["completion_tokens"]
268
+
296
269
  return sanitized
297
270
 
298
271
 
@@ -316,7 +289,9 @@ class DefaultSpanHandler(SpanHandler):
316
289
  )
317
290
  # Create a new trace when there's no parent span
318
291
  span_context_manager = self.tracer.start_as_current_observation(
319
- name=context.trace_name, version=tracing_ctx.get("version"), as_type=root_span_type
292
+ name=context.trace_name,
293
+ version=tracing_ctx.get("version"),
294
+ as_type=root_span_type,
320
295
  )
321
296
 
322
297
  # Create LangfuseSpan which will handle entering the context manager
@@ -340,12 +315,26 @@ class DefaultSpanHandler(SpanHandler):
340
315
  span._span.update_trace(**trace_attrs)
341
316
 
342
317
  return span
343
- elif context.component_type == "ToolInvoker":
344
- return LangfuseSpan(self.tracer.start_as_current_observation(name=context.name, as_type="tool"))
318
+
319
+ span_type = None
320
+
321
+ if context.component_type == "ToolInvoker":
322
+ span_type = "tool"
345
323
  elif context.operation_name == "haystack.agent.run":
346
- return LangfuseSpan(self.tracer.start_as_current_observation(name=context.name, as_type="agent"))
347
- elif context.component_type in _ALL_SUPPORTED_GENERATORS:
348
- return LangfuseSpan(self.tracer.start_as_current_observation(name=context.name, as_type="generation"))
324
+ span_type = "agent"
325
+ elif context.component_type and context.component_type.endswith("Retriever"):
326
+ span_type = "retriever"
327
+ elif context.component_type and context.component_type.endswith("Embedder"):
328
+ span_type = "embedding"
329
+ elif context.component_type and context.component_type.endswith("Generator"):
330
+ span_type = "generation"
331
+
332
+ if span_type:
333
+ return LangfuseSpan(
334
+ self.tracer.start_as_current_observation(
335
+ name=context.name, as_type=cast(ObservationSpanType, span_type)
336
+ )
337
+ )
349
338
  else:
350
339
  return LangfuseSpan(self.tracer.start_as_current_span(name=context.name))
351
340
 
@@ -358,7 +347,7 @@ class DefaultSpanHandler(SpanHandler):
358
347
  span.raw_span().update_trace(input=coerced_input, output=coerced_output)
359
348
  # special case for ToolInvoker (to update the span name to be: `original_component_name - [tool_names]`)
360
349
  if component_type == "ToolInvoker":
361
- tool_names: List[str] = []
350
+ tool_names: list[str] = []
362
351
  messages = span.get_data().get(_COMPONENT_INPUT_KEY, {}).get("messages", [])
363
352
  for message in messages:
364
353
  if isinstance(message, ChatMessage) and message.tool_calls:
@@ -371,14 +360,7 @@ class DefaultSpanHandler(SpanHandler):
371
360
  formatted_names = [f"{name} (x{count})" if count > 1 else name for name, count in tool_counts.items()]
372
361
  span.raw_span().update(name=f"{tool_invoker_name} - {sorted(formatted_names)}")
373
362
 
374
- if component_type in _SUPPORTED_GENERATORS:
375
- meta = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("meta")
376
- if meta:
377
- usage = meta[0].get("usage")
378
- sanitized_usage = _sanitize_usage_data(usage) if usage else None
379
- span.raw_span().update(usage_details=sanitized_usage, model=meta[0].get("model"))
380
-
381
- if component_type in _SUPPORTED_CHAT_GENERATORS:
363
+ if component_type and component_type.endswith("ChatGenerator"):
382
364
  replies = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("replies")
383
365
  if replies:
384
366
  meta = replies[0].meta
@@ -396,6 +378,36 @@ class DefaultSpanHandler(SpanHandler):
396
378
  model=meta.get("model"),
397
379
  completion_start_time=completion_start_time,
398
380
  )
381
+ elif component_type and component_type.endswith("Generator"):
382
+ meta = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("meta")
383
+ if meta:
384
+ usage = meta[0].get("usage")
385
+ sanitized_usage = _sanitize_usage_data(usage) if usage else None
386
+ span.raw_span().update(usage_details=sanitized_usage, model=meta[0].get("model"))
387
+ elif component_type and component_type.endswith("Embedder"):
388
+ # Extract usage data from embedder output
389
+ output = span.get_data().get(_COMPONENT_OUTPUT_KEY, {})
390
+ meta = output.get("meta")
391
+
392
+ if meta and isinstance(meta, dict):
393
+ # Build update parameters with available data
394
+ update_params: dict[str, Any] = {}
395
+
396
+ # Try both common formats: 'usage' (OpenAI) or 'billed_units' (Cohere)
397
+ usage = meta.get("usage") or meta.get("billed_units")
398
+ if usage:
399
+ sanitized_usage = _sanitize_usage_data(usage)
400
+ if sanitized_usage:
401
+ update_params["usage_details"] = sanitized_usage
402
+
403
+ # Some embedders may provide model information
404
+ model = meta.get("model")
405
+ if model and isinstance(model, str):
406
+ update_params["model"] = model
407
+
408
+ # Single update call if we have data to update
409
+ if update_params:
410
+ span.raw_span().update(**update_params)
399
411
 
400
412
 
401
413
  class LangfuseTracer(Tracer):
@@ -429,7 +441,7 @@ class LangfuseTracer(Tracer):
429
441
  )
430
442
  self._tracer = tracer
431
443
  # Keep _context as deprecated shim to avoid AttributeError if anyone uses it
432
- self._context: List[LangfuseSpan] = []
444
+ self._context: list[LangfuseSpan] = []
433
445
  self._name = name
434
446
  self._public = public
435
447
  self.enforce_flush = os.getenv(HAYSTACK_LANGFUSE_ENFORCE_FLUSH_ENV_VAR, "true").lower() == "true"
@@ -438,7 +450,7 @@ class LangfuseTracer(Tracer):
438
450
 
439
451
  @contextlib.contextmanager
440
452
  def trace(
441
- self, operation_name: str, tags: Optional[Dict[str, Any]] = None, parent_span: Optional[Span] = None
453
+ self, operation_name: str, tags: Optional[dict[str, Any]] = None, parent_span: Optional[Span] = None
442
454
  ) -> Iterator[Span]:
443
455
  tags = tags or {}
444
456
  span_name = tags.get(_COMPONENT_NAME_KEY, operation_name)
@@ -470,16 +482,44 @@ class LangfuseTracer(Tracer):
470
482
 
471
483
  try:
472
484
  yield span
473
- finally:
474
- # Always clean up context, even if nested operations fail
485
+ except Exception:
486
+ # Exception occurred - capture exception info and pass to __exit__
487
+ # This allows Langfuse/OpenTelemetry to properly mark the span with ERROR level
488
+ exc_info = sys.exc_info()
475
489
  try:
476
490
  # Process span data (may fail with nested pipeline exceptions)
477
491
  self._span_handler.handle(span, component_type)
478
492
 
479
- # End span (may fail if span data is corrupted)
493
+ # End span with exception info (may fail if span data is corrupted)
494
+ raw_span = span.raw_span()
495
+ if span._context_manager is not None:
496
+ # Pass actual exception info to mark span as failed with ERROR level
497
+ span._context_manager.__exit__(*exc_info)
498
+ elif hasattr(raw_span, "end"):
499
+ # Only call end() if it's not a context manager
500
+ raw_span.end()
501
+ except Exception as cleanup_error:
502
+ # Log cleanup errors but don't let them corrupt context
503
+ logger.warning(
504
+ "Error during span cleanup for {operation_name}: {cleanup_error}",
505
+ operation_name=operation_name,
506
+ cleanup_error=cleanup_error,
507
+ )
508
+
509
+ # Re-raise the original exception
510
+ raise
511
+ else:
512
+ # No exception - clean exit with success status
513
+ # This preserves any manually-set log levels (WARNING, DEBUG)
514
+ try:
515
+ # Process span data
516
+ self._span_handler.handle(span, component_type)
517
+
518
+ # End span successfully
480
519
  raw_span = span.raw_span()
481
520
  # In v3, we need to properly exit context managers
482
521
  if span._context_manager is not None:
522
+ # No exception - pass None to indicate success
483
523
  span._context_manager.__exit__(None, None, None)
484
524
  elif hasattr(raw_span, "end"):
485
525
  # Only call end() if it's not a context manager
@@ -491,9 +531,9 @@ class LangfuseTracer(Tracer):
491
531
  operation_name=operation_name,
492
532
  cleanup_error=cleanup_error,
493
533
  )
494
- finally:
495
- # Restore previous span stack using saved token - ensures proper cleanup
496
- span_stack_var.reset(token)
534
+ finally:
535
+ # Restore previous span stack using saved token
536
+ span_stack_var.reset(token)
497
537
 
498
538
  if self.enforce_flush:
499
539
  self.flush()
@@ -204,20 +204,16 @@ class TestSanitizeUsageData:
204
204
  "completion_tokens": 449,
205
205
  }
206
206
  result = _sanitize_usage_data(usage)
207
- assert result == {
208
- "cache_creation.ephemeral_1h_input_tokens": 0,
209
- "cache_creation.ephemeral_5m_input_tokens": 0,
210
- "cache_creation_input_tokens": 0,
211
- "cache_read_input_tokens": 0,
212
- "prompt_tokens": 25,
213
- "completion_tokens": 449,
214
- }
207
+ assert result["input_tokens"] == 25
208
+ assert result["output_tokens"] == 449
215
209
 
216
210
  def test_openai_usage_preserved(self):
217
211
  """Test OpenAI/Cohere flat dict with only numeric values works unchanged"""
218
212
  usage = {"prompt_tokens": 29, "completion_tokens": 267, "total_tokens": 296}
219
213
  result = _sanitize_usage_data(usage)
220
- assert result == {"prompt_tokens": 29, "completion_tokens": 267, "total_tokens": 296}
214
+ assert result["input_tokens"] == 29
215
+ assert result["output_tokens"] == 267
216
+ assert result["total_tokens"] == 296
221
217
 
222
218
  def test_empty_and_invalid_input(self):
223
219
  """Test edge cases return empty dict"""
@@ -295,6 +291,207 @@ class TestDefaultSpanHandler:
295
291
  "completion_start_time": None,
296
292
  }
297
293
 
294
+ def test_create_span_custom_chat_generator(self):
295
+ """Test that custom chat generators create 'generation' span type."""
296
+ mock_client = Mock()
297
+ mock_client.start_as_current_span = Mock(return_value=MockContextManager())
298
+ mock_client.start_as_current_observation = Mock(return_value=MockContextManager())
299
+
300
+ handler = DefaultSpanHandler()
301
+ handler.init_tracer(mock_client)
302
+
303
+ context = SpanContext(
304
+ name="MistralChatGenerator",
305
+ operation_name="haystack.component.run",
306
+ component_type="MistralChatGenerator",
307
+ tags={},
308
+ parent_span=LangfuseSpan(mock_client.start_as_current_span()),
309
+ )
310
+
311
+ span = handler.create_span(context)
312
+ assert isinstance(span, LangfuseSpan)
313
+ mock_client.start_as_current_observation.assert_called_once_with(
314
+ name="MistralChatGenerator", as_type="generation"
315
+ )
316
+
317
+ def test_create_span_custom_generator(self):
318
+ """Test that custom generators create 'generation' span type."""
319
+ mock_client = Mock()
320
+ mock_client.start_as_current_span = Mock(return_value=MockContextManager())
321
+ mock_client.start_as_current_observation = Mock(return_value=MockContextManager())
322
+
323
+ handler = DefaultSpanHandler()
324
+ handler.init_tracer(mock_client)
325
+
326
+ context = SpanContext(
327
+ name="CustomAPIGenerator",
328
+ operation_name="haystack.component.run",
329
+ component_type="CustomAPIGenerator",
330
+ tags={},
331
+ parent_span=LangfuseSpan(mock_client.start_as_current_span()),
332
+ )
333
+
334
+ span = handler.create_span(context)
335
+ assert isinstance(span, LangfuseSpan)
336
+ mock_client.start_as_current_observation.assert_called_once_with(
337
+ name="CustomAPIGenerator", as_type="generation"
338
+ )
339
+
340
+ def test_create_span_retriever(self):
341
+ """Test that retrievers create 'retriever' span type."""
342
+ mock_client = Mock()
343
+ mock_client.start_as_current_span = Mock(return_value=MockContextManager())
344
+ mock_client.start_as_current_observation = Mock(return_value=MockContextManager())
345
+
346
+ handler = DefaultSpanHandler()
347
+ handler.init_tracer(mock_client)
348
+
349
+ context = SpanContext(
350
+ name="InMemoryBM25Retriever",
351
+ operation_name="haystack.component.run",
352
+ component_type="InMemoryBM25Retriever",
353
+ tags={},
354
+ parent_span=LangfuseSpan(mock_client.start_as_current_span()),
355
+ )
356
+
357
+ span = handler.create_span(context)
358
+ assert isinstance(span, LangfuseSpan)
359
+ mock_client.start_as_current_observation.assert_called_once_with(
360
+ name="InMemoryBM25Retriever", as_type="retriever"
361
+ )
362
+
363
+ def test_create_span_embedder(self):
364
+ """Test that embedders create 'embedding' span type."""
365
+ mock_client = Mock()
366
+ mock_client.start_as_current_span = Mock(return_value=MockContextManager())
367
+ mock_client.start_as_current_observation = Mock(return_value=MockContextManager())
368
+
369
+ handler = DefaultSpanHandler()
370
+ handler.init_tracer(mock_client)
371
+
372
+ context = SpanContext(
373
+ name="SentenceTransformersDocumentEmbedder",
374
+ operation_name="haystack.component.run",
375
+ component_type="SentenceTransformersDocumentEmbedder",
376
+ tags={},
377
+ parent_span=LangfuseSpan(mock_client.start_as_current_span()),
378
+ )
379
+
380
+ span = handler.create_span(context)
381
+ assert isinstance(span, LangfuseSpan)
382
+ mock_client.start_as_current_observation.assert_called_once_with(
383
+ name="SentenceTransformersDocumentEmbedder", as_type="embedding"
384
+ )
385
+
386
+ def test_create_span_non_component(self):
387
+ """Test that non-matching components create regular spans."""
388
+ mock_client = Mock()
389
+ mock_client.start_as_current_span = Mock(return_value=MockContextManager())
390
+ mock_client.start_as_current_observation = Mock(return_value=MockContextManager())
391
+
392
+ handler = DefaultSpanHandler()
393
+ handler.init_tracer(mock_client)
394
+
395
+ context = SpanContext(
396
+ name="DocumentJoiner",
397
+ operation_name="haystack.component.run",
398
+ component_type="DocumentJoiner",
399
+ tags={},
400
+ parent_span=LangfuseSpan(mock_client.start_as_current_span()),
401
+ )
402
+
403
+ span = handler.create_span(context)
404
+ assert isinstance(span, LangfuseSpan)
405
+ # Non-matching components should use start_as_current_span, not start_as_current_observation
406
+ mock_client.start_as_current_observation.assert_not_called()
407
+ # Verify start_as_current_span was called for the actual span creation (not just parent)
408
+ assert mock_client.start_as_current_span.call_count == 2 # Once for parent, once for the span
409
+
410
+ def test_handle_embedder_with_openai_format(self):
411
+ """Test that embedder usage is extracted in OpenAI format."""
412
+ mock_span = Mock()
413
+ mock_span.raw_span.return_value = mock_span
414
+ mock_span.get_data.return_value = {
415
+ "haystack.component.type": "OpenAITextEmbedder",
416
+ "haystack.component.output": {
417
+ "embedding": [0.1, 0.2, 0.3],
418
+ "meta": {"model": "custom-model", "usage": {"prompt_tokens": 15, "total_tokens": 15}},
419
+ },
420
+ }
421
+
422
+ handler = DefaultSpanHandler()
423
+ handler.handle(mock_span, component_type="OpenAITextEmbedder")
424
+
425
+ assert mock_span.update.call_count == 1
426
+ update_args = mock_span.update.call_args_list[0][1]
427
+ assert update_args["model"] == "custom-model"
428
+ assert update_args["usage_details"] == {"input_tokens": 15, "total_tokens": 15}
429
+
430
+ def test_handle_embedder_with_cohere_format(self):
431
+ """Test that embedder usage is extracted in Cohere billed_units format."""
432
+ mock_span = Mock()
433
+ mock_span.raw_span.return_value = mock_span
434
+ mock_span.get_data.return_value = {
435
+ "haystack.component.type": "CohereTextEmbedder",
436
+ "haystack.component.output": {
437
+ "embedding": [0.1, 0.2, 0.3],
438
+ "meta": {"api_version": {"version": "1"}, "billed_units": {"input_tokens": 4}},
439
+ },
440
+ }
441
+
442
+ handler = DefaultSpanHandler()
443
+ handler.handle(mock_span, component_type="CohereTextEmbedder")
444
+
445
+ assert mock_span.update.call_count == 1
446
+ assert mock_span.update.call_args_list[0][1] == {"usage_details": {"input_tokens": 4}}
447
+
448
+ def test_handle_embedder_without_usage(self):
449
+ """Test that embedders without usage data are handled gracefully."""
450
+ mock_span = Mock()
451
+ mock_span.raw_span.return_value = mock_span
452
+ mock_span.get_data.return_value = {
453
+ "haystack.component.type": "SentenceTransformersTextEmbedder",
454
+ "haystack.component.output": {
455
+ "embedding": [0.1, 0.2, 0.3],
456
+ "meta": {}, # No usage data
457
+ },
458
+ }
459
+
460
+ handler = DefaultSpanHandler()
461
+ handler.handle(mock_span, component_type="SentenceTransformersTextEmbedder")
462
+
463
+ # Should not call update when no usage data is available
464
+ assert mock_span.update.call_count == 0
465
+
466
+ def test_handle_embedder_with_nested_usage(self):
467
+ """Test that embedders with nested usage data are sanitized correctly."""
468
+ mock_span = Mock()
469
+ mock_span.raw_span.return_value = mock_span
470
+ mock_span.get_data.return_value = {
471
+ "haystack.component.type": "CustomEmbedder",
472
+ "haystack.component.output": {
473
+ "embedding": [0.1, 0.2, 0.3],
474
+ "meta": {
475
+ "model": "custom-model",
476
+ "usage": {
477
+ "cache_creation": {"input_tokens": 10},
478
+ "cache_read": {"input_tokens": 5},
479
+ "total_tokens": 15,
480
+ },
481
+ },
482
+ },
483
+ }
484
+
485
+ handler = DefaultSpanHandler()
486
+ handler.handle(mock_span, component_type="CustomEmbedder")
487
+
488
+ assert mock_span.update.call_count == 1
489
+ # Only adds total_tokens as Langfuse standard key (no prompt_tokens/completion_tokens to convert)
490
+ assert mock_span.update.call_args_list[0][1] == {
491
+ "usage_details": {"total_tokens": 15},
492
+ "model": "custom-model",
493
+ }
494
+
298
495
 
299
496
  class TestCustomSpanHandler:
300
497
  def test_handle(self):
@@ -338,13 +535,22 @@ class TestLangfuseTracer:
338
535
  mock_raw_span.metadata = {"tag1": "value1", "tag2": "value2"}
339
536
 
340
537
  with patch("haystack_integrations.tracing.langfuse.tracer.LangfuseSpan") as mock_langfuse_span:
538
+ mock_context_manager = MockContextManager()
539
+ mock_context_manager._span = mock_raw_span
540
+
341
541
  mock_span_instance = mock_langfuse_span.return_value
342
542
  mock_span_instance.raw_span.return_value = mock_raw_span
543
+ # Return a proper dict to prevent MagicMock from being truthy in handle() checks.
544
+ # When get_data() returns a MagicMock, `span.get_data().get(key) is not None` is True
545
+ # because MagicMock().get() returns another MagicMock (truthy). This triggers
546
+ # tracing_utils.coerce_tag_value() with MagicMock objects, which can hang on
547
+ # Linux Python 3.9/3.13 due to platform-specific MagicMock iteration behavior.
548
+ mock_span_instance.get_data.return_value = {}
549
+ mock_span_instance._context_manager = mock_context_manager
343
550
 
344
- mock_context_manager = MockContextManager()
345
- mock_context_manager._span = mock_raw_span
346
551
  mock_tracer = MagicMock()
347
552
  mock_tracer.start_as_current_span.return_value = mock_context_manager
553
+ mock_tracer.start_as_current_observation.return_value = mock_context_manager
348
554
 
349
555
  tracer = LangfuseTracer(tracer=mock_tracer, name="Haystack", public=False)
350
556
 
@@ -574,3 +780,54 @@ class TestLangfuseTracer:
574
780
  assert task2_spans[1][2] == task2_inner # current_span during inner
575
781
  assert task2_spans[2][2] == task2_outer # current_span after inner
576
782
  assert task2_spans[3][2] is None # current_span after outer
783
+
784
+ def test_trace_exception_handling(self):
785
+ """
786
+ Test that exceptions are properly captured and passed to span __exit__.
787
+
788
+ This verifies the new exception handling behavior where:
789
+ - Exception case: __exit__() receives (exc_type, exc_val, exc_tb)
790
+ - Success case: __exit__() receives (None, None, None)
791
+ """
792
+ # Create a mock context manager that tracks how __exit__ was called
793
+ mock_exit_calls = []
794
+
795
+ class TrackingContextManager:
796
+ def __init__(self):
797
+ self._span = MockSpan()
798
+
799
+ def __enter__(self):
800
+ return self._span
801
+
802
+ def __exit__(self, exc_type, exc_val, exc_tb):
803
+ # Track what was passed to __exit__
804
+ mock_exit_calls.append((exc_type, exc_val, exc_tb))
805
+ return False # Don't suppress exceptions
806
+
807
+ mock_client = MockLangfuseClient()
808
+ mock_client._mock_context_manager = TrackingContextManager()
809
+
810
+ tracer = LangfuseTracer(tracer=mock_client, name="Test", public=False)
811
+
812
+ # Test 1: Exception case - __exit__ should receive exception info
813
+ mock_exit_calls.clear()
814
+ error_msg = "test error"
815
+ with pytest.raises(ValueError, match="test error"):
816
+ with tracer.trace("test_operation"):
817
+ raise ValueError(error_msg)
818
+
819
+ assert len(mock_exit_calls) == 1
820
+ assert mock_exit_calls[0][0] is ValueError # exc_type
821
+ assert str(mock_exit_calls[0][1]) == error_msg # exc_val
822
+ assert mock_exit_calls[0][2] is not None # exc_tb (traceback)
823
+
824
+ # Test 2: Success case - __exit__ should receive (None, None, None)
825
+ mock_exit_calls.clear()
826
+ with tracer.trace("test_operation"):
827
+ pass # No exception
828
+
829
+ assert len(mock_exit_calls) == 1
830
+ assert mock_exit_calls[0] == (None, None, None)
831
+
832
+ # Test 3: Verify span stack is cleaned up after exception
833
+ assert tracer.current_span() is None
@@ -5,7 +5,7 @@
5
5
  import json
6
6
  import os
7
7
  import time
8
- from typing import Any, Dict, List
8
+ from typing import Any
9
9
  from urllib.parse import urlparse
10
10
 
11
11
  import pytest
@@ -38,7 +38,7 @@ os.environ.setdefault("LANGFUSE_HOST", "https://cloud.langfuse.com")
38
38
  def poll_langfuse(url: str):
39
39
  """Utility function to poll Langfuse API until the trace is ready"""
40
40
  # Initial wait for trace creation
41
- time.sleep(10)
41
+ time.sleep(30)
42
42
 
43
43
  auth = HTTPBasicAuth(os.environ["LANGFUSE_PUBLIC_KEY"], os.environ["LANGFUSE_SECRET_KEY"])
44
44
 
@@ -137,8 +137,8 @@ def test_tracing_with_sub_pipelines():
137
137
  self.sub_pipeline = Pipeline()
138
138
  self.sub_pipeline.add_component("llm", OpenAIChatGenerator())
139
139
 
140
- @component.output_types(replies=List[ChatMessage])
141
- def run(self, messages: List[ChatMessage]) -> Dict[str, Any]:
140
+ @component.output_types(replies=list[ChatMessage])
141
+ def run(self, messages: list[ChatMessage]) -> dict[str, Any]:
142
142
  return {"replies": self.sub_pipeline.run(data={"llm": {"messages": messages}})["llm"]["replies"]}
143
143
 
144
144
  @component
@@ -149,8 +149,8 @@ def test_tracing_with_sub_pipelines():
149
149
  self.sub_pipeline.add_component("sub_llm", SubGenerator())
150
150
  self.sub_pipeline.connect("prompt_builder.prompt", "sub_llm.messages")
151
151
 
152
- @component.output_types(replies=List[ChatMessage])
153
- def run(self, messages: List[ChatMessage]) -> Dict[str, Any]:
152
+ @component.output_types(replies=list[ChatMessage])
153
+ def run(self, messages: list[ChatMessage]) -> dict[str, Any]:
154
154
  return {
155
155
  "replies": self.sub_pipeline.run(
156
156
  data={"prompt_builder": {"template": messages, "template_variables": {"location": "Berlin"}}}
@@ -195,8 +195,8 @@ def test_tracing_with_sub_pipelines():
195
195
  # There should be two observations for the haystack.pipeline.run span: one for each sub pipeline
196
196
  # Main pipeline is stored under the name "Sub-pipeline example"
197
197
  assert len(haystack_pipeline_run_observations) == 2
198
- assert "prompt_builder" in str(haystack_pipeline_run_observations[0])
199
- assert "llm" in str(haystack_pipeline_run_observations[1])
198
+ # Verify both observations are pipeline runs (less brittle than checking for component names)
199
+ assert all(obs["name"] == "haystack.pipeline.run" for obs in haystack_pipeline_run_observations)
200
200
 
201
201
 
202
202
  @pytest.mark.skipif(
@@ -1,30 +0,0 @@
1
- loaders:
2
- - type: haystack_pydoc_tools.loaders.CustomPythonLoader
3
- search_path: [../src]
4
- modules: [
5
- "haystack_integrations.components.connectors.langfuse.langfuse_connector",
6
- "haystack_integrations.tracing.langfuse.tracer",
7
- ]
8
- ignore_when_discovered: ["__init__"]
9
- processors:
10
- - type: filter
11
- expression:
12
- documented_only: true
13
- do_not_filter_modules: false
14
- skip_empty_modules: true
15
- - type: smart
16
- - type: crossref
17
- renderer:
18
- type: haystack_pydoc_tools.renderers.ReadmeIntegrationRenderer
19
- excerpt: Langfuse integration for Haystack
20
- category_slug: integrations-api
21
- title: langfuse
22
- slug: integrations-langfuse
23
- order: 136
24
- markdown:
25
- descriptive_class_title: false
26
- classdef_code_block: false
27
- descriptive_module_title: true
28
- add_method_class_prefix: true
29
- add_member_class_prefix: false
30
- filename: _readme_langfuse.md