langfuse-haystack 3.3.0__tar.gz → 4.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (22) hide show
  1. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/CHANGELOG.md +26 -0
  2. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/PKG-INFO +3 -4
  3. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/pyproject.toml +4 -10
  4. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/src/haystack_integrations/components/connectors/langfuse/langfuse_connector.py +8 -8
  5. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/src/haystack_integrations/tracing/langfuse/tracer.py +51 -38
  6. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/tests/test_tracer.py +102 -13
  7. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/tests/test_tracing.py +3 -3
  8. langfuse_haystack-3.3.0/pydoc/config.yml +0 -30
  9. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/.gitignore +0 -0
  10. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/LICENSE.txt +0 -0
  11. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/README.md +0 -0
  12. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/example/basic_rag.py +0 -0
  13. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/example/chat.py +0 -0
  14. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/example/requirements.txt +0 -0
  15. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/pydoc/config_docusaurus.yml +0 -0
  16. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/src/haystack_integrations/components/connectors/__init__.py +0 -0
  17. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/src/haystack_integrations/components/connectors/langfuse/__init__.py +0 -0
  18. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/src/haystack_integrations/components/connectors/py.typed +0 -0
  19. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/src/haystack_integrations/tracing/langfuse/__init__.py +0 -0
  20. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/src/haystack_integrations/tracing/py.typed +0 -0
  21. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/tests/__init__.py +0 -0
  22. {langfuse_haystack-3.3.0 → langfuse_haystack-4.0.0}/tests/test_langfuse_connector.py +0 -0
@@ -1,5 +1,31 @@
1
1
  # Changelog
2
2
 
3
+ ## [integrations/langfuse-v3.3.1] - 2025-12-10
4
+
5
+ ### 🚀 Features
6
+
7
+ - *(langfuse)* Add embedder usage metrics for langfuse (#2542)
8
+
9
+ ### 🐛 Bug Fixes
10
+
11
+ - Correct token usage accounting, fix trace polling in tests (#2594)
12
+
13
+ ### 🧹 Chores
14
+
15
+ - Remove Readme API CI workflow and configs (#2573)
16
+
17
+
18
+ ## [integrations/langfuse-v3.3.0] - 2025-11-21
19
+
20
+ ### 🚀 Features
21
+
22
+ - *(langfuse)* Embedder, retriever and generator as obs. type (#2497)
23
+
24
+ ### 🌀 Miscellaneous
25
+
26
+ - Enhancement: Adopt PEP 585 type hinting (part 4) (#2527)
27
+ - *(langfuse)* Log levels (#2522)
28
+
3
29
  ## [integrations/langfuse-v3.2.1] - 2025-11-07
4
30
 
5
31
  ### 🌀 Miscellaneous
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: langfuse-haystack
3
- Version: 3.3.0
3
+ Version: 4.0.0
4
4
  Summary: Langfuse integration for Haystack
5
5
  Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/langfuse#readme
6
6
  Project-URL: Issues, https://github.com/deepset-ai/haystack-core-integrations/issues
@@ -10,15 +10,14 @@ License-Expression: Apache-2.0
10
10
  License-File: LICENSE.txt
11
11
  Classifier: Development Status :: 4 - Beta
12
12
  Classifier: Programming Language :: Python
13
- Classifier: Programming Language :: Python :: 3.9
14
13
  Classifier: Programming Language :: Python :: 3.10
15
14
  Classifier: Programming Language :: Python :: 3.11
16
15
  Classifier: Programming Language :: Python :: 3.12
17
16
  Classifier: Programming Language :: Python :: 3.13
18
17
  Classifier: Programming Language :: Python :: Implementation :: CPython
19
18
  Classifier: Programming Language :: Python :: Implementation :: PyPy
20
- Requires-Python: >=3.9
21
- Requires-Dist: haystack-ai>=2.17.1
19
+ Requires-Python: >=3.10
20
+ Requires-Dist: haystack-ai>=2.22.0
22
21
  Requires-Dist: langfuse<4.0.0,>=3.3.1
23
22
  Description-Content-Type: text/markdown
24
23
 
@@ -7,14 +7,13 @@ name = "langfuse-haystack"
7
7
  dynamic = ["version"]
8
8
  description = "Langfuse integration for Haystack"
9
9
  readme = "README.md"
10
- requires-python = ">=3.9"
10
+ requires-python = ">=3.10"
11
11
  license = "Apache-2.0"
12
12
  keywords = []
13
13
  authors = [{ name = "deepset GmbH", email = "info@deepset.ai" }]
14
14
  classifiers = [
15
15
  "Development Status :: 4 - Beta",
16
16
  "Programming Language :: Python",
17
- "Programming Language :: Python :: 3.9",
18
17
  "Programming Language :: Python :: 3.10",
19
18
  "Programming Language :: Python :: 3.11",
20
19
  "Programming Language :: Python :: 3.12",
@@ -22,7 +21,7 @@ classifiers = [
22
21
  "Programming Language :: Python :: Implementation :: CPython",
23
22
  "Programming Language :: Python :: Implementation :: PyPy",
24
23
  ]
25
- dependencies = ["haystack-ai>=2.17.1", "langfuse>=3.3.1, <4.0.0"]
24
+ dependencies = ["haystack-ai>=2.22.0", "langfuse>=3.3.1, <4.0.0"]
26
25
 
27
26
  [project.urls]
28
27
  Documentation = "https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/langfuse#readme"
@@ -46,8 +45,8 @@ installer = "uv"
46
45
  dependencies = ["haystack-pydoc-tools", "ruff"]
47
46
 
48
47
  [tool.hatch.envs.default.scripts]
49
- docs = ["pydoc-markdown pydoc/config.yml"]
50
- fmt = "ruff check --fix {args} && ruff format {args}"
48
+ docs = ["pydoc-markdown pydoc/config_docusaurus.yml"]
49
+ fmt = "ruff check --fix {args}; ruff format {args}"
51
50
  fmt-check = "ruff check {args} && ruff format --check {args}"
52
51
 
53
52
  [tool.hatch.envs.test]
@@ -82,7 +81,6 @@ allow-direct-references = true
82
81
 
83
82
 
84
83
  [tool.ruff]
85
- target-version = "py39"
86
84
  line-length = 120
87
85
 
88
86
  [tool.ruff.lint]
@@ -128,10 +126,6 @@ ignore = [
128
126
  # Asserts
129
127
  "S101",
130
128
  ]
131
- unfixable = [
132
- # Don't touch unused imports
133
- "F401",
134
- ]
135
129
 
136
130
  [tool.ruff.lint.isort]
137
131
  known-first-party = ["haystack_integrations"]
@@ -2,7 +2,7 @@
2
2
  #
3
3
  # SPDX-License-Identifier: Apache-2.0
4
4
 
5
- from typing import Any, Optional
5
+ from typing import Any
6
6
 
7
7
  import httpx
8
8
  from haystack import component, default_from_dict, default_to_dict, logging, tracing
@@ -118,13 +118,13 @@ class LangfuseConnector:
118
118
  self,
119
119
  name: str,
120
120
  public: bool = False,
121
- public_key: Optional[Secret] = Secret.from_env_var("LANGFUSE_PUBLIC_KEY"), # noqa: B008
122
- secret_key: Optional[Secret] = Secret.from_env_var("LANGFUSE_SECRET_KEY"), # noqa: B008
123
- httpx_client: Optional[httpx.Client] = None,
124
- span_handler: Optional[SpanHandler] = None,
121
+ public_key: Secret | None = Secret.from_env_var("LANGFUSE_PUBLIC_KEY"), # noqa: B008
122
+ secret_key: Secret | None = Secret.from_env_var("LANGFUSE_SECRET_KEY"), # noqa: B008
123
+ httpx_client: httpx.Client | None = None,
124
+ span_handler: SpanHandler | None = None,
125
125
  *,
126
- host: Optional[str] = None,
127
- langfuse_client_kwargs: Optional[dict[str, Any]] = None,
126
+ host: str | None = None,
127
+ langfuse_client_kwargs: dict[str, Any] | None = None,
128
128
  ) -> None:
129
129
  """
130
130
  Initialize the LangfuseConnector component.
@@ -172,7 +172,7 @@ class LangfuseConnector:
172
172
  tracing.enable_tracing(self.tracer)
173
173
 
174
174
  @component.output_types(name=str, trace_url=str, trace_id=str)
175
- def run(self, invocation_context: Optional[dict[str, Any]] = None) -> dict[str, str]:
175
+ def run(self, invocation_context: dict[str, Any] | None = None) -> dict[str, str]:
176
176
  """
177
177
  Runs the LangfuseConnector component.
178
178
 
@@ -12,7 +12,7 @@ from contextlib import AbstractContextManager
12
12
  from contextvars import ContextVar
13
13
  from dataclasses import dataclass
14
14
  from datetime import datetime
15
- from typing import Any, Literal, Optional, cast
15
+ from typing import Any, Literal, cast
16
16
 
17
17
  from haystack import default_from_dict, default_to_dict, logging
18
18
  from haystack.dataclasses import ChatMessage
@@ -48,7 +48,7 @@ tracing_context_var: ContextVar[dict[Any, Any]] = ContextVar("tracing_context")
48
48
 
49
49
  # Internal span execution hierarchy for our tracer
50
50
  # Manages parent-child relationships and prevents cross-request span interleaving
51
- span_stack_var: ContextVar[Optional[list["LangfuseSpan"]]] = ContextVar("span_stack", default=None)
51
+ span_stack_var: ContextVar[list["LangfuseSpan"] | None] = ContextVar("span_stack", default=None)
52
52
 
53
53
 
54
54
  class LangfuseSpan(Span):
@@ -150,9 +150,9 @@ class SpanContext:
150
150
 
151
151
  name: str
152
152
  operation_name: str
153
- component_type: Optional[str]
153
+ component_type: str | None
154
154
  tags: dict[str, Any]
155
- parent_span: Optional[Span]
155
+ parent_span: Span | None
156
156
  trace_name: str = "Haystack"
157
157
  public: bool = False
158
158
 
@@ -189,7 +189,7 @@ class SpanHandler(ABC):
189
189
  """
190
190
 
191
191
  def __init__(self) -> None:
192
- self.tracer: Optional[langfuse.Langfuse] = None
192
+ self.tracer: langfuse.Langfuse | None = None
193
193
 
194
194
  def init_tracer(self, tracer: langfuse.Langfuse) -> None:
195
195
  """
@@ -215,7 +215,7 @@ class SpanHandler(ABC):
215
215
  pass
216
216
 
217
217
  @abstractmethod
218
- def handle(self, span: LangfuseSpan, component_type: Optional[str]) -> None:
218
+ def handle(self, span: LangfuseSpan, component_type: str | None) -> None:
219
219
  """
220
220
  Process a span after component execution by attaching metadata and metrics.
221
221
 
@@ -242,41 +242,30 @@ class SpanHandler(ABC):
242
242
 
243
243
  def _sanitize_usage_data(usage: dict[str, Any]) -> dict[str, Any]:
244
244
  """
245
- Sanitize usage data for Langfuse by flattening to a single-level dictionary.
245
+ Sanitize usage data for Langfuse by converting provider-specific keys to Langfuse standard keys.
246
246
 
247
- Langfuse's usage_details must be a flat dictionary with only numeric values. This function:
248
- - Flattens nested dictionaries using dot notation (e.g., cache_creation.input_tokens)
249
- - Keeps int and float values
250
- - Skips None, boolean, string, and other non-numeric types
247
+ Langfuse expects usage_details with standard keys: input_tokens, output_tokens, and total_tokens.
248
+ This function converts provider-specific keys to Langfuse's expected format:
249
+ - prompt_tokens -> input_tokens
250
+ - completion_tokens -> output_tokens
251
+ - total_tokens -> total_tokens (preserved as-is)
251
252
 
252
253
  :param usage: Raw usage dictionary from the provider.
253
- :returns: Flat dictionary with only numeric values (int or float).
254
+ :returns: Dictionary with Langfuse standard keys (input_tokens, output_tokens, total_tokens).
254
255
  """
255
256
  if not isinstance(usage, dict):
256
257
  return {}
257
258
 
258
- sanitized: dict[str, Any] = {}
259
-
260
- def _flatten(data: dict[str, Any], prefix: str = "") -> None:
261
- """Recursively flatten nested dictionaries."""
262
- for key, value in data.items():
263
- full_key = f"{prefix}.{key}" if prefix else key
264
-
265
- if value is None:
266
- # Skip None values (e.g., Anthropic's server_tool_use)
267
- continue
268
- elif isinstance(value, bool):
269
- # Skip boolean values
270
- continue
271
- elif isinstance(value, (int, float)):
272
- # Keep numeric values
273
- sanitized[full_key] = value
274
- elif isinstance(value, dict):
275
- # Recursively flatten nested dicts
276
- _flatten(value, full_key)
277
- # Skip strings and other non-numeric types (e.g., Anthropic's service_tier)
278
-
279
- _flatten(usage)
259
+ # Start with Langfuse standard keys from usage if present
260
+ sanitized: dict[str, Any] = {
261
+ k: v for k, v in usage.items() if k in ("input_tokens", "output_tokens", "total_tokens")
262
+ }
263
+ # Convert provider format to Langfuse standard keys if not already present
264
+ if "input_tokens" not in sanitized and "prompt_tokens" in usage:
265
+ sanitized["input_tokens"] = usage["prompt_tokens"]
266
+ if "output_tokens" not in sanitized and "completion_tokens" in usage:
267
+ sanitized["output_tokens"] = usage["completion_tokens"]
268
+
280
269
  return sanitized
281
270
 
282
271
 
@@ -349,7 +338,7 @@ class DefaultSpanHandler(SpanHandler):
349
338
  else:
350
339
  return LangfuseSpan(self.tracer.start_as_current_span(name=context.name))
351
340
 
352
- def handle(self, span: LangfuseSpan, component_type: Optional[str]) -> None:
341
+ def handle(self, span: LangfuseSpan, component_type: str | None) -> None:
353
342
  # If the span is at the pipeline level, we add input and output keys to the span
354
343
  at_pipeline_level = span.get_data().get(_PIPELINE_INPUT_KEY) is not None
355
344
  if at_pipeline_level:
@@ -395,6 +384,30 @@ class DefaultSpanHandler(SpanHandler):
395
384
  usage = meta[0].get("usage")
396
385
  sanitized_usage = _sanitize_usage_data(usage) if usage else None
397
386
  span.raw_span().update(usage_details=sanitized_usage, model=meta[0].get("model"))
387
+ elif component_type and component_type.endswith("Embedder"):
388
+ # Extract usage data from embedder output
389
+ output = span.get_data().get(_COMPONENT_OUTPUT_KEY, {})
390
+ meta = output.get("meta")
391
+
392
+ if meta and isinstance(meta, dict):
393
+ # Build update parameters with available data
394
+ update_params: dict[str, Any] = {}
395
+
396
+ # Try both common formats: 'usage' (OpenAI) or 'billed_units' (Cohere)
397
+ usage = meta.get("usage") or meta.get("billed_units")
398
+ if usage:
399
+ sanitized_usage = _sanitize_usage_data(usage)
400
+ if sanitized_usage:
401
+ update_params["usage_details"] = sanitized_usage
402
+
403
+ # Some embedders may provide model information
404
+ model = meta.get("model")
405
+ if model and isinstance(model, str):
406
+ update_params["model"] = model
407
+
408
+ # Single update call if we have data to update
409
+ if update_params:
410
+ span.raw_span().update(**update_params)
398
411
 
399
412
 
400
413
  class LangfuseTracer(Tracer):
@@ -407,7 +420,7 @@ class LangfuseTracer(Tracer):
407
420
  tracer: langfuse.Langfuse,
408
421
  name: str = "Haystack",
409
422
  public: bool = False,
410
- span_handler: Optional[SpanHandler] = None,
423
+ span_handler: SpanHandler | None = None,
411
424
  ) -> None:
412
425
  """
413
426
  Initialize a LangfuseTracer instance.
@@ -437,7 +450,7 @@ class LangfuseTracer(Tracer):
437
450
 
438
451
  @contextlib.contextmanager
439
452
  def trace(
440
- self, operation_name: str, tags: Optional[dict[str, Any]] = None, parent_span: Optional[Span] = None
453
+ self, operation_name: str, tags: dict[str, Any] | None = None, parent_span: Span | None = None
441
454
  ) -> Iterator[Span]:
442
455
  tags = tags or {}
443
456
  span_name = tags.get(_COMPONENT_NAME_KEY, operation_name)
@@ -528,7 +541,7 @@ class LangfuseTracer(Tracer):
528
541
  def flush(self) -> None:
529
542
  self._tracer.flush()
530
543
 
531
- def current_span(self) -> Optional[Span]:
544
+ def current_span(self) -> Span | None:
532
545
  """
533
546
  Return the current active span.
534
547
 
@@ -6,7 +6,6 @@ import asyncio
6
6
  import datetime
7
7
  import logging
8
8
  import sys
9
- from typing import Optional
10
9
  from unittest.mock import MagicMock, Mock, patch
11
10
 
12
11
  import pytest
@@ -107,7 +106,7 @@ class MockLangfuseClient:
107
106
 
108
107
 
109
108
  class CustomSpanHandler(DefaultSpanHandler):
110
- def handle(self, span: LangfuseSpan, component_type: Optional[str]) -> None:
109
+ def handle(self, span: LangfuseSpan, component_type: str | None) -> None:
111
110
  if component_type == "OpenAIChatGenerator":
112
111
  output = span.get_data().get(_COMPONENT_OUTPUT_KEY, {})
113
112
  replies = output.get("replies", [])
@@ -204,20 +203,16 @@ class TestSanitizeUsageData:
204
203
  "completion_tokens": 449,
205
204
  }
206
205
  result = _sanitize_usage_data(usage)
207
- assert result == {
208
- "cache_creation.ephemeral_1h_input_tokens": 0,
209
- "cache_creation.ephemeral_5m_input_tokens": 0,
210
- "cache_creation_input_tokens": 0,
211
- "cache_read_input_tokens": 0,
212
- "prompt_tokens": 25,
213
- "completion_tokens": 449,
214
- }
206
+ assert result["input_tokens"] == 25
207
+ assert result["output_tokens"] == 449
215
208
 
216
209
  def test_openai_usage_preserved(self):
217
210
  """Test OpenAI/Cohere flat dict with only numeric values works unchanged"""
218
211
  usage = {"prompt_tokens": 29, "completion_tokens": 267, "total_tokens": 296}
219
212
  result = _sanitize_usage_data(usage)
220
- assert result == {"prompt_tokens": 29, "completion_tokens": 267, "total_tokens": 296}
213
+ assert result["input_tokens"] == 29
214
+ assert result["output_tokens"] == 267
215
+ assert result["total_tokens"] == 296
221
216
 
222
217
  def test_empty_and_invalid_input(self):
223
218
  """Test edge cases return empty dict"""
@@ -411,6 +406,91 @@ class TestDefaultSpanHandler:
411
406
  # Verify start_as_current_span was called for the actual span creation (not just parent)
412
407
  assert mock_client.start_as_current_span.call_count == 2 # Once for parent, once for the span
413
408
 
409
+ def test_handle_embedder_with_openai_format(self):
410
+ """Test that embedder usage is extracted in OpenAI format."""
411
+ mock_span = Mock()
412
+ mock_span.raw_span.return_value = mock_span
413
+ mock_span.get_data.return_value = {
414
+ "haystack.component.type": "OpenAITextEmbedder",
415
+ "haystack.component.output": {
416
+ "embedding": [0.1, 0.2, 0.3],
417
+ "meta": {"model": "custom-model", "usage": {"prompt_tokens": 15, "total_tokens": 15}},
418
+ },
419
+ }
420
+
421
+ handler = DefaultSpanHandler()
422
+ handler.handle(mock_span, component_type="OpenAITextEmbedder")
423
+
424
+ assert mock_span.update.call_count == 1
425
+ update_args = mock_span.update.call_args_list[0][1]
426
+ assert update_args["model"] == "custom-model"
427
+ assert update_args["usage_details"] == {"input_tokens": 15, "total_tokens": 15}
428
+
429
+ def test_handle_embedder_with_cohere_format(self):
430
+ """Test that embedder usage is extracted in Cohere billed_units format."""
431
+ mock_span = Mock()
432
+ mock_span.raw_span.return_value = mock_span
433
+ mock_span.get_data.return_value = {
434
+ "haystack.component.type": "CohereTextEmbedder",
435
+ "haystack.component.output": {
436
+ "embedding": [0.1, 0.2, 0.3],
437
+ "meta": {"api_version": {"version": "1"}, "billed_units": {"input_tokens": 4}},
438
+ },
439
+ }
440
+
441
+ handler = DefaultSpanHandler()
442
+ handler.handle(mock_span, component_type="CohereTextEmbedder")
443
+
444
+ assert mock_span.update.call_count == 1
445
+ assert mock_span.update.call_args_list[0][1] == {"usage_details": {"input_tokens": 4}}
446
+
447
+ def test_handle_embedder_without_usage(self):
448
+ """Test that embedders without usage data are handled gracefully."""
449
+ mock_span = Mock()
450
+ mock_span.raw_span.return_value = mock_span
451
+ mock_span.get_data.return_value = {
452
+ "haystack.component.type": "SentenceTransformersTextEmbedder",
453
+ "haystack.component.output": {
454
+ "embedding": [0.1, 0.2, 0.3],
455
+ "meta": {}, # No usage data
456
+ },
457
+ }
458
+
459
+ handler = DefaultSpanHandler()
460
+ handler.handle(mock_span, component_type="SentenceTransformersTextEmbedder")
461
+
462
+ # Should not call update when no usage data is available
463
+ assert mock_span.update.call_count == 0
464
+
465
+ def test_handle_embedder_with_nested_usage(self):
466
+ """Test that embedders with nested usage data are sanitized correctly."""
467
+ mock_span = Mock()
468
+ mock_span.raw_span.return_value = mock_span
469
+ mock_span.get_data.return_value = {
470
+ "haystack.component.type": "CustomEmbedder",
471
+ "haystack.component.output": {
472
+ "embedding": [0.1, 0.2, 0.3],
473
+ "meta": {
474
+ "model": "custom-model",
475
+ "usage": {
476
+ "cache_creation": {"input_tokens": 10},
477
+ "cache_read": {"input_tokens": 5},
478
+ "total_tokens": 15,
479
+ },
480
+ },
481
+ },
482
+ }
483
+
484
+ handler = DefaultSpanHandler()
485
+ handler.handle(mock_span, component_type="CustomEmbedder")
486
+
487
+ assert mock_span.update.call_count == 1
488
+ # Only adds total_tokens as Langfuse standard key (no prompt_tokens/completion_tokens to convert)
489
+ assert mock_span.update.call_args_list[0][1] == {
490
+ "usage_details": {"total_tokens": 15},
491
+ "model": "custom-model",
492
+ }
493
+
414
494
 
415
495
  class TestCustomSpanHandler:
416
496
  def test_handle(self):
@@ -454,13 +534,22 @@ class TestLangfuseTracer:
454
534
  mock_raw_span.metadata = {"tag1": "value1", "tag2": "value2"}
455
535
 
456
536
  with patch("haystack_integrations.tracing.langfuse.tracer.LangfuseSpan") as mock_langfuse_span:
537
+ mock_context_manager = MockContextManager()
538
+ mock_context_manager._span = mock_raw_span
539
+
457
540
  mock_span_instance = mock_langfuse_span.return_value
458
541
  mock_span_instance.raw_span.return_value = mock_raw_span
542
+ # Return a proper dict to prevent MagicMock from being truthy in handle() checks.
543
+ # When get_data() returns a MagicMock, `span.get_data().get(key) is not None` is True
544
+ # because MagicMock().get() returns another MagicMock (truthy). This triggers
545
+ # tracing_utils.coerce_tag_value() with MagicMock objects, which can hang on
546
+ # Linux Python 3.9/3.13 due to platform-specific MagicMock iteration behavior.
547
+ mock_span_instance.get_data.return_value = {}
548
+ mock_span_instance._context_manager = mock_context_manager
459
549
 
460
- mock_context_manager = MockContextManager()
461
- mock_context_manager._span = mock_raw_span
462
550
  mock_tracer = MagicMock()
463
551
  mock_tracer.start_as_current_span.return_value = mock_context_manager
552
+ mock_tracer.start_as_current_observation.return_value = mock_context_manager
464
553
 
465
554
  tracer = LangfuseTracer(tracer=mock_tracer, name="Haystack", public=False)
466
555
 
@@ -38,7 +38,7 @@ os.environ.setdefault("LANGFUSE_HOST", "https://cloud.langfuse.com")
38
38
  def poll_langfuse(url: str):
39
39
  """Utility function to poll Langfuse API until the trace is ready"""
40
40
  # Initial wait for trace creation
41
- time.sleep(10)
41
+ time.sleep(30)
42
42
 
43
43
  auth = HTTPBasicAuth(os.environ["LANGFUSE_PUBLIC_KEY"], os.environ["LANGFUSE_SECRET_KEY"])
44
44
 
@@ -195,8 +195,8 @@ def test_tracing_with_sub_pipelines():
195
195
  # There should be two observations for the haystack.pipeline.run span: one for each sub pipeline
196
196
  # Main pipeline is stored under the name "Sub-pipeline example"
197
197
  assert len(haystack_pipeline_run_observations) == 2
198
- assert "prompt_builder" in str(haystack_pipeline_run_observations[0])
199
- assert "llm" in str(haystack_pipeline_run_observations[1])
198
+ # Verify both observations are pipeline runs (less brittle than checking for component names)
199
+ assert all(obs["name"] == "haystack.pipeline.run" for obs in haystack_pipeline_run_observations)
200
200
 
201
201
 
202
202
  @pytest.mark.skipif(
@@ -1,30 +0,0 @@
1
- loaders:
2
- - type: haystack_pydoc_tools.loaders.CustomPythonLoader
3
- search_path: [../src]
4
- modules: [
5
- "haystack_integrations.components.connectors.langfuse.langfuse_connector",
6
- "haystack_integrations.tracing.langfuse.tracer",
7
- ]
8
- ignore_when_discovered: ["__init__"]
9
- processors:
10
- - type: filter
11
- expression:
12
- documented_only: true
13
- do_not_filter_modules: false
14
- skip_empty_modules: true
15
- - type: smart
16
- - type: crossref
17
- renderer:
18
- type: haystack_pydoc_tools.renderers.ReadmeIntegrationRenderer
19
- excerpt: Langfuse integration for Haystack
20
- category_slug: integrations-api
21
- title: langfuse
22
- slug: integrations-langfuse
23
- order: 136
24
- markdown:
25
- descriptive_class_title: false
26
- classdef_code_block: false
27
- descriptive_module_title: true
28
- add_method_class_prefix: true
29
- add_member_class_prefix: false
30
- filename: _readme_langfuse.md