langfuse-haystack 3.3.0__tar.gz → 3.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (22) hide show
  1. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/CHANGELOG.md +11 -0
  2. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/PKG-INFO +1 -1
  3. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/pyproject.toml +1 -1
  4. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/src/haystack_integrations/tracing/langfuse/tracer.py +41 -28
  5. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/tests/test_tracer.py +101 -11
  6. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/tests/test_tracing.py +3 -3
  7. langfuse_haystack-3.3.0/pydoc/config.yml +0 -30
  8. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/.gitignore +0 -0
  9. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/LICENSE.txt +0 -0
  10. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/README.md +0 -0
  11. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/example/basic_rag.py +0 -0
  12. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/example/chat.py +0 -0
  13. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/example/requirements.txt +0 -0
  14. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/pydoc/config_docusaurus.yml +0 -0
  15. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/src/haystack_integrations/components/connectors/__init__.py +0 -0
  16. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/src/haystack_integrations/components/connectors/langfuse/__init__.py +0 -0
  17. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/src/haystack_integrations/components/connectors/langfuse/langfuse_connector.py +0 -0
  18. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/src/haystack_integrations/components/connectors/py.typed +0 -0
  19. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/src/haystack_integrations/tracing/langfuse/__init__.py +0 -0
  20. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/src/haystack_integrations/tracing/py.typed +0 -0
  21. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/tests/__init__.py +0 -0
  22. {langfuse_haystack-3.3.0 → langfuse_haystack-3.3.1}/tests/test_langfuse_connector.py +0 -0
@@ -1,5 +1,16 @@
1
1
  # Changelog
2
2
 
3
+ ## [integrations/langfuse-v3.3.0] - 2025-11-21
4
+
5
+ ### 🚀 Features
6
+
7
+ - *(langfuse)* Embedder, retriever and generator as obs. type (#2497)
8
+
9
+ ### 🌀 Miscellaneous
10
+
11
+ - Enhancement: Adopt PEP 585 type hinting (part 4) (#2527)
12
+ - *(langfuse)* Log levels (#2522)
13
+
3
14
  ## [integrations/langfuse-v3.2.1] - 2025-11-07
4
15
 
5
16
  ### 🌀 Miscellaneous
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: langfuse-haystack
3
- Version: 3.3.0
3
+ Version: 3.3.1
4
4
  Summary: Langfuse integration for Haystack
5
5
  Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/langfuse#readme
6
6
  Project-URL: Issues, https://github.com/deepset-ai/haystack-core-integrations/issues
@@ -46,7 +46,7 @@ installer = "uv"
46
46
  dependencies = ["haystack-pydoc-tools", "ruff"]
47
47
 
48
48
  [tool.hatch.envs.default.scripts]
49
- docs = ["pydoc-markdown pydoc/config.yml"]
49
+ docs = ["pydoc-markdown pydoc/config_docusaurus.yml"]
50
50
  fmt = "ruff check --fix {args} && ruff format {args}"
51
51
  fmt-check = "ruff check {args} && ruff format --check {args}"
52
52
 
@@ -242,41 +242,30 @@ class SpanHandler(ABC):
242
242
 
243
243
  def _sanitize_usage_data(usage: dict[str, Any]) -> dict[str, Any]:
244
244
  """
245
- Sanitize usage data for Langfuse by flattening to a single-level dictionary.
245
+ Sanitize usage data for Langfuse by converting provider-specific keys to Langfuse standard keys.
246
246
 
247
- Langfuse's usage_details must be a flat dictionary with only numeric values. This function:
248
- - Flattens nested dictionaries using dot notation (e.g., cache_creation.input_tokens)
249
- - Keeps int and float values
250
- - Skips None, boolean, string, and other non-numeric types
247
+ Langfuse expects usage_details with standard keys: input_tokens, output_tokens, and total_tokens.
248
+ This function converts provider-specific keys to Langfuse's expected format:
249
+ - prompt_tokens -> input_tokens
250
+ - completion_tokens -> output_tokens
251
+ - total_tokens -> total_tokens (preserved as-is)
251
252
 
252
253
  :param usage: Raw usage dictionary from the provider.
253
- :returns: Flat dictionary with only numeric values (int or float).
254
+ :returns: Dictionary with Langfuse standard keys (input_tokens, output_tokens, total_tokens).
254
255
  """
255
256
  if not isinstance(usage, dict):
256
257
  return {}
257
258
 
258
- sanitized: dict[str, Any] = {}
259
-
260
- def _flatten(data: dict[str, Any], prefix: str = "") -> None:
261
- """Recursively flatten nested dictionaries."""
262
- for key, value in data.items():
263
- full_key = f"{prefix}.{key}" if prefix else key
264
-
265
- if value is None:
266
- # Skip None values (e.g., Anthropic's server_tool_use)
267
- continue
268
- elif isinstance(value, bool):
269
- # Skip boolean values
270
- continue
271
- elif isinstance(value, (int, float)):
272
- # Keep numeric values
273
- sanitized[full_key] = value
274
- elif isinstance(value, dict):
275
- # Recursively flatten nested dicts
276
- _flatten(value, full_key)
277
- # Skip strings and other non-numeric types (e.g., Anthropic's service_tier)
278
-
279
- _flatten(usage)
259
+ # Start with Langfuse standard keys from usage if present
260
+ sanitized: dict[str, Any] = {
261
+ k: v for k, v in usage.items() if k in ("input_tokens", "output_tokens", "total_tokens")
262
+ }
263
+ # Convert provider format to Langfuse standard keys if not already present
264
+ if "input_tokens" not in sanitized and "prompt_tokens" in usage:
265
+ sanitized["input_tokens"] = usage["prompt_tokens"]
266
+ if "output_tokens" not in sanitized and "completion_tokens" in usage:
267
+ sanitized["output_tokens"] = usage["completion_tokens"]
268
+
280
269
  return sanitized
281
270
 
282
271
 
@@ -395,6 +384,30 @@ class DefaultSpanHandler(SpanHandler):
395
384
  usage = meta[0].get("usage")
396
385
  sanitized_usage = _sanitize_usage_data(usage) if usage else None
397
386
  span.raw_span().update(usage_details=sanitized_usage, model=meta[0].get("model"))
387
+ elif component_type and component_type.endswith("Embedder"):
388
+ # Extract usage data from embedder output
389
+ output = span.get_data().get(_COMPONENT_OUTPUT_KEY, {})
390
+ meta = output.get("meta")
391
+
392
+ if meta and isinstance(meta, dict):
393
+ # Build update parameters with available data
394
+ update_params: dict[str, Any] = {}
395
+
396
+ # Try both common formats: 'usage' (OpenAI) or 'billed_units' (Cohere)
397
+ usage = meta.get("usage") or meta.get("billed_units")
398
+ if usage:
399
+ sanitized_usage = _sanitize_usage_data(usage)
400
+ if sanitized_usage:
401
+ update_params["usage_details"] = sanitized_usage
402
+
403
+ # Some embedders may provide model information
404
+ model = meta.get("model")
405
+ if model and isinstance(model, str):
406
+ update_params["model"] = model
407
+
408
+ # Single update call if we have data to update
409
+ if update_params:
410
+ span.raw_span().update(**update_params)
398
411
 
399
412
 
400
413
  class LangfuseTracer(Tracer):
@@ -204,20 +204,16 @@ class TestSanitizeUsageData:
204
204
  "completion_tokens": 449,
205
205
  }
206
206
  result = _sanitize_usage_data(usage)
207
- assert result == {
208
- "cache_creation.ephemeral_1h_input_tokens": 0,
209
- "cache_creation.ephemeral_5m_input_tokens": 0,
210
- "cache_creation_input_tokens": 0,
211
- "cache_read_input_tokens": 0,
212
- "prompt_tokens": 25,
213
- "completion_tokens": 449,
214
- }
207
+ assert result["input_tokens"] == 25
208
+ assert result["output_tokens"] == 449
215
209
 
216
210
  def test_openai_usage_preserved(self):
217
211
  """Test OpenAI/Cohere flat dict with only numeric values works unchanged"""
218
212
  usage = {"prompt_tokens": 29, "completion_tokens": 267, "total_tokens": 296}
219
213
  result = _sanitize_usage_data(usage)
220
- assert result == {"prompt_tokens": 29, "completion_tokens": 267, "total_tokens": 296}
214
+ assert result["input_tokens"] == 29
215
+ assert result["output_tokens"] == 267
216
+ assert result["total_tokens"] == 296
221
217
 
222
218
  def test_empty_and_invalid_input(self):
223
219
  """Test edge cases return empty dict"""
@@ -411,6 +407,91 @@ class TestDefaultSpanHandler:
411
407
  # Verify start_as_current_span was called for the actual span creation (not just parent)
412
408
  assert mock_client.start_as_current_span.call_count == 2 # Once for parent, once for the span
413
409
 
410
+ def test_handle_embedder_with_openai_format(self):
411
+ """Test that embedder usage is extracted in OpenAI format."""
412
+ mock_span = Mock()
413
+ mock_span.raw_span.return_value = mock_span
414
+ mock_span.get_data.return_value = {
415
+ "haystack.component.type": "OpenAITextEmbedder",
416
+ "haystack.component.output": {
417
+ "embedding": [0.1, 0.2, 0.3],
418
+ "meta": {"model": "custom-model", "usage": {"prompt_tokens": 15, "total_tokens": 15}},
419
+ },
420
+ }
421
+
422
+ handler = DefaultSpanHandler()
423
+ handler.handle(mock_span, component_type="OpenAITextEmbedder")
424
+
425
+ assert mock_span.update.call_count == 1
426
+ update_args = mock_span.update.call_args_list[0][1]
427
+ assert update_args["model"] == "custom-model"
428
+ assert update_args["usage_details"] == {"input_tokens": 15, "total_tokens": 15}
429
+
430
+ def test_handle_embedder_with_cohere_format(self):
431
+ """Test that embedder usage is extracted in Cohere billed_units format."""
432
+ mock_span = Mock()
433
+ mock_span.raw_span.return_value = mock_span
434
+ mock_span.get_data.return_value = {
435
+ "haystack.component.type": "CohereTextEmbedder",
436
+ "haystack.component.output": {
437
+ "embedding": [0.1, 0.2, 0.3],
438
+ "meta": {"api_version": {"version": "1"}, "billed_units": {"input_tokens": 4}},
439
+ },
440
+ }
441
+
442
+ handler = DefaultSpanHandler()
443
+ handler.handle(mock_span, component_type="CohereTextEmbedder")
444
+
445
+ assert mock_span.update.call_count == 1
446
+ assert mock_span.update.call_args_list[0][1] == {"usage_details": {"input_tokens": 4}}
447
+
448
+ def test_handle_embedder_without_usage(self):
449
+ """Test that embedders without usage data are handled gracefully."""
450
+ mock_span = Mock()
451
+ mock_span.raw_span.return_value = mock_span
452
+ mock_span.get_data.return_value = {
453
+ "haystack.component.type": "SentenceTransformersTextEmbedder",
454
+ "haystack.component.output": {
455
+ "embedding": [0.1, 0.2, 0.3],
456
+ "meta": {}, # No usage data
457
+ },
458
+ }
459
+
460
+ handler = DefaultSpanHandler()
461
+ handler.handle(mock_span, component_type="SentenceTransformersTextEmbedder")
462
+
463
+ # Should not call update when no usage data is available
464
+ assert mock_span.update.call_count == 0
465
+
466
+ def test_handle_embedder_with_nested_usage(self):
467
+ """Test that embedders with nested usage data are sanitized correctly."""
468
+ mock_span = Mock()
469
+ mock_span.raw_span.return_value = mock_span
470
+ mock_span.get_data.return_value = {
471
+ "haystack.component.type": "CustomEmbedder",
472
+ "haystack.component.output": {
473
+ "embedding": [0.1, 0.2, 0.3],
474
+ "meta": {
475
+ "model": "custom-model",
476
+ "usage": {
477
+ "cache_creation": {"input_tokens": 10},
478
+ "cache_read": {"input_tokens": 5},
479
+ "total_tokens": 15,
480
+ },
481
+ },
482
+ },
483
+ }
484
+
485
+ handler = DefaultSpanHandler()
486
+ handler.handle(mock_span, component_type="CustomEmbedder")
487
+
488
+ assert mock_span.update.call_count == 1
489
+ # Only adds total_tokens as Langfuse standard key (no prompt_tokens/completion_tokens to convert)
490
+ assert mock_span.update.call_args_list[0][1] == {
491
+ "usage_details": {"total_tokens": 15},
492
+ "model": "custom-model",
493
+ }
494
+
414
495
 
415
496
  class TestCustomSpanHandler:
416
497
  def test_handle(self):
@@ -454,13 +535,22 @@ class TestLangfuseTracer:
454
535
  mock_raw_span.metadata = {"tag1": "value1", "tag2": "value2"}
455
536
 
456
537
  with patch("haystack_integrations.tracing.langfuse.tracer.LangfuseSpan") as mock_langfuse_span:
538
+ mock_context_manager = MockContextManager()
539
+ mock_context_manager._span = mock_raw_span
540
+
457
541
  mock_span_instance = mock_langfuse_span.return_value
458
542
  mock_span_instance.raw_span.return_value = mock_raw_span
543
+ # Return a proper dict to prevent MagicMock from being truthy in handle() checks.
544
+ # When get_data() returns a MagicMock, `span.get_data().get(key) is not None` is True
545
+ # because MagicMock().get() returns another MagicMock (truthy). This triggers
546
+ # tracing_utils.coerce_tag_value() with MagicMock objects, which can hang on
547
+ # Linux Python 3.9/3.13 due to platform-specific MagicMock iteration behavior.
548
+ mock_span_instance.get_data.return_value = {}
549
+ mock_span_instance._context_manager = mock_context_manager
459
550
 
460
- mock_context_manager = MockContextManager()
461
- mock_context_manager._span = mock_raw_span
462
551
  mock_tracer = MagicMock()
463
552
  mock_tracer.start_as_current_span.return_value = mock_context_manager
553
+ mock_tracer.start_as_current_observation.return_value = mock_context_manager
464
554
 
465
555
  tracer = LangfuseTracer(tracer=mock_tracer, name="Haystack", public=False)
466
556
 
@@ -38,7 +38,7 @@ os.environ.setdefault("LANGFUSE_HOST", "https://cloud.langfuse.com")
38
38
  def poll_langfuse(url: str):
39
39
  """Utility function to poll Langfuse API until the trace is ready"""
40
40
  # Initial wait for trace creation
41
- time.sleep(10)
41
+ time.sleep(30)
42
42
 
43
43
  auth = HTTPBasicAuth(os.environ["LANGFUSE_PUBLIC_KEY"], os.environ["LANGFUSE_SECRET_KEY"])
44
44
 
@@ -195,8 +195,8 @@ def test_tracing_with_sub_pipelines():
195
195
  # There should be two observations for the haystack.pipeline.run span: one for each sub pipeline
196
196
  # Main pipeline is stored under the name "Sub-pipeline example"
197
197
  assert len(haystack_pipeline_run_observations) == 2
198
- assert "prompt_builder" in str(haystack_pipeline_run_observations[0])
199
- assert "llm" in str(haystack_pipeline_run_observations[1])
198
+ # Verify both observations are pipeline runs (less brittle than checking for component names)
199
+ assert all(obs["name"] == "haystack.pipeline.run" for obs in haystack_pipeline_run_observations)
200
200
 
201
201
 
202
202
  @pytest.mark.skipif(
@@ -1,30 +0,0 @@
1
- loaders:
2
- - type: haystack_pydoc_tools.loaders.CustomPythonLoader
3
- search_path: [../src]
4
- modules: [
5
- "haystack_integrations.components.connectors.langfuse.langfuse_connector",
6
- "haystack_integrations.tracing.langfuse.tracer",
7
- ]
8
- ignore_when_discovered: ["__init__"]
9
- processors:
10
- - type: filter
11
- expression:
12
- documented_only: true
13
- do_not_filter_modules: false
14
- skip_empty_modules: true
15
- - type: smart
16
- - type: crossref
17
- renderer:
18
- type: haystack_pydoc_tools.renderers.ReadmeIntegrationRenderer
19
- excerpt: Langfuse integration for Haystack
20
- category_slug: integrations-api
21
- title: langfuse
22
- slug: integrations-langfuse
23
- order: 136
24
- markdown:
25
- descriptive_class_title: false
26
- classdef_code_block: false
27
- descriptive_module_title: true
28
- add_method_class_prefix: true
29
- add_member_class_prefix: false
30
- filename: _readme_langfuse.md