langfuse-haystack 3.1.0__tar.gz → 3.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (22) hide show
  1. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/CHANGELOG.md +19 -0
  2. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/PKG-INFO +1 -1
  3. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/pyproject.toml +1 -1
  4. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/tracing/langfuse/tracer.py +56 -6
  5. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/tests/test_tracer.py +41 -5
  6. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/.gitignore +0 -0
  7. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/LICENSE.txt +0 -0
  8. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/README.md +0 -0
  9. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/example/basic_rag.py +0 -0
  10. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/example/chat.py +0 -0
  11. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/example/requirements.txt +0 -0
  12. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/pydoc/config.yml +0 -0
  13. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/pydoc/config_docusaurus.yml +0 -0
  14. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/components/connectors/__init__.py +0 -0
  15. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/components/connectors/langfuse/__init__.py +0 -0
  16. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/components/connectors/langfuse/langfuse_connector.py +0 -0
  17. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/components/connectors/py.typed +0 -0
  18. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/tracing/langfuse/__init__.py +0 -0
  19. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/tracing/py.typed +0 -0
  20. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/tests/__init__.py +0 -0
  21. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/tests/test_langfuse_connector.py +0 -0
  22. {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/tests/test_tracing.py +0 -0
@@ -1,5 +1,24 @@
1
1
  # Changelog
2
2
 
3
+ ## [integrations/langfuse-v3.1.0] - 2025-10-24
4
+
5
+ ### 🐛 Bug Fixes
6
+
7
+ - Langfuse - add py.typed; fix testing with lowest deps (#2458)
8
+
9
+ ### 📚 Documentation
10
+
11
+ - Add pydoc configurations for Docusaurus (#2411)
12
+
13
+ ### ⚙️ CI
14
+
15
+ - Install dependencies in the `test` environment when testing with lowest direct dependencies and Haystack main (#2418)
16
+
17
+ ### 🧹 Chores
18
+
19
+ - Remove ruff exclude and fix linting in Langfuse integration (#2257)
20
+
21
+
3
22
  ## [integrations/langfuse-v3.0.0] - 2025-09-19
4
23
 
5
24
  ### 🌀 Miscellaneous
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: langfuse-haystack
3
- Version: 3.1.0
3
+ Version: 3.2.0
4
4
  Summary: Langfuse integration for Haystack
5
5
  Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/langfuse#readme
6
6
  Project-URL: Issues, https://github.com/deepset-ai/haystack-core-integrations/issues
@@ -67,7 +67,7 @@ dependencies = [
67
67
  unit = 'pytest -m "not integration" {args:tests}'
68
68
  integration = 'pytest -m "integration" {args:tests}'
69
69
  all = 'pytest {args:tests}'
70
- cov-retry = 'all --cov=haystack_integrations --reruns 3 --reruns-delay 30 -x'
70
+ cov-retry = 'pytest --cov=haystack_integrations --reruns 3 --reruns-delay 30 -x {args:tests}'
71
71
 
72
72
  types = "mypy -p haystack_integrations.components.connectors.langfuse -p haystack_integrations.tracing.langfuse {args}"
73
73
 
@@ -10,7 +10,7 @@ from contextlib import AbstractContextManager
10
10
  from contextvars import ContextVar
11
11
  from dataclasses import dataclass
12
12
  from datetime import datetime
13
- from typing import Any, Dict, Iterator, List, Optional
13
+ from typing import Any, Dict, Iterator, List, Literal, Optional
14
14
 
15
15
  from haystack import default_from_dict, default_to_dict, logging
16
16
  from haystack.dataclasses import ChatMessage
@@ -256,6 +256,46 @@ class SpanHandler(ABC):
256
256
  return default_to_dict(self)
257
257
 
258
258
 
259
+ def _sanitize_usage_data(usage: Dict[str, Any]) -> Dict[str, Any]:
260
+ """
261
+ Sanitize usage data for Langfuse by flattening to a single-level dictionary.
262
+
263
+ Langfuse's usage_details must be a flat dictionary with only numeric values. This function:
264
+ - Flattens nested dictionaries using dot notation (e.g., cache_creation.input_tokens)
265
+ - Keeps int and float values
266
+ - Skips None, boolean, string, and other non-numeric types
267
+
268
+ :param usage: Raw usage dictionary from the provider.
269
+ :returns: Flat dictionary with only numeric values (int or float).
270
+ """
271
+ if not isinstance(usage, dict):
272
+ return {}
273
+
274
+ sanitized: Dict[str, Any] = {}
275
+
276
+ def _flatten(data: Dict[str, Any], prefix: str = "") -> None:
277
+ """Recursively flatten nested dictionaries."""
278
+ for key, value in data.items():
279
+ full_key = f"{prefix}.{key}" if prefix else key
280
+
281
+ if value is None:
282
+ # Skip None values (e.g., Anthropic's server_tool_use)
283
+ continue
284
+ elif isinstance(value, bool):
285
+ # Skip boolean values
286
+ continue
287
+ elif isinstance(value, (int, float)):
288
+ # Keep numeric values
289
+ sanitized[full_key] = value
290
+ elif isinstance(value, dict):
291
+ # Recursively flatten nested dicts
292
+ _flatten(value, full_key)
293
+ # Skip strings and other non-numeric types (e.g., Anthropic's service_tier)
294
+
295
+ _flatten(usage)
296
+ return sanitized
297
+
298
+
259
299
  class DefaultSpanHandler(SpanHandler):
260
300
  """DefaultSpanHandler provides the default Langfuse tracing behavior for Haystack."""
261
301
 
@@ -271,10 +311,12 @@ class DefaultSpanHandler(SpanHandler):
271
311
  # Get external tracing context for root trace creation (correlation metadata)
272
312
  tracing_ctx = tracing_context_var.get({})
273
313
  if not context.parent_span:
314
+ root_span_type: Literal["agent", "span"] = (
315
+ "agent" if context.operation_name == "haystack.agent.run" else "span"
316
+ )
274
317
  # Create a new trace when there's no parent span
275
- span_context_manager = self.tracer.start_as_current_span(
276
- name=context.trace_name,
277
- version=tracing_ctx.get("version"),
318
+ span_context_manager = self.tracer.start_as_current_observation(
319
+ name=context.trace_name, version=tracing_ctx.get("version"), as_type=root_span_type
278
320
  )
279
321
 
280
322
  # Create LangfuseSpan which will handle entering the context manager
@@ -298,6 +340,10 @@ class DefaultSpanHandler(SpanHandler):
298
340
  span._span.update_trace(**trace_attrs)
299
341
 
300
342
  return span
343
+ elif context.component_type == "ToolInvoker":
344
+ return LangfuseSpan(self.tracer.start_as_current_observation(name=context.name, as_type="tool"))
345
+ elif context.operation_name == "haystack.agent.run":
346
+ return LangfuseSpan(self.tracer.start_as_current_observation(name=context.name, as_type="agent"))
301
347
  elif context.component_type in _ALL_SUPPORTED_GENERATORS:
302
348
  return LangfuseSpan(self.tracer.start_as_current_observation(name=context.name, as_type="generation"))
303
349
  else:
@@ -328,7 +374,9 @@ class DefaultSpanHandler(SpanHandler):
328
374
  if component_type in _SUPPORTED_GENERATORS:
329
375
  meta = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("meta")
330
376
  if meta:
331
- span.raw_span().update(usage=meta[0].get("usage") or None, model=meta[0].get("model"))
377
+ usage = meta[0].get("usage")
378
+ sanitized_usage = _sanitize_usage_data(usage) if usage else None
379
+ span.raw_span().update(usage_details=sanitized_usage, model=meta[0].get("model"))
332
380
 
333
381
  if component_type in _SUPPORTED_CHAT_GENERATORS:
334
382
  replies = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("replies")
@@ -341,8 +389,10 @@ class DefaultSpanHandler(SpanHandler):
341
389
  except ValueError:
342
390
  logger.error(f"Failed to parse completion_start_time: {completion_start_time}")
343
391
  completion_start_time = None
392
+ usage = meta.get("usage")
393
+ sanitized_usage = _sanitize_usage_data(usage) if usage else None
344
394
  span.raw_span().update(
345
- usage=meta.get("usage") or None,
395
+ usage_details=sanitized_usage,
346
396
  model=meta.get("model"),
347
397
  completion_start_time=completion_start_time,
348
398
  )
@@ -18,6 +18,7 @@ from haystack_integrations.tracing.langfuse.tracer import (
18
18
  LangfuseSpan,
19
19
  LangfuseTracer,
20
20
  SpanContext,
21
+ _sanitize_usage_data,
21
22
  )
22
23
 
23
24
 
@@ -190,6 +191,41 @@ class TestSpanContext:
190
191
  )
191
192
 
192
193
 
194
+ class TestSanitizeUsageData:
195
+ def test_anthropic_usage_flattens_and_filters(self):
196
+ """Test Anthropic's nested dict with None and strings gets flattened and filtered"""
197
+ usage = {
198
+ "cache_creation": {"ephemeral_1h_input_tokens": 0, "ephemeral_5m_input_tokens": 0},
199
+ "cache_creation_input_tokens": 0,
200
+ "cache_read_input_tokens": 0,
201
+ "server_tool_use": None, # Should be filtered
202
+ "service_tier": "standard", # Should be filtered
203
+ "prompt_tokens": 25,
204
+ "completion_tokens": 449,
205
+ }
206
+ result = _sanitize_usage_data(usage)
207
+ assert result == {
208
+ "cache_creation.ephemeral_1h_input_tokens": 0,
209
+ "cache_creation.ephemeral_5m_input_tokens": 0,
210
+ "cache_creation_input_tokens": 0,
211
+ "cache_read_input_tokens": 0,
212
+ "prompt_tokens": 25,
213
+ "completion_tokens": 449,
214
+ }
215
+
216
+ def test_openai_usage_preserved(self):
217
+ """Test OpenAI/Cohere flat dict with only numeric values works unchanged"""
218
+ usage = {"prompt_tokens": 29, "completion_tokens": 267, "total_tokens": 296}
219
+ result = _sanitize_usage_data(usage)
220
+ assert result == {"prompt_tokens": 29, "completion_tokens": 267, "total_tokens": 296}
221
+
222
+ def test_empty_and_invalid_input(self):
223
+ """Test edge cases return empty dict"""
224
+ assert _sanitize_usage_data({}) == {}
225
+ assert _sanitize_usage_data(None) == {}
226
+ assert _sanitize_usage_data({"only_strings": "value", "only_none": None}) == {}
227
+
228
+
193
229
  class TestDefaultSpanHandler:
194
230
  def test_handle_generator(self):
195
231
  mock_span = Mock()
@@ -203,7 +239,7 @@ class TestDefaultSpanHandler:
203
239
  handler.handle(mock_span, component_type="OpenAIGenerator")
204
240
 
205
241
  assert mock_span.update.call_count == 1
206
- assert mock_span.update.call_args_list[0][1] == {"usage": None, "model": "test_model"}
242
+ assert mock_span.update.call_args_list[0][1] == {"usage_details": None, "model": "test_model"}
207
243
 
208
244
  def test_handle_chat_generator(self):
209
245
  mock_span = Mock()
@@ -225,7 +261,7 @@ class TestDefaultSpanHandler:
225
261
 
226
262
  assert mock_span.update.call_count == 1
227
263
  assert mock_span.update.call_args_list[0][1] == {
228
- "usage": None,
264
+ "usage_details": None,
229
265
  "model": "test_model",
230
266
  "completion_start_time": datetime.datetime( # noqa: DTZ001
231
267
  2021, 7, 27, 16, 2, 8, 12345
@@ -254,7 +290,7 @@ class TestDefaultSpanHandler:
254
290
 
255
291
  assert mock_span.update.call_count == 1
256
292
  assert mock_span.update.call_args_list[0][1] == {
257
- "usage": None,
293
+ "usage_details": None,
258
294
  "model": "test_model",
259
295
  "completion_start_time": None,
260
296
  }
@@ -348,7 +384,7 @@ class TestLangfuseTracer:
348
384
  }
349
385
  with tracer.trace(operation_name="operation_name", tags=tags) as span:
350
386
  ...
351
- assert span.raw_span()._data["usage"] is None
387
+ assert span.raw_span()._data["usage_details"] is None
352
388
  assert span.raw_span()._data["model"] == "test_model"
353
389
  assert span.raw_span()._data["completion_start_time"] == datetime.datetime(2021, 7, 27, 16, 2, 8, 12345) # noqa: DTZ001
354
390
 
@@ -415,7 +451,7 @@ class TestLangfuseTracer:
415
451
  }
416
452
  with tracer.trace(operation_name="operation_name", tags=tags) as span:
417
453
  ...
418
- assert span.raw_span()._data["usage"] is None
454
+ assert span.raw_span()._data["usage_details"] is None
419
455
  assert span.raw_span()._data["model"] == "test_model"
420
456
  assert span.raw_span()._data["completion_start_time"] is None
421
457